- New plants/services/vpc.py: search VPC (vasteplantencatalogus.nl) by keyword and scrape full plant detail on select - Both searches run in parallel via ThreadPoolExecutor - species_results partial shows results from both sources with colour-coded source badges (green = Perenual, blue = VPC) - vpc_select view scrapes the detail page on select, creates Species with bloom_months, max_height_cm, sunlight from the VPC page - Species.vpc_slug field (migration 0005) prevents re-importing the same plant - beautifulsoup4 added to requirements - Handles VPC unavailability gracefully (empty results, no error shown to user) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
196 lines
6.1 KiB
Python
196 lines
6.1 KiB
Python
"""
|
||
Scraper for vasteplantencatalogus.nl (VPC).
|
||
- search_species(query) → list of lightweight result dicts
|
||
- scrape_plant(slug) → full detail dict ready to store as Species
|
||
|
||
Notes on VPC data:
|
||
- Plant titles ARE the Latin names (no separate common name)
|
||
- Bloom months listed in Dutch ("september, oktober")
|
||
- Height in a dedicated "Hoogte in cm" field as plain integer
|
||
- Base URL is https://vasteplantencatalogus.nl (no www)
|
||
"""
|
||
import re
|
||
import requests
|
||
from bs4 import BeautifulSoup
|
||
|
||
_BASE = 'https://vasteplantencatalogus.nl'
|
||
_TIMEOUT = 8
|
||
_HEADERS = {'User-Agent': 'PlantDB/1.0 (personal plant tracker)'}
|
||
|
||
_NL_MONTHS = {
|
||
'januari': 1, 'februari': 2, 'maart': 3, 'april': 4,
|
||
'mei': 5, 'juni': 6, 'juli': 7, 'augustus': 8,
|
||
'september': 9, 'oktober': 10, 'november': 11, 'december': 12,
|
||
}
|
||
|
||
# Also handle Roman numerals (some older pages still use them)
|
||
_ROMAN_MAP = {
|
||
'XII': 12, 'VIII': 8, 'VII': 7, 'VI': 6, 'XI': 11,
|
||
'IX': 9, 'IV': 4, 'III': 3, 'II': 2, 'X': 10, 'V': 5, 'I': 1,
|
||
}
|
||
|
||
|
||
class VPCError(Exception):
|
||
pass
|
||
|
||
|
||
def _parse_months(text):
|
||
"""Parse Dutch or Roman numeral month strings into a sorted list of ints."""
|
||
result = set()
|
||
text_lower = text.lower().replace('–', '-').replace('—', '-')
|
||
|
||
# Try Dutch names first
|
||
for name, month in _NL_MONTHS.items():
|
||
if name in text_lower:
|
||
result.add(month)
|
||
|
||
if result:
|
||
return sorted(result)
|
||
|
||
# Fall back to Roman numerals
|
||
text_upper = text.upper().replace('–', '-').replace('—', '-')
|
||
for part in re.split(r'[,;\s]+', text_upper):
|
||
part = part.strip()
|
||
if not part:
|
||
continue
|
||
if '-' in part:
|
||
lo_str, _, hi_str = part.partition('-')
|
||
lo = _ROMAN_MAP.get(lo_str.strip())
|
||
hi = _ROMAN_MAP.get(hi_str.strip())
|
||
if lo and hi:
|
||
if lo <= hi:
|
||
result.update(range(lo, hi + 1))
|
||
else:
|
||
result.update(range(lo, 13))
|
||
result.update(range(1, hi + 1))
|
||
else:
|
||
m = _ROMAN_MAP.get(part)
|
||
if m:
|
||
result.add(m)
|
||
|
||
return sorted(result)
|
||
|
||
|
||
def _slug_from_url(url):
|
||
m = re.search(r'/soorten/([^/?#]+)/?', url)
|
||
return m.group(1) if m else ''
|
||
|
||
|
||
def search_species(query):
|
||
"""
|
||
Search VPC and return up to 8 result dicts:
|
||
{slug, common_name, scientific_name, thumbnail_url, source}
|
||
|
||
On VPC, the title IS the Latin name so both name fields are set to it.
|
||
Raises VPCError on network failure.
|
||
"""
|
||
try:
|
||
resp = requests.get(
|
||
f'{_BASE}/',
|
||
params={'s': query, 'post_type': 'product'},
|
||
timeout=_TIMEOUT,
|
||
headers=_HEADERS,
|
||
)
|
||
resp.raise_for_status()
|
||
except requests.Timeout:
|
||
raise VPCError('VPC timed out.')
|
||
except Exception as e:
|
||
raise VPCError(f'VPC unreachable: {e}')
|
||
|
||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||
seen_slugs = set()
|
||
results = []
|
||
|
||
for item in soup.select('li.product, article.product'):
|
||
# Get the first product link (skip "Lees verder" duplicates)
|
||
for link in item.find_all('a', href=True):
|
||
slug = _slug_from_url(link['href'])
|
||
if slug and slug not in seen_slugs:
|
||
seen_slugs.add(slug)
|
||
title_el = item.find(['h2', 'h3'], class_=re.compile(r'title|product', re.I))
|
||
title = title_el.get_text(strip=True) if title_el else link.get_text(strip=True)
|
||
if title.lower() in ('lees verder', 'read more', ''):
|
||
continue
|
||
img = item.find('img')
|
||
thumbnail = img.get('src', '') if img else ''
|
||
results.append({
|
||
'slug': slug,
|
||
'common_name': title,
|
||
'scientific_name': title,
|
||
'thumbnail_url': thumbnail,
|
||
'source': 'vpc',
|
||
})
|
||
break
|
||
|
||
if len(results) >= 8:
|
||
break
|
||
|
||
return results
|
||
|
||
|
||
def scrape_plant(slug):
|
||
"""
|
||
Scrape a VPC plant detail page and return a dict with:
|
||
common_name, scientific_name, bloom_months, max_height_cm,
|
||
sunlight, growth_rate, description, api_image_url, vpc_slug
|
||
|
||
Raises VPCError on failure.
|
||
"""
|
||
url = f'{_BASE}/soorten/{slug}/'
|
||
try:
|
||
resp = requests.get(url, timeout=_TIMEOUT, headers=_HEADERS)
|
||
resp.raise_for_status()
|
||
except requests.Timeout:
|
||
raise VPCError('VPC timed out while fetching plant page.')
|
||
except Exception as e:
|
||
raise VPCError(f'Could not fetch plant page: {e}')
|
||
|
||
soup = BeautifulSoup(resp.text, 'html.parser')
|
||
|
||
# Title IS the Latin name on VPC
|
||
h1 = soup.find('h1')
|
||
name = h1.get_text(strip=True) if h1 else slug.replace('-', ' ').title()
|
||
|
||
# Hero image
|
||
hero_img = ''
|
||
for sel in ['.woocommerce-product-gallery img', '.product-images img', 'article img']:
|
||
img = soup.select_one(sel)
|
||
if img and img.get('src', '').startswith('http'):
|
||
hero_img = img['src']
|
||
break
|
||
|
||
result = {
|
||
'common_name': name,
|
||
'scientific_name': name,
|
||
'bloom_months': [],
|
||
'max_height_cm': None,
|
||
'sunlight': '',
|
||
'growth_rate': '',
|
||
'description': '',
|
||
'api_image_url': hero_img,
|
||
'vpc_slug': slug,
|
||
}
|
||
|
||
# Parse <b>Label:</b> + next sibling text
|
||
for b in soup.find_all('b'):
|
||
label = b.get_text(strip=True).rstrip(':').strip()
|
||
sibling = b.next_sibling
|
||
if sibling is None:
|
||
continue
|
||
value = str(sibling).strip().lstrip(':').strip()
|
||
if not value:
|
||
continue
|
||
|
||
if label in ('Bloeitijd', 'Bloeimaanden'):
|
||
result['bloom_months'] = _parse_months(value)
|
||
elif label == 'Hoogte in cm':
|
||
try:
|
||
result['max_height_cm'] = int(re.search(r'\d+', value).group())
|
||
except (AttributeError, ValueError):
|
||
pass
|
||
elif label == 'Standplaats':
|
||
result['sunlight'] = value
|
||
elif label in ('Groeisnelheid', 'Groeivorm'):
|
||
result['growth_rate'] = value
|
||
|
||
return result
|