feat: VPC species search alongside Perenual

- New plants/services/vpc.py: search VPC (vasteplantencatalogus.nl) by
  keyword and scrape full plant detail on select
- Both searches run in parallel via ThreadPoolExecutor
- species_results partial shows results from both sources with colour-coded
  source badges (green = Perenual, blue = VPC)
- vpc_select view scrapes the detail page on select, creates Species with
  bloom_months, max_height_cm, sunlight from the VPC page
- Species.vpc_slug field (migration 0005) prevents re-importing the same plant
- beautifulsoup4 added to requirements
- Handles VPC unavailability gracefully (empty results, no error shown to user)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Stephan Kerkman 2026-05-28 18:44:06 +02:00
parent b88ba03884
commit a57415983b
7 changed files with 305 additions and 14 deletions

View file

@ -0,0 +1,16 @@
from django.db import migrations, models
class Migration(migrations.Migration):
dependencies = [
('plants', '0004_location_plantphoto'),
]
operations = [
migrations.AddField(
model_name='species',
name='vpc_slug',
field=models.CharField(blank=True, db_index=True, max_length=200),
),
]

View file

@ -25,6 +25,7 @@ class Species(models.Model):
bloom_months = models.JSONField(default=list) bloom_months = models.JSONField(default=list)
api_image_url = models.URLField(blank=True) api_image_url = models.URLField(blank=True)
perenual_id = models.IntegerField(unique=True, null=True, blank=True) perenual_id = models.IntegerField(unique=True, null=True, blank=True)
vpc_slug = models.CharField(max_length=200, blank=True, db_index=True)
class Meta: class Meta:
ordering = ['common_name'] ordering = ['common_name']

196
plants/services/vpc.py Normal file
View file

@ -0,0 +1,196 @@
"""
Scraper for vasteplantencatalogus.nl (VPC).
- search_species(query) list of lightweight result dicts
- scrape_plant(slug) full detail dict ready to store as Species
Notes on VPC data:
- Plant titles ARE the Latin names (no separate common name)
- Bloom months listed in Dutch ("september, oktober")
- Height in a dedicated "Hoogte in cm" field as plain integer
- Base URL is https://vasteplantencatalogus.nl (no www)
"""
import re
import requests
from bs4 import BeautifulSoup
_BASE = 'https://vasteplantencatalogus.nl'
_TIMEOUT = 8
_HEADERS = {'User-Agent': 'PlantDB/1.0 (personal plant tracker)'}
_NL_MONTHS = {
'januari': 1, 'februari': 2, 'maart': 3, 'april': 4,
'mei': 5, 'juni': 6, 'juli': 7, 'augustus': 8,
'september': 9, 'oktober': 10, 'november': 11, 'december': 12,
}
# Also handle Roman numerals (some older pages still use them)
_ROMAN_MAP = {
'XII': 12, 'VIII': 8, 'VII': 7, 'VI': 6, 'XI': 11,
'IX': 9, 'IV': 4, 'III': 3, 'II': 2, 'X': 10, 'V': 5, 'I': 1,
}
class VPCError(Exception):
pass
def _parse_months(text):
"""Parse Dutch or Roman numeral month strings into a sorted list of ints."""
result = set()
text_lower = text.lower().replace('', '-').replace('', '-')
# Try Dutch names first
for name, month in _NL_MONTHS.items():
if name in text_lower:
result.add(month)
if result:
return sorted(result)
# Fall back to Roman numerals
text_upper = text.upper().replace('', '-').replace('', '-')
for part in re.split(r'[,;\s]+', text_upper):
part = part.strip()
if not part:
continue
if '-' in part:
lo_str, _, hi_str = part.partition('-')
lo = _ROMAN_MAP.get(lo_str.strip())
hi = _ROMAN_MAP.get(hi_str.strip())
if lo and hi:
if lo <= hi:
result.update(range(lo, hi + 1))
else:
result.update(range(lo, 13))
result.update(range(1, hi + 1))
else:
m = _ROMAN_MAP.get(part)
if m:
result.add(m)
return sorted(result)
def _slug_from_url(url):
m = re.search(r'/soorten/([^/?#]+)/?', url)
return m.group(1) if m else ''
def search_species(query):
"""
Search VPC and return up to 8 result dicts:
{slug, common_name, scientific_name, thumbnail_url, source}
On VPC, the title IS the Latin name so both name fields are set to it.
Raises VPCError on network failure.
"""
try:
resp = requests.get(
f'{_BASE}/',
params={'s': query, 'post_type': 'product'},
timeout=_TIMEOUT,
headers=_HEADERS,
)
resp.raise_for_status()
except requests.Timeout:
raise VPCError('VPC timed out.')
except Exception as e:
raise VPCError(f'VPC unreachable: {e}')
soup = BeautifulSoup(resp.text, 'html.parser')
seen_slugs = set()
results = []
for item in soup.select('li.product, article.product'):
# Get the first product link (skip "Lees verder" duplicates)
for link in item.find_all('a', href=True):
slug = _slug_from_url(link['href'])
if slug and slug not in seen_slugs:
seen_slugs.add(slug)
title_el = item.find(['h2', 'h3'], class_=re.compile(r'title|product', re.I))
title = title_el.get_text(strip=True) if title_el else link.get_text(strip=True)
if title.lower() in ('lees verder', 'read more', ''):
continue
img = item.find('img')
thumbnail = img.get('src', '') if img else ''
results.append({
'slug': slug,
'common_name': title,
'scientific_name': title,
'thumbnail_url': thumbnail,
'source': 'vpc',
})
break
if len(results) >= 8:
break
return results
def scrape_plant(slug):
"""
Scrape a VPC plant detail page and return a dict with:
common_name, scientific_name, bloom_months, max_height_cm,
sunlight, growth_rate, description, api_image_url, vpc_slug
Raises VPCError on failure.
"""
url = f'{_BASE}/soorten/{slug}/'
try:
resp = requests.get(url, timeout=_TIMEOUT, headers=_HEADERS)
resp.raise_for_status()
except requests.Timeout:
raise VPCError('VPC timed out while fetching plant page.')
except Exception as e:
raise VPCError(f'Could not fetch plant page: {e}')
soup = BeautifulSoup(resp.text, 'html.parser')
# Title IS the Latin name on VPC
h1 = soup.find('h1')
name = h1.get_text(strip=True) if h1 else slug.replace('-', ' ').title()
# Hero image
hero_img = ''
for sel in ['.woocommerce-product-gallery img', '.product-images img', 'article img']:
img = soup.select_one(sel)
if img and img.get('src', '').startswith('http'):
hero_img = img['src']
break
result = {
'common_name': name,
'scientific_name': name,
'bloom_months': [],
'max_height_cm': None,
'sunlight': '',
'growth_rate': '',
'description': '',
'api_image_url': hero_img,
'vpc_slug': slug,
}
# Parse <b>Label:</b> + next sibling text
for b in soup.find_all('b'):
label = b.get_text(strip=True).rstrip(':').strip()
sibling = b.next_sibling
if sibling is None:
continue
value = str(sibling).strip().lstrip(':').strip()
if not value:
continue
if label in ('Bloeitijd', 'Bloeimaanden'):
result['bloom_months'] = _parse_months(value)
elif label == 'Hoogte in cm':
try:
result['max_height_cm'] = int(re.search(r'\d+', value).group())
except (AttributeError, ValueError):
pass
elif label == 'Standplaats':
result['sunlight'] = value
elif label in ('Groeisnelheid', 'Groeivorm'):
result['growth_rate'] = value
return result

View file

@ -1,4 +1,8 @@
{% for result in results %} {% if error %}
<p class="list-group-item text-warning small">⚠️ Perenual: {{ error }}</p>
{% endif %}
{% for result in perenual_results %}
<div class="list-group-item"> <div class="list-group-item">
<div class="d-flex align-items-center gap-2 mb-1"> <div class="d-flex align-items-center gap-2 mb-1">
{% if result.default_image and result.default_image.thumbnail %} {% if result.default_image and result.default_image.thumbnail %}
@ -6,10 +10,11 @@
{% else %} {% else %}
<div style="width:40px; height:40px; background:#d8f3dc; border-radius:.375rem; flex-shrink:0; display:flex; align-items:center; justify-content:center;">🌿</div> <div style="width:40px; height:40px; background:#d8f3dc; border-radius:.375rem; flex-shrink:0; display:flex; align-items:center; justify-content:center;">🌿</div>
{% endif %} {% endif %}
<div> <div class="flex-grow-1">
<div class="fw-semibold">{{ result.common_name }}</div> <div class="fw-semibold">{{ result.common_name }}</div>
<small class="text-muted fst-italic">{{ result.scientific_name|join:", " }}</small> <small class="text-muted fst-italic">{{ result.scientific_name|join:", " }}</small>
</div> </div>
<span class="badge bg-success" style="font-size:.65rem;">Perenual</span>
</div> </div>
<form method="POST" action="{% url 'species_select' %}"> <form method="POST" action="{% url 'species_select' %}">
{% csrf_token %} {% csrf_token %}
@ -23,10 +28,30 @@
<button type="submit" class="btn btn-sm btn-success">Select →</button> <button type="submit" class="btn btn-sm btn-success">Select →</button>
</form> </form>
</div> </div>
{% empty %} {% endfor %}
{% if error %}
<p class="list-group-item text-warning">⚠️ {{ error }}</p> {% for result in vpc_results %}
{% elif q %} <div class="list-group-item">
<div class="d-flex align-items-center gap-2 mb-1">
{% if result.thumbnail_url %}
<img src="{{ result.thumbnail_url }}" width="40" height="40" class="rounded" style="object-fit:cover; flex-shrink:0;">
{% else %}
<div style="width:40px; height:40px; background:#e8f5e9; border-radius:.375rem; flex-shrink:0; display:flex; align-items:center; justify-content:center;">🌱</div>
{% endif %}
<div class="flex-grow-1">
<div class="fw-semibold">{{ result.common_name }}</div>
{% if result.scientific_name %}<small class="text-muted fst-italic">{{ result.scientific_name }}</small>{% endif %}
</div>
<span class="badge" style="background:#3a86ff; font-size:.65rem;">VPC</span>
</div>
<form method="POST" action="{% url 'vpc_select' %}">
{% csrf_token %}
<input type="hidden" name="vpc_slug" value="{{ result.slug }}">
<button type="submit" class="btn btn-sm btn-primary">Select →</button>
</form>
</div>
{% endfor %}
{% if not perenual_results and not vpc_results and not error and q %}
<p class="list-group-item text-muted">No results for "{{ q }}"</p> <p class="list-group-item text-muted">No results for "{{ q }}"</p>
{% endif %} {% endif %}
{% endfor %}

View file

@ -15,4 +15,5 @@ urlpatterns = [
path('pruning/', pruning.pruning_calendar, name='pruning_calendar'), path('pruning/', pruning.pruning_calendar, name='pruning_calendar'),
path('species/search/', species.species_search, name='species_search'), path('species/search/', species.species_search, name='species_search'),
path('species/select/', species.species_select, name='species_select'), path('species/select/', species.species_select, name='species_select'),
path('species/select/vpc/', species.vpc_select, name='vpc_select'),
] ]

View file

@ -1,7 +1,10 @@
from concurrent.futures import ThreadPoolExecutor
from django.shortcuts import render, redirect from django.shortcuts import render, redirect
from django.urls import reverse from django.urls import reverse
from plants.models import Species from plants.models import Species
from plants.services.perenual import search_species, PerenualError from plants.services.perenual import search_species as perenual_search, PerenualError
from plants.services import vpc as vpc_service
from plants.services.vpc import VPCError
_MONTH_MAP = { _MONTH_MAP = {
'January': 1, 'February': 2, 'March': 3, 'April': 4, 'January': 1, 'February': 2, 'March': 3, 'April': 4,
@ -12,15 +15,34 @@ _MONTH_MAP = {
def species_search(request): def species_search(request):
q = request.GET.get('q', '').strip() q = request.GET.get('q', '').strip()
results = [] perenual_results = []
vpc_results = []
error = None error = None
if len(q) >= 2: if len(q) >= 2:
def _run_perenual():
try: try:
results = search_species(q) return perenual_search(q), None
except PerenualError as e: except PerenualError as e:
error = str(e) return [], str(e)
def _run_vpc():
try:
return vpc_service.search_species(q)
except VPCError:
return []
with ThreadPoolExecutor(max_workers=2) as ex:
f_p = ex.submit(_run_perenual)
f_v = ex.submit(_run_vpc)
perenual_results, error = f_p.result()
vpc_results = f_v.result()
return render(request, 'plants/partials/species_results.html', { return render(request, 'plants/partials/species_results.html', {
'results': results, 'q': q, 'error': error, 'perenual_results': perenual_results,
'vpc_results': vpc_results,
'q': q,
'error': error,
}) })
@ -44,3 +66,32 @@ def species_select(request):
) )
species = Species.objects.get(perenual_id=perenual_id) species = Species.objects.get(perenual_id=perenual_id)
return redirect(f"{reverse('plant_add')}?species_id={species.pk}") return redirect(f"{reverse('plant_add')}?species_id={species.pk}")
def vpc_select(request):
slug = request.POST.get('vpc_slug', '').strip()
if not slug:
return redirect(reverse('plant_add'))
existing = Species.objects.filter(vpc_slug=slug).first()
if existing:
return redirect(f"{reverse('plant_add')}?species_id={existing.pk}")
try:
data = vpc_service.scrape_plant(slug)
except VPCError:
# Fall back to plant_add without species pre-fill
return redirect(f"{reverse('plant_add')}?species_id=skip")
species = Species.objects.create(
common_name=data['common_name'],
scientific_name=data['scientific_name'],
bloom_months=data['bloom_months'],
max_height_cm=data['max_height_cm'],
sunlight=data['sunlight'],
growth_rate=data['growth_rate'],
description=data['description'],
api_image_url=data['api_image_url'],
vpc_slug=slug,
)
return redirect(f"{reverse('plant_add')}?species_id={species.pk}")

View file

@ -1,4 +1,5 @@
asgiref==3.11.1 asgiref==3.11.1
beautifulsoup4==4.14.3
certifi==2026.5.20 certifi==2026.5.20
charset-normalizer==3.4.7 charset-normalizer==3.4.7
Django==5.2.1 Django==5.2.1