Update error handling for recipe parsing failures to return 422 status code

This commit is contained in:
jableader 2025-11-02 19:46:37 +11:00
parent e801c9056d
commit 7b7daa1c3c
4 changed files with 164 additions and 8 deletions

View file

@ -252,7 +252,8 @@ async def parse_from_url(
# Build an unsaved Recipe object using NLP parsing and product matching; do not insert
r = await recipes.parse_recipe(conn, user, body.url)
if not r:
return error_response(None, 404, "Recipe data not found at URL")
# Parsing failed: treat as unprocessable rather than not-found
return error_response(None, 422, "Unable to parse recipe from URL")
# Return the same shape a client would POST to create
return RecipeCreate(
name=r.name,

View file

@ -1,5 +1,5 @@
import json
from typing import Optional, Iterable
from typing import Optional, Iterable, List
import httpx
from bs4 import BeautifulSoup
@ -89,7 +89,19 @@ async def scrape_recipe_ldata(url: str) -> Optional[dict]:
soup = BeautifulSoup(response.text, "html.parser")
for ld in soup.find_all("script", type="application/ld+json"):
try:
data = json.loads(ld.text)
text = ld.text.strip()
# Some sites embed multiple JSON objects without an array. Try to coerce if needed.
try:
data = json.loads(text)
except json.decoder.JSONDecodeError:
# Attempt to split objects and wrap in a list (best-effort)
# Very conservative: only try if it looks like multiple root objects.
if text.count("{") > 1 and "}\n{" in text:
parts = [p for p in text.split("\n") if p.strip()]
maybe = "[" + ",".join(parts) + "]"
data = json.loads(maybe)
else:
raise
# _dump_json_data_to_log(data)
if _is_recipe_ldata(data):
return data
@ -107,6 +119,16 @@ async def scrape_recipe_ldata(url: str) -> Optional[dict]:
except (json.decoder.JSONDecodeError, KeyError):
pass
# Fallback 1: microdata (itemtype Recipe, itemprop recipeIngredient)
md = _extract_recipe_from_microdata(soup)
if md:
return md
# Fallback 2: heuristic DOM extraction around "Ingredients" section
hd = _extract_recipe_heuristic(soup, candidate)
if hd:
return hd
return None
# Fallback return to satisfy static analysis
@ -134,3 +156,136 @@ def _dump_json_data_to_log(data: dict) -> str:
with open(full_path, "w") as f:
json.dump(data, f, indent=4)
return full_path
def _extract_text(el) -> str:
return " ".join(el.get_text(" ", strip=True).split()) if el else ""
def _extract_recipe_from_microdata(soup: BeautifulSoup) -> Optional[dict]:
import re
candidates = []
for node in soup.find_all(attrs={"itemtype": re.compile(r"schema\.org/Recipe", re.I)}):
candidates.append(node)
# Some sites omit itemtype but use itemprop markers globally
if not candidates:
candidates = [soup]
def get_itemprop(node, prop) -> Optional[str]:
el = node.find(attrs={"itemprop": prop})
if not el:
return None
# Support <meta content>, <img src>, or text
if el.has_attr("content"):
return el["content"]
if el.has_attr("src"):
return el["src"]
if el.has_attr("href"):
return el["href"]
return _extract_text(el)
for node in candidates:
# Gather ingredients by itemprop
ingredients: List[str] = []
for prop in ("recipeIngredient", "ingredients"):
for ing_el in node.find_all(attrs={"itemprop": prop}):
t = _extract_text(ing_el)
if t:
ingredients.append(t)
name = get_itemprop(node, "name") or get_itemprop(soup, "og:title") or None
image = (
get_itemprop(node, "image")
or get_itemprop(soup, "og:image")
or get_itemprop(soup, "twitter:image")
)
if image:
images = [image]
else:
images = []
ry = get_itemprop(node, "recipeYield") or get_itemprop(soup, "recipeYield")
# Minimal viability: at least one ingredient and a name
if ingredients and (name or len(ingredients) >= 2):
out = {"@type": "Recipe", "recipeIngredient": ingredients}
if name:
out["name"] = name
if images:
out["image"] = images
if ry:
out["recipeYield"] = ry
return out
return None
def _extract_recipe_heuristic(soup: BeautifulSoup, page_url: str) -> Optional[dict]:
"""Best-effort heuristic extraction when no structured data is present.
- Finds an Ingredients heading and the nearest list (ul/ol)
- Or any element with class/id containing 'ingredient' and list items
- Derives name from og:title or <title>
- Derives image from og:image if available
"""
import re
# Try explicit ingredients container by class/id
candidates = []
for el in soup.find_all(True, attrs={"class": re.compile("ingredient", re.I)}):
candidates.append(el)
for el in soup.find_all(True, attrs={"id": re.compile("ingredient", re.I)}):
candidates.append(el)
def collect_list_items(node) -> List[str]:
items: List[str] = []
for ul in node.find_all(["ul", "ol"]):
for li in ul.find_all("li"):
t = _extract_text(li)
if t:
items.append(t)
return items
ingredients: List[str] = []
for node in candidates:
ingredients = collect_list_items(node)
if ingredients:
break
# Try heading-based approach
if not ingredients:
for h in soup.find_all(["h1", "h2", "h3", "h4", "h5", "h6"]):
if "ingredient" in _extract_text(h).lower():
# Try next sibling lists or parent lists
sib = h.find_next_sibling()
tries = [sib, h.parent]
for t in tries:
if not t:
continue
lst = collect_list_items(t)
if lst:
ingredients = lst
break
if ingredients:
break
if not ingredients:
return None
# Name and image fallbacks
name = None
og_title = soup.find("meta", attrs={"property": "og:title"})
if og_title and og_title.has_attr("content"):
name = og_title["content"]
if not name and soup.title:
name = _extract_text(soup.title)
image = None
og_image = soup.find("meta", attrs={"property": "og:image"})
if og_image and og_image.has_attr("content"):
image = og_image["content"]
out = {"@type": "Recipe", "recipeIngredient": ingredients}
out["name"] = name or page_url
if image:
out["image"] = [image]
return out

View file

@ -107,9 +107,9 @@ class TestRecipesParseFromUrlIntegrationV2(unittest.IsolatedAsyncioTestCase):
headers=self.headers,
json={"url": SAMPLE_URL},
)
assert r.status_code == 404, r.text
assert r.status_code == 422, r.text
body = r.json()
assert body.get("status") == 404
assert body.get("status") == 422
finally:
scraping.httpx.AsyncClient = orig_client

View file

@ -67,7 +67,7 @@ class TestRecipesParseFromUrlV2(unittest.IsolatedAsyncioTestCase):
finally:
recipes_pkg._scrape_recipe_ldata = orig
# Not found case
# Unprocessable (parse failure) case
recipes_pkg._scrape_recipe_ldata = fake_scrape_none
try:
r2 = self.client.post(
@ -75,8 +75,8 @@ class TestRecipesParseFromUrlV2(unittest.IsolatedAsyncioTestCase):
headers=self.headers,
json={"url": "https://example.com/missing"},
)
assert r2.status_code == 404, r2.text
assert r2.status_code == 422, r2.text
pb = r2.json()
assert pb.get("status") == 404
assert pb.get("status") == 422
finally:
recipes_pkg._scrape_recipe_ldata = orig