from bs4 import BeautifulSoup import httpx import json HEADERS = { 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8', 'Accept-Language': 'en-US,en;q=0.5', 'DNT': '1', 'Sec-GPC': '1', 'Connection': 'keep-alive', 'Upgrade-Insecure-Requests': '1', 'Sec-Fetch-Dest': 'document', 'Sec-Fetch-Mode': 'navigate', 'Sec-Fetch-Site': 'none', 'Sec-Fetch-User': '?1', 'Priority': 'u=1', 'Pragma': 'no-cache', 'Cache-Control': 'no-cache', } def _is_recipe_ldata(ldata_node): if '@type' in ldata_node: typ = ldata_node['@type'] if isinstance(typ, list): typ = typ[0] if isinstance(typ, str) and typ.lower() == 'recipe': return True return None async def scrape_recipe_ldata(url: str) -> dict: # Load the requested URL with headers async with httpx.AsyncClient() as client: response = await client.get(url, headers=HEADERS, follow_redirects=True) if response.status_code >= 300: return None # Extract the recipe ld+json data soup = BeautifulSoup(response.text, 'html.parser') for ld in soup.find_all('script', type='application/ld+json'): try: data = json.loads(ld.text) #_dump_json_data_to_log(data) if _is_recipe_ldata(data): return data if '@graph' in data: for item in data['@graph']: if _is_recipe_ldata(item): return item if isinstance(data, list): for item in data: if _is_recipe_ldata(item): return item except (json.decoder.JSONDecodeError, KeyError): pass return None def _dump_json_data_to_log(data: dict) -> str: import os, re dir = './data/dump' if not os.path.exists(dir): os.makedirs(dir) prefix = 'ldata_' suffix = '.json' file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)] id = max(file_ids) + 1 if file_ids else 0 filename = f'{prefix}{id}{suffix}' with open(os.path.join(dir, filename), 'w') as f: json.dump(data, f, indent=4)