From 75ecb5d0f78c54ac6c6eeb7b0f15303817ae3f13 Mon Sep 17 00:00:00 2001 From: jableader Date: Sat, 13 Jan 2024 14:21:38 +1100 Subject: [PATCH] Pydantic update --- db.py | 4 ++ main.py | 6 +-- model.py | 31 +++------------ product/db.py | 27 +++++++------ recipe.py | 97 ---------------------------------------------- recipe/__init__.py | 60 ++++++++++++++++++++++++++++ recipe/db.py | 79 +++++++++++++++++++++++++++++++++++++ recipe/scraping.py | 49 +++++++++++++++++++++++ 8 files changed, 213 insertions(+), 140 deletions(-) delete mode 100644 recipe.py create mode 100644 recipe/__init__.py create mode 100644 recipe/db.py create mode 100644 recipe/scraping.py diff --git a/db.py b/db.py index 4eb10c2..6164f10 100644 --- a/db.py +++ b/db.py @@ -7,6 +7,10 @@ async def create(): import product.db as product_db conn = await connect() await product_db.create(conn) + + import recipe.db as recipe_db + await recipe_db.create(conn) + await conn.commit() await conn.close() diff --git a/main.py b/main.py index 86bda7c..8ed844c 100644 --- a/main.py +++ b/main.py @@ -26,7 +26,7 @@ async def get_db(): await sql_db.close() @app.get("/recipes/parse") -async def parse_recipe_handler(url: str, conn: sqlite3.Connection = Depends(get_db)): +async def parse_recipe_handler(url: str, conn: sqlite3.Connection = Depends(get_db)) -> recipe.Recipe: return await recipe.parse_recipe(conn, url) @app.get("/recipes/ingredients/parse") @@ -34,7 +34,7 @@ async def parse_ingredients(lines: Annotated[ List[str], Query(alias="ingredients", title="Array of ingredients to parse")], - conn: sqlite3.Connection = Depends(get_db)): + conn: sqlite3.Connection = Depends(get_db)) -> List[recipe.Ingredient]: ingredients = recipe.parse_ingredient_from_nlp(lines) recipe.match_existing_products(conn, ingredients) return ingredients @@ -44,5 +44,5 @@ class ProductUrl(BaseModel): tags: List[str] = [] @app.post("/products/") -async def create_product(url: ProductUrl, conn: sqlite3.Connection = Depends(get_db)): +async def create_product(url: ProductUrl, conn: sqlite3.Connection = Depends(get_db)) -> product.Product: return await product.get_or_create(conn, url.url, url.tags) \ No newline at end of file diff --git a/model.py b/model.py index b3ee126..3bca6d4 100644 --- a/model.py +++ b/model.py @@ -1,34 +1,12 @@ -from typing import List, Union -from datetime import datetime - -from product import Product +from typing import List, ClassVar +from pydantic import BaseModel +""" class Person: def __init__(self, id: int, name: str) -> None: self.id = id self.name = name -class Measurement: - def __init__(self, qty: int, unit: str) -> None: - self.qty = qty - self.unit = unit - -class Ingredient: - def __init__(self, id: int, line: str, name: str, product: Product, measure: Measurement, preparation: str) -> None: - self.id = id - self.name = name - self.product = product - self.measure = measure - self.line = line - self.preparation = preparation - -class Recipe: - def __init__(self, id: int, name: str, link: str, ingredients: List[Ingredient]) -> None: - self.id = id - self.name = name - self.link = link - self.ingredients = ingredients - class Meal: def __init__(self, id: int, date: datetime, chef: List[Person], cleanup: List[Person], consumers: List[Person], recipes: List[Recipe]) -> None: @@ -49,4 +27,5 @@ class ShoppingListItem: class ShoppingList: def __init__(self, date: datetime, items: List[ShoppingListItem]) -> None: self.date = date - self.items = items \ No newline at end of file + self.items = items +""" diff --git a/product/db.py b/product/db.py index e989a21..b7a1ec7 100644 --- a/product/db.py +++ b/product/db.py @@ -1,18 +1,17 @@ import aiosqlite -from typing import List +from typing import List, ClassVar +from pydantic import BaseModel -class Product: - KEYS = ['id', 'product_id', 'link', 'name', 'img_small', 'img_large', 'raw_data'] - - def __init__(self, id: int, product_id: str, link: str, name: str, img_small: str, img_large: str, raw_data: dict) -> None: - self.id = id - self.product_id = product_id - self.link = link - self.name = name - self.img_small = img_small - self.img_large = img_large - self.raw_data = raw_data +class Product(BaseModel): + KEYS: ClassVar[List[str]] = ['id', 'product_id', 'link', 'name', 'img_small', 'img_large', 'raw_data'] + id: int + product_id: str + link: str + name: str + img_small: str + img_large: str + raw_data: str async def create(conn): await conn.execute(''' @@ -43,7 +42,7 @@ async def find_product_by_tag(conn, tag: str) -> List[Product]: ) ''', (tag,)) as cursor: async for row in cursor: - yield Product(*row) + yield Product(**{k:v for k,v in zip(Product.KEYS, row)}) async def find_product_by_product_id(conn, product_id: str) -> Product: async with conn.execute(f''' @@ -52,7 +51,7 @@ async def find_product_by_product_id(conn, product_id: str) -> Product: LIMIT 1 ''', (product_id,)) as cursor: async for row in cursor: - return Product(*row) + return Product(**{k:v for k,v in zip(Product.KEYS, row)}) async def insert_product(conn, product: Product): async with conn.execute(''' diff --git a/recipe.py b/recipe.py deleted file mode 100644 index def90b8..0000000 --- a/recipe.py +++ /dev/null @@ -1,97 +0,0 @@ -from model import Product, Ingredient, Recipe, Measurement -from ingredient_parser import parse_multiple_ingredients -from typing import List -from bs4 import BeautifulSoup - -import httpx -import json -import product.db as db - -async def find_existing_product(conn, ingredient: str) -> Product: - async for item in db.find_product_by_tag(conn, ingredient): - return item - return None - -def parse_ingredient_from_nlp(ingredients: List[str]) -> Ingredient: - results = [] - for ingredient in parse_multiple_ingredients(ingredients): - name = ingredient.name.text if ingredient.name else '' - - if ingredient.amount: - measure = Measurement(ingredient.amount[0].quantity, ingredient.amount[0].unit) - else: - measure = Measurement(0, '') - - results.append(Ingredient( - id=0, - line=ingredient.sentence, - name=name, - product=None, - measure=measure, - preparation=ingredient.preparation.text if ingredient.preparation else '' - )) - - return results - -async def match_existing_products(conn, ingredients: List[Ingredient]) -> List[Ingredient]: - for ingredient in ingredients: - if not ingredient.product: - existing = await find_existing_product(conn, ingredient.name) - if existing: - ingredient.product = existing - return ingredients - -async def get_recipe_from_ldata(conn, url: str, ldata: dict) -> dict: - ingredients = parse_ingredient_from_nlp(ldata['recipeIngredient']) - ingredients = await match_existing_products(conn, ingredients) - return Recipe( - id=0, - name=ldata['name'], - link=url, - ingredients=ingredients - ) - -async def parse_recipe(conn, url: str) -> dict: - headers = { - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36", - "Accept-Language": "en-US,en;q=0.9", - "Referer": "https://www.google.com/", - } - - # Load the requested URL with headers - async with httpx.AsyncClient() as client: - response = await client.get(url, headers=headers, follow_redirects=True) - response.raise_for_status() - - # Extract the recipe ld+json data - soup = BeautifulSoup(response.text, 'html.parser') - for ld in soup.find_all('script', type='application/ld+json'): - try: - data = json.loads(ld.text) - #_dump_json_data_to_log(data) - if '@type' in data and data['@type'].lower() == 'recipe': - return { 'raw': data, 'recipe': await get_recipe_from_ldata(conn, url, data) } - - if '@graph' in data: - for item in data['@graph']: - if '@type' in item and item['@type'].lower() == 'recipe': - return { 'raw': data, 'recipe': await get_recipe_from_ldata(conn, url, item) } - - except (json.decoder.JSONDecodeError, KeyError): - pass - - return None - -def _dump_json_data_to_log(data: dict) -> str: - import os, re - dir = './dump' - if not os.path.exists(dir): - os.makedirs(dir) - - prefix = 'ldata_' - suffix = '.json' - file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)] - id = max(file_ids) + 1 if file_ids else 0 - filename = f'{prefix}{id}{suffix}' - with open(os.path.join(dir, filename), 'w') as f: - json.dump(data, f, indent=4) \ No newline at end of file diff --git a/recipe/__init__.py b/recipe/__init__.py new file mode 100644 index 0000000..a22a8df --- /dev/null +++ b/recipe/__init__.py @@ -0,0 +1,60 @@ +from product import Product +from recipe.db import Recipe, Ingredient +from recipe.scraping import scrape_recipe + +from ingredient_parser import parse_multiple_ingredients +from typing import List +from product import find_product_by_tag + +import json + +async def find_existing_product(conn, ingredient: str) -> Product: + async for item in find_product_by_tag(conn, ingredient): + return item + return None + +def parse_ingredient_from_nlp(ingredients: List[str]) -> Ingredient: + results = [] + for ingredient in parse_multiple_ingredients(ingredients): + name = ingredient.name.text if ingredient.name else '' + + quantity, unit = None, None + if ingredient.amount: + amount = ingredient.amount[0] + quantity, unit = amount.quantity, amount.unit + + results.append(Ingredient(id=0, + line=ingredient.sentence, + name=name, + quantity=quantity, + unit=unit, + preparation=ingredient.preparation.text if ingredient.preparation else '', + product_id=0 + )) + + return results + +async def match_existing_products(conn, ingredients: List[Ingredient]) -> List[Ingredient]: + for ingredient in ingredients: + if not ingredient.product: + existing = await find_existing_product(conn, ingredient.name) + if existing: + ingredient.product = existing + return ingredients + +async def _get_recipe_from_ldata(conn, url: str, ldata: dict) -> dict: + ingredients = parse_ingredient_from_nlp(ldata['recipeIngredient']) + ingredients = await match_existing_products(conn, ingredients) + return Recipe( + id=0, + name=ldata['name'], + link=url, + raw_data=json.dumps(ldata), + ingredients=ingredients + ) + +async def parse_recipe(conn, url: str) -> dict: + ldata = await scrape_recipe(conn, url) + if ldata: + return await _get_recipe_from_ldata(conn, url, ldata) + return None \ No newline at end of file diff --git a/recipe/db.py b/recipe/db.py new file mode 100644 index 0000000..ab57735 --- /dev/null +++ b/recipe/db.py @@ -0,0 +1,79 @@ +import aiosqlite + +from product import Product + +from pydantic import BaseModel +from typing import List, ClassVar + +class Ingredient(BaseModel): + KEYS: ClassVar[List[str]] = ['id', 'name', 'line', 'preparation', 'unit', 'quantity', 'product_id'] + id: int + name: str + line: str + unit: str + quantity: float + preparation: str + product_id: int + product: Product = None + +class Recipe(BaseModel): + KEYS: ClassVar[List[str]] = ['id', 'name', 'link', 'raw_data'] + id: int + name: str + link: str + raw_data: str + ingredients: List[Ingredient] = [] + +async def create(conn): + await conn.execute(''' + CREATE TABLE IF NOT EXISTS Recipe ( + id INTEGER PRIMARY KEY, + name TEXT, + link TEXT, + raw_data TEXT + );''') + + await conn.execute(''' + CREATE TABLE IF NOT EXISTS Ingredient ( + id INTEGER PRIMARY KEY, + name TEXT, + line TEXT, + preparation TEXT, + unit TEXT, + quantity REAL, + product_id INTEGER, + recipe_id INTEGER, + FOREIGN KEY (product_id) REFERENCES Product(id), + FOREIGN KEY (recipe_id) REFERENCES Recipe(id) + );''') + +async def insert_ingredient(conn, ingredient: Ingredient): + async with conn.execute(''' + INSERT INTO Ingredient (name, line, preparation, unit, quantity, product_id, recipe_id) + VALUES (?, ?, ?, ?, ?, ?, ?) + ''', (ingredient.name, ingredient.line, ingredient.preparation, ingredient.unit, ingredient.quantity, ingredient.product_id, ingredient.recipe_id)) as cursor: + ingredient.id = cursor.lastrowid + +async def insert_recipe(conn, recipe: Recipe): + async with conn.execute(''' + INSERT INTO Recipe (name, link, raw_data) + VALUES (?, ?, ?) + ''', (recipe.name, recipe.link, recipe.raw_data)) as cursor: + recipe.id = cursor.lastrowid + +async def find_recipe_by_id(conn, recipe_id: int) -> Recipe: + async with conn.execute(f''' + SELECT {','.join(Recipe.KEYS)} FROM Recipe + WHERE id = ? + LIMIT 1 + ''', (recipe_id,)) as cursor: + async for row in cursor: + return Recipe(**{k:v for k,v in zip(Recipe.KEYS, row)}) + +async def get_ingredients(conn, recipe_id: int) -> List[Ingredient]: + async with conn.execute(f''' + SELECT {','.join(Ingredient.KEYS)} FROM Ingredient + WHERE recipe_id = ? + ''', (recipe_id,)) as cursor: + async for row in cursor: + yield Ingredient(**{k:v for k,v in zip(Ingredient.KEYS, row)}) \ No newline at end of file diff --git a/recipe/scraping.py b/recipe/scraping.py new file mode 100644 index 0000000..1947b3f --- /dev/null +++ b/recipe/scraping.py @@ -0,0 +1,49 @@ +from bs4 import BeautifulSoup +import httpx +import json + + +async def scrape_recipe(conn, url: str) -> dict: + headers = { + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36", + "Accept-Language": "en-US,en;q=0.9", + "Referer": "https://www.google.com/", + } + + # Load the requested URL with headers + async with httpx.AsyncClient() as client: + response = await client.get(url, headers=headers, follow_redirects=True) + response.raise_for_status() + + # Extract the recipe ld+json data + soup = BeautifulSoup(response.text, 'html.parser') + for ld in soup.find_all('script', type='application/ld+json'): + try: + data = json.loads(ld.text) + #_dump_json_data_to_log(data) + if '@type' in data and data['@type'].lower() == 'recipe': + return data + + if '@graph' in data: + for item in data['@graph']: + if '@type' in item and item['@type'].lower() == 'recipe': + return item + + except (json.decoder.JSONDecodeError, KeyError): + pass + + return None + +def _dump_json_data_to_log(data: dict) -> str: + import os, re + dir = './dump' + if not os.path.exists(dir): + os.makedirs(dir) + + prefix = 'ldata_' + suffix = '.json' + file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)] + id = max(file_ids) + 1 if file_ids else 0 + filename = f'{prefix}{id}{suffix}' + with open(os.path.join(dir, filename), 'w') as f: + json.dump(data, f, indent=4) \ No newline at end of file