import json from typing import List, Optional, Tuple from products import coles, woolworths from products.models import Product from products.repository import ( add_tag, find_product_by_id as find_product_by_id, find_product_by_key, find_product_by_tag as find_product_by_tag, get_tags, insert_product, ) SCRAPERS = {"woolworths": woolworths, "coles": coles} def _get_shop_key(link: str) -> Tuple[Optional[str], Optional[str]]: # (shop_code, product_id) for shop_code, shop_scraper in SCRAPERS.items(): product_id = shop_scraper.get_product_id(link) if product_id: return shop_code, product_id return None, None async def add_missing_tags(conn, product: Product, tags: List[str]): existing_tags = {tag async for tag in get_tags(conn, product)} remaining_tags = set(tags) - existing_tags if not remaining_tags: return False for tag in remaining_tags: await add_tag(conn, product, tag) return product async def get_or_create(conn, url: str, tags: List[str]) -> Optional[Product]: shop_code, product_id = _get_shop_key(url) if not shop_code or not product_id: return None existing = await find_product_by_key(conn, shop_code, product_id) if existing: await add_missing_tags(conn, existing, tags) return existing scraper = SCRAPERS[shop_code] product_data, raw_response = await scraper.scrape(product_id) product = Product(id=-1, shop_code=shop_code, product_id=product_id, link=url, **product_data) await insert_product(conn, product, raw_response) await add_missing_tags(conn, product, tags) return product def _dump_json_data_to_log(data: dict, product_id: str) -> str: import os import re dir = "./data/dump" if not os.path.exists(dir): os.makedirs(dir) prefix = f"product_{product_id}" suffix = ".json" file_ids = [ int(re.findall(r"\d+", f)[0]) for f in os.listdir(dir) if re.match(prefix + r"\d+" + suffix, f) ] id = max(file_ids) + 1 if file_ids else 0 filename = f"{prefix}{id}{suffix}" full_path = os.path.join(dir, filename) with open(full_path, "w") as f: json.dump(data, f, indent=4) return full_path