import json from products.db import Product, find_product_by_tag, find_product_by_key, insert_product, get_tags, add_tag, find_product_by_id from products import woolworths, coles SCRAPERS = { 'woolworths': woolworths, 'coles': coles } from typing import List, Union import re def _get_shop_key(link: str) -> Union[str, str]: # (shop_code, product_id) for shop_code, shop_scraper in SCRAPERS.items(): product_id = shop_scraper.get_product_id(link) if product_id: return shop_code, product_id return None, None async def add_missing_tags(conn, product: Product, tags: List[str]): existing_tags = set() async for tag in get_tags(conn, product): existing_tags.add(tag) remaining_tags = set(tags) - existing_tags if not remaining_tags: return False for tag in remaining_tags: await add_tag(conn, product, tag) return product async def get_or_create(conn, url: str, tags: List[str]) -> Product: shop_code, product_id = _get_shop_key(url) if not product_id: return None existing = await find_product_by_key(conn, shop_code, product_id) if existing: await add_missing_tags(conn, existing, tags) return existing product_data, raw_response = await SCRAPERS[shop_code].scrape(product_id) product = Product( id=-1, shop_code=shop_code, product_id=product_id, link=url, **product_data ) await insert_product(conn, product, raw_response) await add_missing_tags(conn, product, tags) return product def _dump_json_data_to_log(data: dict, product_id: str) -> str: import os, re dir = './data/dump' if not os.path.exists(dir): os.makedirs(dir) prefix = f'product_{product_id}' suffix = '.json' file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)] id = max(file_ids) + 1 if file_ids else 0 filename = f'{prefix}{id}{suffix}' with open(os.path.join(dir, filename), 'w') as f: json.dump(data, f, indent=4)