munch-ease-backend/products/__init__.py

68 lines
No EOL
2.1 KiB
Python

import json
from products.db import Product, find_product_by_tag, find_product_by_key, insert_product, get_tags, add_tag, find_product_by_id
from products import woolworths, coles
SCRAPERS = { 'woolworths': woolworths, 'coles': coles }
from typing import List, Union
import re
def _get_shop_key(link: str) -> Union[str, str]: # (shop_code, product_id)
for shop_code, shop_scraper in SCRAPERS.items():
product_id = shop_scraper.get_product_id(link)
if product_id:
return shop_code, product_id
return None, None
async def add_missing_tags(conn, product: Product, tags: List[str]):
existing_tags = set()
async for tag in get_tags(conn, product):
existing_tags.add(tag)
remaining_tags = set(tags) - existing_tags
if not remaining_tags:
return False
for tag in remaining_tags:
await add_tag(conn, product, tag)
return product
async def get_or_create(conn, url: str, tags: List[str]) -> Product:
shop_code, product_id = _get_shop_key(url)
if not product_id:
return None
existing = await find_product_by_key(conn, shop_code, product_id)
if existing:
await add_missing_tags(conn, existing, tags)
return existing
product_data, raw_response = await SCRAPERS[shop_code].scrape(product_id)
product = Product(
id=-1,
shop_code=shop_code,
product_id=product_id,
link=url,
**product_data
)
await insert_product(conn, product, raw_response)
await add_missing_tags(conn, product, tags)
return product
def _dump_json_data_to_log(data: dict, product_id: str) -> str:
import os, re
dir = './data/dump'
if not os.path.exists(dir):
os.makedirs(dir)
prefix = f'product_{product_id}'
suffix = '.json'
file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)]
id = max(file_ids) + 1 if file_ids else 0
filename = f'{prefix}{id}{suffix}'
with open(os.path.join(dir, filename), 'w') as f:
json.dump(data, f, indent=4)