2024-01-13 01:54:04 +00:00
|
|
|
import json
|
2025-10-18 03:26:42 +00:00
|
|
|
from typing import List, Optional, Tuple
|
2024-01-13 01:54:04 +00:00
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
from products import coles, woolworths
|
|
|
|
|
from products.db import (
|
|
|
|
|
Product,
|
|
|
|
|
add_tag,
|
|
|
|
|
find_product_by_id as find_product_by_id,
|
|
|
|
|
find_product_by_key,
|
|
|
|
|
find_product_by_tag as find_product_by_tag,
|
|
|
|
|
get_tags,
|
|
|
|
|
insert_product,
|
|
|
|
|
)
|
2024-01-13 01:54:04 +00:00
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
SCRAPERS = {"woolworths": woolworths, "coles": coles}
|
2024-05-19 12:35:37 +00:00
|
|
|
|
|
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
def _get_shop_key(link: str) -> Tuple[Optional[str], Optional[str]]: # (shop_code, product_id)
|
2024-09-29 05:04:10 +00:00
|
|
|
for shop_code, shop_scraper in SCRAPERS.items():
|
|
|
|
|
product_id = shop_scraper.get_product_id(link)
|
|
|
|
|
if product_id:
|
|
|
|
|
return shop_code, product_id
|
|
|
|
|
return None, None
|
2024-01-18 08:28:26 +00:00
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
|
2024-01-13 01:54:04 +00:00
|
|
|
async def add_missing_tags(conn, product: Product, tags: List[str]):
|
2025-07-29 07:15:38 +00:00
|
|
|
existing_tags = {tag async for tag in get_tags(conn, product)}
|
2024-01-13 01:54:04 +00:00
|
|
|
remaining_tags = set(tags) - existing_tags
|
|
|
|
|
if not remaining_tags:
|
|
|
|
|
return False
|
2025-10-18 03:26:42 +00:00
|
|
|
|
2024-01-13 01:54:04 +00:00
|
|
|
for tag in remaining_tags:
|
|
|
|
|
await add_tag(conn, product, tag)
|
2025-10-18 03:26:42 +00:00
|
|
|
|
2024-01-13 01:54:04 +00:00
|
|
|
return product
|
|
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
|
|
|
|
|
async def get_or_create(conn, url: str, tags: List[str]) -> Optional[Product]:
|
2024-09-29 05:04:10 +00:00
|
|
|
shop_code, product_id = _get_shop_key(url)
|
2025-10-18 03:26:42 +00:00
|
|
|
if not shop_code or not product_id:
|
2024-01-13 01:54:04 +00:00
|
|
|
return None
|
2025-10-18 03:26:42 +00:00
|
|
|
|
2024-09-29 05:04:10 +00:00
|
|
|
existing = await find_product_by_key(conn, shop_code, product_id)
|
2024-01-13 01:54:04 +00:00
|
|
|
if existing:
|
|
|
|
|
await add_missing_tags(conn, existing, tags)
|
|
|
|
|
return existing
|
2025-10-18 03:26:42 +00:00
|
|
|
|
|
|
|
|
scraper = SCRAPERS[shop_code]
|
|
|
|
|
product_data, raw_response = await scraper.scrape(product_id)
|
|
|
|
|
product = Product(id=-1, shop_code=shop_code, product_id=product_id, link=url, **product_data)
|
2024-09-29 05:04:10 +00:00
|
|
|
|
|
|
|
|
await insert_product(conn, product, raw_response)
|
|
|
|
|
await add_missing_tags(conn, product, tags)
|
2024-01-13 01:54:04 +00:00
|
|
|
|
|
|
|
|
return product
|
2024-05-19 12:35:37 +00:00
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
|
2024-05-19 12:35:37 +00:00
|
|
|
def _dump_json_data_to_log(data: dict, product_id: str) -> str:
|
2025-10-18 03:26:42 +00:00
|
|
|
import os
|
|
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
dir = "./data/dump"
|
2024-05-19 12:35:37 +00:00
|
|
|
if not os.path.exists(dir):
|
|
|
|
|
os.makedirs(dir)
|
|
|
|
|
|
2025-10-18 03:26:42 +00:00
|
|
|
prefix = f"product_{product_id}"
|
|
|
|
|
suffix = ".json"
|
|
|
|
|
file_ids = [
|
|
|
|
|
int(re.findall(r"\d+", f)[0])
|
|
|
|
|
for f in os.listdir(dir)
|
|
|
|
|
if re.match(prefix + r"\d+" + suffix, f)
|
|
|
|
|
]
|
2024-05-19 12:35:37 +00:00
|
|
|
id = max(file_ids) + 1 if file_ids else 0
|
2025-10-18 03:26:42 +00:00
|
|
|
filename = f"{prefix}{id}{suffix}"
|
|
|
|
|
full_path = os.path.join(dir, filename)
|
|
|
|
|
with open(full_path, "w") as f:
|
|
|
|
|
json.dump(data, f, indent=4)
|
|
|
|
|
return full_path
|