munch-ease-backend/products/__init__.py

79 lines
2.2 KiB
Python
Raw Normal View History

2024-01-13 01:54:04 +00:00
import json
2025-10-18 03:26:42 +00:00
from typing import List, Optional, Tuple
2024-01-13 01:54:04 +00:00
2025-10-18 03:26:42 +00:00
from products import coles, woolworths
Squashed commit of the following: commit fcd005b8624023547f28b7b28e59e6099bcfc7d4 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 20:24:07 2025 +1100 Openapi tightening commit f93bd8f641d561052c7bd075bae321b4ff3b676d Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 19:03:52 2025 +1100 Removed refactor strategy doc commit 0c5a61092f522be0c47cbbe86917c8a7e4e2d339 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 17:48:33 2025 +1100 mypy & ruff checks commit 23d66d6b18984127e17c73c3063f6120385935e9 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 16:49:35 2025 +1100 Final removal of db.py files commit f454aed1ca9783cc558cc203f29a7fe31b62a975 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:42:31 2025 +1100 Finalise restructure, remove db.py files commit 7187f6dd89489521538791c6bdebb426514beb99 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:34:54 2025 +1100 commit 6fea227ae20d32b8eb1e7a4885006a620fcc7bb1 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:32:53 2025 +1100 commit 27415e7e02d89195ad514cb017a9dbbf84d7a5e4 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:31:10 2025 +1100 commit b773428033d855f9ad82005602e049c1a2e3c585 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:28:58 2025 +1100 commit 116592c95278d995f4c516e87f2cea43cf5b7735 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:25:21 2025 +1100 commit 03ec565faea088971968ee2f9bb83e2de16b21f3 Author: jableader <jacobdunk@gmail.com> Date: Sun Oct 19 15:21:29 2025 +1100 Plan
2025-10-19 09:24:23 +00:00
from products.models import Product
from products.repository import (
2025-10-18 03:26:42 +00:00
add_tag,
find_product_by_id as find_product_by_id,
find_product_by_key,
find_product_by_tag as find_product_by_tag,
get_tags,
insert_product,
)
2024-01-13 01:54:04 +00:00
2025-10-18 03:26:42 +00:00
SCRAPERS = {"woolworths": woolworths, "coles": coles}
2024-05-19 12:35:37 +00:00
2025-10-18 03:26:42 +00:00
def _get_shop_key(link: str) -> Tuple[Optional[str], Optional[str]]: # (shop_code, product_id)
2024-09-29 05:04:10 +00:00
for shop_code, shop_scraper in SCRAPERS.items():
product_id = shop_scraper.get_product_id(link)
if product_id:
return shop_code, product_id
return None, None
2024-01-18 08:28:26 +00:00
2025-10-18 03:26:42 +00:00
2024-01-13 01:54:04 +00:00
async def add_missing_tags(conn, product: Product, tags: List[str]):
2025-07-29 07:15:38 +00:00
existing_tags = {tag async for tag in get_tags(conn, product)}
2024-01-13 01:54:04 +00:00
remaining_tags = set(tags) - existing_tags
if not remaining_tags:
return False
2025-10-18 03:26:42 +00:00
2024-01-13 01:54:04 +00:00
for tag in remaining_tags:
await add_tag(conn, product, tag)
2025-10-18 03:26:42 +00:00
2024-01-13 01:54:04 +00:00
return product
2025-10-18 03:26:42 +00:00
async def get_or_create(conn, url: str, tags: List[str]) -> Optional[Product]:
2024-09-29 05:04:10 +00:00
shop_code, product_id = _get_shop_key(url)
2025-10-18 03:26:42 +00:00
if not shop_code or not product_id:
2024-01-13 01:54:04 +00:00
return None
2025-10-18 03:26:42 +00:00
2024-09-29 05:04:10 +00:00
existing = await find_product_by_key(conn, shop_code, product_id)
2024-01-13 01:54:04 +00:00
if existing:
await add_missing_tags(conn, existing, tags)
return existing
2025-10-18 03:26:42 +00:00
scraper = SCRAPERS[shop_code]
product_data, raw_response = await scraper.scrape(product_id)
product = Product(id=-1, shop_code=shop_code, product_id=product_id, link=url, **product_data)
2024-09-29 05:04:10 +00:00
await insert_product(conn, product, raw_response)
await add_missing_tags(conn, product, tags)
2024-01-13 01:54:04 +00:00
return product
2024-05-19 12:35:37 +00:00
2025-10-18 03:26:42 +00:00
2024-05-19 12:35:37 +00:00
def _dump_json_data_to_log(data: dict, product_id: str) -> str:
2025-10-18 03:26:42 +00:00
import os
import re
dir = "./data/dump"
2024-05-19 12:35:37 +00:00
if not os.path.exists(dir):
os.makedirs(dir)
2025-10-18 03:26:42 +00:00
prefix = f"product_{product_id}"
suffix = ".json"
file_ids = [
int(re.findall(r"\d+", f)[0])
for f in os.listdir(dir)
if re.match(prefix + r"\d+" + suffix, f)
]
2024-05-19 12:35:37 +00:00
id = max(file_ids) + 1 if file_ids else 0
2025-10-18 03:26:42 +00:00
filename = f"{prefix}{id}{suffix}"
full_path = os.path.join(dir, filename)
with open(full_path, "w") as f:
json.dump(data, f, indent=4)
return full_path