2024-01-13 01:54:04 +00:00
|
|
|
import json
|
|
|
|
|
|
2024-01-13 07:44:48 +00:00
|
|
|
from products.db import Product, find_product_by_tag, find_product_by_product_id, insert_product, get_tags, add_tag, find_product_by_id
|
|
|
|
|
from products.scraping import scrape_woolies_data, get_product_id, get_product_details_url
|
2024-01-13 01:54:04 +00:00
|
|
|
|
|
|
|
|
from typing import List
|
2024-05-19 12:35:37 +00:00
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
def get_package_size(data: dict) -> str:
|
|
|
|
|
size = data['Product']['PackageSize']
|
|
|
|
|
if size:
|
|
|
|
|
match = re.match(r'(\d+)(.*)', size)
|
|
|
|
|
if match:
|
|
|
|
|
return int(match.group(1)), match.group(2)
|
|
|
|
|
|
|
|
|
|
return 1, 'items'
|
2024-01-13 01:54:04 +00:00
|
|
|
|
|
|
|
|
async def create_product(link: str) -> Product:
|
|
|
|
|
product_id = get_product_id(link)
|
2024-05-20 10:23:53 +00:00
|
|
|
if not product_id:
|
2024-01-18 08:28:26 +00:00
|
|
|
return None, None
|
2024-01-13 01:54:04 +00:00
|
|
|
|
|
|
|
|
product_url = get_product_details_url(product_id)
|
2024-01-18 08:28:26 +00:00
|
|
|
product, data = None, await scrape_woolies_data(product_url)
|
2024-01-13 01:54:04 +00:00
|
|
|
if data:
|
2024-05-19 12:35:37 +00:00
|
|
|
_dump_json_data_to_log(data, product_id)
|
|
|
|
|
|
|
|
|
|
quantity, unit = get_package_size(data)
|
2024-01-18 08:28:26 +00:00
|
|
|
product = Product(
|
2024-01-13 01:54:04 +00:00
|
|
|
id=0,
|
|
|
|
|
product_id=product_id,
|
|
|
|
|
name=data['Product']['Name'],
|
|
|
|
|
link=link,
|
2024-05-19 12:35:37 +00:00
|
|
|
quantity=quantity,
|
|
|
|
|
unit=unit,
|
2024-01-13 01:54:04 +00:00
|
|
|
img_small=data['Product']['SmallImageFile'],
|
|
|
|
|
img_large=data['Product']['LargeImageFile'],
|
|
|
|
|
)
|
2024-01-18 08:28:26 +00:00
|
|
|
|
|
|
|
|
return product, data
|
2024-01-13 01:54:04 +00:00
|
|
|
|
|
|
|
|
async def add_missing_tags(conn, product: Product, tags: List[str]):
|
|
|
|
|
existing_tags = set()
|
|
|
|
|
async for tag in get_tags(conn, product):
|
|
|
|
|
existing_tags.add(tag)
|
|
|
|
|
|
|
|
|
|
remaining_tags = set(tags) - existing_tags
|
|
|
|
|
if not remaining_tags:
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
for tag in remaining_tags:
|
|
|
|
|
await add_tag(conn, product, tag)
|
|
|
|
|
|
|
|
|
|
return product
|
|
|
|
|
|
|
|
|
|
async def get_or_create(conn, url: str, tags: List[str]) -> Product:
|
|
|
|
|
product_id = get_product_id(url)
|
2024-05-20 10:23:53 +00:00
|
|
|
if not product_id:
|
2024-01-13 01:54:04 +00:00
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
existing = await find_product_by_product_id(conn, product_id)
|
|
|
|
|
if existing:
|
|
|
|
|
await add_missing_tags(conn, existing, tags)
|
|
|
|
|
return existing
|
|
|
|
|
|
2024-01-18 08:28:26 +00:00
|
|
|
product, data = await create_product(url)
|
2024-01-13 01:54:04 +00:00
|
|
|
if product:
|
2024-01-18 08:28:26 +00:00
|
|
|
await insert_product(conn, product, data)
|
2024-01-13 01:54:04 +00:00
|
|
|
await add_missing_tags(conn, product, tags)
|
|
|
|
|
|
|
|
|
|
return product
|
2024-05-19 12:35:37 +00:00
|
|
|
|
|
|
|
|
def _dump_json_data_to_log(data: dict, product_id: str) -> str:
|
|
|
|
|
import os, re
|
|
|
|
|
dir = './data/dump'
|
|
|
|
|
if not os.path.exists(dir):
|
|
|
|
|
os.makedirs(dir)
|
|
|
|
|
|
|
|
|
|
prefix = f'product_{product_id}'
|
|
|
|
|
suffix = '.json'
|
|
|
|
|
file_ids = [int(re.findall(r'\d+', f)[0]) for f in os.listdir(dir) if re.match(prefix + r'\d+' + suffix, f)]
|
|
|
|
|
id = max(file_ids) + 1 if file_ids else 0
|
|
|
|
|
filename = f'{prefix}{id}{suffix}'
|
|
|
|
|
with open(os.path.join(dir, filename), 'w') as f:
|
|
|
|
|
json.dump(data, f, indent=4)
|