import re, httpx from typing import Union HEADERS = { 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:121.0) Gecko/20100101 Firefox/121.0', 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8', 'Accept-Language': 'en-US,en;q=0.5', 'Accept-Encoding': 'gzip, deflate, br', 'DNT': '1', 'Sec-GPC': '1', 'Connection': 'keep-alive', 'Upgrade-Insecure-Requests': '1', 'Sec-Fetch-Dest': 'document', 'Sec-Fetch-Mode': 'navigate', 'Sec-Fetch-Site': 'none', 'Sec-Fetch-User': '?1', 'Pragma': 'no-cache', 'Cache-Control': 'no-cache', } async def _get_cookies(client): # Make a request to https://www.coles.com.au/ as if we were a normal browser, then return the cookies response = await client.get('https://www.coles.com.au/product/coles-strawberries-250g-5191256', headers=HEADERS, follow_redirects=True) version = re.findall(r'202[4-9][01]\d[0-2]\d.02_v\d.\d\d.\d', response.text) if len(version) == 0: raise Exception('Could not find the Coles API') return dict(response.cookies), version[0] def _get_package_size(size: str) -> Union[int, str]: if size: match = re.match(r'(\d+)(.*)', size) if match: return int(match.group(1)), match.group(2) return 1, 'items' def _get_client(): return httpx.AsyncClient() api_details = None async def _request_details(product_id: str) -> dict: global api_details async with _get_client() as client: if not api_details: api_details = await _get_cookies(client) cookies, api_version = api_details url = _get_product_details_url(api_version, product_id) try: response = await client.get(url, headers=HEADERS, follow_redirects=True, cookies=cookies) response.raise_for_status() return response.json() except httpx.NetworkError as ne: print(ne) api_details = None return None def _get_product_details_url(version: str, product_id: str) -> str: # https://www.coles.com.au/_next/data/20240926.02_v4.18.0/en/product/cadbury-favourites-boxed-chocolate-340g-3571992.json?slug=cadbury-favourites-boxed-chocolate-340g-3571992 # https://www.coles.com.au/_next/data/20240926.02_v4.18.0/en/product/coles-blueberries-170g-3571948.json?slug=coles-blueberries-170g-3571948 # return f'https://www.coles.com.au/_next/data/{version}/en/product/{product_id}.json' def get_product_id(url: str) -> str: # https://www.coles.com.au/product/cadbury-favourites-boxed-chocolate-340g-3571992 regex = r'https://www.coles.com.au/product/([^/]+)/?.*' match = re.match(regex, url) if match: return match.group(1) return None async def scrape(product_id: str) -> Union[dict, dict]: raw_data = await _request_details(product_id) product = raw_data['pageProps']['product'] quantity, unit = _get_package_size(product['size']) img_prefix = 'https://shop.coles.com.au' images = product['images'][0] product_data = { 'name': product['name'], 'quantity': quantity, 'unit': unit, 'img_small': (img_prefix + images['thumb']['path']) if images else None, 'img_large': (img_prefix + images['full']['path']) if images else None, } return product_data, raw_data