""" Redefine Meals menu scraper. Navigates to the menu page using a headless browser, waits for the Vue.js SPA to render, and extracts structured meal data from the DOM. Also intercepts network requests to detect any JSON API that could replace the browser scrape in the future. """ import json import sys import os from datetime import datetime from pathlib import Path from playwright.sync_api import sync_playwright, TimeoutError as PwTimeout CONFIG_PATH = Path(__file__).resolve().parents[2] / "config.json" def load_config(): with open(CONFIG_PATH) as f: return json.load(f) def scrape_menu(url: str, *, headless: bool = True, timeout_ms: int = 60_000) -> dict: """ Returns { "scraped_at": ISO timestamp, "menu_url": str, "api_endpoints_found": [str], "meals": [ { name, price, calories, protein, dietary_tags, image_url, is_new, description } ] } Raises RuntimeError if the page fails to load or no meals are found. """ api_endpoints = [] with sync_playwright() as p: browser = p.chromium.launch(headless=headless) context = browser.new_context( user_agent=( "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " "AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/120.0.0.0 Safari/537.36" ) ) page = context.new_page() def on_response(response): ct = response.headers.get("content-type", "") if "json" in ct and "/api/" in response.url and "cart" not in response.url: api_endpoints.append(response.url) page.on("response", on_response) try: page.goto(url, wait_until="networkidle", timeout=timeout_ms) except PwTimeout: browser.close() raise RuntimeError(f"Timed out loading {url}") page.wait_for_timeout(3000) articles = page.query_selector_all("article.editorial_card") if not articles: browser.close() raise RuntimeError( "No meal cards found on page. The site layout may have changed. " "Run recon.py to inspect the current structure." ) meals = [] for article in articles: meal = _extract_card(article) if meal: meals.append(meal) # Try to get descriptions via Quick View modals _enrich_with_descriptions(page, articles, meals) browser.close() if not meals: raise RuntimeError("Scraped 0 meals — extraction selectors are likely broken.") for meal in meals: meal["dietary_tags"] = _clean_tags(meal["dietary_tags"]) return { "scraped_at": datetime.now().isoformat(), "menu_url": url, "api_endpoints_found": api_endpoints, "meal_count": len(meals), "meals": meals, } def _extract_card(article) -> dict | None: try: name_el = article.query_selector("h2.meal_title") if not name_el: return None name = name_el.inner_text().strip() price_el = article.query_selector(".meal_price") price_text = price_el.inner_text().strip() if price_el else "" price = _parse_price(price_text) cal_el = article.query_selector(".card_macros_brief") calories = None protein = None if cal_el: macros_text = cal_el.inner_text() calories, protein = _parse_macros(macros_text) tag_els = article.query_selector_all(".diet_mini_tag") dietary_tags = [t.inner_text().strip().title() for t in tag_els if t.inner_text().strip()] img_el = article.query_selector("img.main_meal_img") image_url = img_el.get_attribute("src") if img_el else None is_new = article.query_selector(".new_badge_pulse") is not None return { "name": name, "price": price, "calories": calories, "protein": protein, "dietary_tags": dietary_tags, "image_url": image_url, "is_new": is_new, "description": None, } except Exception as e: print(f" Warning: failed to extract a card: {e}", file=sys.stderr) return None def _enrich_with_descriptions(page, articles, meals): """Click each meal's Quick View overlay to grab the description.""" for i, article in enumerate(articles): if i >= len(meals): break try: overlay = article.query_selector(".card_overlay") if not overlay: continue article.query_selector(".card_media_wrap").click() page.wait_for_timeout(800) modal = page.query_selector(".modal.show, [class*='modal'][class*='show'], [class*='quickview']") if not modal: # Try broader selector modal = page.query_selector("[class*='modal']:not([style*='display: none'])") if modal and modal.is_visible(): desc_el = modal.query_selector("[class*='description'], [class*='desc'], .meal_description") if desc_el: desc_text = desc_el.inner_text().strip() # Filter out price strings and very short text if desc_text and len(desc_text) > 10 and not desc_text.startswith("$"): meals[i]["description"] = desc_text # Grab full macro details if available detail_tags = modal.query_selector_all(".diet_mini_tag, [class*='lifestyle'] span") for tag_el in detail_tags: tag_text = tag_el.inner_text().strip().title() if tag_text and tag_text not in meals[i]["dietary_tags"]: meals[i]["dietary_tags"].append(tag_text) # Close modal close_btn = modal.query_selector("button[class*='close'], [aria-label='Close'], .btn-close") if close_btn: close_btn.click() else: page.keyboard.press("Escape") page.wait_for_timeout(300) except Exception as e: print(f" Warning: Quick View failed for meal {i} ({meals[i]['name']}): {e}", file=sys.stderr) try: page.keyboard.press("Escape") page.wait_for_timeout(300) except Exception: pass KNOWN_TAGS = ["Gluten Free", "Dairy Free", "Grass-Fed", "Low Carb", "Keto", "Vegan", "Vegetarian", "Nut Free"] def _clean_tags(raw_tags: list[str]) -> list[str]: """Split concatenated tags and deduplicate.""" import re cleaned = set() for raw in raw_tags: # Split on known tag boundaries (e.g., "Gluten Freedairy Free" → "Gluten Free", "Dairy Free") remaining = raw for known in KNOWN_TAGS: if known.lower() in remaining.lower(): cleaned.add(known) remaining = re.sub(re.escape(known), "", remaining, flags=re.IGNORECASE).strip() if remaining and len(remaining) > 2: cleaned.add(remaining.strip().title()) return sorted(cleaned) def _parse_price(text: str) -> float | None: text = text.replace("$", "").replace(",", "").strip() try: return float(text) except ValueError: return None def _parse_macros(text: str) -> tuple[int | None, str | None]: """Parse '570cal • 39gP' into (570, '39g').""" import re cal_match = re.search(r"(\d+)\s*cal", text, re.IGNORECASE) prot_match = re.search(r"(\d+g?)\s*P", text) calories = int(cal_match.group(1)) if cal_match else None protein = prot_match.group(1) if prot_match else None if protein and not protein.endswith("g"): protein += "g" return calories, protein def main(): config = load_config() url = config.get("menu_url", "https://www.redefinemeals.com/menu") output_dir = Path(__file__).resolve().parents[2] / config.get("output_dir", "output") output_dir.mkdir(exist_ok=True) print(f"Scraping menu from {url} ...") result = scrape_menu(url) week_str = datetime.now().strftime("%Y-W%U") output_file = output_dir / f"menu-{week_str}.json" with open(output_file, "w") as f: json.dump(result, f, indent=2) print(f"\nScraped {result['meal_count']} meals") if result["api_endpoints_found"]: print(f"API endpoints detected (potential future shortcut):") for ep in result["api_endpoints_found"]: print(f" {ep}") print(f"Output saved to {output_file}") # Print summary table print(f"\n{'Name':<40} {'Price':>7} {'Cal':>5} {'Prot':>5} {'Tags'}") print("-" * 90) for m in result["meals"]: tags = ", ".join(m["dietary_tags"]) if m["dietary_tags"] else "" new = " *NEW*" if m["is_new"] else "" price = f"${m['price']:.2f}" if m["price"] else "?" cal = str(m["calories"]) if m["calories"] else "?" prot = m["protein"] or "?" print(f"{(m['name'] + new):<40} {price:>7} {cal:>5} {prot:>5} {tags}") if __name__ == "__main__": main()