meal-order-manager/src/scraper/scrape_menu.py

260 lines
9 KiB
Python
Raw Normal View History

"""
Redefine Meals menu scraper.
Navigates to the menu page using a headless browser, waits for the
Vue.js SPA to render, and extracts structured meal data from the DOM.
Also intercepts network requests to detect any JSON API that could
replace the browser scrape in the future.
"""
import json
import sys
import os
from datetime import datetime
from pathlib import Path
from playwright.sync_api import sync_playwright, TimeoutError as PwTimeout
CONFIG_PATH = Path(__file__).resolve().parents[2] / "config.json"
def load_config():
with open(CONFIG_PATH) as f:
return json.load(f)
def scrape_menu(url: str, *, headless: bool = True, timeout_ms: int = 60_000) -> dict:
"""
Returns {
"scraped_at": ISO timestamp,
"menu_url": str,
"api_endpoints_found": [str],
"meals": [ { name, price, calories, protein, dietary_tags,
image_url, is_new, description } ]
}
Raises RuntimeError if the page fails to load or no meals are found.
"""
api_endpoints = []
with sync_playwright() as p:
browser = p.chromium.launch(headless=headless)
context = browser.new_context(
user_agent=(
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36"
)
)
page = context.new_page()
def on_response(response):
ct = response.headers.get("content-type", "")
if "json" in ct and "/api/" in response.url and "cart" not in response.url:
api_endpoints.append(response.url)
page.on("response", on_response)
try:
page.goto(url, wait_until="networkidle", timeout=timeout_ms)
except PwTimeout:
browser.close()
raise RuntimeError(f"Timed out loading {url}")
page.wait_for_timeout(3000)
articles = page.query_selector_all("article.editorial_card")
if not articles:
browser.close()
raise RuntimeError(
"No meal cards found on page. The site layout may have changed. "
"Run recon.py to inspect the current structure."
)
meals = []
for article in articles:
meal = _extract_card(article)
if meal:
meals.append(meal)
# Try to get descriptions via Quick View modals
_enrich_with_descriptions(page, articles, meals)
browser.close()
if not meals:
raise RuntimeError("Scraped 0 meals — extraction selectors are likely broken.")
for meal in meals:
meal["dietary_tags"] = _clean_tags(meal["dietary_tags"])
return {
"scraped_at": datetime.now().isoformat(),
"menu_url": url,
"api_endpoints_found": api_endpoints,
"meal_count": len(meals),
"meals": meals,
}
def _extract_card(article) -> dict | None:
try:
name_el = article.query_selector("h2.meal_title")
if not name_el:
return None
name = name_el.inner_text().strip()
price_el = article.query_selector(".meal_price")
price_text = price_el.inner_text().strip() if price_el else ""
price = _parse_price(price_text)
cal_el = article.query_selector(".card_macros_brief")
calories = None
protein = None
if cal_el:
macros_text = cal_el.inner_text()
calories, protein = _parse_macros(macros_text)
tag_els = article.query_selector_all(".diet_mini_tag")
dietary_tags = [t.inner_text().strip().title() for t in tag_els if t.inner_text().strip()]
img_el = article.query_selector("img.main_meal_img")
image_url = img_el.get_attribute("src") if img_el else None
is_new = article.query_selector(".new_badge_pulse") is not None
return {
"name": name,
"price": price,
"calories": calories,
"protein": protein,
"dietary_tags": dietary_tags,
"image_url": image_url,
"is_new": is_new,
"description": None,
}
except Exception as e:
print(f" Warning: failed to extract a card: {e}", file=sys.stderr)
return None
def _enrich_with_descriptions(page, articles, meals):
"""Click each meal's Quick View overlay to grab the description."""
for i, article in enumerate(articles):
if i >= len(meals):
break
try:
overlay = article.query_selector(".card_overlay")
if not overlay:
continue
article.query_selector(".card_media_wrap").click()
page.wait_for_timeout(800)
modal = page.query_selector(".modal.show, [class*='modal'][class*='show'], [class*='quickview']")
if not modal:
# Try broader selector
modal = page.query_selector("[class*='modal']:not([style*='display: none'])")
if modal and modal.is_visible():
desc_el = modal.query_selector("[class*='description'], [class*='desc'], .meal_description")
if desc_el:
desc_text = desc_el.inner_text().strip()
# Filter out price strings and very short text
if desc_text and len(desc_text) > 10 and not desc_text.startswith("$"):
meals[i]["description"] = desc_text
# Grab full macro details if available
detail_tags = modal.query_selector_all(".diet_mini_tag, [class*='lifestyle'] span")
for tag_el in detail_tags:
tag_text = tag_el.inner_text().strip().title()
if tag_text and tag_text not in meals[i]["dietary_tags"]:
meals[i]["dietary_tags"].append(tag_text)
# Close modal
close_btn = modal.query_selector("button[class*='close'], [aria-label='Close'], .btn-close")
if close_btn:
close_btn.click()
else:
page.keyboard.press("Escape")
page.wait_for_timeout(300)
except Exception as e:
print(f" Warning: Quick View failed for meal {i} ({meals[i]['name']}): {e}", file=sys.stderr)
try:
page.keyboard.press("Escape")
page.wait_for_timeout(300)
except Exception:
pass
KNOWN_TAGS = ["Gluten Free", "Dairy Free", "Grass-Fed", "Low Carb", "Keto", "Vegan", "Vegetarian", "Nut Free"]
def _clean_tags(raw_tags: list[str]) -> list[str]:
"""Split concatenated tags and deduplicate."""
import re
cleaned = set()
for raw in raw_tags:
# Split on known tag boundaries (e.g., "Gluten Freedairy Free" → "Gluten Free", "Dairy Free")
remaining = raw
for known in KNOWN_TAGS:
if known.lower() in remaining.lower():
cleaned.add(known)
remaining = re.sub(re.escape(known), "", remaining, flags=re.IGNORECASE).strip()
if remaining and len(remaining) > 2:
cleaned.add(remaining.strip().title())
return sorted(cleaned)
def _parse_price(text: str) -> float | None:
text = text.replace("$", "").replace(",", "").strip()
try:
return float(text)
except ValueError:
return None
def _parse_macros(text: str) -> tuple[int | None, str | None]:
"""Parse '570cal • 39gP' into (570, '39g')."""
import re
cal_match = re.search(r"(\d+)\s*cal", text, re.IGNORECASE)
prot_match = re.search(r"(\d+g?)\s*P", text)
calories = int(cal_match.group(1)) if cal_match else None
protein = prot_match.group(1) if prot_match else None
if protein and not protein.endswith("g"):
protein += "g"
return calories, protein
def main():
config = load_config()
url = config.get("menu_url", "https://www.redefinemeals.com/menu")
output_dir = Path(__file__).resolve().parents[2] / config.get("output_dir", "output")
output_dir.mkdir(exist_ok=True)
print(f"Scraping menu from {url} ...")
result = scrape_menu(url)
week_str = datetime.now().strftime("%Y-W%U")
output_file = output_dir / f"menu-{week_str}.json"
with open(output_file, "w") as f:
json.dump(result, f, indent=2)
print(f"\nScraped {result['meal_count']} meals")
if result["api_endpoints_found"]:
print(f"API endpoints detected (potential future shortcut):")
for ep in result["api_endpoints_found"]:
print(f" {ep}")
print(f"Output saved to {output_file}")
# Print summary table
print(f"\n{'Name':<40} {'Price':>7} {'Cal':>5} {'Prot':>5} {'Tags'}")
print("-" * 90)
for m in result["meals"]:
tags = ", ".join(m["dietary_tags"]) if m["dietary_tags"] else ""
new = " *NEW*" if m["is_new"] else ""
price = f"${m['price']:.2f}" if m["price"] else "?"
cal = str(m["calories"]) if m["calories"] else "?"
prot = m["protein"] or "?"
print(f"{(m['name'] + new):<40} {price:>7} {cal:>5} {prot:>5} {tags}")
if __name__ == "__main__":
main()