mirror of
https://github.com/Sea-Haven-Industries/meal-order-manager.git
synced 2026-09-30 08:53:13 +00:00
260 lines
9 KiB
Python
260 lines
9 KiB
Python
|
|
"""
|
||
|
|
Redefine Meals menu scraper.
|
||
|
|
|
||
|
|
Navigates to the menu page using a headless browser, waits for the
|
||
|
|
Vue.js SPA to render, and extracts structured meal data from the DOM.
|
||
|
|
Also intercepts network requests to detect any JSON API that could
|
||
|
|
replace the browser scrape in the future.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
import sys
|
||
|
|
import os
|
||
|
|
from datetime import datetime
|
||
|
|
from pathlib import Path
|
||
|
|
from playwright.sync_api import sync_playwright, TimeoutError as PwTimeout
|
||
|
|
|
||
|
|
CONFIG_PATH = Path(__file__).resolve().parents[2] / "config.json"
|
||
|
|
|
||
|
|
|
||
|
|
def load_config():
|
||
|
|
with open(CONFIG_PATH) as f:
|
||
|
|
return json.load(f)
|
||
|
|
|
||
|
|
|
||
|
|
def scrape_menu(url: str, *, headless: bool = True, timeout_ms: int = 60_000) -> dict:
|
||
|
|
"""
|
||
|
|
Returns {
|
||
|
|
"scraped_at": ISO timestamp,
|
||
|
|
"menu_url": str,
|
||
|
|
"api_endpoints_found": [str],
|
||
|
|
"meals": [ { name, price, calories, protein, dietary_tags,
|
||
|
|
image_url, is_new, description } ]
|
||
|
|
}
|
||
|
|
Raises RuntimeError if the page fails to load or no meals are found.
|
||
|
|
"""
|
||
|
|
api_endpoints = []
|
||
|
|
|
||
|
|
with sync_playwright() as p:
|
||
|
|
browser = p.chromium.launch(headless=headless)
|
||
|
|
context = browser.new_context(
|
||
|
|
user_agent=(
|
||
|
|
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||
|
|
"Chrome/120.0.0.0 Safari/537.36"
|
||
|
|
)
|
||
|
|
)
|
||
|
|
page = context.new_page()
|
||
|
|
|
||
|
|
def on_response(response):
|
||
|
|
ct = response.headers.get("content-type", "")
|
||
|
|
if "json" in ct and "/api/" in response.url and "cart" not in response.url:
|
||
|
|
api_endpoints.append(response.url)
|
||
|
|
|
||
|
|
page.on("response", on_response)
|
||
|
|
|
||
|
|
try:
|
||
|
|
page.goto(url, wait_until="networkidle", timeout=timeout_ms)
|
||
|
|
except PwTimeout:
|
||
|
|
browser.close()
|
||
|
|
raise RuntimeError(f"Timed out loading {url}")
|
||
|
|
|
||
|
|
page.wait_for_timeout(3000)
|
||
|
|
|
||
|
|
articles = page.query_selector_all("article.editorial_card")
|
||
|
|
if not articles:
|
||
|
|
browser.close()
|
||
|
|
raise RuntimeError(
|
||
|
|
"No meal cards found on page. The site layout may have changed. "
|
||
|
|
"Run recon.py to inspect the current structure."
|
||
|
|
)
|
||
|
|
|
||
|
|
meals = []
|
||
|
|
for article in articles:
|
||
|
|
meal = _extract_card(article)
|
||
|
|
if meal:
|
||
|
|
meals.append(meal)
|
||
|
|
|
||
|
|
# Try to get descriptions via Quick View modals
|
||
|
|
_enrich_with_descriptions(page, articles, meals)
|
||
|
|
|
||
|
|
browser.close()
|
||
|
|
|
||
|
|
if not meals:
|
||
|
|
raise RuntimeError("Scraped 0 meals — extraction selectors are likely broken.")
|
||
|
|
|
||
|
|
for meal in meals:
|
||
|
|
meal["dietary_tags"] = _clean_tags(meal["dietary_tags"])
|
||
|
|
|
||
|
|
return {
|
||
|
|
"scraped_at": datetime.now().isoformat(),
|
||
|
|
"menu_url": url,
|
||
|
|
"api_endpoints_found": api_endpoints,
|
||
|
|
"meal_count": len(meals),
|
||
|
|
"meals": meals,
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
def _extract_card(article) -> dict | None:
|
||
|
|
try:
|
||
|
|
name_el = article.query_selector("h2.meal_title")
|
||
|
|
if not name_el:
|
||
|
|
return None
|
||
|
|
name = name_el.inner_text().strip()
|
||
|
|
|
||
|
|
price_el = article.query_selector(".meal_price")
|
||
|
|
price_text = price_el.inner_text().strip() if price_el else ""
|
||
|
|
price = _parse_price(price_text)
|
||
|
|
|
||
|
|
cal_el = article.query_selector(".card_macros_brief")
|
||
|
|
calories = None
|
||
|
|
protein = None
|
||
|
|
if cal_el:
|
||
|
|
macros_text = cal_el.inner_text()
|
||
|
|
calories, protein = _parse_macros(macros_text)
|
||
|
|
|
||
|
|
tag_els = article.query_selector_all(".diet_mini_tag")
|
||
|
|
dietary_tags = [t.inner_text().strip().title() for t in tag_els if t.inner_text().strip()]
|
||
|
|
|
||
|
|
img_el = article.query_selector("img.main_meal_img")
|
||
|
|
image_url = img_el.get_attribute("src") if img_el else None
|
||
|
|
|
||
|
|
is_new = article.query_selector(".new_badge_pulse") is not None
|
||
|
|
|
||
|
|
return {
|
||
|
|
"name": name,
|
||
|
|
"price": price,
|
||
|
|
"calories": calories,
|
||
|
|
"protein": protein,
|
||
|
|
"dietary_tags": dietary_tags,
|
||
|
|
"image_url": image_url,
|
||
|
|
"is_new": is_new,
|
||
|
|
"description": None,
|
||
|
|
}
|
||
|
|
except Exception as e:
|
||
|
|
print(f" Warning: failed to extract a card: {e}", file=sys.stderr)
|
||
|
|
return None
|
||
|
|
|
||
|
|
|
||
|
|
def _enrich_with_descriptions(page, articles, meals):
|
||
|
|
"""Click each meal's Quick View overlay to grab the description."""
|
||
|
|
for i, article in enumerate(articles):
|
||
|
|
if i >= len(meals):
|
||
|
|
break
|
||
|
|
try:
|
||
|
|
overlay = article.query_selector(".card_overlay")
|
||
|
|
if not overlay:
|
||
|
|
continue
|
||
|
|
article.query_selector(".card_media_wrap").click()
|
||
|
|
page.wait_for_timeout(800)
|
||
|
|
|
||
|
|
modal = page.query_selector(".modal.show, [class*='modal'][class*='show'], [class*='quickview']")
|
||
|
|
if not modal:
|
||
|
|
# Try broader selector
|
||
|
|
modal = page.query_selector("[class*='modal']:not([style*='display: none'])")
|
||
|
|
if modal and modal.is_visible():
|
||
|
|
desc_el = modal.query_selector("[class*='description'], [class*='desc'], .meal_description")
|
||
|
|
if desc_el:
|
||
|
|
desc_text = desc_el.inner_text().strip()
|
||
|
|
# Filter out price strings and very short text
|
||
|
|
if desc_text and len(desc_text) > 10 and not desc_text.startswith("$"):
|
||
|
|
meals[i]["description"] = desc_text
|
||
|
|
|
||
|
|
# Grab full macro details if available
|
||
|
|
detail_tags = modal.query_selector_all(".diet_mini_tag, [class*='lifestyle'] span")
|
||
|
|
for tag_el in detail_tags:
|
||
|
|
tag_text = tag_el.inner_text().strip().title()
|
||
|
|
if tag_text and tag_text not in meals[i]["dietary_tags"]:
|
||
|
|
meals[i]["dietary_tags"].append(tag_text)
|
||
|
|
|
||
|
|
# Close modal
|
||
|
|
close_btn = modal.query_selector("button[class*='close'], [aria-label='Close'], .btn-close")
|
||
|
|
if close_btn:
|
||
|
|
close_btn.click()
|
||
|
|
else:
|
||
|
|
page.keyboard.press("Escape")
|
||
|
|
page.wait_for_timeout(300)
|
||
|
|
except Exception as e:
|
||
|
|
print(f" Warning: Quick View failed for meal {i} ({meals[i]['name']}): {e}", file=sys.stderr)
|
||
|
|
try:
|
||
|
|
page.keyboard.press("Escape")
|
||
|
|
page.wait_for_timeout(300)
|
||
|
|
except Exception:
|
||
|
|
pass
|
||
|
|
|
||
|
|
|
||
|
|
KNOWN_TAGS = ["Gluten Free", "Dairy Free", "Grass-Fed", "Low Carb", "Keto", "Vegan", "Vegetarian", "Nut Free"]
|
||
|
|
|
||
|
|
|
||
|
|
def _clean_tags(raw_tags: list[str]) -> list[str]:
|
||
|
|
"""Split concatenated tags and deduplicate."""
|
||
|
|
import re
|
||
|
|
cleaned = set()
|
||
|
|
for raw in raw_tags:
|
||
|
|
# Split on known tag boundaries (e.g., "Gluten Freedairy Free" → "Gluten Free", "Dairy Free")
|
||
|
|
remaining = raw
|
||
|
|
for known in KNOWN_TAGS:
|
||
|
|
if known.lower() in remaining.lower():
|
||
|
|
cleaned.add(known)
|
||
|
|
remaining = re.sub(re.escape(known), "", remaining, flags=re.IGNORECASE).strip()
|
||
|
|
if remaining and len(remaining) > 2:
|
||
|
|
cleaned.add(remaining.strip().title())
|
||
|
|
return sorted(cleaned)
|
||
|
|
|
||
|
|
|
||
|
|
def _parse_price(text: str) -> float | None:
|
||
|
|
text = text.replace("$", "").replace(",", "").strip()
|
||
|
|
try:
|
||
|
|
return float(text)
|
||
|
|
except ValueError:
|
||
|
|
return None
|
||
|
|
|
||
|
|
|
||
|
|
def _parse_macros(text: str) -> tuple[int | None, str | None]:
|
||
|
|
"""Parse '570cal • 39gP' into (570, '39g')."""
|
||
|
|
import re
|
||
|
|
cal_match = re.search(r"(\d+)\s*cal", text, re.IGNORECASE)
|
||
|
|
prot_match = re.search(r"(\d+g?)\s*P", text)
|
||
|
|
calories = int(cal_match.group(1)) if cal_match else None
|
||
|
|
protein = prot_match.group(1) if prot_match else None
|
||
|
|
if protein and not protein.endswith("g"):
|
||
|
|
protein += "g"
|
||
|
|
return calories, protein
|
||
|
|
|
||
|
|
|
||
|
|
def main():
|
||
|
|
config = load_config()
|
||
|
|
url = config.get("menu_url", "https://www.redefinemeals.com/menu")
|
||
|
|
output_dir = Path(__file__).resolve().parents[2] / config.get("output_dir", "output")
|
||
|
|
output_dir.mkdir(exist_ok=True)
|
||
|
|
|
||
|
|
print(f"Scraping menu from {url} ...")
|
||
|
|
result = scrape_menu(url)
|
||
|
|
|
||
|
|
week_str = datetime.now().strftime("%Y-W%U")
|
||
|
|
output_file = output_dir / f"menu-{week_str}.json"
|
||
|
|
with open(output_file, "w") as f:
|
||
|
|
json.dump(result, f, indent=2)
|
||
|
|
|
||
|
|
print(f"\nScraped {result['meal_count']} meals")
|
||
|
|
if result["api_endpoints_found"]:
|
||
|
|
print(f"API endpoints detected (potential future shortcut):")
|
||
|
|
for ep in result["api_endpoints_found"]:
|
||
|
|
print(f" {ep}")
|
||
|
|
print(f"Output saved to {output_file}")
|
||
|
|
|
||
|
|
# Print summary table
|
||
|
|
print(f"\n{'Name':<40} {'Price':>7} {'Cal':>5} {'Prot':>5} {'Tags'}")
|
||
|
|
print("-" * 90)
|
||
|
|
for m in result["meals"]:
|
||
|
|
tags = ", ".join(m["dietary_tags"]) if m["dietary_tags"] else ""
|
||
|
|
new = " *NEW*" if m["is_new"] else ""
|
||
|
|
price = f"${m['price']:.2f}" if m["price"] else "?"
|
||
|
|
cal = str(m["calories"]) if m["calories"] else "?"
|
||
|
|
prot = m["protein"] or "?"
|
||
|
|
print(f"{(m['name'] + new):<40} {price:>7} {cal:>5} {prot:>5} {tags}")
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
main()
|