meal-order-manager/src/scraper/recon_deep.py
Adam Moussa 440117c9a1
Fix ruff lint and format violations, update README (#5)
Apply ruff check --fix and ruff format across all Python files to
pass CI pipeline. Remove unused imports (os, sys), fix f-strings
without placeholders. Update README to reflect sync-roster Lambda,
corrected shared layer path, and current project structure.
2026-05-12 19:31:27 -04:00

137 lines
4.5 KiB
Python

"""
Deep recon: extract the full structure of a single meal card
and check for menu-related API endpoints.
"""
import json
from playwright.sync_api import sync_playwright
def run():
api_hits = []
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
context = browser.new_context(
user_agent="Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
)
page = context.new_page()
def capture_response(response):
ct = response.headers.get("content-type", "")
if "json" in ct or "/api/" in response.url:
try:
body = response.json()
api_hits.append({"url": response.url, "body": body})
except Exception:
pass
page.on("response", capture_response)
page.goto(
"https://www.redefinemeals.com/menu",
wait_until="networkidle",
timeout=60000,
)
page.wait_for_timeout(3000)
# Extract detailed structure from first few article cards
articles = page.query_selector_all("article.editorial_card")
print(f"Found {len(articles)} article.editorial_card elements\n")
for i, article in enumerate(articles[:3]):
html = article.evaluate("el => el.outerHTML")
text = article.inner_text()
print(f"--- Article {i} ---")
print(f"Text:\n{text}\n")
print(f"HTML:\n{html[:3000]}\n")
# Check for dietary tag elements
print("\n--- Dietary/tag elements ---")
for sel in [
"[class*='tag']",
"[class*='diet']",
"[class*='label']",
"[class*='badge']",
"[class*='filter']",
]:
els = page.query_selector_all(sel)
if els:
texts = [
e.inner_text().strip() for e in els[:10] if e.inner_text().strip()
]
print(f" {sel}: {len(els)} elements, samples: {texts}")
# Check for filter/category buttons
print("\n--- Filter/category buttons ---")
for sel in [
"button",
"[class*='filter']",
"[class*='category']",
"[class*='tab']",
]:
els = page.query_selector_all(sel)
if els:
texts = [
e.inner_text().strip() for e in els[:20] if e.inner_text().strip()
]
print(f" {sel}: {texts}")
# Try known API patterns
print("\n--- Trying API endpoints ---")
for path in [
"/api/menu",
"/api/products",
"/api/meals",
"/api/items",
"/api/categories",
]:
try:
resp = page.evaluate(f"""
async () => {{
const r = await fetch('{path}');
if (r.ok) return await r.text();
return `${{r.status}}`;
}}
""")
if resp and resp not in ["404", "500", "403"]:
print(f" {path}: {resp[:500]}")
else:
print(f" {path}: {resp}")
except Exception as e:
print(f" {path}: error - {e}")
# Check for Quick View modal data
print("\n--- Quick View data (click first meal) ---")
quick_view_btns = page.query_selector_all("[class*='quick']")
if quick_view_btns:
print(f" Found {len(quick_view_btns)} Quick View buttons")
try:
quick_view_btns[0].click()
page.wait_for_timeout(2000)
# Look for modal content
for sel in [
".modal",
"[class*='modal']",
"[class*='popup']",
"[class*='quick-view']",
"[class*='quickview']",
]:
modal = page.query_selector(sel)
if modal and modal.is_visible():
print(f" Modal selector: {sel}")
print(f" Modal text:\n{modal.inner_text()[:1000]}")
break
except Exception as e:
print(f" Quick View click failed: {e}")
print("\n--- API calls captured ---")
for hit in api_hits:
print(f" {hit['url']}")
print(f" {json.dumps(hit['body'])[:500]}\n")
browser.close()
if __name__ == "__main__":
run()