meal-order-manager/src/scraper/recon.py
Adam Moussa 440117c9a1
Fix ruff lint and format violations, update README (#5)
Apply ruff check --fix and ruff format across all Python files to
pass CI pipeline. Remove unused imports (os, sys), fix f-strings
without placeholders. Update README to reflect sync-roster Lambda,
corrected shared layer path, and current project structure.
2026-05-12 19:31:27 -04:00

156 lines
5.3 KiB
Python

"""
Reconnaissance script: loads redefinemeals.com/menu in a real browser,
intercepts all network requests, and dumps the page structure.
Run this once to understand the site before building the production scraper.
"""
import json
from playwright.sync_api import sync_playwright
def run_recon():
api_responses = []
all_requests = []
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
context = browser.new_context(
user_agent="Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
page = context.new_page()
def handle_response(response):
url = response.url
content_type = response.headers.get("content-type", "")
all_requests.append(
{
"url": url,
"status": response.status,
"content_type": content_type,
}
)
if "json" in content_type or "graphql" in url.lower():
try:
body = response.json()
api_responses.append(
{
"url": url,
"status": response.status,
"body_preview": json.dumps(body)[:2000],
}
)
except Exception:
pass
elif url.endswith(".json") or "/api/" in url:
try:
body = response.text()
api_responses.append(
{
"url": url,
"status": response.status,
"body_preview": body[:2000],
}
)
except Exception:
pass
page.on("response", handle_response)
print("Navigating to https://redefinemeals.com/menu ...")
page.goto(
"https://redefinemeals.com/menu", wait_until="networkidle", timeout=60000
)
page.wait_for_timeout(5000)
print(f"\n{'=' * 60}")
print("PAGE TITLE:", page.title())
print(f"{'=' * 60}")
print(f"\n--- All network requests ({len(all_requests)} total) ---")
for req in all_requests:
if any(
ext in req["url"]
for ext in [
".png",
".jpg",
".jpeg",
".gif",
".svg",
".ico",
".woff",
".woff2",
".ttf",
".css",
]
):
continue
print(
f" [{req['status']}] {req['content_type'][:30]:30s} {req['url'][:120]}"
)
print(f"\n--- JSON/API responses ({len(api_responses)} found) ---")
for resp in api_responses:
print(f"\n URL: {resp['url']}")
print(f" Status: {resp['status']}")
print(f" Body preview:\n {resp['body_preview'][:500]}")
print("\n--- Page structure (meal-related elements) ---")
for selector in [
"[class*='meal']",
"[class*='menu']",
"[class*='product']",
"[class*='item']",
"[class*='card']",
"[data-product]",
"[data-item]",
".grid > div",
"article",
]:
elements = page.query_selector_all(selector)
if elements:
print(f"\n Selector '{selector}': {len(elements)} elements")
if elements:
first = elements[0]
print(f" Tag: {first.evaluate('el => el.tagName')}")
print(f" Classes: {first.evaluate('el => el.className')}")
inner = first.inner_text()
print(f" Text preview: {inner[:200]}")
# Also dump outer HTML of likely meal containers
print("\n--- Raw HTML sample (first product/card element) ---")
for selector in [
"[class*='product']",
"[class*='card']",
"[class*='meal']",
"[class*='menu-item']",
]:
els = page.query_selector_all(selector)
if els:
html = els[0].evaluate("el => el.outerHTML")
print(f"\n Selector: {selector}")
print(f" Count: {len(els)}")
print(f" First element HTML:\n{html[:1500]}")
break
# Check for Shopify, WooCommerce, or other known platforms
print("\n--- Platform detection ---")
platform_checks = {
"Shopify": "Shopify" in page.content(),
"WooCommerce": "woocommerce" in page.content().lower(),
"Squarespace": "squarespace" in page.content().lower(),
"Wix": "wix" in page.content().lower(),
}
for name, found in platform_checks.items():
if found:
print(f" Detected: {name}")
# Try to get the full page URL after any redirects
print(f"\n Final URL: {page.url}")
browser.close()
return api_responses
if __name__ == "__main__":
run_recon()