mirror of
https://github.com/Sea-Haven-Industries/meal-order-manager.git
synced 2026-10-05 09:31:59 +00:00
123 lines
4.7 KiB
Python
123 lines
4.7 KiB
Python
|
|
"""
|
||
|
|
Reconnaissance script: loads redefinemeals.com/menu in a real browser,
|
||
|
|
intercepts all network requests, and dumps the page structure.
|
||
|
|
Run this once to understand the site before building the production scraper.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
import sys
|
||
|
|
from playwright.sync_api import sync_playwright
|
||
|
|
|
||
|
|
|
||
|
|
def run_recon():
|
||
|
|
api_responses = []
|
||
|
|
all_requests = []
|
||
|
|
|
||
|
|
with sync_playwright() as p:
|
||
|
|
browser = p.chromium.launch(headless=True)
|
||
|
|
context = browser.new_context(
|
||
|
|
user_agent="Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||
|
|
)
|
||
|
|
page = context.new_page()
|
||
|
|
|
||
|
|
def handle_response(response):
|
||
|
|
url = response.url
|
||
|
|
content_type = response.headers.get("content-type", "")
|
||
|
|
all_requests.append({
|
||
|
|
"url": url,
|
||
|
|
"status": response.status,
|
||
|
|
"content_type": content_type,
|
||
|
|
})
|
||
|
|
if "json" in content_type or "graphql" in url.lower():
|
||
|
|
try:
|
||
|
|
body = response.json()
|
||
|
|
api_responses.append({
|
||
|
|
"url": url,
|
||
|
|
"status": response.status,
|
||
|
|
"body_preview": json.dumps(body)[:2000],
|
||
|
|
})
|
||
|
|
except Exception:
|
||
|
|
pass
|
||
|
|
elif url.endswith(".json") or "/api/" in url:
|
||
|
|
try:
|
||
|
|
body = response.text()
|
||
|
|
api_responses.append({
|
||
|
|
"url": url,
|
||
|
|
"status": response.status,
|
||
|
|
"body_preview": body[:2000],
|
||
|
|
})
|
||
|
|
except Exception:
|
||
|
|
pass
|
||
|
|
|
||
|
|
page.on("response", handle_response)
|
||
|
|
|
||
|
|
print("Navigating to https://redefinemeals.com/menu ...")
|
||
|
|
page.goto("https://redefinemeals.com/menu", wait_until="networkidle", timeout=60000)
|
||
|
|
page.wait_for_timeout(5000)
|
||
|
|
|
||
|
|
print(f"\n{'='*60}")
|
||
|
|
print("PAGE TITLE:", page.title())
|
||
|
|
print(f"{'='*60}")
|
||
|
|
|
||
|
|
print(f"\n--- All network requests ({len(all_requests)} total) ---")
|
||
|
|
for req in all_requests:
|
||
|
|
if any(ext in req["url"] for ext in [".png", ".jpg", ".jpeg", ".gif", ".svg", ".ico", ".woff", ".woff2", ".ttf", ".css"]):
|
||
|
|
continue
|
||
|
|
print(f" [{req['status']}] {req['content_type'][:30]:30s} {req['url'][:120]}")
|
||
|
|
|
||
|
|
print(f"\n--- JSON/API responses ({len(api_responses)} found) ---")
|
||
|
|
for resp in api_responses:
|
||
|
|
print(f"\n URL: {resp['url']}")
|
||
|
|
print(f" Status: {resp['status']}")
|
||
|
|
print(f" Body preview:\n {resp['body_preview'][:500]}")
|
||
|
|
|
||
|
|
print(f"\n--- Page structure (meal-related elements) ---")
|
||
|
|
for selector in [
|
||
|
|
"[class*='meal']", "[class*='menu']", "[class*='product']",
|
||
|
|
"[class*='item']", "[class*='card']", "[data-product]",
|
||
|
|
"[data-item]", ".grid > div", "article",
|
||
|
|
]:
|
||
|
|
elements = page.query_selector_all(selector)
|
||
|
|
if elements:
|
||
|
|
print(f"\n Selector '{selector}': {len(elements)} elements")
|
||
|
|
if elements:
|
||
|
|
first = elements[0]
|
||
|
|
print(f" Tag: {first.evaluate('el => el.tagName')}")
|
||
|
|
print(f" Classes: {first.evaluate('el => el.className')}")
|
||
|
|
inner = first.inner_text()
|
||
|
|
print(f" Text preview: {inner[:200]}")
|
||
|
|
|
||
|
|
# Also dump outer HTML of likely meal containers
|
||
|
|
print(f"\n--- Raw HTML sample (first product/card element) ---")
|
||
|
|
for selector in ["[class*='product']", "[class*='card']", "[class*='meal']", "[class*='menu-item']"]:
|
||
|
|
els = page.query_selector_all(selector)
|
||
|
|
if els:
|
||
|
|
html = els[0].evaluate("el => el.outerHTML")
|
||
|
|
print(f"\n Selector: {selector}")
|
||
|
|
print(f" Count: {len(els)}")
|
||
|
|
print(f" First element HTML:\n{html[:1500]}")
|
||
|
|
break
|
||
|
|
|
||
|
|
# Check for Shopify, WooCommerce, or other known platforms
|
||
|
|
print(f"\n--- Platform detection ---")
|
||
|
|
platform_checks = {
|
||
|
|
"Shopify": "Shopify" in page.content(),
|
||
|
|
"WooCommerce": "woocommerce" in page.content().lower(),
|
||
|
|
"Squarespace": "squarespace" in page.content().lower(),
|
||
|
|
"Wix": "wix" in page.content().lower(),
|
||
|
|
}
|
||
|
|
for name, found in platform_checks.items():
|
||
|
|
if found:
|
||
|
|
print(f" Detected: {name}")
|
||
|
|
|
||
|
|
# Try to get the full page URL after any redirects
|
||
|
|
print(f"\n Final URL: {page.url}")
|
||
|
|
|
||
|
|
browser.close()
|
||
|
|
|
||
|
|
return api_responses
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
run_recon()
|