diff --git a/.github/workflows/deploy-api.yaml b/.github/workflows/deploy-api.yaml index d663953..527eab9 100644 --- a/.github/workflows/deploy-api.yaml +++ b/.github/workflows/deploy-api.yaml @@ -18,7 +18,6 @@ on: - "terraform/**" - "docs/**" - "*.md" - - ".github/workflows/weekly-menu.yml" - ".github/workflows/ci.yml" release: types: [published] diff --git a/.github/workflows/weekly-menu.yml b/.github/workflows/weekly-menu.yml deleted file mode 100644 index 02198e7..0000000 --- a/.github/workflows/weekly-menu.yml +++ /dev/null @@ -1,179 +0,0 @@ -name: Weekly Menu Scrape & Publish - -on: - schedule: - # Monday 7:30am EST = 12:30 UTC - - cron: '30 12 * * 1' - # Monday 7:30am EDT = 11:30 UTC - - cron: '30 11 * * 1' - workflow_dispatch: - -permissions: - id-token: write - contents: read - -concurrency: - group: weekly-menu - cancel-in-progress: false - -jobs: - scrape-and-publish: - runs-on: ubuntu-latest - # A hung Playwright scrape would otherwise hold the weekly-menu concurrency - # group for the 360-minute default. - timeout-minutes: 30 - env: - AWS_REGION: us-east-1 - - steps: - - name: Timezone guard - if: github.event_name == 'schedule' - env: - CRON: ${{ github.event.schedule }} - run: | - OFFSET=$(TZ='America/New_York' date +%z) - echo "Cron: $CRON | Eastern offset: $OFFSET" - if { [ "$OFFSET" = "-0400" ] && [ "$CRON" = "30 12 * * 1" ]; } || \ - { [ "$OFFSET" = "-0500" ] && [ "$CRON" = "30 11 * * 1" ]; }; then - echo "Wrong-timezone cron fired — skipping" - echo "SKIP_RUN=true" >> "$GITHUB_ENV" - fi - - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - if: env.SKIP_RUN != 'true' - - - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 - if: env.SKIP_RUN != 'true' - with: - python-version: '3.12.14' - - - name: Install dependencies - if: env.SKIP_RUN != 'true' - run: | - pip install -r requirements.txt - playwright install chromium --with-deps - - - name: Configure AWS credentials - if: env.SKIP_RUN != 'true' - uses: aws-actions/configure-aws-credentials@e1253824e5c10ff9df46874f81ed3ec929e19cfd # v6.3.0 - with: - role-to-assume: ${{ secrets.AWS_WEEKLY_MENU_ROLE_ARN }} - aws-region: us-east-1 - - - name: Scrape menu - if: env.SKIP_RUN != 'true' - run: python3 src/scraper/scrape_menu.py - - # Deploy targets come from Parameter Store, written by Terraform - # (terraform/ssm.tf). They replace the CloudFormation stack outputs this - # job used to read; there is no CloudFormation stack any more. - - name: Get deploy parameters - if: env.SKIP_RUN != 'true' - id: stack - run: | - set -euo pipefail - get_param() { - aws ssm get-parameter --name "$1" --query 'Parameter.Value' --output text - } - API_URL=$(get_param /meal-order-manager/deploy/api-url) - FORM_BUCKET=$(get_param /meal-order-manager/deploy/form-bucket) - DIST_ID=$(get_param /meal-order-manager/deploy/distribution-id) - FORM_URL=$(get_param /meal-order-manager/deploy/form-url) - for v in "$API_URL" "$FORM_BUCKET" "$DIST_ID" "$FORM_URL"; do - if [ -z "$v" ] || [ "$v" = "None" ]; then - echo "A /meal-order-manager/deploy/* parameter is missing; has Terraform been applied?" >&2 - exit 1 - fi - done - echo "api_url=$API_URL" >> "$GITHUB_OUTPUT" - echo "form_bucket=$FORM_BUCKET" >> "$GITHUB_OUTPUT" - echo "dist_id=$DIST_ID" >> "$GITHUB_OUTPUT" - echo "form_url=$FORM_URL" >> "$GITHUB_OUTPUT" - - - name: Get discount settings - if: env.SKIP_RUN != 'true' - id: discount - env: - API_URL: ${{ steps.stack.outputs.api_url }} - run: | - set -euo pipefail - MEALS_PUBLISH_KEY=$(aws ssm get-parameter \ - --name /meal-order-manager/publish-key \ - --with-decryption \ - --query 'Parameter.Value' \ - --output text) - SETTINGS=$(MEALS_PUBLISH_KEY="$MEALS_PUBLISH_KEY" python3 scripts/upload_menu.py settings --api-url "$API_URL") - BULK=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["bulk_discount_percent"])' "$SETTINGS") - SUBSIDY=$(python3 -c 'import json,sys; print(json.loads(sys.argv[1])["company_subsidy_percent"])' "$SETTINGS") - echo "bulk_discount=$BULK" >> "$GITHUB_OUTPUT" - echo "company_subsidy=$SUBSIDY" >> "$GITHUB_OUTPUT" - - - name: Get Google Client ID - if: env.SKIP_RUN != 'true' - id: google - run: | - GOOGLE_CLIENT_ID=$(aws ssm get-parameter \ - --name /meal-order-manager/google-client-id \ - --query 'Parameter.Value' \ - --output text) - if [ "$GOOGLE_CLIENT_ID" = "None" ] || [ -z "$GOOGLE_CLIENT_ID" ]; then - echo "Google client ID is required for cloud form generation" >&2 - exit 1 - fi - echo "client_id=$GOOGLE_CLIENT_ID" >> "$GITHUB_OUTPUT" - - - name: Generate order form - if: env.SKIP_RUN != 'true' - env: - BULK_DISCOUNT: ${{ steps.discount.outputs.bulk_discount }} - COMPANY_SUBSIDY: ${{ steps.discount.outputs.company_subsidy }} - GOOGLE_CLIENT_ID: ${{ steps.google.outputs.client_id }} - run: | - # Relative /api paths so the form stays same-origin on CloudFront - # after the ALB origin swap. Do not bake the ALB DNS into HTML. - python3 src/server/generate_form.py \ - --bulk-discount "$BULK_DISCOUNT" \ - --company-subsidy "$COMPANY_SUBSIDY" \ - --google-client-id "$GOOGLE_CLIENT_ID" - - - name: Publish menu through API - if: env.SKIP_RUN != 'true' - env: - API_URL: ${{ steps.stack.outputs.api_url }} - run: | - set -euo pipefail - MEALS_PUBLISH_KEY=$(aws ssm get-parameter \ - --name /meal-order-manager/publish-key \ - --with-decryption \ - --query 'Parameter.Value' \ - --output text) - MEALS_PUBLISH_KEY="$MEALS_PUBLISH_KEY" python3 scripts/upload_menu.py publish --api-url "$API_URL" - - - name: Upload form to S3 - if: env.SKIP_RUN != 'true' - env: - FORM_BUCKET: ${{ steps.stack.outputs.form_bucket }} - run: | - WEEK=$(date +%Y-W%U) - aws s3 cp "output/order-form-$WEEK.html" \ - "s3://${FORM_BUCKET}/index.html" \ - --content-type "text/html" \ - --cache-control "no-cache" - aws s3 cp "output/order-form-$WEEK.html" \ - "s3://${FORM_BUCKET}/archive/$WEEK.html" \ - --content-type "text/html" - - - name: Invalidate CloudFront cache - if: env.SKIP_RUN != 'true' - env: - DIST_ID: ${{ steps.stack.outputs.dist_id }} - run: | - aws cloudfront create-invalidation \ - --distribution-id "$DIST_ID" \ - --paths "/index.html" - - - name: Notify Slack - if: env.SKIP_RUN != 'true' - env: - FORM_URL: ${{ steps.stack.outputs.form_url }} - run: python3 scripts/notify_slack.py "$FORM_URL" diff --git a/README.md b/README.md index 3b2e3ec..e627425 100644 --- a/README.md +++ b/README.md @@ -10,26 +10,21 @@ Automates weekly meal ordering from [Redefine Meals](https://www.redefinemeals.c ## Architecture ``` -Monday 7:30am ET Employees (Mon–Thu) Thursday 11:59pm ET -┌─────────────────┐ ┌──────────────────┐ ┌──────────────────┐ -│ GitHub Actions │ │ orders.seahaven │ │ EventBridge │ -│ - Scrape menu │────S3 upload───▶│ .com │ │ Scheduler (ET) │ -│ - HMAC publish │ │ (CloudFront+S3) │──POST───┐ │ → jobs SQS │ -│ - Slack notify │ └──────────────────┘ │ └─────────┬────────┘ -└─────────────────┘ ▼ │ - ┌──────────┐ │ -Thu 10am: Slack DM │ ALB + │◀──────────┘ -reminders to employees │ Fargate │ -who haven't ordered │ Flask │ - └────┬─────┘ - ▼ - ┌──────────┐ - │ DynamoDB │ - │ orders │ - └──────────┘ +EventBridge Scheduler (America/New_York) Employees +┌──────────────────────────────────────┐ ┌────────────────────┐ +│ Mon 6:55 roster, Mon 7:30 menu │ │ orders.seahaven.com│ +│ Thu 10:00 reminder, Thu 23:59 close │ │ CloudFront + S3 │ +└──────────────────┬───────────────────┘ └─────────┬──────────┘ + │ jobs SQS │ POST /api + ▼ ▼ + ┌──────────────────────────────────────────────────┐ + │ Fargate: Flask + SQS worker │ + │ menu → DynamoDB │ + │ form HTML → S3, then invalidate /index.html │ + └──────────────────────────────────────────────────┘ ``` -The production HTTP app is `src/server/app.py` (gunicorn). Close, aggregate/PDF, Slack reminder, and roster sync run in the same task from a dedicated SQS consumer (`src/server/worker.py`). Playwright scrape stays in GitHub Actions. +The production HTTP app is `src/server/app.py` (gunicorn). Menu publish, close, aggregate/PDF, Slack reminder, and roster sync run in the same task from a dedicated SQS consumer (`src/server/worker.py`). Menu publish fetches the Redefine HTML catalog. It does not run a browser. ### Form frontend decisions @@ -45,21 +40,21 @@ the generated HTML, and the generated deployment artifact remains self-contained | When | What | How | |------|------|-----| | Monday 6:55am ET | Sync employee roster from Slack channel membership | EventBridge Scheduler → jobs SQS → Fargate | -| Monday 7:30am ET | Scrape menu, generate form, HMAC-publish menu, upload form to S3, post link to Slack | GitHub Actions cron | +| Monday 7:30am ET | Fetch menu, generate form, write menu, upload form to S3, invalidate CloudFront, post link to Slack | EventBridge Scheduler → jobs SQS | | Mon–Thu | Employees visit `orders.seahaven.com` and submit orders | S3 form → CloudFront `/api/*` → ALB → Flask → DynamoDB | | Thursday 10am ET | DM employees who haven't ordered yet | EventBridge Scheduler → jobs SQS | | Thursday 11:59pm ET | Close form, aggregate orders, write CSV reports + weekly summary PDF, post Redefine order summary to Slack | EventBridge Scheduler → jobs SQS | -### Weekly menu publication boundary +### Weekly menu publication -The scheduled GitHub workflow has no DynamoDB permissions. It sends two HMAC -requests with `X-Meals-Publish-Key` from Parameter Store to the ALB: +Monday 7:30am Eastern, EventBridge Scheduler enqueues `publish_menu`. The Fargate worker fetches the Redefine menu HTML, parses the embedded catalog, writes that Eastern-time week's menu to DynamoDB, renders the form, uploads it to the form bucket, invalidates `/index.html`, and posts to Slack. + +`scripts/upload_menu.py` remains a manual HMAC fallback for one production Monday: - `GET /api/publish/settings` returns only the bulk discount and company subsidy. - `POST /api/publish/menu` validates and writes the current Eastern-time week's menu. -Publish routes are omitted from CloudFront. The generated form uses relative -`/api/...` paths so it stays same-origin on `orders.seahaven.com`. +Publish routes are omitted from CloudFront. The generated form uses relative `/api/...` paths so it stays same-origin on `orders.seahaven.com`. ### Reports (written to `meal-order-manager-reports-*` at Thursday close) @@ -81,7 +76,7 @@ Workspace: `meal-order-manager-prod` / `meal-order-manager-dev` (us-east-1) - **S3** — `meal-order-manager-form-*` (static form hosting), `meal-order-manager-reports-*` (CSV reports + weekly summary PDF) - **CloudFront** — HTTPS distribution with custom domain `orders.seahaven.com`. API origin is the ALB (HTTP-only). - **DynamoDB** — `meal-order-manager-orders` (orders, menu, roster, config) -- **SQS** — `meal-order-manager-jobs` (+ DLQ). EventBridge Scheduler in `America/New_York` enqueues close, reminder, and roster sync. +- **SQS** — `meal-order-manager-jobs` (+ DLQ). EventBridge Scheduler in `America/New_York` enqueues menu publish, close, reminder, and roster sync. - **Secrets Manager** — Slack bot token - **CloudWatch Alarms** — ALB 5xx, ECS CPU, jobs DLQ, DynamoDB throttles, all notifying `site-alerts` - **HCP Terraform** — workspace `meal-order-manager-` in project `seahaven-`. Working directory `terraform/`. VCS file triggers should be `terraform/**` only after the image deploy workflow owns `src/`. Do not `terraform apply` locally to prod. @@ -143,10 +138,11 @@ Local `:5050` is the same Flask app as production. DynamoDB is used when AWS cre python3 -m venv .venv source .venv/bin/activate pip install -r requirements.txt -r requirements-api.txt -playwright install chromium PYTHONPATH=src:src/shared python3 -m server.app ``` +Form tests and `src/scraper/recon.py` need `playwright install chromium`. Menu publish does not. + ### Deploy to AWS Image deploys are GitHub Actions `deploy-api.yaml` (push to `main` → dev, GitHub Release → prod). Infrastructure applies through HCP Terraform. First apply of the new `tf-managed` IAM policies needs the hcptf-bootstrap window. @@ -186,7 +182,6 @@ python3 src/aggregator/aggregate.py # generate CSV reports ``` meal-order-manager/ ├── .github/workflows/ -│ ├── weekly-menu.yml # Monday cron: scrape + HMAC publish + notify │ ├── deploy-api.yaml # Image CD to Fargate │ └── ci.yml # PR checks (lint, pytest, template JS, terraform, ci-complete) ├── terraform/ # HCP Terraform (cluster, ALB, ECR, jobs queue) @@ -194,7 +189,7 @@ meal-order-manager/ ├── .redocly.yaml ├── Dockerfile ├── src/ -│ ├── scraper/ # Playwright menu scraper +│ ├── scraper/ # HTML menu parser (recon scripts still use Playwright) │ ├── server/ # Flask API, form generator, job handlers │ ├── aggregator/ # Order aggregation + CSV reports │ └── shared/shared/ # db, secrets, slack, pdf helpers diff --git a/src/scraper/__init__.py b/src/scraper/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/scraper/parse_menu.py b/src/scraper/parse_menu.py new file mode 100644 index 0000000..5d6bd96 --- /dev/null +++ b/src/scraper/parse_menu.py @@ -0,0 +1,162 @@ +"""Parse the Redefine Meals menu from the HTML catalog embedded on the page. + +The menu document includes an element whose :products and +:newest-ids attributes are JSON. That replaces the Playwright DOM scrape. +""" + +from __future__ import annotations + +import html +import json +import re +import urllib.error +import urllib.request +from datetime import datetime +from zoneinfo import ZoneInfo + +EASTERN = ZoneInfo("America/New_York") +PRODUCTS_MARKER = ":products='" +NEWEST_MARKER = ":newest-ids='" +_TAG_RE = re.compile(r"<[^>]+>") +_SPACE_RE = re.compile(r"\s+") +_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/120.0.0.0 Safari/537.36" +) + + +class MenuParseError(RuntimeError): + """The menu page did not contain a usable catalog.""" + + +def fetch_menu(url: str, *, timeout: float = 30) -> dict: + request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT}) + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + page = response.read().decode("utf-8", "replace") + except urllib.error.URLError as exc: + raise MenuParseError(f"Failed to fetch {url}") from exc + return parse_menu_html(page, menu_url=url) + + +def parse_menu_html(page: str, *, menu_url: str, scraped_at: str | None = None) -> dict: + products = _extract_json_attr(page, PRODUCTS_MARKER) + newest = _extract_json_attr(page, NEWEST_MARKER) + if not isinstance(products, list): + raise MenuParseError(":products must be a JSON array") + if not isinstance(newest, list): + raise MenuParseError(":newest-ids must be a JSON array") + if not products: + raise MenuParseError("Menu catalog is empty") + + newest_ids = {str(item) for item in newest} + meals = [] + for product in products: + if not isinstance(product, dict): + raise MenuParseError("Each catalog entry must be an object") + if product.get("available") is False: + continue + meals.append(_meal_from_product(product, newest_ids)) + + if not meals: + raise MenuParseError("Scraped 0 meals") + + return { + "scraped_at": scraped_at or datetime.now(EASTERN).isoformat(), + "menu_url": menu_url, + "meal_count": len(meals), + "meals": meals, + } + + +def _extract_json_attr(page: str, marker: str): + index = page.find(marker) + if index < 0: + raise MenuParseError(f"Menu page is missing {marker}") + raw = html.unescape(page[index + len(marker) :]) + try: + value, _end = json.JSONDecoder().raw_decode(raw) + except json.JSONDecodeError as exc: + raise MenuParseError(f"Menu page has invalid JSON after {marker}") from exc + return value + + +def _meal_from_product(product: dict, newest_ids: set[str]) -> dict: + name = str(product.get("name") or "").strip() + if not name: + raise MenuParseError("A catalog entry is missing a name") + + details = product.get("details") if isinstance(product.get("details"), dict) else {} + concerns = product.get("dietary_concerns") or [] + if not isinstance(concerns, list): + raise MenuParseError(f"{name} dietary_concerns must be a list") + + return { + "name": name, + "price": _price(product), + "calories": _calories(details), + "protein": _protein(details), + "dietary_tags": [str(tag).strip() for tag in concerns if str(tag).strip()], + "image_url": _image_url(product), + "is_new": str(product.get("id")) in newest_ids, + "description": _plain_text(product.get("description")), + } + + +def _price(product: dict) -> float: + raw = ( + product.get("sale_price") + if product.get("on_sale") is True + else product.get("price") + ) + name = str(product.get("name") or "meal") + if isinstance(raw, bool) or raw is None or str(raw).strip() == "": + raise MenuParseError(f"{name} is missing a price") + try: + price = float(raw) + except (TypeError, ValueError) as exc: + raise MenuParseError(f"{name} has an invalid price") from exc + if price < 0: + raise MenuParseError(f"{name} has a negative price") + return price + + +def _calories(details: dict) -> int | None: + raw = details.get("calories") + if raw is None or str(raw).strip() == "": + return None + text = str(raw).strip().lower().removesuffix("cal").strip() + try: + return int(text) + except ValueError: + return None + + +def _protein(details: dict) -> str | None: + raw = details.get("protein") + if raw is None or str(raw).strip() == "": + return None + text = str(raw).strip() + if text.lower().endswith("g"): + return text + return f"{text}g" + + +def _image_url(product: dict) -> str | None: + media = product.get("media_urls") + images = media.get("images") if isinstance(media, dict) else None + if not images or not isinstance(images, list): + return None + first = images[0] + if not isinstance(first, dict): + return None + return first.get("original") or first.get("thumbnail") + + +def _plain_text(value) -> str | None: + if value is None: + return None + text = html.unescape(str(value)) + text = _SPACE_RE.sub(" ", _TAG_RE.sub(" ", text)).strip() + return text or None diff --git a/src/scraper/scrape_menu.py b/src/scraper/scrape_menu.py index 53b8e64..8340d14 100644 --- a/src/scraper/scrape_menu.py +++ b/src/scraper/scrape_menu.py @@ -1,17 +1,14 @@ -""" -Redefine Meals menu scraper. - -Navigates to the menu page using a headless browser, waits for the -Vue.js SPA to render, and extracts structured meal data from the DOM. -Also intercepts network requests to detect any JSON API that could -replace the browser scrape in the future. -""" +"""Fetch the Redefine Meals menu and write menu JSON for local form generation.""" import json import sys from datetime import datetime from pathlib import Path -from playwright.sync_api import sync_playwright, TimeoutError as PwTimeout + +try: + from scraper.parse_menu import MenuParseError, fetch_menu +except ImportError: + from parse_menu import MenuParseError, fetch_menu CONFIG_PATH = Path(__file__).resolve().parents[2] / "config.json" @@ -21,236 +18,8 @@ def load_config(): return json.load(f) -def scrape_menu(url: str, *, headless: bool = True, timeout_ms: int = 60_000) -> dict: - """ - Returns { - "scraped_at": ISO timestamp, - "menu_url": str, - "api_endpoints_found": [str], - "meals": [ { name, price, calories, protein, dietary_tags, - image_url, is_new, description } ] - } - Raises RuntimeError if the page fails to load or no meals are found. - """ - api_endpoints = [] - - with sync_playwright() as p: - browser = p.chromium.launch(headless=headless) - context = browser.new_context( - user_agent=( - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/120.0.0.0 Safari/537.36" - ) - ) - page = context.new_page() - - def on_response(response): - ct = response.headers.get("content-type", "") - if "json" in ct and "/api/" in response.url and "cart" not in response.url: - api_endpoints.append(response.url) - - page.on("response", on_response) - - try: - page.goto(url, wait_until="networkidle", timeout=timeout_ms) - except PwTimeout: - browser.close() - raise RuntimeError(f"Timed out loading {url}") - - page.wait_for_timeout(3000) - - articles = page.query_selector_all("article.editorial_card") - if not articles: - browser.close() - raise RuntimeError( - "No meal cards found on page. The site layout may have changed. " - "Run recon.py to inspect the current structure." - ) - - meals = [] - for article in articles: - meal = _extract_card(article) - if meal: - meals.append(meal) - - # Try to get descriptions via Quick View modals - _enrich_with_descriptions(page, articles, meals) - - browser.close() - - if not meals: - raise RuntimeError("Scraped 0 meals — extraction selectors are likely broken.") - - for meal in meals: - meal["dietary_tags"] = _clean_tags(meal["dietary_tags"]) - - return { - "scraped_at": datetime.now().isoformat(), - "menu_url": url, - "api_endpoints_found": api_endpoints, - "meal_count": len(meals), - "meals": meals, - } - - -def _extract_card(article) -> dict | None: - try: - name_el = article.query_selector("h2.meal_title") - if not name_el: - return None - name = name_el.inner_text().strip() - - price_el = article.query_selector(".meal_price") - price_text = price_el.inner_text().strip() if price_el else "" - price = _parse_price(price_text) - - cal_el = article.query_selector(".card_macros_brief") - calories = None - protein = None - if cal_el: - macros_text = cal_el.inner_text() - calories, protein = _parse_macros(macros_text) - - tag_els = article.query_selector_all(".diet_mini_tag") - dietary_tags = [ - t.inner_text().strip().title() for t in tag_els if t.inner_text().strip() - ] - - img_el = article.query_selector("img.main_meal_img") - image_url = img_el.get_attribute("src") if img_el else None - - is_new = article.query_selector(".new_badge_pulse") is not None - - return { - "name": name, - "price": price, - "calories": calories, - "protein": protein, - "dietary_tags": dietary_tags, - "image_url": image_url, - "is_new": is_new, - "description": None, - } - except Exception as e: - print(f" Warning: failed to extract a card: {e}", file=sys.stderr) - return None - - -def _enrich_with_descriptions(page, articles, meals): - """Click each meal's Quick View overlay to grab the description.""" - for i, article in enumerate(articles): - if i >= len(meals): - break - try: - overlay = article.query_selector(".card_overlay") - if not overlay: - continue - article.query_selector(".card_media_wrap").click() - page.wait_for_timeout(800) - - modal = page.query_selector( - ".modal.show, [class*='modal'][class*='show'], [class*='quickview']" - ) - if not modal: - # Try broader selector - modal = page.query_selector( - "[class*='modal']:not([style*='display: none'])" - ) - if modal and modal.is_visible(): - desc_el = modal.query_selector( - "[class*='description'], [class*='desc'], .meal_description" - ) - if desc_el: - desc_text = desc_el.inner_text().strip() - # Filter out price strings and very short text - if ( - desc_text - and len(desc_text) > 10 - and not desc_text.startswith("$") - ): - meals[i]["description"] = desc_text - - # Grab full macro details if available - detail_tags = modal.query_selector_all( - ".diet_mini_tag, [class*='lifestyle'] span" - ) - for tag_el in detail_tags: - tag_text = tag_el.inner_text().strip().title() - if tag_text and tag_text not in meals[i]["dietary_tags"]: - meals[i]["dietary_tags"].append(tag_text) - - # Close modal - close_btn = modal.query_selector( - "button[class*='close'], [aria-label='Close'], .btn-close" - ) - if close_btn: - close_btn.click() - else: - page.keyboard.press("Escape") - page.wait_for_timeout(300) - except Exception as e: - print( - f" Warning: Quick View failed for meal {i} ({meals[i]['name']}): {e}", - file=sys.stderr, - ) - try: - page.keyboard.press("Escape") - page.wait_for_timeout(300) - except Exception: - pass - - -KNOWN_TAGS = [ - "Gluten Free", - "Dairy Free", - "Grass-Fed", - "Low Carb", - "Keto", - "Vegan", - "Vegetarian", - "Nut Free", -] - - -def _clean_tags(raw_tags: list[str]) -> list[str]: - """Split concatenated tags and deduplicate.""" - import re - - cleaned = set() - for raw in raw_tags: - # Split on known tag boundaries (e.g., "Gluten Freedairy Free" → "Gluten Free", "Dairy Free") - remaining = raw - for known in KNOWN_TAGS: - if known.lower() in remaining.lower(): - cleaned.add(known) - remaining = re.sub( - re.escape(known), "", remaining, flags=re.IGNORECASE - ).strip() - if remaining and len(remaining) > 2: - cleaned.add(remaining.strip().title()) - return sorted(cleaned) - - -def _parse_price(text: str) -> float | None: - text = text.replace("$", "").replace(",", "").strip() - try: - return float(text) - except ValueError: - return None - - -def _parse_macros(text: str) -> tuple[int | None, str | None]: - """Parse '570cal • 39gP' into (570, '39g').""" - import re - - cal_match = re.search(r"(\d+)\s*cal", text, re.IGNORECASE) - prot_match = re.search(r"(\d+g?)\s*P", text) - calories = int(cal_match.group(1)) if cal_match else None - protein = prot_match.group(1) if prot_match else None - if protein and not protein.endswith("g"): - protein += "g" - return calories, protein +def scrape_menu(url: str, **_kwargs) -> dict: + return fetch_menu(url) def main(): @@ -261,31 +30,30 @@ def main(): ) output_dir.mkdir(exist_ok=True) - print(f"Scraping menu from {url} ...") - result = scrape_menu(url) + print(f"Fetching menu from {url} ...") + try: + result = scrape_menu(url) + except MenuParseError as exc: + print(f"Error: {exc}", file=sys.stderr) + sys.exit(1) week_str = datetime.now().strftime("%Y-W%U") output_file = output_dir / f"menu-{week_str}.json" with open(output_file, "w") as f: json.dump(result, f, indent=2) - print(f"\nScraped {result['meal_count']} meals") - if result["api_endpoints_found"]: - print("API endpoints detected (potential future shortcut):") - for ep in result["api_endpoints_found"]: - print(f" {ep}") + print(f"\nFetched {result['meal_count']} meals") print(f"Output saved to {output_file}") - # Print summary table print(f"\n{'Name':<40} {'Price':>7} {'Cal':>5} {'Prot':>5} {'Tags'}") print("-" * 90) - for m in result["meals"]: - tags = ", ".join(m["dietary_tags"]) if m["dietary_tags"] else "" - new = " *NEW*" if m["is_new"] else "" - price = f"${m['price']:.2f}" if m["price"] else "?" - cal = str(m["calories"]) if m["calories"] else "?" - prot = m["protein"] or "?" - print(f"{(m['name'] + new):<40} {price:>7} {cal:>5} {prot:>5} {tags}") + for meal in result["meals"]: + tags = ", ".join(meal["dietary_tags"]) if meal["dietary_tags"] else "" + new = " *NEW*" if meal["is_new"] else "" + price = f"${meal['price']:.2f}" if meal["price"] is not None else "?" + cal = str(meal["calories"]) if meal["calories"] else "?" + prot = meal["protein"] or "?" + print(f"{(meal['name'] + new):<40} {price:>7} {cal:>5} {prot:>5} {tags}") if __name__ == "__main__": diff --git a/src/server/generate_form.py b/src/server/generate_form.py index 0cc2c22..e5e546e 100644 --- a/src/server/generate_form.py +++ b/src/server/generate_form.py @@ -54,13 +54,15 @@ def generate_form( bulk_discount: float = 0, company_subsidy: float = 0, google_client_id: str = "", + week: str | None = None, ) -> str: google_client_id = google_client_id.strip() api_url = (api_url or "").rstrip("/") if api_url and not google_client_id: raise ValueError("Google client ID is required in cloud mode") - week = datetime.now().strftime("%Y-W%U") + if not week: + week = datetime.now().strftime("%Y-W%U") scraped_at = menu.get("scraped_at", "unknown") deadline = config.get("order_deadline", "Thursday 11:59 PM") has_discount = bulk_discount > 0 or company_subsidy > 0 diff --git a/src/server/http_api.py b/src/server/http_api.py index 5e96618..18a93dd 100644 --- a/src/server/http_api.py +++ b/src/server/http_api.py @@ -363,7 +363,7 @@ def _require_publish_key(event) -> dict | None: def handle_publish_settings(event=None): - """Return only the pricing fields needed by the weekly-menu workflow.""" + """Return only the pricing fields needed by the manual HMAC publish fallback.""" denied = _require_publish_key(event or {}) if denied: return denied diff --git a/src/server/jobs/__init__.py b/src/server/jobs/__init__.py index 116abb8..a67f21c 100644 --- a/src/server/jobs/__init__.py +++ b/src/server/jobs/__init__.py @@ -42,6 +42,10 @@ def run_job(payload: dict) -> dict: if event_type == "sync_roster": from server.jobs.sync_roster import lambda_handler + return lambda_handler(payload, None) + if event_type == "publish_menu": + from server.jobs.publish_menu import lambda_handler + return lambda_handler(payload, None) from server.jobs.notify import lambda_handler diff --git a/src/server/jobs/publish_menu.py b/src/server/jobs/publish_menu.py new file mode 100644 index 0000000..664c0d5 --- /dev/null +++ b/src/server/jobs/publish_menu.py @@ -0,0 +1,133 @@ +"""Monday menu publish: fetch the catalog, write it, upload the form, notify.""" + +from __future__ import annotations + +import os + +import boto3 + +from scraper.parse_menu import fetch_menu +from server.generate_form import generate_form +from shared.db import current_week, get_settings, put_menu +from shared.secrets import get_parameter + +DEFAULT_MENU_URL = "https://www.redefinemeals.com/menu" +DEFAULT_DEADLINE = "Thursday at 11:59 PM" +FORM_BUCKET_PARAM = "/meal-order-manager/deploy/form-bucket" +DISTRIBUTION_ID_PARAM = "/meal-order-manager/deploy/distribution-id" + +# Same announcement the GitHub weekly-menu workflow posted. +_SLACK_TEXT = "This week's meal order is open! Deadline: Thursday 6pm." + + +def lambda_handler(event, context): + del event, context + menu_url = os.environ.get("MENU_URL", DEFAULT_MENU_URL).strip() or DEFAULT_MENU_URL + menu = fetch_menu(menu_url) + week = current_week() + + settings = get_settings() + bulk_discount = float(settings.get("bulk_discount_percent") or 0) + company_subsidy = float(settings.get("company_subsidy_percent") or 0) + google_client_id = _google_client_id() + bucket = _configured("FORM_BUCKET", "FORM_BUCKET_PARAM", FORM_BUCKET_PARAM) + distribution_id = _configured( + "DISTRIBUTION_ID", "DISTRIBUTION_ID_PARAM", DISTRIBUTION_ID_PARAM + ) + form_url = os.environ.get("FORM_URL", "").strip() + if not form_url: + raise RuntimeError("FORM_URL is required") + + html = generate_form( + menu, + {"order_deadline": DEFAULT_DEADLINE, "roster": []}, + bulk_discount=bulk_discount, + company_subsidy=company_subsidy, + google_client_id=google_client_id, + week=week, + ) + put_menu(week, menu) + _upload_form(bucket, week, html) + _invalidate(distribution_id, week, menu.get("scraped_at") or week) + _notify(form_url) + return { + "status": "published", + "week": week, + "meal_count": menu["meal_count"], + } + + +def _google_client_id() -> str: + direct = os.environ.get("GOOGLE_CLIENT_ID", "").strip() + if direct: + return direct + param = os.environ.get("GOOGLE_CLIENT_ID_PARAM", "").strip() + if not param: + raise RuntimeError("GOOGLE_CLIENT_ID_PARAM is required") + client_id = get_parameter(param).strip() + if not client_id: + raise RuntimeError("Google client ID is empty") + return client_id + + +def _configured(env_name: str, param_env: str, default_param: str) -> str: + direct = os.environ.get(env_name, "").strip() + if direct: + return direct + param = os.environ.get(param_env, default_param).strip() or default_param + value = get_parameter(param).strip() + if not value: + raise RuntimeError(f"{env_name} is empty") + return value + + +def _upload_form(bucket: str, week: str, html: str) -> None: + body = html.encode("utf-8") + s3 = boto3.client("s3") + s3.put_object( + Bucket=bucket, + Key="index.html", + Body=body, + ContentType="text/html", + CacheControl="no-cache", + ) + s3.put_object( + Bucket=bucket, + Key=f"archive/{week}.html", + Body=body, + ContentType="text/html", + ) + + +def _invalidate(distribution_id: str, week: str, scraped_at: str) -> None: + reference = f"publish-menu-{week}-{scraped_at}".replace(":", "-") + boto3.client("cloudfront").create_invalidation( + DistributionId=distribution_id, + InvalidationBatch={ + "Paths": {"Quantity": 1, "Items": ["/index.html"]}, + "CallerReference": reference[:128], + }, + ) + + +def _notify(form_url: str) -> None: + from shared.slack import post_channel_message + + blocks = [ + { + "type": "header", + "text": {"type": "plain_text", "text": "Meal Order Open"}, + }, + { + "type": "section", + "text": { + "type": "mrkdwn", + "text": ( + f"*<{form_url}|Place your order>*\n\n*Deadline:* Thursday 6pm\n" + ), + }, + }, + ] + result = post_channel_message(_SLACK_TEXT, blocks) + if not result.get("ok"): + raise RuntimeError(f"Slack API error: {result.get('error')}") diff --git a/terraform/hcp_iam.tf b/terraform/hcp_iam.tf index b765ac5..c5ca1d7 100644 --- a/terraform/hcp_iam.tf +++ b/terraform/hcp_iam.tf @@ -225,6 +225,8 @@ data "aws_iam_policy_document" "hcptf_scoped_iam" { resources = ["arn:aws:iam::${local.account_id}:role/tf-managed/githubdeploy-meal-order-manager"] } + # Kept so this apply can delete githubdeploy-meal-order-manager-weekly-menu. + # Drop the statement after that role is gone. statement { sid = "WriteDeployRoles" effect = "Allow" diff --git a/terraform/iam.tf b/terraform/iam.tf index d8112cc..4a1828b 100644 --- a/terraform/iam.tf +++ b/terraform/iam.tf @@ -38,6 +38,23 @@ data "aws_iam_policy_document" "ecs_task_boundary" { resources = ["${aws_s3_bucket.reports.arn}/*"] } + statement { + sid = "FormObjects" + effect = "Allow" + actions = ["s3:PutObject"] + resources = [ + "${aws_s3_bucket.form.arn}/index.html", + "${aws_s3_bucket.form.arn}/archive/*.html", + ] + } + + statement { + sid = "InvalidateForm" + effect = "Allow" + actions = ["cloudfront:CreateInvalidation"] + resources = [aws_cloudfront_distribution.form.arn] + } + statement { sid = "JobsQueue" effect = "Allow" diff --git a/terraform/iam_github_weekly_menu.tf b/terraform/iam_github_weekly_menu.tf deleted file mode 100644 index a5ea4eb..0000000 --- a/terraform/iam_github_weekly_menu.tf +++ /dev/null @@ -1,113 +0,0 @@ -# GitHub Actions OIDC role for .github/workflows/weekly-menu.yml. -# -# Trust is pinned three ways (aud, sub to main, job_workflow_ref to the -# weekly-menu workflow at main) so no other workflow in the repo can assume it. -# Permissions mirror the mgmt github-oidc-deploy-roles weekly-menu role, retargeted -# to prod resources and without form-api-key (SigV4 publish path). -# -# OIDC provider ARN is literal (not a data source): hcptf-meal-order-manager-plan -# lacks iam:GetOpenIDConnectProvider, and the provider is account-stable. - -data "aws_iam_policy_document" "weekly_menu_assume" { - statement { - effect = "Allow" - actions = ["sts:AssumeRoleWithWebIdentity"] - - principals { - type = "Federated" - identifiers = [local.github_oidc_provider_arn] - } - - condition { - test = "StringEquals" - variable = "token.actions.githubusercontent.com:aud" - values = ["sts.amazonaws.com"] - } - - condition { - test = "StringEquals" - variable = "token.actions.githubusercontent.com:sub" - values = ["repo:Sea-Haven-Industries/meal-order-manager:ref:refs/heads/main"] - } - - condition { - test = "StringEquals" - variable = "token.actions.githubusercontent.com:job_workflow_ref" - values = ["Sea-Haven-Industries/meal-order-manager/.github/workflows/weekly-menu.yml@refs/heads/main"] - } - } -} - -resource "aws_iam_role" "weekly_menu" { - name = "githubdeploy-meal-order-manager-weekly-menu" - path = "/tf-managed/" - description = "GitHub Actions weekly-menu scrape/publish for meal-order-manager" - assume_role_policy = data.aws_iam_policy_document.weekly_menu_assume.json - max_session_duration = 3600 - - # Not a Lambda execution role. Config omits permissions_boundary so a later - # apply will not PutRolePermissionsBoundary the Lambda ceiling back. Live still - # has seahaven-lambda-execution-boundary; omitting without ignore_changes would - # plan DeleteRolePermissionsBoundary, which hcptf-meal-order-manager is denied - # (DenyBoundaryTampering). Ignore the attribute so this apply does not touch - # the ceiling. An administrator deletes the live attachment, then a follow-up - # drops this lifecycle after refresh-only updates state to null. - lifecycle { - ignore_changes = [permissions_boundary] - } -} - -data "aws_iam_policy_document" "weekly_menu" { - statement { - sid = "SlackBotSecret" - effect = "Allow" - actions = [ - "secretsmanager:GetSecretValue", - ] - resources = [var.slack_bot_secret_arn] - } - - statement { - sid = "DeployAndAppParams" - effect = "Allow" - actions = [ - "ssm:GetParameter", - ] - resources = [ - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/deploy/api-url", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/deploy/form-bucket", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/deploy/distribution-id", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/deploy/form-url", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.google_client_id_param}", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.slack_channel_param}", - "arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/publish-key", - ] - } - - statement { - sid = "FormObjects" - effect = "Allow" - actions = [ - "s3:PutObject", - ] - resources = [ - "${aws_s3_bucket.form.arn}/index.html", - "${aws_s3_bucket.form.arn}/archive/*.html", - ] - } - - statement { - sid = "InvalidateForm" - effect = "Allow" - actions = [ - "cloudfront:CreateInvalidation", - ] - resources = [aws_cloudfront_distribution.form.arn] - } -} - -resource "aws_iam_role_policy" "weekly_menu" { - name = "weekly-menu-publish" - role = aws_iam_role.weekly_menu.id - policy = data.aws_iam_policy_document.weekly_menu.json -} diff --git a/terraform/outputs.tf b/terraform/outputs.tf index 766e28c..4d951ec 100644 --- a/terraform/outputs.tf +++ b/terraform/outputs.tf @@ -43,11 +43,6 @@ output "orders_table_name" { value = aws_dynamodb_table.orders.name } -output "weekly_menu_role_arn" { - description = "OIDC role ARN for .github/workflows/weekly-menu.yml (repo secret AWS_WEEKLY_MENU_ROLE_ARN)." - value = aws_iam_role.weekly_menu.arn -} - output "github_deploy_role_arn" { description = "OIDC role ARN for .github/workflows/deploy-api.yaml (Environment DEPLOY_ROLE_ARN)." value = aws_iam_role.github_deploy.arn diff --git a/terraform/scheduler.tf b/terraform/scheduler.tf index 9cdfbb6..a798ffc 100644 --- a/terraform/scheduler.tf +++ b/terraform/scheduler.tf @@ -17,6 +17,11 @@ locals { schedule = "cron(55 6 ? * MON *)" event = "sync_roster" } + publish-menu = { + description = "Publish the weekly menu Monday 7:30am Eastern" + schedule = "cron(30 7 ? * MON *)" + event = "publish_menu" + } } } diff --git a/terraform/ssm.tf b/terraform/ssm.tf index 307b3e7..79b5c1a 100644 --- a/terraform/ssm.tf +++ b/terraform/ssm.tf @@ -50,9 +50,8 @@ resource "aws_ssm_parameter" "portal_cognito_trust" { # Deploy-time lookups # --------------------------------------------------------------------------- # -# These replace the CloudFormation stack outputs that -# .github/workflows/weekly-menu.yml used to read, so the job can resolve its -# deploy targets without a CloudFormation stack. +# Deploy targets for the image workflow and the publish_menu job. +# There is no CloudFormation stack. resource "aws_ssm_parameter" "deploy_api_url" { name = "${local.ssm_prefix}/deploy/api-url" @@ -100,19 +99,19 @@ resource "aws_ssm_parameter" "deploy_form_bucket" { name = "${local.ssm_prefix}/deploy/form-bucket" type = "String" value = aws_s3_bucket.form.id - description = "S3 bucket holding the order form; sync target for the weekly-menu deploy job" + description = "S3 bucket holding the order form; upload target for the publish_menu job" } resource "aws_ssm_parameter" "deploy_distribution_id" { name = "${local.ssm_prefix}/deploy/distribution-id" type = "String" value = aws_cloudfront_distribution.form.id - description = "CloudFront distribution ID; cache-invalidation target for the weekly-menu deploy job" + description = "CloudFront distribution ID; cache-invalidation target for the publish_menu job" } resource "aws_ssm_parameter" "deploy_form_url" { name = "${local.ssm_prefix}/deploy/form-url" type = "String" value = local.form_url - description = "Public order form URL; reported by the weekly-menu deploy job" + description = "Public order form URL; linked from the publish_menu Slack post" } diff --git a/tests/fixtures/menu_page.html b/tests/fixtures/menu_page.html new file mode 100644 index 0000000..db87a62 --- /dev/null +++ b/tests/fixtures/menu_page.html @@ -0,0 +1,9 @@ + + + + + + diff --git a/tests/test_generate_form.py b/tests/test_generate_form.py index f54477b..315c6bc 100644 --- a/tests/test_generate_form.py +++ b/tests/test_generate_form.py @@ -158,9 +158,11 @@ class TestGenerateFormStructural: assert "x-api-key" not in html def test_cloud_configuration_has_no_form_api_key(self): - workflow = (REPO_ROOT / ".github" / "workflows" / "weekly-menu.yml").read_text() http_api = (REPO_ROOT / "src" / "server" / "http_api.py").read_text() - for text in (workflow, http_api): + publish_job = ( + REPO_ROOT / "src" / "server" / "jobs" / "publish_menu.py" + ).read_text() + for text in (http_api, publish_job): assert "FORM_APIKEY" not in text assert "form-api-key" not in text assert "x-api-key" not in text @@ -171,19 +173,30 @@ class TestGenerateFormStructural: assert "SUBMIT_RATE_PER_SEC = 5.0" in http_api assert "def _allow_submit()" in http_api - def test_weekly_menu_uses_hmac_publish_without_dynamodb(self): - workflow = (REPO_ROOT / ".github" / "workflows" / "weekly-menu.yml").read_text() + def test_manual_publish_stays_hmac_and_scheduled_job_is_in_process(self): script = (REPO_ROOT / "scripts" / "upload_menu.py").read_text() locals_tf = (REPO_ROOT / "terraform" / "locals.tf").read_text() + scheduler = (REPO_ROOT / "terraform" / "scheduler.tf").read_text() + publish_job = ( + REPO_ROOT / "src" / "server" / "jobs" / "publish_menu.py" + ).read_text() for route in ("/api/publish/settings", "/api/publish/menu"): assert route in script assert 'authorizer = "HMAC"' in locals_tf assert "X-Meals-Publish-Key" in script - assert "scripts/upload_menu.py settings" in workflow - assert "scripts/upload_menu.py publish" in workflow - assert "aws dynamodb" not in workflow assert 'boto3.resource("dynamodb")' not in script + assert 'event = "publish_menu"' in scheduler + assert 'schedule = "cron(30 7 ? * MON *)"' in scheduler + assert "put_menu" in publish_job + assert not (REPO_ROOT / ".github" / "workflows" / "weekly-menu.yml").exists() + + def test_explicit_week_is_embedded_in_the_form(self): + menu, config = _load_fixtures() + html = generate_form(menu, config, week="2026-W38") + config_blob = _extract_config(html) + assert config_blob["week"] == "2026-W38" + assert "/api/form-status/2026-W38" in html def test_local_and_google_render(self): local = _render(google=False) diff --git a/tests/test_parse_menu.py b/tests/test_parse_menu.py new file mode 100644 index 0000000..a06ad61 --- /dev/null +++ b/tests/test_parse_menu.py @@ -0,0 +1,60 @@ +"""Parser for the catalog embedded on the Redefine menu page.""" + +from pathlib import Path + +import pytest + +from scraper.parse_menu import MenuParseError, parse_menu_html + +FIXTURE = Path(__file__).resolve().parent / "fixtures" / "menu_page.html" + + +def test_parse_menu_html_maps_catalog_fields(): + page = FIXTURE.read_text() + menu = parse_menu_html( + page, + menu_url="https://www.redefinemeals.com/menu", + scraped_at="2026-09-21T07:30:00-04:00", + ) + + assert menu["meal_count"] == 2 + assert menu["scraped_at"] == "2026-09-21T07:30:00-04:00" + korean, sale = menu["meals"] + + assert korean["name"] == "Korean Steak Bowl" + assert korean["price"] == 12.49 + assert korean["calories"] == 590 + assert korean["protein"] == "50g" + assert korean["dietary_tags"] == ["Gluten Free", "Dairy Free"] + assert korean["image_url"] == "https://example.com/korean.png" + assert korean["is_new"] is True + assert korean["description"] == "Shaved ribeye & rice." + + assert sale["name"] == "Sale Bowl" + assert sale["price"] == 9.50 + assert sale["calories"] == 400 + assert sale["protein"] == "30g" + assert sale["is_new"] is False + assert sale["description"] == "On sale." + + +def test_parse_menu_html_rejects_a_missing_catalog(): + with pytest.raises(MenuParseError, match="missing :products"): + parse_menu_html("", menu_url="https://example.com/menu") + + +def test_parse_menu_html_rejects_an_empty_catalog(): + page = "" + with pytest.raises(MenuParseError, match="empty"): + parse_menu_html(page, menu_url="https://example.com/menu") + + +def test_parse_menu_html_rejects_a_meal_without_a_price(): + page = """ + + """ + with pytest.raises(MenuParseError, match="missing a price"): + parse_menu_html(page, menu_url="https://example.com/menu") diff --git a/tests/test_publish_menu.py b/tests/test_publish_menu.py new file mode 100644 index 0000000..8ff09de --- /dev/null +++ b/tests/test_publish_menu.py @@ -0,0 +1,142 @@ +"""publish_menu writes the menu, then the form, then Slack.""" + +from unittest.mock import MagicMock, patch + +import pytest + +from server.jobs import publish_menu +from server.jobs import run_job + + +MENU = { + "scraped_at": "2026-09-21T07:30:00-04:00", + "menu_url": "https://www.redefinemeals.com/menu", + "meal_count": 1, + "meals": [{"name": "Korean Steak Bowl", "price": 12.49}], +} + +ENV = { + "MENU_URL": "https://www.redefinemeals.com/menu", + "GOOGLE_CLIENT_ID": "client.apps.googleusercontent.com", + "FORM_BUCKET": "meal-order-manager-form-test", + "DISTRIBUTION_ID": "E123", + "FORM_URL": "https://orders.seahaven.com", + "TABLE_NAME": "meal-order-manager-orders", +} + + +def _clients(s3, cloudfront): + def client(name, **_kwargs): + if name == "s3": + return s3 + if name == "cloudfront": + return cloudfront + raise AssertionError(name) + + return client + + +@patch("shared.slack.post_channel_message", return_value={"ok": True}) +@patch("server.jobs.publish_menu.put_menu") +@patch( + "server.jobs.publish_menu.get_settings", + return_value={"bulk_discount_percent": 10, "company_subsidy_percent": 50}, +) +@patch("server.jobs.publish_menu.generate_form", return_value="form") +@patch("server.jobs.publish_menu.current_week", return_value="2026-W38") +@patch("server.jobs.publish_menu.fetch_menu", return_value=MENU) +@patch("server.jobs.publish_menu.boto3.client") +def test_publish_menu_uploads_after_the_menu_write( + mock_client, + mock_fetch, + mock_week, + mock_generate, + mock_settings, + mock_put, + mock_slack, +): + s3 = MagicMock() + cloudfront = MagicMock() + mock_client.side_effect = _clients(s3, cloudfront) + order = [] + mock_put.side_effect = lambda *args, **kwargs: order.append("menu") + s3.put_object.side_effect = lambda **kwargs: order.append(kwargs["Key"]) + cloudfront.create_invalidation.side_effect = lambda **kwargs: order.append( + "invalidate" + ) + mock_slack.side_effect = lambda *args, **kwargs: ( + order.append("slack") or {"ok": True} + ) + + with patch.dict("os.environ", ENV, clear=False): + result = publish_menu.lambda_handler({}, None) + + assert result == {"status": "published", "week": "2026-W38", "meal_count": 1} + mock_fetch.assert_called_once_with("https://www.redefinemeals.com/menu") + mock_week.assert_called_once() + mock_settings.assert_called_once() + mock_generate.assert_called_once() + assert mock_generate.call_args.kwargs["week"] == "2026-W38" + assert mock_generate.call_args.kwargs["google_client_id"] == ENV["GOOGLE_CLIENT_ID"] + assert mock_generate.call_args.kwargs["bulk_discount"] == 10 + assert mock_generate.call_args.kwargs["company_subsidy"] == 50 + mock_put.assert_called_once_with("2026-W38", MENU) + assert s3.put_object.call_count == 2 + index = s3.put_object.call_args_list[0].kwargs + archive = s3.put_object.call_args_list[1].kwargs + assert index["Bucket"] == ENV["FORM_BUCKET"] + assert index["Key"] == "index.html" + assert index["CacheControl"] == "no-cache" + assert archive["Key"] == "archive/2026-W38.html" + invalidation = cloudfront.create_invalidation.call_args.kwargs + assert invalidation["DistributionId"] == "E123" + assert invalidation["InvalidationBatch"]["Paths"]["Items"] == ["/index.html"] + mock_slack.assert_called_once() + assert ( + "https://orders.seahaven.com" in mock_slack.call_args.args[1][1]["text"]["text"] + ) + assert order == [ + "menu", + "index.html", + "archive/2026-W38.html", + "invalidate", + "slack", + ] + + +@patch("shared.slack.post_channel_message", return_value={"ok": True}) +@patch("server.jobs.publish_menu.put_menu") +@patch("server.jobs.publish_menu.get_settings", return_value={}) +@patch("server.jobs.publish_menu.generate_form", return_value="form") +@patch("server.jobs.publish_menu.current_week", return_value="2026-W38") +@patch("server.jobs.publish_menu.fetch_menu", return_value=MENU) +@patch("server.jobs.publish_menu.boto3.client") +def test_publish_menu_does_not_notify_when_upload_fails( + mock_client, + _fetch, + _week, + _generate, + _settings, + mock_put, + mock_slack, +): + s3 = MagicMock() + s3.put_object.side_effect = RuntimeError("s3 down") + mock_client.side_effect = _clients(s3, MagicMock()) + + with patch.dict("os.environ", ENV, clear=False): + with pytest.raises(RuntimeError, match="s3 down"): + publish_menu.lambda_handler({}, None) + + mock_put.assert_called_once() + mock_slack.assert_not_called() + + +def test_run_job_dispatches_publish_menu(): + with patch( + "server.jobs.publish_menu.lambda_handler", return_value={"status": "published"} + ) as handler: + result = run_job({"event": "publish_menu"}) + + assert result == {"status": "published"} + handler.assert_called_once() diff --git a/tests/test_terraform_iam.py b/tests/test_terraform_iam.py index de69433..d70c065 100644 --- a/tests/test_terraform_iam.py +++ b/tests/test_terraform_iam.py @@ -5,6 +5,14 @@ from pathlib import Path IAM = Path(__file__).resolve().parents[1] / "terraform" / "iam.tf" +def test_task_boundary_can_publish_the_order_form(): + text = IAM.read_text() + assert 'sid = "FormObjects"' in text + assert "cloudfront:CreateInvalidation" in text + assert "${aws_s3_bucket.form.arn}/index.html" in text + assert not (IAM.parent / "iam_github_weekly_menu.tf").exists() + + def test_checkcomponents_send_omitted_when_queue_arn_empty(): text = IAM.read_text() assert "compact([var.checkcomponents_queue_arn])" not in text