"""Unit tests for ``scripts/assert_no_write_token.py`` (P3 no-write-token audit). The audit asserts the box invariant from ``docs/P3-PHASE0-DESIGN.md`` and ``ci/README.md``: the ``pull-requests: write`` GitHub App credential (``AGENT_APPLY_APP_ID`` / ``AGENT_APPLY_APP_PRIVATE_KEY``) lives ONLY as an Actions secret, and the always-on box holds **no standing write token**. These tests never touch real box paths: they grep temp files and a patched ``os.environ``. ``scripts/`` is not a package, so (matching the ``run-team.py`` idiom) the module is loaded from its file path via :mod:`importlib`. """ from __future__ import annotations import importlib.util import os from pathlib import Path from types import ModuleType import pytest _SCRIPT_PATH = ( Path(__file__).resolve().parents[1] / "scripts" / "assert_no_write_token.py" ) # A sample PEM private key header for each algorithm the design calls out. We # assert the regex covers RSA / EC / OPENSSH (and a bare PKCS#8 block). The PEM # markers are ASSEMBLED at runtime (not written as contiguous literals) so this # test data does not itself trip the repo's gitleaks pre-push backstop — the # values are still full PEM blocks at runtime, which is what the detector sees. _BEGIN = "-----BEGIN " _END = "-----END " _PEM_BODY = "\nMIIB...redacted...\n" def _pem(alg: str) -> str: label = f"{alg} PRIVATE KEY-----" if alg else "PRIVATE KEY-----" return f"{_BEGIN}{label}{_PEM_BODY}{_END}{label}" _RSA_KEY = _pem("RSA") _EC_KEY = _pem("EC") _OPENSSH_KEY = _pem("OPENSSH") _PKCS8_KEY = _pem("") def _load_script() -> ModuleType: """Load the (non-package) audit script from its file path.""" spec = importlib.util.spec_from_file_location("assert_no_write_token", _SCRIPT_PATH) assert spec is not None and spec.loader is not None module = importlib.util.module_from_spec(spec) spec.loader.exec_module(module) return module @pytest.fixture(scope="module") def mod() -> ModuleType: """The loaded audit module (loaded once per test module).""" return _load_script() # --------------------------------------------------------------------------- # # scan_environ # --------------------------------------------------------------------------- # def test_clean_environ_has_no_findings(mod: ModuleType) -> None: env = {"PATH": "/usr/bin", "AGENT_TEAM_REPO_OWNER": "Sea-Haven-Industries"} assert mod.scan_environ(env) == [] def test_forbidden_app_id_env_var_is_flagged(mod: ModuleType) -> None: findings = mod.scan_environ({"AGENT_APPLY_APP_ID": "123456"}) assert len(findings) == 1 assert "AGENT_APPLY_APP_ID" in findings[0] def test_forbidden_app_private_key_env_var_is_flagged(mod: ModuleType) -> None: # The var trips both the forbidden-name check AND the PEM-value check (SEC-03) # — both are legitimate findings; assert it is flagged (not an exact count). findings = mod.scan_environ({"AGENT_APPLY_APP_PRIVATE_KEY": _RSA_KEY}) assert findings assert all("AGENT_APPLY_APP_PRIVATE_KEY" in fn for fn in findings) assert any("PRIVATE KEY" in fn for fn in findings) def test_write_token_shaped_name_is_flagged(mod: ModuleType) -> None: for name in ( "GITHUB_APP_TOKEN", "GH_PAT_TOKEN", "GITHUB_WRITE_TOKEN", "GH_APPLY_KEY", ): findings = mod.scan_environ({name: "whatever"}) assert findings, f"{name} should be flagged by the write-name heuristic" def test_write_capable_token_value_is_flagged_regardless_of_name( mod: ModuleType, ) -> None: # A benign-looking var name but a write-capable PAT value. findings = mod.scan_environ({"SOME_VAR": "ghp_" + "a" * 36}) assert len(findings) == 1 assert "SOME_VAR" in findings[0] findings = mod.scan_environ({"OTHER": "github_pat_" + "b" * 30}) assert len(findings) == 1 findings = mod.scan_environ({"INSTALL": "ghs_" + "c" * 36}) assert len(findings) == 1 # user-to-server (ghu_) and refresh (ghr_) tokens are also write-risk. findings = mod.scan_environ({"U2S": "ghu_" + "d" * 36}) assert len(findings) == 1 findings = mod.scan_environ({"REFRESH": "ghr_" + "e" * 36}) assert len(findings) == 1 def test_read_only_tokens_are_not_flagged(mod: ModuleType) -> None: # Known read-only / non-GitHub tokens must not trip the audit. env = { "AGENT_TEAM_CI_READ_TOKEN": "ghp_" + "r" * 36, # read-only by contract, allowlisted "AGENT_TEAM_API_TOKEN": "secret-bearer", "SLACK_APP_TOKEN": "xapp-1-abc", "SLACK_BOT_TOKEN": "xoxb-abc", "CLAUDE_CODE_OAUTH_TOKEN": "oauth-abc", "ANTHROPIC_API_KEY": "sk-ant-abc", } assert mod.scan_environ(env) == [] def test_plain_github_token_fallback_is_not_flagged_by_name(mod: ModuleType) -> None: # The read-only GITHUB_TOKEN runtime fallback should not match the *name* # heuristic (it carries no GH_*_WRITE/APP/PAT shape). assert mod.scan_environ({"GITHUB_TOKEN": "a-non-write-shaped-value"}) == [] def test_github_token_value_is_still_write_value_scanned(mod: ModuleType) -> None: # SEC-04: GITHUB_TOKEN is name-exempt from the write *name* heuristic, but a # write-capable token VALUE parked in it must still be flagged. findings = mod.scan_environ({"GITHUB_TOKEN": "ghp_" + "a" * 36}) assert len(findings) == 1 assert "GITHUB_TOKEN" in findings[0] # SEC-03: a PEM private key exported under a *benign* env name must be flagged # (the App key value check is NOT name-exempt). def test_pem_private_key_under_benign_env_name_is_flagged(mod: ModuleType) -> None: findings = mod.scan_environ({"GH_APP_KEY": _RSA_KEY}) assert any("PRIVATE KEY" in fn for fn in findings) assert any("GH_APP_KEY" in fn for fn in findings) @pytest.mark.parametrize("key", [_RSA_KEY, _EC_KEY, _OPENSSH_KEY, _PKCS8_KEY]) def test_pem_private_key_value_is_flagged_for_each_algo( mod: ModuleType, key: str ) -> None: # Even a totally unremarkable var name carrying any PEM algorithm is flagged. findings = mod.scan_environ({"HARMLESS": key}) assert any("PRIVATE KEY" in fn for fn in findings) def test_pem_private_key_in_allowlisted_env_var_is_still_flagged( mod: ModuleType, ) -> None: # The value-level PEM check is not exempted by the read-only allowlist. findings = mod.scan_environ({"AGENT_TEAM_CI_READ_TOKEN": _PKCS8_KEY}) assert any("PRIVATE KEY" in fn for fn in findings) # SEC-03: the configured App-id appearing as an env VALUE (under any name) must # be flagged — mirroring scan_env_file's App-id grep. def test_configured_app_id_value_in_env_is_flagged(mod: ModuleType) -> None: findings = mod.scan_environ({"SOME_VAR": "installed-987654-here"}, app_id="987654") assert any("App id value" in fn for fn in findings) assert any("SOME_VAR" in fn for fn in findings) def test_app_id_value_check_is_skipped_without_app_id(mod: ModuleType) -> None: # No configured app_id -> the value-substring check does not fire. assert mod.scan_environ({"SOME_VAR": "987654"}) == [] def test_audit_flags_app_id_value_in_environ(mod: ModuleType, tmp_path: Path) -> None: # End-to-end: AGENT_APPLY_APP_ID in environ both flags itself and is matched # against other env VALUES (a benign var carrying the same id is flagged). clean = tmp_path / "secrev.env" clean.write_text("ok\n") findings = mod.audit( environ={"AGENT_APPLY_APP_ID": "424242", "BENIGN": "id-is-424242"}, env_files=[str(clean)], ) assert any("AGENT_APPLY_APP_ID" in fn for fn in findings) assert any("BENIGN" in fn and "App id value" in fn for fn in findings) # --------------------------------------------------------------------------- # # scan_config # --------------------------------------------------------------------------- # def test_scan_config_none_and_empty(mod: ModuleType) -> None: assert mod.scan_config(None) == [] assert mod.scan_config({}) == [] def test_scan_config_flags_forbidden_key(mod: ModuleType) -> None: findings = mod.scan_config({"AGENT_APPLY_APP_PRIVATE_KEY": _EC_KEY}) assert len(findings) == 1 assert "AGENT_APPLY_APP_PRIVATE_KEY" in findings[0] def test_scan_config_flags_write_value(mod: ModuleType) -> None: findings = mod.scan_config({"token": "ghp_" + "z" * 36}) assert len(findings) == 1 def test_scan_config_ignores_non_string_values(mod: ModuleType) -> None: assert mod.scan_config({"timeout": 15, "enabled": True}) == [] # --------------------------------------------------------------------------- # # scan_env_file # --------------------------------------------------------------------------- # def test_missing_file_is_not_a_finding(mod: ModuleType, tmp_path: Path) -> None: assert mod.scan_env_file(tmp_path / "nope.env") == [] def test_clean_file_is_not_a_finding(mod: ModuleType, tmp_path: Path) -> None: f = tmp_path / "secrev.env" f.write_text("CLAUDE_CODE_OAUTH_TOKEN=abc\nAGENT_TEAM_CI_READ_TOKEN=def\n") assert mod.scan_env_file(f) == [] @pytest.mark.parametrize("key", [_RSA_KEY, _EC_KEY, _OPENSSH_KEY, _PKCS8_KEY]) def test_private_key_header_is_flagged_for_each_algo( mod: ModuleType, tmp_path: Path, key: str ) -> None: f = tmp_path / "orchestrator.env" f.write_text("HARMLESS=1\n" + key + "\n") findings = mod.scan_env_file(f) assert any("PRIVATE KEY" in fn for fn in findings) def test_app_credential_name_in_file_is_flagged( mod: ModuleType, tmp_path: Path ) -> None: f = tmp_path / "secrev.env" f.write_text("AGENT_APPLY_APP_ID=987654\n") findings = mod.scan_env_file(f) assert any("AGENT_APPLY_APP_ID" in fn for fn in findings) def test_configured_app_id_value_in_file_is_flagged( mod: ModuleType, tmp_path: Path ) -> None: f = tmp_path / "secrev.env" # The literal id value leaking under any var name. f.write_text("SOMETHING=installed-as-987654-here\n") findings = mod.scan_env_file(f, app_id="987654") assert any("App id value" in fn for fn in findings) def test_unreadable_present_file_is_a_finding( mod: ModuleType, tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: f = tmp_path / "secrev.env" f.write_text("ok\n") def _boom(*_a: object, **_k: object) -> str: raise OSError("permission denied") monkeypatch.setattr(Path, "read_text", _boom) findings = mod.scan_env_file(f) assert len(findings) == 1 assert "could not read" in findings[0] # --------------------------------------------------------------------------- # # audit() orchestration + file resolution # --------------------------------------------------------------------------- # def test_audit_clean_box_returns_empty(mod: ModuleType, tmp_path: Path) -> None: clean = tmp_path / "secrev.env" clean.write_text("CLAUDE_CODE_OAUTH_TOKEN=abc\n") findings = mod.audit( environ={"PATH": "/usr/bin"}, config={"timeout": 15}, env_files=[str(clean)], ) assert findings == [] def test_audit_aggregates_findings_across_scanners( mod: ModuleType, tmp_path: Path ) -> None: leaky = tmp_path / "orchestrator.env" leaky.write_text(_OPENSSH_KEY + "\n") findings = mod.audit( environ={"AGENT_APPLY_APP_ID": "42", "GH_PAT_TOKEN": "x"}, config={"token": "ghp_" + "q" * 36}, env_files=[str(leaky)], ) # env (2) + config (1) + file (1) assert len(findings) >= 4 def test_audit_uses_configured_app_id_from_environ( mod: ModuleType, tmp_path: Path ) -> None: f = tmp_path / "secrev.env" f.write_text("LEAK=value-555-leaked\n") findings = mod.audit( environ={"AGENT_APPLY_APP_ID": "555"}, env_files=[str(f)], ) # AGENT_APPLY_APP_ID in environ is itself a finding, and "555" in the file # is a second. assert any("AGENT_APPLY_APP_ID" in fn for fn in findings) assert any("App id value" in fn for fn in findings) def test_env_file_resolution_prefers_cli(mod: ModuleType, tmp_path: Path) -> None: a = tmp_path / "a.env" paths = mod._resolve_env_files([str(a)], {}) assert paths == [a] def test_env_file_resolution_uses_env_var(mod: ModuleType, tmp_path: Path) -> None: a = tmp_path / "a.env" b = tmp_path / "b.env" env = {"ASSERT_NO_WRITE_TOKEN_ENV_FILES": os.pathsep.join([str(a), str(b)])} paths = mod._resolve_env_files(None, env) assert paths == [a, b] def test_env_file_resolution_defaults_and_expands_home(mod: ModuleType) -> None: paths = mod._resolve_env_files(None, {}) # Defaults to ~/secrev.env and ~/orchestrator/.env, expanded (no literal ~). assert len(paths) == 2 assert all("~" not in str(p) for p in paths) assert paths[0].name == "secrev.env" # --------------------------------------------------------------------------- # # main() exit codes + messaging # --------------------------------------------------------------------------- # def test_main_clean_exits_zero( mod: ModuleType, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture, ) -> None: clean = tmp_path / "secrev.env" clean.write_text("CLAUDE_CODE_OAUTH_TOKEN=abc\n") # Patch the live environ so the real process env can't leak into the scan. monkeypatch.setattr(os, "environ", {"PATH": "/usr/bin"}) rc = mod.main(["--env-file", str(clean)]) assert rc == 0 out = capsys.readouterr().out assert "OK" in out assert "no standing write token" in out def test_main_finding_exits_nonzero( mod: ModuleType, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture, ) -> None: leaky = tmp_path / "secrev.env" leaky.write_text(_RSA_KEY + "\n") monkeypatch.setattr(os, "environ", {"PATH": "/usr/bin"}) rc = mod.main(["--env-file", str(leaky)]) assert rc == 1 err = capsys.readouterr().err assert "FAIL" in err assert "Actions secret" in err def test_main_flags_write_token_in_live_environ( mod: ModuleType, tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: clean = tmp_path / "secrev.env" clean.write_text("ok\n") monkeypatch.setattr(os, "environ", {"AGENT_APPLY_APP_PRIVATE_KEY": _RSA_KEY}) rc = mod.main(["--env-file", str(clean)]) assert rc == 1