mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 08:36:25 +00:00
The two-line comment above `documents/interview/**` says interview prep and experience records live there. Nothing has ever written to that directory: /interview saves its pack to documents/applications/<company>_<role>/interview_prep_<stage>.md, covered by the documents/applications/** rule. `git grep documents/interview` returns only the two declarations of the rule itself (.gitignore and REQUIRED_IGNORE_RULES), `git log --all -- 'documents/interview*'` is empty, and documents/README.md documents the applications path outright. Nothing leaks - the comment is the defect, and it is the misleading kind. It is the one dedicated, well-argued line about interview material in the personal-data block, so an auditor checking that the framework's most sensitive artifact is covered reads it and stops, at the only path in the block with no writer. The comment's description of what needs protecting was always right; only its location was wrong. It now sits above documents/applications/**, the rule that actually provides that protection, so a reader auditing the block finds the reasoning attached to the rule doing the work. documents/interview/** stays - REQUIRED_IGNORE_RULES pins it, so dropping it from .gitignore alone turns CI red, and it is harmless defence in depth - relabelled in both files as belt-and-braces rather than the primary guard. The new check-ignore case in GitignorePatternBehaviorTests derives the prep-pack path from /interview's own spec instead of hardcoding it. That distinction is the whole value of the test: a hardcoded path pins only that documents/applications/** still matches that shape, which security_guards.py already catches first, and stays green if /interview moves its output - leaving the corrected comment stale exactly the way this issue found it. Since #329 the spec states the location in two pieces - Step 1 derives the archive folder, Step 3 names interview_prep_<stage>.md - so the test pins both fragments separately and composes the concrete path from them. Mutation- verified on each half: repointing the folder at documents/prep_packs/, and renaming the file, both fail this test while `python3 tools/security_guards.py` still reports OK. The class's temp-repo setup moved to setUp for the second case. ayobamiseun reviewed the pre-rebase branch and called all three rebase hazards in advance: the split literal, the released CHANGELOG context, and the setUp re-merge. Reached independently here during the rebase; the review was posted first.
291 lines
13 KiB
Python
291 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""Supply-chain guards for the template's riskiest surfaces.
|
|
|
|
Run from anywhere: python tools/security_guards.py
|
|
|
|
This repo ships pre-approved Claude Code permissions and CLI code that every
|
|
fork user executes. These guards make the dangerous changes LOUD, not
|
|
impossible: a PR that intentionally needs one of them must update the
|
|
allowlists in this file in the same diff, so the change is explicit and
|
|
reviewable rather than buried.
|
|
|
|
Checks:
|
|
1. .claude/settings.json — every permissions.allow entry must be in the exact
|
|
allowlist below. Catches permission widening (e.g. Bash(*), Bash(curl:*)),
|
|
which would auto-approve commands on every fork. The same file's `hooks`
|
|
key is held to an allowlist too: a hook runs automatically when its event
|
|
fires, with no prompt, so it is strictly more dangerous than a pre-approved
|
|
permission.
|
|
2. .gitignore — the personal-data ignore rules must all still be present,
|
|
and no un-allowlisted negation (!pattern) may re-include them. Catches
|
|
weakening that would make future users silently commit their tracker,
|
|
profile exports, or application archives.
|
|
3. .agents/**/package.json — no npm/bun lifecycle scripts (preinstall,
|
|
install, postinstall, prepare, prepack) and no trustedDependencies.
|
|
Catches code execution smuggled into `bun install`.
|
|
|
|
Stdlib only. Exit 0 on success, 1 with a failure list otherwise.
|
|
"""
|
|
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
errors: list[str] = []
|
|
|
|
# The exact permission entries the template ships. A PR that adds or changes
|
|
# an entry must add it here too - that is the point: the diff shows both.
|
|
ALLOWED_PERMISSIONS = {
|
|
"Skill(job-application-assistant)",
|
|
# Narrowed from the upstream template's blanket Bash(bun run:*), which
|
|
# pre-approved `bun run <any file>`. One entry per shipped portal CLI,
|
|
# matching what each SKILL.md already declares in its allowed-tools.
|
|
# A portal added by /add-portal needs its own entry here and in
|
|
# .claude/settings.json - that review step is the point.
|
|
"Bash(bun run .agents/skills/jobbank-search/cli/src/cli.ts:*)",
|
|
"Bash(bun run .agents/skills/jobdanmark-search/cli/src/cli.ts:*)",
|
|
"Bash(bun run .agents/skills/jobindex-search/cli/src/cli.ts:*)",
|
|
"Bash(bun run .agents/skills/jobnet-search/cli/src/cli.ts:*)",
|
|
"Bash(bun run .agents/skills/linkedin-search/cli/src/cli.ts:*)",
|
|
"Bash(bun run .agents/skills/freehire-search/cli/src/cli.ts:*)",
|
|
"Bash(python salary_lookup.py:*)",
|
|
"Bash(python3 salary_lookup.py:*)",
|
|
"Bash(python tools/verify_pdf.py:*)",
|
|
"Bash(python3 tools/verify_pdf.py:*)",
|
|
"Bash(pdftotext:*)",
|
|
}
|
|
|
|
# Personal-data ignore rules that must never disappear from .gitignore.
|
|
REQUIRED_IGNORE_RULES = [
|
|
"salary_data.json",
|
|
# Depth-independent: the job-scraper skill resolves `job_scraper/` relative
|
|
# to its own directory, so the state file lands under .claude/skills/... and
|
|
# a repo-rooted rule silently fails to match it.
|
|
"**/job_scraper/seen_jobs.json",
|
|
"**/job_scraper/notion_sync.json",
|
|
"**/job_scraper/*.md",
|
|
"*_BehavioralReport.pdf",
|
|
"linkedin_Profile.pdf",
|
|
"cv/main_*.*",
|
|
"!cv/main_example.tex",
|
|
# ATS text extractions (/apply step 5d) carry the CV's full text.
|
|
"cv/*.txt",
|
|
"cover_letters/cover_*.*",
|
|
# /apply also recognizes the uppercase Cover_* naming variant.
|
|
"cover_letters/Cover_*.*",
|
|
"documents/cv/**",
|
|
"documents/linkedin/**",
|
|
"documents/diplomas/**",
|
|
"documents/references/**",
|
|
"documents/applications/**",
|
|
"documents/postings/**",
|
|
# Belt-and-braces, not the primary guard: nothing writes here.
|
|
# /interview's prep packs land under documents/applications/**, above.
|
|
"documents/interview/**",
|
|
"job_search_tracker.csv",
|
|
"gmail_sync/",
|
|
"reports/",
|
|
"upskill/*.md",
|
|
# Depth-independent twin of the rule above. The upskill *skill* resolves
|
|
# `upskill/` relative to its own directory - the same observed behavior
|
|
# the **/job_scraper rules exist for - so reports can land at
|
|
# .claude/skills/upskill/upskill/*.md where the rooted rule cannot see
|
|
# them. `**/upskill/*.md` would also ignore the skill's own SKILL.md
|
|
# (the directory shares the name), so the report-file prefix is pinned.
|
|
"**/upskill/report-*.md",
|
|
# Not personal data but the same failure mode: /add-portal can generate a
|
|
# skill for a portal that only returns usable content through a paid
|
|
# fetching service, and that skill reads an API token from the environment.
|
|
".env",
|
|
".env.*",
|
|
# Company research cache (/apply Step 3, /interview Step 2). Referenced
|
|
# from commands, not a skill, so a plain rooted rule is correct here -
|
|
# unlike the **/-prefixed job_scraper/upskill rules above.
|
|
"company_research/*.json",
|
|
]
|
|
|
|
# Negation (re-include) rules the template legitimately ships. .gitignore is
|
|
# order-sensitive: a later `!pattern` re-includes a path an earlier rule
|
|
# excluded, so a rule can be physically present in REQUIRED_IGNORE_RULES yet
|
|
# no longer ignored (e.g. adding `!salary_data.json`). Set membership on the
|
|
# required rules cannot see that. Any negation outside this allowlist is a
|
|
# failure - add an intentional one here in the same PR, exactly as with
|
|
# ALLOWED_PERMISSIONS, so the widening is explicit and reviewable.
|
|
ALLOWED_IGNORE_NEGATIONS = {
|
|
"!cover_letters/OpenFonts/fonts/**",
|
|
"!cv/main_example.tex",
|
|
"!cover_letters/cover_example.tex",
|
|
"!documents/**/.gitkeep",
|
|
}
|
|
|
|
# Hook commands the template legitimately ships, as "<Event>:<command>" strings.
|
|
# Empty by design - the template ships no hooks at all.
|
|
#
|
|
# A hook is strictly more dangerous than a permissions.allow entry. A permission
|
|
# pre-approves something Claude may choose to do; a hook runs unconditionally when
|
|
# its event fires, with no prompt and no model decision in between. Cloning a repo
|
|
# and opening it is enough. This is the vector the Shai-Hulud worm used in its
|
|
# August 2026 wave, planting a SessionStart hook in .claude/settings.json that
|
|
# executed on session start:
|
|
# https://research.jfrog.com/post/shai-hulud-is-back-august/
|
|
ALLOWED_HOOKS: set[str] = set()
|
|
|
|
FORBIDDEN_SCRIPTS = {"preinstall", "install", "postinstall", "prepare", "prepack"}
|
|
|
|
|
|
def _hook_commands(event: str, entries: object):
|
|
"""Yield "<Event>:<command>" for every command a hook event would run.
|
|
|
|
Fails closed: any shape this does not recognise yields a marker that cannot
|
|
be in the allowlist, so an unfamiliar hook layout is rejected rather than
|
|
silently skipped.
|
|
"""
|
|
unrecognised = f"{event}:<unrecognised hook shape>"
|
|
if not isinstance(entries, list):
|
|
yield unrecognised
|
|
return
|
|
for entry in entries:
|
|
if not isinstance(entry, dict):
|
|
yield unrecognised
|
|
continue
|
|
inner = entry.get("hooks")
|
|
if not isinstance(inner, list):
|
|
yield unrecognised
|
|
continue
|
|
for hook in inner:
|
|
command = hook.get("command") if isinstance(hook, dict) else None
|
|
yield f"{event}:{command}" if isinstance(command, str) else unrecognised
|
|
|
|
|
|
def check_permissions() -> None:
|
|
path = ROOT / ".claude" / "settings.json"
|
|
try:
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError) as exc:
|
|
errors.append(f".claude/settings.json: unreadable or invalid JSON: {exc}")
|
|
return
|
|
if not isinstance(data, dict):
|
|
errors.append(".claude/settings.json: top-level JSON value must be an object")
|
|
return
|
|
|
|
# Checked before the permissions shape guards below, so a file that pairs a
|
|
# malformed permissions block with a hook cannot return early and skip this.
|
|
hooks = data.get("hooks", {})
|
|
if hooks:
|
|
if not isinstance(hooks, dict):
|
|
errors.append(".claude/settings.json: hooks must be an object")
|
|
else:
|
|
for event, entries in hooks.items():
|
|
for command in _hook_commands(str(event), entries):
|
|
if command not in ALLOWED_HOOKS:
|
|
errors.append(
|
|
f".claude/settings.json: hook not in the reviewed allowlist: "
|
|
f"{command!r}. A hook runs automatically when its event fires - it "
|
|
"is never gated by the permissions prompt, so it executes on every "
|
|
"fork without the user agreeing to anything. If this hook is "
|
|
"intentional, add it to ALLOWED_HOOKS in tools/security_guards.py "
|
|
"in the same PR so the addition is explicit and reviewable."
|
|
)
|
|
|
|
permissions = data.get("permissions", {})
|
|
if not isinstance(permissions, dict):
|
|
errors.append(".claude/settings.json: permissions must be an object")
|
|
return
|
|
allow = permissions.get("allow", [])
|
|
if not isinstance(allow, list) or not all(isinstance(entry, str) for entry in allow):
|
|
errors.append(".claude/settings.json: permissions.allow must be a list of strings")
|
|
return
|
|
for entry in allow:
|
|
if entry not in ALLOWED_PERMISSIONS:
|
|
errors.append(
|
|
f".claude/settings.json: permission not in the reviewed allowlist: {entry!r}. "
|
|
"Pre-approved permissions run without prompting on every fork. If this entry is "
|
|
"intentional, add it to ALLOWED_PERMISSIONS in tools/security_guards.py in the "
|
|
"same PR so the widening is explicit and reviewable."
|
|
)
|
|
for entry in ALLOWED_PERMISSIONS - set(allow):
|
|
# Not an error: settings may legitimately drop an entry. But an
|
|
# allowlist entry that no longer exists should be pruned.
|
|
print(f"note: allowlisted permission not present in settings.json: {entry!r}")
|
|
|
|
|
|
def check_gitignore() -> None:
|
|
path = ROOT / ".gitignore"
|
|
try:
|
|
lines = [line.strip() for line in path.read_text(encoding="utf-8").splitlines()]
|
|
except OSError as exc:
|
|
errors.append(f".gitignore: unreadable: {exc}")
|
|
return
|
|
rules = set(lines)
|
|
for rule in REQUIRED_IGNORE_RULES:
|
|
if rule not in rules:
|
|
errors.append(
|
|
f".gitignore: required personal-data rule missing: {rule!r}. "
|
|
"These rules keep fork users from committing personal data. If the rule moved "
|
|
"or was renamed intentionally, update REQUIRED_IGNORE_RULES in "
|
|
"tools/security_guards.py in the same PR."
|
|
)
|
|
for line in lines:
|
|
if line.startswith("!") and line not in ALLOWED_IGNORE_NEGATIONS:
|
|
errors.append(
|
|
f".gitignore: negation rule not in the reviewed allowlist: {line!r}. "
|
|
"A negation re-includes a path an earlier rule excluded and can silently "
|
|
"re-expose personal data (a required ignore rule stays present but stops "
|
|
"taking effect). If this negation is intentional, add it to "
|
|
"ALLOWED_IGNORE_NEGATIONS in tools/security_guards.py in the same PR."
|
|
)
|
|
|
|
|
|
def check_package_manifests() -> None:
|
|
manifests = [
|
|
p for p in ROOT.glob(".agents/**/package.json") if "node_modules" not in p.parts
|
|
]
|
|
if not manifests:
|
|
errors.append(".agents: no package.json files found - glob roots are wrong or the tree moved")
|
|
for manifest in manifests:
|
|
relpath = manifest.relative_to(ROOT)
|
|
try:
|
|
data = json.loads(manifest.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError) as exc:
|
|
errors.append(f"{relpath}: unreadable or invalid JSON: {exc}")
|
|
continue
|
|
if not isinstance(data, dict):
|
|
errors.append(f"{relpath}: top-level JSON value must be an object")
|
|
continue
|
|
scripts = data.get("scripts", {})
|
|
if not isinstance(scripts, dict):
|
|
errors.append(f"{relpath}: scripts must be an object")
|
|
continue
|
|
bad = FORBIDDEN_SCRIPTS & set(scripts)
|
|
if bad:
|
|
errors.append(
|
|
f"{relpath}: lifecycle script(s) {sorted(bad)} are forbidden - they execute "
|
|
"arbitrary code during `bun install` on every fork user's machine."
|
|
)
|
|
if "trustedDependencies" in data:
|
|
errors.append(
|
|
f"{relpath}: trustedDependencies is forbidden - it re-enables dependency "
|
|
"lifecycle scripts that bun blocks by default."
|
|
)
|
|
|
|
|
|
def main() -> int:
|
|
check_permissions()
|
|
check_gitignore()
|
|
check_package_manifests()
|
|
if errors:
|
|
print(f"security_guards: {len(errors)} failure(s)")
|
|
for err in errors:
|
|
print(f" - {err}")
|
|
return 1
|
|
print(
|
|
"security_guards: OK (permissions allowlist, hooks allowlist, gitignore rules, "
|
|
"package manifests)"
|
|
)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|