mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 00:26:26 +00:00
* fix(rank): move seen_jobs.json read/write off the state file's own critical path (#395) /rank's Step 1 read the whole of seen_jobs.json into the conversation to select candidates by eye, and Step 4 emitted it back to record scores. That cost is paid on every run regardless of how many jobs are scored, and it grows for the life of the workspace, since the file is append-only and most stored entries are `skipped`. tools/rank_state.py moves that traffic into code: - `candidates` selects entries per Step 1's existing rules (status filter, tracker exclusion, focus filter, `--limit`/`--all` from #424) and projects only the fields a scoring agent needs. - `sweep` runs rule 6's expiry pass over entries the run did not re-score - a stored-date comparison, no fetch, no agent - preserving its defensive parsing of non-ISO deadlines and its `--all` reversibility. - `apply` writes scoring results back atomically and prints the ranked/vetoed/expired rows Step 5's report is built from, preserving Step 4's existing write-back rules exactly: the `location` -> `location_verdict` legacy migration, the deadline null-is-not-a-correction rule, verbatim strengths/gaps persistence, and idempotent re-scoring. Step 1, Step 3's rule 6, and Step 4 now route through the tool instead of describing a manual read/write. Nothing about scoring policy changes - no new status, no new persisted field, no change to what counts as a veto. The tracker stays read-only and every write is atomic (temp file + rename). tests/test_rank_state.py (25 tests) covers the three subcommands directly. The new spec-guard class in test_rank_command.py derives the fields Step 4 must preserve from Step 2's own JSON schema block rather than retyping them as a second list, so a future edit to that contract is what the test reads instead of something that can drift from it. * fix(rank): add CHANGELOG entry and remove the undefined $SCRATCHPAD reference Two mechanical fixes from review: - Step 4 named the results hand-off file via $SCRATCHPAD, a variable nothing in the repo defines - a reader following the spec literally has no path to substitute. Named the location in prose instead (a temporary file outside the repo tree, never committed) and replaced the shell-variable-looking path in the example command with an explicit placeholder. - Added the [Unreleased] entry this change was missing; the one already in the diff belongs to #424. * changelog: fold the #395 entry into the existing Fixed section Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013fqqLgQSnwgWkv98twQhHi --------- Co-authored-by: nox <nox@Mac.home> Co-authored-by: Mads Lorentzen <madslorentzen17@gmail.com> Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5.1
nox
Mads Lorentzen
parent
71674d0220
commit
3bf41149e0
@@ -0,0 +1,380 @@
|
||||
#!/usr/bin/env python3
|
||||
"""State helper for /rank: select candidates and write results back.
|
||||
|
||||
/rank reads the whole of seen_jobs.json into the model's context to filter it
|
||||
by eye (Step 1), then re-emits the whole file to record scores (Step 4). That
|
||||
cost is paid on every run regardless of how many jobs are actually scored, and
|
||||
it grows for the life of the workspace, since seen_jobs.json is append-only by
|
||||
design and most stored entries are `skipped`.
|
||||
|
||||
This moves the state-file traffic into code. Three subcommands:
|
||||
|
||||
candidates select the eligible entries for this run and project only the
|
||||
fields a scoring agent needs
|
||||
sweep rule 6's expiry pass over entries this run did not re-score -
|
||||
a stored-date comparison, no fetch, no agent
|
||||
apply write scoring results back to seen_jobs.json and print the
|
||||
ranked/vetoed/expired rows Step 5's report is built from
|
||||
|
||||
Selection and projection follow Step 1's existing rules exactly (status
|
||||
filter, tracker exclusion, focus filter, `--limit`/`--all`); the write-back
|
||||
follows Step 4's existing rules exactly (the `location` -> `location_verdict`
|
||||
legacy migration, the deadline null-is-not-a-correction rule, verbatim
|
||||
strengths/gaps persistence, idempotent skip of already-ranked entries); the
|
||||
sweep follows rule 6 exactly (defensive date parsing, an absent deadline left
|
||||
alone, `--all` making a retired entry revivable).
|
||||
|
||||
Nothing here fetches a posting or judges a fit. Scoring stays with the model;
|
||||
this only removes the state file from the conversation.
|
||||
|
||||
Usage:
|
||||
python3 tools/rank_state.py candidates [--all] [--focus TEXT] [--limit N]
|
||||
python3 tools/rank_state.py sweep [--write] [--exclude KEY,KEY]
|
||||
python3 tools/rank_state.py apply --results results.json [--dry-run]
|
||||
|
||||
Both subcommands print JSON on stdout. Exit 0 on success, 1 on a usage or
|
||||
state error, or on `apply` when any result could not be written.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
from datetime import date, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
STATE = ROOT / "job_scraper" / "seen_jobs.json"
|
||||
TRACKER = ROOT / "job_search_tracker.csv"
|
||||
|
||||
# 04-job-evaluation.md
|
||||
WEIGHTS = {"technical": 0.30, "experience": 0.25, "behavioral": 0.15, "career": 0.30}
|
||||
BANDS = ((75, "Strong Fit"), (60, "Good Fit"), (45, "Moderate Fit"), (30, "Weak Fit"), (0, "Poor Fit"))
|
||||
|
||||
DEFAULT_LIMIT = 10
|
||||
URGENT_DAYS = 7
|
||||
ISO = re.compile(r"^\d{4}-\d{2}-\d{2}$")
|
||||
|
||||
|
||||
def load_state(path: Path) -> tuple[dict, dict]:
|
||||
"""Return (document, seen-map). The map is mutated in place by callers."""
|
||||
if not path.is_file():
|
||||
sys.exit(f"{path} not found - run /scrape first")
|
||||
try:
|
||||
doc = json.loads(path.read_text(encoding="utf-8"))
|
||||
except json.JSONDecodeError as exc:
|
||||
sys.exit(f"{path} is not valid JSON: {exc}")
|
||||
seen = doc.get("seen") if isinstance(doc, dict) and "seen" in doc else doc
|
||||
if not isinstance(seen, dict):
|
||||
sys.exit(f"{path}: expected an object of job entries")
|
||||
return doc, seen
|
||||
|
||||
|
||||
def save_state(path: Path, doc: dict) -> None:
|
||||
"""Atomic replace: a half-written seen_jobs.json loses the scrape history."""
|
||||
fd, tmp = tempfile.mkstemp(dir=str(path.parent), prefix=".seen_jobs.", suffix=".tmp")
|
||||
try:
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as fh:
|
||||
json.dump(doc, fh, indent=2, ensure_ascii=False)
|
||||
fh.write("\n")
|
||||
os.replace(tmp, path)
|
||||
except BaseException:
|
||||
Path(tmp).unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
|
||||
def parse_iso(value) -> date | None:
|
||||
"""Rule 6's defensive-parse rule: anything that is not YYYY-MM-DD is treated
|
||||
exactly like an absent value - never compared, never guessed at."""
|
||||
if not isinstance(value, str) or not ISO.match(value.strip()):
|
||||
return None
|
||||
try:
|
||||
return date.fromisoformat(value.strip())
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def norm(text) -> str:
|
||||
return re.sub(r"[^a-z0-9]", "", str(text or "").lower())
|
||||
|
||||
|
||||
def tracker_pairs(path: Path) -> set[tuple[str, str]]:
|
||||
"""company+role pairs already in the tracker - out of scope for ranking."""
|
||||
if not path.is_file():
|
||||
return set()
|
||||
import csv
|
||||
|
||||
pairs = set()
|
||||
with path.open(encoding="utf-8", newline="") as fh:
|
||||
for row in csv.DictReader(fh):
|
||||
company, role = norm(row.get("company")), norm(row.get("role"))
|
||||
if company:
|
||||
pairs.add((company, role))
|
||||
return pairs
|
||||
|
||||
|
||||
def entry_location_verdict(entry: dict) -> str | None:
|
||||
"""location_verdict, falling back to a legacy verdict stored under `location`
|
||||
(Step 4: "an entry ranked before this rename may carry a legacy PASS/FAIL/
|
||||
FLAG string in `location`")."""
|
||||
verdict = entry.get("location_verdict")
|
||||
if verdict:
|
||||
return verdict
|
||||
legacy = entry.get("location")
|
||||
return legacy if legacy in ("PASS", "FAIL", "FLAG") else None
|
||||
|
||||
|
||||
def cmd_candidates(args) -> int:
|
||||
_, seen = load_state(args.state)
|
||||
excluded = tracker_pairs(args.tracker)
|
||||
|
||||
selected, skipped_tracker = [], 0
|
||||
for key, entry in seen.items():
|
||||
status = entry.get("status")
|
||||
if args.all:
|
||||
if status == "skipped":
|
||||
continue
|
||||
elif status != "new":
|
||||
continue
|
||||
if (norm(entry.get("company")), norm(entry.get("title"))) in excluded:
|
||||
skipped_tracker += 1
|
||||
continue
|
||||
if args.focus:
|
||||
haystack = " ".join(
|
||||
[str(entry.get("title") or ""), str(entry.get("company") or "")]
|
||||
+ [str(b) for b in entry.get("strengths") or []]
|
||||
+ [str(b) for b in entry.get("gaps") or []]
|
||||
).lower()
|
||||
if args.focus.lower() not in haystack:
|
||||
continue
|
||||
selected.append(
|
||||
{
|
||||
"key": key,
|
||||
"title": entry.get("title"),
|
||||
"company": entry.get("company"),
|
||||
"url": entry.get("url"),
|
||||
"portal": entry.get("portal"),
|
||||
"deadline": entry.get("deadline"),
|
||||
"posted_date": entry.get("posted_date"),
|
||||
}
|
||||
)
|
||||
|
||||
eligible = len(selected)
|
||||
if args.limit > 0:
|
||||
selected = selected[: args.limit]
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"eligible": eligible,
|
||||
"selected": selected,
|
||||
"deferred": max(0, eligible - len(selected)),
|
||||
"excluded_by_tracker": skipped_tracker,
|
||||
"total_entries": len(seen),
|
||||
},
|
||||
indent=2,
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_sweep(args) -> int:
|
||||
doc, seen = load_state(args.state)
|
||||
today = args.today
|
||||
exclude = {k for k in (args.exclude or "").split(",") if k}
|
||||
|
||||
expired, closing, unparseable, checked = [], [], [], 0
|
||||
for key, entry in seen.items():
|
||||
if entry.get("status") != "ranked" or key in exclude:
|
||||
continue
|
||||
checked += 1
|
||||
raw = entry.get("deadline")
|
||||
if raw in (None, ""):
|
||||
continue
|
||||
parsed = parse_iso(raw)
|
||||
if parsed is None:
|
||||
unparseable.append({"key": key, "portal": entry.get("portal"), "deadline": raw})
|
||||
continue
|
||||
row = {
|
||||
"key": key,
|
||||
"title": entry.get("title"),
|
||||
"company": entry.get("company"),
|
||||
"url": entry.get("url"),
|
||||
"deadline": raw,
|
||||
}
|
||||
if parsed < today:
|
||||
expired.append(row)
|
||||
elif (parsed - today).days <= URGENT_DAYS:
|
||||
closing.append(row)
|
||||
|
||||
if args.write and expired:
|
||||
for row in expired:
|
||||
seen[row["key"]]["status"] = "expired"
|
||||
save_state(args.state, doc)
|
||||
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"swept": checked,
|
||||
"newly_expired": expired,
|
||||
"closing_soon": sorted(closing, key=lambda r: r["deadline"]),
|
||||
"unparseable_deadlines": unparseable,
|
||||
"written": bool(args.write and expired),
|
||||
},
|
||||
indent=2,
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
def overall_score(scores: dict) -> int:
|
||||
total = 0.0
|
||||
for dim, weight in WEIGHTS.items():
|
||||
value = scores.get(dim)
|
||||
if not isinstance(value, (int, float)):
|
||||
raise ValueError(f"missing or non-numeric score '{dim}'")
|
||||
total += float(value) * weight
|
||||
return int(total + 0.5)
|
||||
|
||||
|
||||
def band(score: int) -> str:
|
||||
for floor, name in BANDS:
|
||||
if score >= floor:
|
||||
return name
|
||||
return "Poor Fit"
|
||||
|
||||
|
||||
def cmd_apply(args) -> int:
|
||||
doc, seen = load_state(args.state)
|
||||
today = args.today
|
||||
try:
|
||||
results = json.loads(Path(args.results).read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
sys.exit(f"cannot read results file {args.results}: {exc}")
|
||||
if isinstance(results, dict):
|
||||
results = results.get("results", [])
|
||||
if not isinstance(results, list):
|
||||
sys.exit("results file must be a JSON array of scoring objects")
|
||||
|
||||
rows, expired, errors = [], [], []
|
||||
for result in results:
|
||||
key = result.get("key")
|
||||
entry = seen.get(key)
|
||||
if entry is None:
|
||||
errors.append({"key": key, "error": "no such key in seen_jobs.json"})
|
||||
continue
|
||||
|
||||
if result.get("status") == "expired":
|
||||
entry["status"] = "expired"
|
||||
expired.append(
|
||||
{"key": key, "title": entry.get("title"), "company": entry.get("company"), "url": entry.get("url")}
|
||||
)
|
||||
continue
|
||||
|
||||
try:
|
||||
score = overall_score(result.get("scores") or {})
|
||||
except ValueError as exc:
|
||||
errors.append({"key": key, "error": str(exc)})
|
||||
continue
|
||||
|
||||
legacy = entry_location_verdict(entry)
|
||||
if entry.get("location") in ("PASS", "FAIL", "FLAG"):
|
||||
entry.pop("location", None) # legacy verdict, never a place
|
||||
entry["status"] = "ranked"
|
||||
entry["rank_score"] = score
|
||||
entry["rank_verdict"] = band(score)
|
||||
entry["rank_date"] = today.isoformat()
|
||||
entry["location_verdict"] = result.get("location_verdict") or legacy or "PASS"
|
||||
entry["language_gate"] = result.get("language_gate") or "PASS"
|
||||
if entry["language_gate"] == "PASS":
|
||||
entry.pop("language_note", None)
|
||||
else:
|
||||
entry["language_note"] = result.get("language_note")
|
||||
# Absence is not a correction: a fetch that degraded to a listing page
|
||||
# returns no deadline, and blanking a stored one would erase a real
|
||||
# date and make the entry immortal to rule 6's sweep.
|
||||
if result.get("deadline"):
|
||||
entry["deadline"] = result["deadline"]
|
||||
for field in ("strengths", "gaps"):
|
||||
value = result.get(field)
|
||||
if isinstance(value, list):
|
||||
entry[field] = [str(b) for b in value][:3]
|
||||
|
||||
parsed = parse_iso(entry.get("deadline"))
|
||||
rows.append(
|
||||
{
|
||||
"key": key,
|
||||
"title": entry.get("title"),
|
||||
"company": entry.get("company"),
|
||||
"location": entry.get("location"),
|
||||
"url": entry.get("url"),
|
||||
"score": score,
|
||||
"verdict": entry["rank_verdict"],
|
||||
"location_verdict": entry["location_verdict"],
|
||||
"language_gate": entry["language_gate"],
|
||||
"language_note": entry.get("language_note"),
|
||||
"deadline": entry.get("deadline"),
|
||||
"posted_date": entry.get("posted_date"),
|
||||
"urgent": bool(parsed and today <= parsed <= today + timedelta(days=URGENT_DAYS)),
|
||||
"strengths": entry.get("strengths", []),
|
||||
"gaps": entry.get("gaps", []),
|
||||
}
|
||||
)
|
||||
|
||||
if not args.dry_run:
|
||||
save_state(args.state, doc)
|
||||
|
||||
rows.sort(key=lambda r: (r["score"], r["urgent"]), reverse=True)
|
||||
veto = lambda r: r["location_verdict"] == "FAIL" or r["language_gate"] == "FAIL"
|
||||
vetoed = [r for r in rows if veto(r)]
|
||||
ranked = [r for r in rows if not veto(r)]
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"ranked": ranked,
|
||||
"vetoed": vetoed,
|
||||
"expired": expired,
|
||||
"errors": errors,
|
||||
"written": not args.dry_run,
|
||||
},
|
||||
indent=2,
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
return 1 if errors else 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
common = argparse.ArgumentParser(add_help=False)
|
||||
common.add_argument("--state", type=Path, default=STATE)
|
||||
common.add_argument("--today", type=date.fromisoformat, default=date.today())
|
||||
|
||||
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||
sub = ap.add_subparsers(dest="command", required=True)
|
||||
|
||||
cand = sub.add_parser("candidates", parents=[common], help="select the entries to score")
|
||||
cand.add_argument("--tracker", type=Path, default=TRACKER)
|
||||
cand.add_argument("--all", action="store_true", help="include every non-skipped status")
|
||||
cand.add_argument("--focus", help="substring filter over title, company and stored fit notes")
|
||||
cand.add_argument("--limit", type=int, default=DEFAULT_LIMIT, help="0 for no cap")
|
||||
cand.set_defaults(func=cmd_candidates)
|
||||
|
||||
sweep = sub.add_parser("sweep", parents=[common], help="rule 6's expiry pass, no fetch")
|
||||
sweep.add_argument("--write", action="store_true", help="persist the expiries")
|
||||
sweep.add_argument("--exclude", help="comma-separated keys re-scored this run")
|
||||
sweep.set_defaults(func=cmd_sweep)
|
||||
|
||||
app = sub.add_parser("apply", parents=[common], help="write scoring results back and print the ranking")
|
||||
app.add_argument("--results", required=True, help="JSON array from the scoring agents")
|
||||
app.add_argument("--dry-run", action="store_true")
|
||||
app.set_defaults(func=cmd_apply)
|
||||
|
||||
args = ap.parse_args()
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user