Files
matchmaking-v2/tests/run_regression.py
raulgupta eb6ded5b3d Engine Phase 2: fix the pre-Opus funnel (lifts ALL matches)
The sift was feeding Opus a magnitude-blind, under-trusted, truncated shortlist — only 68% of the
genuinely-best jobs reached the curator. Fixes:
- embed.py: embed the FULL description (was desc[:600]) + role_category + industry; richer profile
  text (current_role + experience summary + seniority). The best semantic signal, fed real content.
- sift.py: magnitude-preserving min-max fusion (was rank-position, which flattened cosine 0.95 vs
  0.72) and embedding-LED weights (W_VIBE 0.6 / W_WHITEBOX 0.4). Embed runs FIRST so the white-box
  f_semantic reuses the real cosine (threaded as job["_vibe_cosine"]) instead of token overlap.
- config.py: SIFT_TOP_K 18 → 28 (the gate was cutting good jobs before Opus).
- curate.py: richer Opus briefs — full responsibilities (was desc[:500]) + role_category/industry/
  seniority/pay, so even the shortlist is fully described.

Regression (120 pairs): sift membership recall 0.68 → 0.92, score MAE 24.7 → 18.4. 31 tests pass.
2026-06-25 15:50:14 +05:30

94 lines
4.2 KiB
Python

"""Regression runner — the measurement gate for every engine-improvement phase.
The fixture `tests/fixtures/regression_set.json` freezes a set of CASES, each = {prefs, user_context,
jobs[], targets{job_id: recruiter_score}}. The jobs are captured from a real sweep so the run is
reproducible WITHOUT re-hitting Apify (deterministic, cheap, comparable across phases).
For each case this runs the REAL engine (rank → sift) over the frozen jobs and reports two metrics:
• SCORE fidelity — MAE / correlation of the engine's matchScore vs the human/judge target.
• SIFT TOP-K MEMBERSHIP RECALL — of the genuinely-best jobs (target ≥ 75), how many reach the top-K
pool Opus actually sees. This is the "is the funnel feeding Opus the best jobs?" metric.
Usage: python tests/run_regression.py # prints the table for the current engine
python tests/run_regression.py --json # machine-readable (to diff phases)
"""
from __future__ import annotations
import json
import sys
from pathlib import Path
from app.engine import rank as _rank
from app.engine import sift as _sift
from app.engine import embed as _embed
from app.config import get_settings
FIXTURE = Path(__file__).parent / "fixtures" / "regression_set.json"
GREAT = 75 # target ≥ this = a "genuinely good" job that SHOULD reach Opus
def _mae(pairs):
return round(sum(abs(a - b) for a, b in pairs) / len(pairs), 1) if pairs else None
def run(emit_json: bool = False):
data = json.loads(FIXTURE.read_text())
cases = data["cases"]
k = get_settings().SIFT_TOP_K
rows, all_score_pairs, recalls = [], [], []
for c in cases:
prefs, ctx, jobs = c["prefs"], c.get("user_context"), c["jobs"]
targets = {str(j): int(s) for j, s in c["targets"].items()}
# engine score = the funnel's white-box AFTER the embedding cosine is threaded in (what the
# fallback shows + feeds the sift) — replicate embed→thread→rank so MAE reflects the real funnel.
jc = [dict(j) for j in jobs]
vibe = _embed.vibe_scores(prefs, ctx, jc)
if vibe:
for j in jc:
if j["id"] in vibe:
j["_vibe_cosine"] = vibe[j["id"]]
ranked = _rank.rank(prefs, ctx, jc)["opportunities"]
eng = {j["id"]: j.get("matchScore", 0) for j in ranked}
# the sift top-K the curator actually sees
top, _ = _sift.sift(prefs, ctx, [dict(j) for j in jobs], k=k)
top_ids = {j["id"] for j in top}
score_pairs = [(eng[jid], targets[jid]) for jid in eng if jid in targets]
all_score_pairs += score_pairs
# membership recall: of target-great jobs, how many made the sift top-K
great = [jid for jid, t in targets.items() if t >= GREAT and jid in eng]
in_top = [jid for jid in great if jid in top_ids]
recall = round(len(in_top) / len(great), 2) if great else None
if recall is not None:
recalls.append(recall)
rows.append({
"case": c["label"], "n_jobs": len(jobs), "n_great": len(great),
"great_in_topK": len(in_top), "membership_recall": recall,
"score_mae": _mae(score_pairs),
})
agg = {
"sift_top_k": k,
"overall_score_mae": _mae(all_score_pairs),
"overall_membership_recall": round(sum(recalls) / len(recalls), 2) if recalls else None,
"n_cases": len(cases), "n_pairs": len(all_score_pairs),
}
if emit_json:
print(json.dumps({"cases": rows, "aggregate": agg}, indent=2))
return
print(f"\n Regression — engine scoring vs target (SIFT_TOP_K={k})")
print(" " + "-" * 78)
print(f" {'case':<40}{'jobs':>5}{'great':>7}{'in-topK':>9}{'recall':>8}{'MAE':>6}")
for r in rows:
print(f" {r['case'][:40]:<40}{r['n_jobs']:>5}{r['n_great']:>7}{r['great_in_topK']:>9}"
f"{str(r['membership_recall']):>8}{str(r['score_mae']):>6}")
print(" " + "-" * 78)
print(f" AGGREGATE → score MAE {agg['overall_score_mae']} · "
f"sift membership recall {agg['overall_membership_recall']} "
f"({agg['n_pairs']} pairs, {agg['n_cases']} cases)\n")
if __name__ == "__main__":
run(emit_json="--json" in sys.argv)