#!/usr/bin/env python3 """Fold observed verification outcomes back into proficiency. The eval harness measures models on a fixed 23-task benchmark. This measures them on YOUR traffic, which is more predictive of routing quality and accumulates for free as you work. python feedback.py --dry-run # project what would change python feedback.py --dry-run --csv # the same projection, as CSV python feedback.py --dry-run --db copy.db # project against a snapshot python feedback.py # apply -- IRREVERSIBLE ``--dry-run`` projects the fold rather than describing it: it copies the database into memory, runs the real ``add_outcome`` against the copy, and reports the before/after ``blended_score`` per row. It opens the source read-only and writes nothing to it. See ``feedback_preview.py``. Client outcomes (kind=`client_outcome`) feed proficiency in both directions. A client reporting `succeeded` is strong — the tests ran or the answer was used, and the work was correct. A client reporting `failed` is definitive — the work did not succeed. This is the only ground truth available here, and the only signal that survives streaming, where a retry cannot reach. Checks (`structural`, `local_llm`) no longer feed proficiency. A structural 'ok' only means the code parsed, not that it was correct, and a model emitting syntactically valid nonsense would score 1.0. A `truncated` or `malformed` verdict would bias `outcome_score` downward by checker frequency (checkers fire on every response while only failures are evidence). Structural and `local_llm` verdicts remain visible in `coverage()` for diagnostics only. Client failures now route to `add_outcome` (outcome_score / outcome_samples) instead of `add_self_eval`, keeping proficiency data on the outcome path that the empirical-Bayes converter expects. """ from __future__ import annotations import argparse import sqlite3 import sys from collections import defaultdict from config import load_config from proficiency_store import add_outcome # Only client-reported "failed" counts — structural and local_llm verdicts # (truncated, malformed) are diagnostics only. Including them in proficiency # would bias outcome_score downward by checker frequency since checkers fire # on every response and structural/local_llm passes are not evidence of quality. FAILURE_VERDICTS = ("failed",) # The one verdict that counts as a positive sample. A parser's 'ok' does not # qualify — it means the code parsed. This means the client ran it. SUCCESS_VERDICTS = ("succeeded",) # Failures the model did not cause are excluded. The obvious case: a client # that sets max_tokens=40 and gets a truncated answer caused that itself, and # counting it would let any agent with a tight cap drag down whatever model it # happened to route to. Found by forcing exactly that during testing. def unapplied_failures(conn: sqlite3.Connection) -> list[sqlite3.Row]: """Rows that should move a model's score, with the sample each contributes. Named for what it mostly is. Successes only enter via client outcomes; a check passing is not evidence the answer was right. """ conn.row_factory = sqlite3.Row scored = FAILURE_VERDICTS + SUCCESS_VERDICTS placeholders = ",".join("?" * len(scored)) return conn.execute( f""" SELECT id, model_id, provider, task_category, kind, verdict, detail FROM verifications WHERE verdict IN ({placeholders}) AND applied_at IS NULL AND task_category IS NOT NULL AND model_attributable = 1 ORDER BY id """, scored, ).fetchall() def summarize( rows: list[sqlite3.Row], ) -> dict[tuple[str, str, str], list[tuple[int, float]]]: """Group by (model, provider, category) as (row id, sample score) pairs.""" grouped: dict[tuple[str, str, str], list[tuple[int, float]]] = defaultdict(list) for r in rows: score = 1.0 if r["verdict"] in SUCCESS_VERDICTS else 0.0 grouped[(r["model_id"], r["provider"], r["task_category"])].append( (r["id"], score) ) return grouped def apply_failures(conn: sqlite3.Connection, cfg, grouped, dry_run: bool) -> int: applied = 0 for (model_id, provider, category), pairs in sorted(grouped.items()): ids = [i for i, _ in pairs] scores = [sc for _, sc in pairs] wins = sum(1 for sc in scores if sc == 1.0) print( f" {model_id:24s} {category:18s} " f"{len(scores) - wins} failure(s), {wins} success(es)" ) if dry_run: continue # Routes to add_outcome (outcome_score/outcome_samples) so the # empirical-Bayes converter can fold client evidence into blended_score. add_outcome(conn, cfg, model_id, provider, category, scores) conn.executemany( "UPDATE verifications SET applied_at = datetime('now') WHERE id = ?", [(i,) for i in ids], ) applied += len(ids) if not dry_run: conn.commit() return applied def coverage(conn: sqlite3.Connection) -> None: """Report what verification has seen, so its usefulness stays visible.""" conn.row_factory = sqlite3.Row rows = conn.execute( """ SELECT kind, verdict, COUNT(*) n FROM verifications GROUP BY kind, verdict ORDER BY kind, verdict """ ).fetchall() if not rows: print(" no verifications recorded yet") return total = sum(r["n"] for r in rows) print(f" {'kind':16s}{'verdict':16s}{'n':>6}{'share':>9}") for r in rows: print(f" {r['kind']:16s}{r['verdict']:16s}{r['n']:>6}{r['n']/total*100:>8.1f}%") unver = sum(r["n"] for r in rows if r["verdict"] == "unverifiable") if total and unver / total > 0.8: print( f"\n NOTE: {unver/total*100:.0f}% of responses were unverifiable. " "Structural checking is not earning much here —\n" " most traffic is prose. The local LLM check covers that, but only " "above the size threshold." ) def preview(db_path: str, cfg, as_csv: bool) -> int: """Project the fold against a read-only copy and report. Writes nothing.""" import feedback_preview as fp # Read-only handle, not a copy: coverage and the emptiness check only read, # and copy_database moves the whole file into memory -- 41 MB on the live # database, which project() is about to pay for again a few lines down. conn = sqlite3.connect(fp.read_only_uri(db_path), uri=True) try: rows = unapplied_failures(conn) if not as_csv: print("verification coverage so far:") coverage(conn) print() finally: conn.close() if not rows: if as_csv: fp.write_csv([], sys.stdout) else: print("no unapplied signals — nothing to fold in") return 0 changes, suppressed, grouped = fp.project(db_path, cfg) if as_csv: fp.write_csv(changes, sys.stdout) return 0 sample_count = sum(len(v) for v in grouped.values()) print(fp.format_projection(changes, suppressed, grouped, sample_count)) print() print( "nothing was written. this fold is IRREVERSIBLE -- to apply it:\n" " PYTHONPATH=src python -m feedback" ) return 0 def main() -> int: ap = argparse.ArgumentParser(description=__doc__) ap.add_argument( "--dry-run", action="store_true", help="project what the fold would change; write nothing", ) ap.add_argument( "--db", metavar="PATH", help="database to use (default: config database.path). With --dry-run " "this may safely be a copy of the live DB.", ) ap.add_argument( "--csv", action="store_true", help="with --dry-run, emit the projection as CSV instead of the table", ) args = ap.parse_args() # --csv only shapes the projection, so without --dry-run it would be # accepted and discarded -- and the run it was silently dropped from is the # irreversible one. Refuse instead: this project has already shipped one # config key that loaded cleanly and did nothing. if args.csv and not args.dry_run: ap.error("--csv only applies to --dry-run; the apply path prints a log") cfg = load_config("config/config.yaml") db_path = args.db or cfg.database.path if args.dry_run: return preview(db_path, cfg, args.csv) conn = sqlite3.connect(db_path) print("verification coverage so far:") coverage(conn) print() rows = unapplied_failures(conn) if not rows: print("no unapplied signals — nothing to fold in") conn.close() return 0 grouped = summarize(rows) print(f"applying {len(rows)} sample(s) across " f"{len(grouped)} (model, category) pair(s):") applied = apply_failures(conn, cfg, grouped, dry_run=False) conn.close() print(f"\napplied {applied} sample(s)") return 0 if __name__ == "__main__": raise SystemExit(main())