- Defect 1: preview() header sample count now uses sum of grouped values from fp.project()'s in-memory copy, not len(rows) from the live ro connection. (Two snapshots were diverging under concurrent writes.) - Defect 2: --dry-run --csv on empty backlog now emits the CSV header line via fp.write_csv([]) before early-returning, instead of printing zero bytes. - Defect 3: test_main_contract_dry_run_does_not_apply patches proficiency_store.add_outcome (the entry point fold_onto actually calls) with wraps so the real function executes but the call is observable. Also verifies applied_at remains NULL on the on-disk DB.
244 lines
9.0 KiB
Python
244 lines
9.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Fold observed verification outcomes back into proficiency.
|
|
|
|
The eval harness measures models on a fixed 23-task benchmark. This measures
|
|
them on YOUR traffic, which is more predictive of routing quality and
|
|
accumulates for free as you work.
|
|
|
|
python feedback.py --dry-run # project what would change
|
|
python feedback.py --dry-run --csv # the same projection, as CSV
|
|
python feedback.py --dry-run --db copy.db # project against a snapshot
|
|
python feedback.py # apply -- IRREVERSIBLE
|
|
|
|
``--dry-run`` projects the fold rather than describing it: it copies the
|
|
database into memory, runs the real ``add_outcome`` against the copy, and
|
|
reports the before/after ``blended_score`` per row. It opens the source
|
|
read-only and writes nothing to it. See ``feedback_preview.py``.
|
|
|
|
Client outcomes (kind=`client_outcome`) feed proficiency in both directions.
|
|
A client reporting `succeeded` is strong — the tests ran or the answer was
|
|
used, and the work was correct. A client reporting `failed` is definitive —
|
|
the work did not succeed. This is the only ground truth available here, and
|
|
the only signal that survives streaming, where a retry cannot reach.
|
|
|
|
Checks (`structural`, `local_llm`) no longer feed proficiency. A structural
|
|
'ok' only means the code parsed, not that it was correct, and a model emitting
|
|
syntactically valid nonsense would score 1.0. A `truncated` or `malformed`
|
|
verdict would bias `outcome_score` downward by checker frequency (checkers
|
|
fire on every response while only failures are evidence). Structural and
|
|
`local_llm` verdicts remain visible in `coverage()` for diagnostics only.
|
|
|
|
Client failures now route to `add_outcome` (outcome_score / outcome_samples)
|
|
instead of `add_self_eval`, keeping proficiency data on the outcome path that
|
|
the empirical-Bayes converter expects.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import sqlite3
|
|
import sys
|
|
from collections import defaultdict
|
|
|
|
from config import load_config
|
|
from proficiency_store import add_outcome
|
|
|
|
# Only client-reported "failed" counts — structural and local_llm verdicts
|
|
# (truncated, malformed) are diagnostics only. Including them in proficiency
|
|
# would bias outcome_score downward by checker frequency since checkers fire
|
|
# on every response and structural/local_llm passes are not evidence of quality.
|
|
FAILURE_VERDICTS = ("failed",)
|
|
|
|
# The one verdict that counts as a positive sample. A parser's 'ok' does not
|
|
# qualify — it means the code parsed. This means the client ran it.
|
|
SUCCESS_VERDICTS = ("succeeded",)
|
|
|
|
# Failures the model did not cause are excluded. The obvious case: a client
|
|
# that sets max_tokens=40 and gets a truncated answer caused that itself, and
|
|
# counting it would let any agent with a tight cap drag down whatever model it
|
|
# happened to route to. Found by forcing exactly that during testing.
|
|
|
|
|
|
def unapplied_failures(conn: sqlite3.Connection) -> list[sqlite3.Row]:
|
|
"""Rows that should move a model's score, with the sample each contributes.
|
|
|
|
Named for what it mostly is. Successes only enter via client outcomes; a
|
|
check passing is not evidence the answer was right.
|
|
"""
|
|
conn.row_factory = sqlite3.Row
|
|
scored = FAILURE_VERDICTS + SUCCESS_VERDICTS
|
|
placeholders = ",".join("?" * len(scored))
|
|
return conn.execute(
|
|
f"""
|
|
SELECT id, model_id, provider, task_category, kind, verdict, detail
|
|
FROM verifications
|
|
WHERE verdict IN ({placeholders})
|
|
AND applied_at IS NULL
|
|
AND task_category IS NOT NULL
|
|
AND model_attributable = 1
|
|
ORDER BY id
|
|
""",
|
|
scored,
|
|
).fetchall()
|
|
|
|
|
|
def summarize(
|
|
rows: list[sqlite3.Row],
|
|
) -> dict[tuple[str, str, str], list[tuple[int, float]]]:
|
|
"""Group by (model, provider, category) as (row id, sample score) pairs."""
|
|
grouped: dict[tuple[str, str, str], list[tuple[int, float]]] = defaultdict(list)
|
|
for r in rows:
|
|
score = 1.0 if r["verdict"] in SUCCESS_VERDICTS else 0.0
|
|
grouped[(r["model_id"], r["provider"], r["task_category"])].append(
|
|
(r["id"], score)
|
|
)
|
|
return grouped
|
|
|
|
|
|
def apply_failures(conn: sqlite3.Connection, cfg, grouped, dry_run: bool) -> int:
|
|
applied = 0
|
|
for (model_id, provider, category), pairs in sorted(grouped.items()):
|
|
ids = [i for i, _ in pairs]
|
|
scores = [sc for _, sc in pairs]
|
|
wins = sum(1 for sc in scores if sc == 1.0)
|
|
print(
|
|
f" {model_id:24s} {category:18s} "
|
|
f"{len(scores) - wins} failure(s), {wins} success(es)"
|
|
)
|
|
if dry_run:
|
|
continue
|
|
# Routes to add_outcome (outcome_score/outcome_samples) so the
|
|
# empirical-Bayes converter can fold client evidence into blended_score.
|
|
add_outcome(conn, cfg, model_id, provider, category, scores)
|
|
conn.executemany(
|
|
"UPDATE verifications SET applied_at = datetime('now') WHERE id = ?",
|
|
[(i,) for i in ids],
|
|
)
|
|
applied += len(ids)
|
|
if not dry_run:
|
|
conn.commit()
|
|
return applied
|
|
|
|
|
|
def coverage(conn: sqlite3.Connection) -> None:
|
|
"""Report what verification has seen, so its usefulness stays visible."""
|
|
conn.row_factory = sqlite3.Row
|
|
rows = conn.execute(
|
|
"""
|
|
SELECT kind, verdict, COUNT(*) n FROM verifications
|
|
GROUP BY kind, verdict ORDER BY kind, verdict
|
|
"""
|
|
).fetchall()
|
|
if not rows:
|
|
print(" no verifications recorded yet")
|
|
return
|
|
total = sum(r["n"] for r in rows)
|
|
print(f" {'kind':16s}{'verdict':16s}{'n':>6}{'share':>9}")
|
|
for r in rows:
|
|
print(f" {r['kind']:16s}{r['verdict']:16s}{r['n']:>6}{r['n']/total*100:>8.1f}%")
|
|
unver = sum(r["n"] for r in rows if r["verdict"] == "unverifiable")
|
|
if total and unver / total > 0.8:
|
|
print(
|
|
f"\n NOTE: {unver/total*100:.0f}% of responses were unverifiable. "
|
|
"Structural checking is not earning much here —\n"
|
|
" most traffic is prose. The local LLM check covers that, but only "
|
|
"above the size threshold."
|
|
)
|
|
|
|
|
|
def preview(db_path: str, cfg, as_csv: bool) -> int:
|
|
"""Project the fold against a read-only copy and report. Writes nothing."""
|
|
import feedback_preview as fp
|
|
|
|
# Read-only handle, not a copy: coverage and the emptiness check only read,
|
|
# and copy_database moves the whole file into memory -- 41 MB on the live
|
|
# database, which project() is about to pay for again a few lines down.
|
|
conn = sqlite3.connect(fp.read_only_uri(db_path), uri=True)
|
|
try:
|
|
rows = unapplied_failures(conn)
|
|
if not as_csv:
|
|
print("verification coverage so far:")
|
|
coverage(conn)
|
|
print()
|
|
finally:
|
|
conn.close()
|
|
|
|
if not rows:
|
|
if as_csv:
|
|
fp.write_csv([], sys.stdout)
|
|
else:
|
|
print("no unapplied signals — nothing to fold in")
|
|
return 0
|
|
|
|
changes, suppressed, grouped = fp.project(db_path, cfg)
|
|
if as_csv:
|
|
fp.write_csv(changes, sys.stdout)
|
|
return 0
|
|
|
|
sample_count = sum(len(v) for v in grouped.values())
|
|
print(fp.format_projection(changes, suppressed, grouped, sample_count))
|
|
print()
|
|
print(
|
|
"nothing was written. this fold is IRREVERSIBLE -- to apply it:\n"
|
|
" PYTHONPATH=src python -m feedback"
|
|
)
|
|
return 0
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument(
|
|
"--dry-run",
|
|
action="store_true",
|
|
help="project what the fold would change; write nothing",
|
|
)
|
|
ap.add_argument(
|
|
"--db",
|
|
metavar="PATH",
|
|
help="database to use (default: config database.path). With --dry-run "
|
|
"this may safely be a copy of the live DB.",
|
|
)
|
|
ap.add_argument(
|
|
"--csv",
|
|
action="store_true",
|
|
help="with --dry-run, emit the projection as CSV instead of the table",
|
|
)
|
|
args = ap.parse_args()
|
|
|
|
# --csv only shapes the projection, so without --dry-run it would be
|
|
# accepted and discarded -- and the run it was silently dropped from is the
|
|
# irreversible one. Refuse instead: this project has already shipped one
|
|
# config key that loaded cleanly and did nothing.
|
|
if args.csv and not args.dry_run:
|
|
ap.error("--csv only applies to --dry-run; the apply path prints a log")
|
|
|
|
cfg = load_config("config/config.yaml")
|
|
db_path = args.db or cfg.database.path
|
|
|
|
if args.dry_run:
|
|
return preview(db_path, cfg, args.csv)
|
|
|
|
conn = sqlite3.connect(db_path)
|
|
|
|
print("verification coverage so far:")
|
|
coverage(conn)
|
|
print()
|
|
|
|
rows = unapplied_failures(conn)
|
|
if not rows:
|
|
print("no unapplied signals — nothing to fold in")
|
|
conn.close()
|
|
return 0
|
|
|
|
grouped = summarize(rows)
|
|
print(f"applying {len(rows)} sample(s) across "
|
|
f"{len(grouped)} (model, category) pair(s):")
|
|
applied = apply_failures(conn, cfg, grouped, dry_run=False)
|
|
conn.close()
|
|
print(f"\napplied {applied} sample(s)")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|