Files
6krrt/src/feedback.py
adlee-was-taken 67ae9011fe fix(feedback): align dry-run preview count with projected snapshot; emit CSV header on empty backlog
- Defect 1: preview() header sample count now uses sum of grouped values from
  fp.project()'s in-memory copy, not len(rows) from the live ro connection.
  (Two snapshots were diverging under concurrent writes.)
- Defect 2: --dry-run --csv on empty backlog now emits the CSV header line
  via fp.write_csv([]) before early-returning, instead of printing zero bytes.
- Defect 3: test_main_contract_dry_run_does_not_apply patches
  proficiency_store.add_outcome (the entry point fold_onto actually calls)
  with wraps so the real function executes but the call is observable. Also
  verifies applied_at remains NULL on the on-disk DB.
2026-09-18 00:00:19 -04:00

244 lines
9.0 KiB
Python

#!/usr/bin/env python3
"""Fold observed verification outcomes back into proficiency.
The eval harness measures models on a fixed 23-task benchmark. This measures
them on YOUR traffic, which is more predictive of routing quality and
accumulates for free as you work.
python feedback.py --dry-run # project what would change
python feedback.py --dry-run --csv # the same projection, as CSV
python feedback.py --dry-run --db copy.db # project against a snapshot
python feedback.py # apply -- IRREVERSIBLE
``--dry-run`` projects the fold rather than describing it: it copies the
database into memory, runs the real ``add_outcome`` against the copy, and
reports the before/after ``blended_score`` per row. It opens the source
read-only and writes nothing to it. See ``feedback_preview.py``.
Client outcomes (kind=`client_outcome`) feed proficiency in both directions.
A client reporting `succeeded` is strong — the tests ran or the answer was
used, and the work was correct. A client reporting `failed` is definitive —
the work did not succeed. This is the only ground truth available here, and
the only signal that survives streaming, where a retry cannot reach.
Checks (`structural`, `local_llm`) no longer feed proficiency. A structural
'ok' only means the code parsed, not that it was correct, and a model emitting
syntactically valid nonsense would score 1.0. A `truncated` or `malformed`
verdict would bias `outcome_score` downward by checker frequency (checkers
fire on every response while only failures are evidence). Structural and
`local_llm` verdicts remain visible in `coverage()` for diagnostics only.
Client failures now route to `add_outcome` (outcome_score / outcome_samples)
instead of `add_self_eval`, keeping proficiency data on the outcome path that
the empirical-Bayes converter expects.
"""
from __future__ import annotations
import argparse
import sqlite3
import sys
from collections import defaultdict
from config import load_config
from proficiency_store import add_outcome
# Only client-reported "failed" counts — structural and local_llm verdicts
# (truncated, malformed) are diagnostics only. Including them in proficiency
# would bias outcome_score downward by checker frequency since checkers fire
# on every response and structural/local_llm passes are not evidence of quality.
FAILURE_VERDICTS = ("failed",)
# The one verdict that counts as a positive sample. A parser's 'ok' does not
# qualify — it means the code parsed. This means the client ran it.
SUCCESS_VERDICTS = ("succeeded",)
# Failures the model did not cause are excluded. The obvious case: a client
# that sets max_tokens=40 and gets a truncated answer caused that itself, and
# counting it would let any agent with a tight cap drag down whatever model it
# happened to route to. Found by forcing exactly that during testing.
def unapplied_failures(conn: sqlite3.Connection) -> list[sqlite3.Row]:
"""Rows that should move a model's score, with the sample each contributes.
Named for what it mostly is. Successes only enter via client outcomes; a
check passing is not evidence the answer was right.
"""
conn.row_factory = sqlite3.Row
scored = FAILURE_VERDICTS + SUCCESS_VERDICTS
placeholders = ",".join("?" * len(scored))
return conn.execute(
f"""
SELECT id, model_id, provider, task_category, kind, verdict, detail
FROM verifications
WHERE verdict IN ({placeholders})
AND applied_at IS NULL
AND task_category IS NOT NULL
AND model_attributable = 1
ORDER BY id
""",
scored,
).fetchall()
def summarize(
rows: list[sqlite3.Row],
) -> dict[tuple[str, str, str], list[tuple[int, float]]]:
"""Group by (model, provider, category) as (row id, sample score) pairs."""
grouped: dict[tuple[str, str, str], list[tuple[int, float]]] = defaultdict(list)
for r in rows:
score = 1.0 if r["verdict"] in SUCCESS_VERDICTS else 0.0
grouped[(r["model_id"], r["provider"], r["task_category"])].append(
(r["id"], score)
)
return grouped
def apply_failures(conn: sqlite3.Connection, cfg, grouped, dry_run: bool) -> int:
applied = 0
for (model_id, provider, category), pairs in sorted(grouped.items()):
ids = [i for i, _ in pairs]
scores = [sc for _, sc in pairs]
wins = sum(1 for sc in scores if sc == 1.0)
print(
f" {model_id:24s} {category:18s} "
f"{len(scores) - wins} failure(s), {wins} success(es)"
)
if dry_run:
continue
# Routes to add_outcome (outcome_score/outcome_samples) so the
# empirical-Bayes converter can fold client evidence into blended_score.
add_outcome(conn, cfg, model_id, provider, category, scores)
conn.executemany(
"UPDATE verifications SET applied_at = datetime('now') WHERE id = ?",
[(i,) for i in ids],
)
applied += len(ids)
if not dry_run:
conn.commit()
return applied
def coverage(conn: sqlite3.Connection) -> None:
"""Report what verification has seen, so its usefulness stays visible."""
conn.row_factory = sqlite3.Row
rows = conn.execute(
"""
SELECT kind, verdict, COUNT(*) n FROM verifications
GROUP BY kind, verdict ORDER BY kind, verdict
"""
).fetchall()
if not rows:
print(" no verifications recorded yet")
return
total = sum(r["n"] for r in rows)
print(f" {'kind':16s}{'verdict':16s}{'n':>6}{'share':>9}")
for r in rows:
print(f" {r['kind']:16s}{r['verdict']:16s}{r['n']:>6}{r['n']/total*100:>8.1f}%")
unver = sum(r["n"] for r in rows if r["verdict"] == "unverifiable")
if total and unver / total > 0.8:
print(
f"\n NOTE: {unver/total*100:.0f}% of responses were unverifiable. "
"Structural checking is not earning much here —\n"
" most traffic is prose. The local LLM check covers that, but only "
"above the size threshold."
)
def preview(db_path: str, cfg, as_csv: bool) -> int:
"""Project the fold against a read-only copy and report. Writes nothing."""
import feedback_preview as fp
# Read-only handle, not a copy: coverage and the emptiness check only read,
# and copy_database moves the whole file into memory -- 41 MB on the live
# database, which project() is about to pay for again a few lines down.
conn = sqlite3.connect(fp.read_only_uri(db_path), uri=True)
try:
rows = unapplied_failures(conn)
if not as_csv:
print("verification coverage so far:")
coverage(conn)
print()
finally:
conn.close()
if not rows:
if as_csv:
fp.write_csv([], sys.stdout)
else:
print("no unapplied signals — nothing to fold in")
return 0
changes, suppressed, grouped = fp.project(db_path, cfg)
if as_csv:
fp.write_csv(changes, sys.stdout)
return 0
sample_count = sum(len(v) for v in grouped.values())
print(fp.format_projection(changes, suppressed, grouped, sample_count))
print()
print(
"nothing was written. this fold is IRREVERSIBLE -- to apply it:\n"
" PYTHONPATH=src python -m feedback"
)
return 0
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument(
"--dry-run",
action="store_true",
help="project what the fold would change; write nothing",
)
ap.add_argument(
"--db",
metavar="PATH",
help="database to use (default: config database.path). With --dry-run "
"this may safely be a copy of the live DB.",
)
ap.add_argument(
"--csv",
action="store_true",
help="with --dry-run, emit the projection as CSV instead of the table",
)
args = ap.parse_args()
# --csv only shapes the projection, so without --dry-run it would be
# accepted and discarded -- and the run it was silently dropped from is the
# irreversible one. Refuse instead: this project has already shipped one
# config key that loaded cleanly and did nothing.
if args.csv and not args.dry_run:
ap.error("--csv only applies to --dry-run; the apply path prints a log")
cfg = load_config("config/config.yaml")
db_path = args.db or cfg.database.path
if args.dry_run:
return preview(db_path, cfg, args.csv)
conn = sqlite3.connect(db_path)
print("verification coverage so far:")
coverage(conn)
print()
rows = unapplied_failures(conn)
if not rows:
print("no unapplied signals — nothing to fold in")
conn.close()
return 0
grouped = summarize(rows)
print(f"applying {len(rows)} sample(s) across "
f"{len(grouped)} (model, category) pair(s):")
applied = apply_failures(conn, cfg, grouped, dry_run=False)
conn.close()
print(f"\napplied {applied} sample(s)")
return 0
if __name__ == "__main__":
raise SystemExit(main())