From 1f74895910184fef6b15da33fdf76b1198e33581 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sun, 6 Sep 2026 20:34:01 -0400 Subject: [PATCH] fix(classifier): local_encoder default model is gated, switch to bart-large-mnli MoritzLaurer/deberta-v3-base-zeroshot-v2 (the local_encoder default) started returning 401 on an unauthenticated GET of its own model page -- gated or moved sometime after this project picked it, discovered live 2026-09-06 when it took production down entirely (crash-loop, exit code 1). The startup check (ensure_available) did its job -- refused to boot rather than fail opaquely on the first request -- but the service still couldn't come up until the default was fixed. Switched to facebook/bart-large-mnli: HuggingFace's own reference model for the zero-shot-classification pipeline, confirmed publicly accessible live. Larger than the old default (~407M vs ~184M params) but that's the honest tradeoff for "still works." Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U --- admin/frontend/controls.html | 2 +- config/config.yaml | 2 +- docs/local-models.md | 22 ++++++++++++++++++---- src/config.py | 10 +++++++++- tests/test_classifier_modes_config.py | 2 +- 5 files changed, 30 insertions(+), 8 deletions(-) diff --git a/admin/frontend/controls.html b/admin/frontend/controls.html index f46a4e5..40b14a1 100644 --- a/admin/frontend/controls.html +++ b/admin/frontend/controls.html @@ -867,7 +867,7 @@ function classifierModeFieldsHtml(mode, data) {
diff --git a/config/config.yaml b/config/config.yaml index 1353f20..abfef97 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -707,7 +707,7 @@ classifier: # Read only when mode: local_encoder. Every field has a default, so # `encoder: {}` is enough to opt in. # encoder: - # model: MoritzLaurer/deberta-v3-base-zeroshot-v2 + # model: facebook/bart-large-mnli # device: cpu # confidence_threshold: 0.5 diff --git a/docs/local-models.md b/docs/local-models.md index d05d425..f5293a3 100644 --- a/docs/local-models.md +++ b/docs/local-models.md @@ -213,10 +213,24 @@ rarer each time, but never structurally impossible, because a chat-completion model can always in principle spend its budget thinking instead of answering. `classifier.mode: local_encoder` sidesteps the whole failure class instead of picking around it: a zero-shot NLI encoder -(`classifier.encoder.model`, default `MoritzLaurer/deberta-v3-base- -zeroshot-v2`) scores the task directly against `proficiency.categories` and -returns a label plus a confidence — there is no generation step, so there is -no trace to run away. +(`classifier.encoder.model`, default `facebook/bart-large-mnli`) scores the +task directly against `proficiency.categories` and returns a label plus a +confidence — there is no generation step, so there is no trace to run away. +(Was `MoritzLaurer/deberta-v3-base-zeroshot-v2` — smaller, ~184M vs ~407M +params — until that repo started returning 401 on an unauthenticated GET of +its own model page sometime after this project picked it, discovered live +2026-09-06 when it took production down: the startup check correctly +refused to boot rather than fail opaquely on the first request, but the +service still crash-looped until the default was corrected.) + +`classifier.encoder.confidence_threshold` is a `[0.0, 1.0]` probability +(`classify_zero_shot`'s own output), not a percent — the admin UI takes 0-100 +for a human to type and converts at the save boundary, but a config file +edit or any other caller must use the raw probability. Config load now +validates the range; a value like `80` used to be silently accepted and +would make every real confidence score read as below-threshold, since none +can exceed `1.0` (also caught live 2026-09-06, before a restart made it +active). The trade is real, not free. It only produces `task_category` — no `task_tier` signal exists in a zero-shot label score, so tier falls back to diff --git a/src/config.py b/src/config.py index 3a0af00..3cfdc8e 100644 --- a/src/config.py +++ b/src/config.py @@ -791,7 +791,15 @@ class LocalEncoderConfig(StrictModel): learn from without a new, separate opt-in data-capture feature. """ - model: str = "MoritzLaurer/deberta-v3-base-zeroshot-v2" + # facebook/bart-large-mnli -- the reference model HuggingFace's own docs + # use for this exact pipeline. Was MoritzLaurer/deberta-v3-base-zeroshot-v2 + # (smaller, ~184M vs ~407M params) until that repo started returning 401 + # even on an unauthenticated GET of its model page -- gated or moved + # sometime after this project picked it. Caught live 2026-09-06: the + # startup check (ensure_available) correctly refused to boot rather than + # fail opaquely on the first request, but it still took production down + # until the default was fixed. + model: str = "facebook/bart-large-mnli" device: Literal["cpu", "cuda"] = "cpu" # Below this, the classification is treated as a FAILURE, not a low- # confidence answer -- the caller cascades exactly as it would for a diff --git a/tests/test_classifier_modes_config.py b/tests/test_classifier_modes_config.py index f2e7089..e77d528 100644 --- a/tests/test_classifier_modes_config.py +++ b/tests/test_classifier_modes_config.py @@ -122,7 +122,7 @@ def test_local_encoder_with_empty_encoder_block_loads_with_defaults(raw): cfg["classifier"]["mode"] = "local_encoder" cfg["classifier"]["encoder"] = {} loaded = RouterConfig(**cfg) - assert loaded.classifier.encoder.model == "MoritzLaurer/deberta-v3-base-zeroshot-v2" + assert loaded.classifier.encoder.model == "facebook/bart-large-mnli" assert loaded.classifier.encoder.device == "cpu" assert loaded.classifier.encoder.confidence_threshold == 0.5 -- 2.49.1