diff --git a/admin/frontend/controls.html b/admin/frontend/controls.html index 094ef53..f46a4e5 100644 --- a/admin/frontend/controls.html +++ b/admin/frontend/controls.html @@ -877,9 +877,12 @@ function classifierModeFieldsHtml(mode, data) {
- +
+ + % +
`; } @@ -967,8 +970,10 @@ function collectClassifierConfigBody() { const model = document.getElementById('classifier-encoder-model').value.trim(); if (model) encoder.model = model; encoder.device = document.getElementById('classifier-encoder-device').value; - const threshold = document.getElementById('classifier-encoder-threshold').value; - if (threshold !== '') encoder.confidence_threshold = parseFloat(threshold); + const thresholdPct = document.getElementById('classifier-encoder-threshold').value; + // The field is 0-100 for a human to type ("80" meaning 80%); the backend + // wants the 0.0-1.0 probability classify_zero_shot actually returns. + if (thresholdPct !== '') encoder.confidence_threshold = parseFloat(thresholdPct) / 100; body.encoder = encoder; } return body; diff --git a/src/config.py b/src/config.py index 4754b29..3a0af00 100644 --- a/src/config.py +++ b/src/config.py @@ -796,9 +796,27 @@ class LocalEncoderConfig(StrictModel): # Below this, the classification is treated as a FAILURE, not a low- # confidence answer -- the caller cascades exactly as it would for a # local-LLM parse failure, rather than confidently mis-routing on a - # guess the encoder itself was unsure about. + # guess the encoder itself was unsure about. classify_zero_shot returns + # a 0.0-1.0 probability, so this must be too -- a percent-style value + # (e.g. 80 meaning "80%") silently makes every real confidence score + # read as below-threshold, since no probability exceeds 1.0. Caught live + # 2026-09-06: the admin UI took a raw number with no conversion or + # bound, so typing the intuitive "80" broke classification on every + # request. The UI now converts 0-100 to 0.0-1.0 before saving; this + # validator is the fail-closed backstop for any other caller. confidence_threshold: float = 0.5 + @field_validator("confidence_threshold") + @classmethod + def confidence_threshold_in_range(cls, v: float) -> float: + if not (0.0 <= v <= 1.0): + raise ValueError( + f"classifier.encoder.confidence_threshold must be in [0.0, 1.0], " + f"got {v!r} -- classify_zero_shot returns a 0-1 probability, " + f"not a percent" + ) + return v + class ClassifierConfig(StrictModel): provider: str diff --git a/tests/test_admin_frontend.py b/tests/test_admin_frontend.py index 7e434af..9fcb2a3 100644 --- a/tests/test_admin_frontend.py +++ b/tests/test_admin_frontend.py @@ -133,6 +133,21 @@ def test_admin_controls_has_a_dedicated_classifier_card(admin_client): assert "(baseUrl && model) ? { base_url: baseUrl, model } : null" in text +def test_admin_controls_confidence_threshold_field_is_percent_with_conversion(admin_client): + """The encoder confidence_threshold field takes 0-100 (a human types "80" + meaning 80%) and converts to the 0.0-1.0 probability classify_zero_shot + actually returns -- typing the intuitive percent value used to be stored + literally, so no real confidence score could ever clear the threshold + and every classification silently failed (live 2026-09-06).""" + resp = admin_client.get("/admin/controls") + text = resp.text + assert 'id="classifier-encoder-threshold"' in text + assert 'max="100"' in text + assert "parseFloat(thresholdPct) / 100" in text + # Loading back a stored 0.0-1.0 value must display it as a percent. + assert "Math.round(enc.confidence_threshold * 100)" in text + + def test_admin_models_returns_html_with_availability_marker(admin_client): """GET /admin/models returns 200, text/html, and contains the Model Availability card title.""" diff --git a/tests/test_classifier_modes_config.py b/tests/test_classifier_modes_config.py index c40fa7c..f2e7089 100644 --- a/tests/test_classifier_modes_config.py +++ b/tests/test_classifier_modes_config.py @@ -139,6 +139,36 @@ def test_local_encoder_rejects_unknown_device(raw): RouterConfig(**cfg) +def test_local_encoder_rejects_percent_style_confidence_threshold(raw): + """classify_zero_shot returns a 0.0-1.0 probability, so a percent-style + value (e.g. 80 meaning "80%") must be rejected -- otherwise no real + confidence score can ever clear the threshold and every classification + silently fails. Caught live 2026-09-06 via the admin UI taking a raw + number with no conversion.""" + cfg = copy.deepcopy(raw) + cfg["classifier"]["mode"] = "local_encoder" + cfg["classifier"]["encoder"] = {"confidence_threshold": 80} + with pytest.raises(ValueError, match=r"must be in \[0.0, 1.0\]"): + RouterConfig(**cfg) + + +def test_local_encoder_rejects_negative_confidence_threshold(raw): + cfg = copy.deepcopy(raw) + cfg["classifier"]["mode"] = "local_encoder" + cfg["classifier"]["encoder"] = {"confidence_threshold": -0.1} + with pytest.raises(ValueError, match=r"must be in \[0.0, 1.0\]"): + RouterConfig(**cfg) + + +def test_local_encoder_accepts_confidence_threshold_at_bounds(raw): + cfg = copy.deepcopy(raw) + cfg["classifier"]["mode"] = "local_encoder" + cfg["classifier"]["encoder"] = {"confidence_threshold": 0.0} + assert RouterConfig(**cfg).classifier.encoder.confidence_threshold == 0.0 + cfg["classifier"]["encoder"] = {"confidence_threshold": 1.0} + assert RouterConfig(**cfg).classifier.encoder.confidence_threshold == 1.0 + + def test_local_encoder_unaffected_by_the_cloud_llm_validator(raw): """A local_encoder config leaving cloud_primary/auto both unset must not trip the cloud_llm validator -- it's scoped to mode == cloud_llm."""