From 6d36eeff746db15ddd27f3f0ed61079a3e1bd1d6 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sun, 30 Aug 2026 00:52:30 -0400 Subject: [PATCH 1/4] docs: document Gitea tea CLI for PR workflow (gh not available) --- AGENTS.md | 41 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index 2388947..ac5c423 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -161,6 +161,47 @@ PYTHONPATH=src python -m router_cli "Refactor this Django view" PYTHONPATH=src python -m poller && PYTHONPATH=src python -m tier ``` +## Gitea / `tea` CLI (not `gh`) + +This repo is hosted on **self-hosted Gitea** (`git.adlee.work/alee/6krrt`). +The GitHub CLI `gh` is **not installed** and will fail — use `tea` instead. + +- **Login:** `tea login` as `alee`. The `alee` login is NOT the default tea + login — always pass `--remote origin` so tea resolves the right context. + +- **Remote:** `origin` resolves to `git.adlee.work/alee/6krrt`. + +- **List open PRs:** `tea pulls --remote origin` + +- **PR metadata (JSON):** + ``` + tea pulls --remote origin --fields index,state,draft,title,mergeable,base,head,body -o json + ``` + Use the `base` and `head` from this output for local diff commands. + +- **PR creation / update / maintenance:** use the `tea pr` family (e.g. `tea pr create --base main --title ... --description ...`). Confirm against `tea pr --help` for the current flag set. + +- **Post a comment to a PR** (uses the `tea api` proxy to Gitea's REST API): + ``` + tea api --remote origin "/repos/{owner}/{repo}/issues//comments" -F body=@- + ``` + - **CRITICAL:** use `-F` (typed field), **not** `-f`. `-f key=@file` writes the literal string `@file`; only `-F` reads stdin or the contents of a file. + - The endpoint **must** include the `/repos/` prefix — omitting it returns a 404. + - `tea api` already prints JSON to stdout — do **not** pass `-o json` on top of that (you would write the body to a file literally named `json`). + - Fallback if stdin is awkward: `-F body=@/tmp/review-body.md`. + +- **Getting a PR diff:** prefer local `git diff origin/...origin/` + (base and head from the PR JSON). In tea v0.14.1 the `diff`/`patch` fields + return empty even when requested — do not rely on them. + +- **Gitea source links** (not GitHub `blob` links): + `https://git.adlee.work/alee/6krrt/src/commit//#L-L` + (segment is `src/commit`, not `blob`). + +- **Code-review plugin:** this repo ships a project-local fork of the code-review + plugin (`code-review-tea`) that already wires these `tea` calls. Prefer + invoking it (e.g. `/code-review-tea:code-review`) over hand-rolling. + ## What's NOT built yet (open items) 1. **Leaderboard priors unfilled** — `leaderboards.yaml` ships empty. -- 2.49.1 From 69662f066ed7132caa89fdfbb61b4804403498cc Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sun, 30 Aug 2026 10:56:14 -0400 Subject: [PATCH 2/4] feat: enable circuit_breaker, pinch, and pinch.relevance by default Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- config/config.yaml | 25 +++++++++++-------------- src/config.py | 10 +++++----- src/dispatcher.py | 12 +++++++++--- tests/test_admin_config.py | 4 +++- tests/test_config_endpoints.py | 10 +++++----- tests/test_dispatcher_helpers.py | 24 ++++++++++++++++++++++-- 6 files changed, 55 insertions(+), 30 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index b0559ce..23630d4 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -166,12 +166,11 @@ pinch: # they carry the bulk of a long agent session's tokens and are least needed # in full by the time the next turn is answered. # - # It does NOT touch the classifier's input, and it defaults off — enable it - # only if long sessions are shipping more prompt tokens than you want to pay - # for. Tool results can only be dropped safely because tool outputs are - # idempotent enough for a placeholder; a wrong guess here loses context, so - # start conservative (large budget, small reduction). - enabled: false + # On by default. Tool results can only be dropped safely because tool + # outputs are idempotent enough for a placeholder; a wrong guess here loses + # context, so start conservative (large budget, small reduction) and watch + # route_decisions / pinch stats on real traffic before widening it. + enabled: true budget_tokens: 50000 keep_last_turns: 4 max_summarize_chars: 4000 @@ -180,7 +179,7 @@ pinch: # project — and specifically requires pinch.enabled too, since this has no # effect otherwise. Ship it, watch route_decisions / pinch stats on real # traffic, then decide the default. - enabled: false + enabled: true # An EMBEDDING model, not a chat model — this must not point at # classifier.model or verification.model. Pull one on the same Ollama: # ollama pull nomic-embed-text @@ -207,13 +206,11 @@ session_cache: staleness_minutes: 20 circuit_breaker: - # Passive availability circuit breaker. Off by default, matching every other - # new-and-unproven knob in this project. When enabled, a model that returns - # 5xx is temporarily excluded from routing with exponential backoff; recovery - # is passive (the next real request becomes the probe once the cooldown - # passes). Unlike most new knobs this one has a low-risk failure mode even - # when wrong, so it's a reasonable candidate to flip on sooner. - enabled: false + # Passive availability circuit breaker, on by default. When enabled, a model + # that returns 5xx is temporarily excluded from routing with exponential + # backoff; recovery is passive (the next real request becomes the probe once + # the cooldown passes). It has a low-risk failure mode even when wrong. + enabled: true initial_cooldown_seconds: 30 max_cooldown_seconds: 600 backoff_multiplier: 2.0 diff --git a/src/config.py b/src/config.py index 393b46f..35a1aa1 100644 --- a/src/config.py +++ b/src/config.py @@ -306,12 +306,12 @@ class PinchRelevanceConfig(StrictModel): When enabled (and ``pinch.enabled`` is also true), the dispatcher embeds the current-turn query with the old tool-result candidates and trims the - least relevant first, so a relevant-but-old result survives. Off by + least relevant first, so a relevant-but-old result survives. On by default; any failure reverts to uniform trimming. This must point at an EMBEDDING model, never ``classifier.model`` or ``verification.model``. """ - enabled: bool = False + enabled: bool = True # OpenAI-compatible embeddings endpoint on the same local Ollama. model: str = "nomic-embed-text" base_url: str = "http://localhost:11434/v1" @@ -343,7 +343,7 @@ class PinchConfig(StrictModel): are summarized or dropped (they carry the bulk of a long session's tokens). """ - enabled: bool = False + enabled: bool = True budget_tokens: int = 50000 # How many recent user turns (plus their assistant replies and tool results) # are protected from pruning. @@ -406,10 +406,10 @@ class CircuitBreakerConfig(StrictModel): When enabled, a model that returns 5xx is temporarily skipped by routing (with exponential backoff). Recovery is passive: a real request that would - have picked it becomes the probe once the cooldown passes. Off by default. + have picked it becomes the probe once the cooldown passes. On by default. """ - enabled: bool = False + enabled: bool = True initial_cooldown_seconds: int = 30 max_cooldown_seconds: int = 600 backoff_multiplier: float = 2.0 diff --git a/src/dispatcher.py b/src/dispatcher.py index bf86f3e..806493c 100644 --- a/src/dispatcher.py +++ b/src/dispatcher.py @@ -59,6 +59,7 @@ from capabilities import detect_capabilities, iter_image_url_values from config import FlexPreference, RouterConfig, load_config from context_prune import ( extract_text, + estimate_tokens, order_by_relevance, prune_context, trim_candidates, @@ -1818,7 +1819,7 @@ def _embed_for_relevance( return order_by_relevance(embeddings[0], embeddings[1:]) -def _relevance_order_for(messages: list[dict], cfg) -> Optional[list[int]]: +def _relevance_order_for(messages: list[dict], cfg, budget_tokens: int) -> Optional[list[int]]: """Compute the relevance_order for prune_context, or None (uniform). When pinch.relevance is off, or candidate count is below min_candidates, @@ -1831,6 +1832,11 @@ def _relevance_order_for(messages: list[dict], cfg) -> Optional[list[int]]: candidates, _, _ = trim_candidates(messages, cfg.pinch.keep_last_turns) if len(candidates) < rel.min_candidates: return None + # Skip the embed when the conversation is under budget — prune_context would + # be a no-op, so ranking is pointless. + orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + if orig_tokens <= budget_tokens: + return None query = _last_user_text(messages) candidate_texts = [_text_only(messages[i]) for i in candidates] return _embed_for_relevance(query, candidate_texts, cfg) @@ -2275,7 +2281,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks): budget_tokens=cfg.pinch.budget_tokens, keep_last_turns=cfg.pinch.keep_last_turns, max_summarize_chars=cfg.pinch.max_summarize_chars, - relevance_order=_relevance_order_for(messages, cfg), + relevance_order=_relevance_order_for(messages, cfg, budget_tokens=cfg.pinch.budget_tokens), ) logs.debug( "pinch", @@ -2507,7 +2513,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks): budget_tokens=cfg.pinch.budget_tokens, keep_last_turns=cfg.pinch.keep_last_turns, max_summarize_chars=cfg.pinch.max_summarize_chars, - relevance_order=_relevance_order_for(messages, cfg), + relevance_order=_relevance_order_for(messages, cfg, budget_tokens=cfg.pinch.budget_tokens), ) logs.debug( "pinch", diff --git a/tests/test_admin_config.py b/tests/test_admin_config.py index 2d70477..57c66c0 100644 --- a/tests/test_admin_config.py +++ b/tests/test_admin_config.py @@ -20,6 +20,7 @@ import threading from pathlib import Path import pytest +import yaml from fastapi import FastAPI from starlette.testclient import TestClient @@ -138,7 +139,8 @@ def test_config_POST_creates_backup_before_write(client, tmp_path): # The backup captured the sentinel comment and the PRE-write value. backup_text = backups[0].read_text() assert _SENTINEL in backup_text - assert re.search(r"^\s*enabled:\s*false\s*$", backup_text, re.MULTILINE) is not None + backup_yaml = yaml.safe_load(backup_text) + assert backup_yaml["circuit_breaker"]["enabled"] is True def test_config_concurrent_writes_are_atomic_no_zero_byte_backups(tmp_path): diff --git a/tests/test_config_endpoints.py b/tests/test_config_endpoints.py index 56ca62c..f2fd9fe 100644 --- a/tests/test_config_endpoints.py +++ b/tests/test_config_endpoints.py @@ -237,7 +237,7 @@ def test_nonpositive_local_timeout_is_rejected(raw): def test_the_shipped_config_loads_the_pinch_section(raw): loaded = RouterConfig(**raw) - assert loaded.pinch.enabled is False + assert loaded.pinch.enabled is True assert loaded.pinch.budget_tokens > 0 assert loaded.pinch.keep_last_turns > 0 @@ -246,7 +246,7 @@ def test_pinch_defaults_when_absent(raw): cfg = copy.deepcopy(raw) cfg.pop("pinch") loaded = RouterConfig(**cfg) - assert loaded.pinch.enabled is False + assert loaded.pinch.enabled is True assert loaded.pinch.budget_tokens == 50000 assert loaded.pinch.keep_last_turns == 4 @@ -294,7 +294,7 @@ def test_pinch_max_summarize_chars_accepts_default(raw): def test_pinch_relevance_defaults_load(raw): loaded = RouterConfig(**raw) - assert loaded.pinch.relevance.enabled is False + assert loaded.pinch.relevance.enabled is True assert loaded.pinch.relevance.model == "nomic-embed-text" assert loaded.pinch.relevance.min_candidates == 2 @@ -303,7 +303,7 @@ def test_pinch_relevance_defaults_when_pinch_absent(raw): cfg = copy.deepcopy(raw) cfg.pop("pinch") loaded = RouterConfig(**cfg) - assert loaded.pinch.relevance.enabled is False + assert loaded.pinch.relevance.enabled is True assert loaded.pinch.relevance.timeout_seconds == 10 @@ -323,7 +323,7 @@ def test_nonpositive_pinch_relevance_min_candidates_is_rejected(raw): def test_circuit_breaker_defaults_load(raw): loaded = RouterConfig(**raw) - assert loaded.circuit_breaker.enabled is False + assert loaded.circuit_breaker.enabled is True assert loaded.circuit_breaker.initial_cooldown_seconds == 30 assert loaded.circuit_breaker.max_cooldown_seconds == 600 assert loaded.circuit_breaker.backoff_multiplier == 2.0 diff --git a/tests/test_dispatcher_helpers.py b/tests/test_dispatcher_helpers.py index 5d14daf..ec2cc1f 100644 --- a/tests/test_dispatcher_helpers.py +++ b/tests/test_dispatcher_helpers.py @@ -120,7 +120,7 @@ def test_embed_for_relevance_request_exception_returns_none(monkeypatch): def test_relevance_order_for_returns_none_when_disabled(): cfg = _cfg_with_relevance(enabled=False) - assert dispatcher._relevance_order_for([], cfg) is None + assert dispatcher._relevance_order_for([], cfg, budget_tokens=100) is None def test_relevance_order_for_below_min_candidates_skips_embedding(monkeypatch): @@ -138,7 +138,27 @@ def test_relevance_order_for_below_min_candidates_skips_embedding(monkeypatch): {"role": "tool", "name": "read", "content": "x" * 6000}, {"role": "assistant", "content": "a"}, ] - assert dispatcher._relevance_order_for(messages, cfg) is None + assert dispatcher._relevance_order_for(messages, cfg, budget_tokens=100) is None + assert called["n"] == 0 + + +def test_relevance_order_for_skips_embedding_when_under_budget(monkeypatch): + called = {"n": 0} + + def fake_post(*a, **k): + called["n"] += 1 + return _FakeResp(status_code=200, json_payload={"data": None}) + + monkeypatch.setattr(dispatcher.requests, "post", fake_post) + cfg = _cfg_with_relevance(enabled=True, min_candidates=2) + # Two short tool-results -> token estimate under budget -> no embed call. + messages = [ + {"role": "user", "content": "q"}, + {"role": "tool", "name": "read", "content": "short"}, + {"role": "tool", "name": "read", "content": "also short"}, + {"role": "assistant", "content": "a"}, + ] + assert dispatcher._relevance_order_for(messages, cfg, budget_tokens=50000) is None assert called["n"] == 0 -- 2.49.1 From e289f5005b99cfa32c9e324ca567701d7ed606ac Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sun, 30 Aug 2026 10:56:18 -0400 Subject: [PATCH 3/4] docs: mark pinch on-by-default in AGENTS.md Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- AGENTS.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/AGENTS.md b/AGENTS.md index ac5c423..509c26a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -35,7 +35,7 @@ code. | `events.py` | In-memory decision-event broker | Pure stdlib (`queue`, `collections.deque`). Thread-safe. No Textual import. The SSE endpoint lives in `dispatcher.py` and calls `events.subscribe()`/`publish_decision()`. | | `config.py` | Pydantic models + YAML loader | All config models inherit `StrictModel` (`extra="forbid"`). Unknown keys fail at load. | | `capabilities.py` | Request-side capability detection (`tools`, `images`, `json_mode`, `reasoning`) | Reads from the OpenAI-format request body, not from the classifier. | -| `context_prune.py` | Relevance-based context pruning (`pinch`) | Ships **disabled** (`pinch.enabled: false`). Imports `PinchConfig` from `config.py` (no circular import: `config.py` doesn't import `context_prune`). | +| `context_prune.py` | Relevance-based context pruning (`pinch`) | Ships **enabled** by default (`pinch.enabled: true`; `pinch.relevance.enabled: true`; budget-gated: the embed only fires when the conversation exceeds `budget_tokens`). Imports `PinchConfig` from `config.py` (no circular import: `config.py` doesn't import `context_prune`). | | `logs.py` | Structured logging (logfmt, journald, ContextVar trace ids) | `logs.bind()` exists for StreamingResponse generators that lose the ContextVar. | ### Pure scoring/routing modules (no I/O) -- 2.49.1 From 6c8151f02b5d7ee81736ab0d1c5af01a353ffa39 Mon Sep 17 00:00:00 2001 From: adlee-was-taken Date: Sun, 30 Aug 2026 11:04:47 -0400 Subject: [PATCH 4/4] test: isolate session_cache in chat fixture; import subprocess/sys in health test Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus --- tests/test_admin_health.py | 2 ++ tests/test_chat_completions.py | 3 +++ 2 files changed, 5 insertions(+) diff --git a/tests/test_admin_health.py b/tests/test_admin_health.py index 96c6ee8..dce39bc 100644 --- a/tests/test_admin_health.py +++ b/tests/test_admin_health.py @@ -11,6 +11,8 @@ from __future__ import annotations import os import sqlite3 +import subprocess +import sys from datetime import datetime, timedelta, timezone from pathlib import Path diff --git a/tests/test_chat_completions.py b/tests/test_chat_completions.py index 235d92d..7608d47 100644 --- a/tests/test_chat_completions.py +++ b/tests/test_chat_completions.py @@ -104,6 +104,9 @@ def router(tmp_path, monkeypatch): # The local vision fallback is off unless a test opts in; without this it # would fire for every image request and try to reach localhost. monkeypatch.setattr(dispatcher.cfg.local_vision, "enabled", False) + # Isolate the shared module-level session cache: a real cache keyed on the + # same fingerprint leaks a classification from one test into the next. + monkeypatch.setattr(dispatcher.cfg.session_cache, "enabled", False) monkeypatch.setenv("NEURALWATT_API_KEY", "test-key") calls = [] -- 2.49.1