feat: detect context window size dynamically from model ID

Hardcoded 200K window caused 101% pressure at 201K tokens on 1M
models. Now detects model from request payload and sets window_size
accordingly (1M for opus-4-6/sonnet-4-6/sonnet-4-5, 200K for others).
Falls back to 200K for unknown models.
This commit is contained in:
Joey Yakimowich-Payne 2026-03-15 19:36:05 -06:00
commit 4ca9c58920

View file

@ -1137,6 +1137,8 @@ class Session:
self.page_store,
log_path=log_dir / f"violations_{sid}.jsonl" if log_dir else None,
)
# Default 200K; updated dynamically from actual model context window
# after the first API response via _update_window_size().
self.fidelity_manager = FidelityManager(window_size=200_000)
# Maps content keys → fidelity object IDs for lookup during apply
self._fidelity_content_map: dict[str, str] = {}
@ -1210,6 +1212,7 @@ class Session:
# Summary cache: (object_id, FidelityLevel) → summary text
# Avoids re-summarizing the same content on repeated _apply_fidelity calls
self._summary_cache: dict[tuple[str, int], str] = {}
self._window_size_detected: bool = False
def track_usage(self, usage: dict) -> None:
"""Update token state from API response usage."""
@ -2337,6 +2340,31 @@ def create_app(
file=sys.stderr,
)
# Model → context window size mapping. Used to set the FM's window_size
# dynamically after the first API response reveals which model is in use.
_MODEL_WINDOW_SIZES: dict[str, int] = {
# 1M context models
"claude-opus-4-6": 1_000_000,
"claude-sonnet-4-6": 1_000_000,
"claude-sonnet-4-5": 1_000_000, # 1M in beta
# 200K context models
"claude-opus-4-5": 200_000,
"claude-sonnet-4": 200_000,
"claude-haiku-4-5": 200_000,
"claude-haiku-3-5": 200_000,
}
def _detect_window_size(model: str) -> int | None:
"""Return context window size for a model, or None if unknown."""
# Try exact match first, then prefix match (strips date suffix)
if model in _MODEL_WINDOW_SIZES:
return _MODEL_WINDOW_SIZES[model]
# Strip date suffix: "claude-sonnet-4-6-20260217" → "claude-sonnet-4-6"
for prefix, size in _MODEL_WINDOW_SIZES.items():
if model.startswith(prefix):
return size
return None
def _update_fidelity_pressure(usage: dict, session: Session) -> None:
"""Update FidelityManager window understanding from actual API usage.
@ -2351,6 +2379,19 @@ def create_app(
fm = session.fidelity_manager
turn = session.token_state.get("turn", 0)
# Detect window size from model on first response
if not getattr(session, "_window_size_detected", False):
model = session.token_state.get("model", "")
detected = _detect_window_size(model)
if detected and detected != fm.window_size:
fm.window_size = detected
print(
f" {_DIM}[{session.id}] window_size set to {detected:,} "
f"for model {model}{_RESET}",
file=sys.stderr,
)
session._window_size_detected = True
pressure_ratio = input_tokens / fm.window_size if fm.window_size > 0 else 1.0
if pressure_ratio >= fm.threshold_caution:
@ -2527,6 +2568,8 @@ def create_app(
req = adapter.normalize_request(payload)
req = pipeline.run(req)
# Store model for dynamic window_size detection
session.token_state["model"] = req.model
duplication_score = _duplication_score(req)
outgoing_body = adapter.denormalize_request(req)
if endpoint == "count_tokens":