From 65e4e38a98ffcab0b8c6c75f2294e11f81d59365 Mon Sep 17 00:00:00 2001 From: Joey Yakimowich-Payne Date: Fri, 13 Mar 2026 21:13:38 -0600 Subject: [PATCH] fix: scale FM window_size to match real API pressure for fidelity degradation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The FidelityManager's internal pressure calculation uses its own tracked object tokens divided by window_size, which is always tiny compared to the real context. Temporarily scale window_size so the FM's pressure matches the actual API input_tokens/window ratio, triggering L0→L1→L2 degradations when context exceeds 50%. --- src/mnemosyne/gateway.py | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/src/mnemosyne/gateway.py b/src/mnemosyne/gateway.py index 89fa156..e8057de 100644 --- a/src/mnemosyne/gateway.py +++ b/src/mnemosyne/gateway.py @@ -2122,8 +2122,8 @@ def create_app( """Update FidelityManager window understanding from actual API usage. After receiving the response, we know the real input token count. - Update the fidelity manager's window_size understanding and schedule - degradation if pressure is above NORMAL for the next turn. + Scale the FM's window_size so its internal pressure calculation + reflects the real API token usage, then trigger degradation. """ input_tokens = usage.get("input_tokens", 0) if input_tokens <= 0: @@ -2132,14 +2132,22 @@ def create_app( fm = session.fidelity_manager turn = session.token_state.get("turn", 0) - # The FidelityManager tracks its own token budget via registered objects. - # Here we use the real API token count to check if we need proactive degradation. - # If real usage exceeds the fidelity window threshold, trigger degradation now - # so the NEXT turn benefits from reduced content. pressure_ratio = input_tokens / fm.window_size if fm.window_size > 0 else 1.0 if pressure_ratio >= fm.threshold_caution: - transitions = fm.degrade(turn) + # The FM's internal pressure uses total_tokens()/window_size, but + # total_tokens() only counts registered objects (a fraction of the + # real context). Temporarily scale window_size down so the FM's + # pressure matches the real API pressure, then restore it. + obj_tokens = fm.total_tokens() + if obj_tokens > 0: + # Set window_size so obj_tokens/window_size == pressure_ratio + saved_ws = fm.window_size + fm.window_size = max(1, int(obj_tokens / pressure_ratio)) + transitions = fm.degrade(turn) + fm.window_size = saved_ws + else: + transitions = fm.degrade(turn) if transitions: zone = fm.current_pressure() for _obj_id, old_level, new_level in transitions: