Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions docs/system-specs/modules/memory-skills-hooks.md
Original file line number Diff line number Diff line change
Expand Up @@ -4724,6 +4724,12 @@ and Kiro Crew resets its context-usage accounting at that chokepoint. Separately
consecutive turn FAILURES for a session key and resets the session; that counter
tracks failures, not compactions.

On the dashboard, confirmed provider-native and manual `/compact` completion
arms `SessionManager.mark_needs_reinjection` for the effective session key.
The next dashboard turn consumes that one-shot flag to restore the skills
context. Failed deferred compaction does not arm it. This completion hook does
not add skills reinjection to messaging surfaces or the task runner.

#### Dynamic budget scaling (per active model context window)

The `_CONTEXT_BUDGET_BASE` (165k) and its derived per-section caps above are the **1M-reference** values — the base was hand-tuned for a 1M-token window, so each section has a fixed *share of that window*. When a session runs on a **smaller-window** model (e.g. Opus 4.8 200K), injecting the same absolute char counts would consume ~5× the proportional share and accelerate compaction. `build_session_context()` / `build_message()` / `compress_thread_history()` / `build_session_replay()` therefore take an optional `model_window` (tokens); `_resolve_caps(window)` re-derives every cap against a base scaled linearly to that window (`base = _CONTEXT_BUDGET_BASE × window / _REFERENCE_WINDOW_TOKENS`, `_REFERENCE_WINDOW_TOKENS`=1,000,000). This keeps each section's **share of the window invariant across models** — a section that is 20% of a 1M window stays 20% of a 200K window (i.e. one-fifth the chars). Results are `functools.lru_cache`d per distinct window; `_ResolvedCaps.max_context` is a computed property, and the module constant `_MAX_CONTEXT_CHARS` is *derived* from `_resolve_caps(_REFERENCE_WINDOW_TOKENS)` so the section-sum lives in one place.
Expand Down
13 changes: 13 additions & 0 deletions src/kiro_crew/dashboard/chat_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -6730,6 +6730,16 @@ def _stop_pressed() -> bool:
#: message, which can carry a path or a credential.
_ledger_error = ""

def _restore_skills_context_after_compaction() -> None:
"""Re-inject the session-start skills context on the next turn."""
try:
state.sessions.mark_needs_reinjection(session_key)
except Exception:
logger.warning(
"post-compaction skills context reinjection could not be armed",
exc_info=True,
)

# Time-to-first-token clock: starts when the user's message reaches the
# runner, stops at the first visible model output (text OR thinking chunk).
# This is the end-to-end latency eager spawn / warm pooling exist to cut —
Expand Down Expand Up @@ -11156,6 +11166,7 @@ def _require_current_binding() -> None:
saw_compaction = True
if event.text == "completed":
_compaction_completed = True
_restore_skills_context_after_compaction()
_produced_visible_output = True
if not event.synthesized:
# A REAL mid-turn terminal IS a segment boundary: text
Expand Down Expand Up @@ -11993,6 +12004,7 @@ def _emit_error(msg: str, *, will_retry: bool = False) -> None:
# backend that reports asynchronously loses the notice, and awaiting one
# that already finished strands the waiter for its whole timeout.
if capabilities_of(client).compacts_inline:
_restore_skills_context_after_compaction()
msg = "✅ Conversation compacted."
_append_compaction_notice(state, slot, msg)
state.broadcast_context_usage(slot.key, _context_usage_payload(slot.key, client))
Expand All @@ -12008,6 +12020,7 @@ def _emit_error(msg: str, *, will_retry: bool = False) -> None:
compaction_result = await client.wait_for_compaction()
logger.info("Deferred compaction result: %s", compaction_result)
if compaction_result["type"] == "completed":
_restore_skills_context_after_compaction()
summary, _ = redact_credentials(compaction_result.get("summary", ""))
summary, _ = redact_exfiltration_urls(summary)
msg = (
Expand Down
25 changes: 25 additions & 0 deletions test/test_dashboard_chat.py
Original file line number Diff line number Diff line change
Expand Up @@ -6071,6 +6071,7 @@ async def test_claude_backend_skips_wait_for_compaction(self, tmp_path, monkeypa
await _run_chat(state, slot, "/compact")

client.wait_for_compaction.assert_not_called()
state.sessions.mark_needs_reinjection.assert_called_once_with("dashboard:s1")
assistant_msgs = [m for m in slot.messages if m.get("role") == "assistant"]
assert any("Conversation compacted" in m["content"] for m in assistant_msgs)
assert not any("timed out" in m["content"] for m in assistant_msgs)
Expand Down Expand Up @@ -6121,6 +6122,7 @@ async def test_kiro_backend_still_waits_for_compaction(self, tmp_path, monkeypat
await _run_chat(state, slot, "/compact")

client.wait_for_compaction.assert_awaited_once()
state.sessions.mark_needs_reinjection.assert_called_once_with("dashboard:s1")
assistant_msgs = [m for m in slot.messages if m.get("role") == "assistant"]
assert any("summary text" in m["content"] for m in assistant_msgs)
# A completed deferred compaction must send the `reset` form — the
Expand All @@ -6137,6 +6139,27 @@ async def test_kiro_backend_still_waits_for_compaction(self, tmp_path, monkeypat
assert compaction_msgs
assert all(m.get("meta", {}).get("kind") == "compaction" for m in compaction_msgs)

@pytest.mark.asyncio
async def test_completed_status_arms_skills_context_reinjection(self, tmp_path, monkeypatch):
"""A provider-native completed status restores skills context once."""
from kiro_crew.providers.base import EVENT_COMPACTION_STATUS, EVENT_COMPLETE, LLMEvent

state = self._make_state_for_run_chat(tmp_path, monkeypatch)
slot = state.get_or_create_slot("s1")
client = self._make_mock_client(
[
LLMEvent(kind=EVENT_COMPACTION_STATUS, text="completed"),
LLMEvent(kind=EVENT_COMPLETE),
]
)
state.sessions.get_or_create = AsyncMock(return_value=(client, True, False))

from kiro_crew.dashboard.chat import _run_chat

await _run_chat(state, slot, "continue after provider compaction")

state.sessions.mark_needs_reinjection.assert_called_once_with("dashboard:s1")

@pytest.mark.asyncio
async def test_kiro_backend_broadcasts_real_post_compaction_usage(self, tmp_path, monkeypatch):
"""When the wait_for_compaction grace drain captured kiro's fresh
Expand All @@ -6163,6 +6186,7 @@ async def test_kiro_backend_broadcasts_real_post_compaction_usage(self, tmp_path

await _run_chat(state, slot, "/compact")

state.sessions.mark_needs_reinjection.assert_called_once_with("dashboard:s1")
usage_calls = [
c for c in state.broadcast_ws.call_args_list if c.args and c.args[0] == "context_usage"
]
Expand Down Expand Up @@ -6200,6 +6224,7 @@ async def test_kiro_backend_failed_compaction_keeps_meter(self, tmp_path, monkey

await _run_chat(state, slot, "/compact")

state.sessions.mark_needs_reinjection.assert_not_called()
usage_calls = [
c for c in state.broadcast_ws.call_args_list if c.args and c.args[0] == "context_usage"
]
Expand Down
Loading