Implemented the next parity slice: prompt-budget preflight and context collapse.
Core changes:
- Added claw-code/src/token_budget.py for projected prompt size, chat-framing overhead, output reserve, and soft/hard input limits.
- Wired preflight prompt-length validation and auto-compact/context collapse into claw-code/src/agent_runtime.py.
- Extended claw-code/src/compact.py so compaction reports usage back to the runtime.
- Added inspection surfaces in claw-code/src/agent_slash_commands.py and claw-code/src/main.py:
- /token-budget and /budget
- token-budget
- Hardened claw-code/src/tokenizer_runtime.py so arbitrary simple model names fall back cleanly instead of trying a slow Transformers
lookup.
- Exported the new helpers in claw-code/src/__init__.py.
Docs and tracking:
- Updated claw-code/PARITY_CHECKLIST.md to mark prompt-length validation, token-budget calculation, and auto-compact/context collapse as
done.
- Updated claw-code/README.md and claw-code/TESTING_GUIDE.md with the new commands and behavior.
Tests:
- Added claw-code/tests/test_token_budget.py.
- Updated claw-code/tests/test_agent_runtime.py, claw-code/tests/test_agent_slash_commands.py, claw-code/tests/test_main.py, and claw-code/
tests/test_agent_context_usage.py.
- Verified with:
- /data/fs201059/aa17626/miniconda3/bin/python3 -m compileall src tests
- /data/fs201059/aa17626/miniconda3/bin/python3 -m unittest -v tests.test_token_budget
tests.test_agent_runtime.AgentRuntimeTests.test_agent_rejects_prompt_before_backend_when_preflight_input_budget_is_exceeded
tests.test_agent_runtime.AgentRuntimeTests.test_agent_auto_compacts_context_before_next_model_call tests.test_agent_slash_commands
tests.test_main tests.test_compact tests.test_tokenizer_runtime tests.test_agent_context_usage
- Result: 71 tests, OK
This commit is contained in:
@@ -0,0 +1,100 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from src.agent_session import AgentSessionState
|
||||
from src.agent_context_usage import ContextUsageReport, MessageBreakdown
|
||||
from src.agent_types import BudgetConfig
|
||||
from src.token_budget import calculate_token_budget, format_token_budget
|
||||
|
||||
|
||||
class TokenBudgetTests(unittest.TestCase):
|
||||
def test_calculate_token_budget_reports_soft_and_hard_limits(self) -> None:
|
||||
session = AgentSessionState.create(
|
||||
['# System\nYou are helpful.'],
|
||||
'Inspect the repository and summarize the current implementation status.',
|
||||
user_context={'currentDate': "Today's date is 2026-04-11."},
|
||||
system_context={'gitStatus': 'Current branch: main'},
|
||||
)
|
||||
session.append_assistant('Reading files and checking runtime state.')
|
||||
|
||||
fake_usage = ContextUsageReport(
|
||||
model='test-model',
|
||||
total_tokens=200,
|
||||
raw_max_tokens=128_000,
|
||||
percentage=0.15,
|
||||
strategy='token_budget',
|
||||
message_count=len(session.messages),
|
||||
categories=(),
|
||||
system_prompt_sections=(),
|
||||
user_context_entries=(),
|
||||
system_context_entries=(),
|
||||
memory_files=(),
|
||||
message_breakdown=MessageBreakdown(
|
||||
user_message_tokens=50,
|
||||
assistant_message_tokens=50,
|
||||
tool_call_tokens=0,
|
||||
tool_result_tokens=0,
|
||||
user_context_tokens=10,
|
||||
tool_calls_by_type=(),
|
||||
),
|
||||
token_counter_backend='heuristic',
|
||||
token_counter_source='test',
|
||||
token_counter_accurate=False,
|
||||
)
|
||||
with patch('src.token_budget.collect_context_usage', return_value=fake_usage):
|
||||
snapshot = calculate_token_budget(
|
||||
session=session,
|
||||
model='test-model',
|
||||
budget_config=BudgetConfig(),
|
||||
)
|
||||
rendered = format_token_budget(snapshot)
|
||||
|
||||
self.assertGreater(snapshot.projected_input_tokens, 0)
|
||||
self.assertGreater(snapshot.hard_input_limit_tokens, snapshot.soft_input_limit_tokens)
|
||||
self.assertGreater(snapshot.chat_overhead_tokens, 0)
|
||||
self.assertIn('# Token Budget', rendered)
|
||||
self.assertIn('Hard input limit', rendered)
|
||||
self.assertIn('Auto-compact buffer', rendered)
|
||||
|
||||
def test_calculate_token_budget_honors_explicit_max_input_tokens(self) -> None:
|
||||
session = AgentSessionState.create(
|
||||
['# System\nYou are helpful.'],
|
||||
'This prompt is deliberately longer than the tiny configured input budget. ' * 4,
|
||||
)
|
||||
|
||||
fake_usage = ContextUsageReport(
|
||||
model='test-model',
|
||||
total_tokens=120,
|
||||
raw_max_tokens=128_000,
|
||||
percentage=0.09,
|
||||
strategy='token_budget',
|
||||
message_count=len(session.messages),
|
||||
categories=(),
|
||||
system_prompt_sections=(),
|
||||
user_context_entries=(),
|
||||
system_context_entries=(),
|
||||
memory_files=(),
|
||||
message_breakdown=MessageBreakdown(
|
||||
user_message_tokens=80,
|
||||
assistant_message_tokens=0,
|
||||
tool_call_tokens=0,
|
||||
tool_result_tokens=0,
|
||||
user_context_tokens=0,
|
||||
tool_calls_by_type=(),
|
||||
),
|
||||
token_counter_backend='heuristic',
|
||||
token_counter_source='test',
|
||||
token_counter_accurate=False,
|
||||
)
|
||||
with patch('src.token_budget.collect_context_usage', return_value=fake_usage):
|
||||
snapshot = calculate_token_budget(
|
||||
session=session,
|
||||
model='test-model',
|
||||
budget_config=BudgetConfig(max_input_tokens=20),
|
||||
)
|
||||
|
||||
self.assertTrue(snapshot.exceeds_hard_limit)
|
||||
self.assertGreater(snapshot.overflow_tokens, 0)
|
||||
self.assertLessEqual(snapshot.hard_input_limit_tokens, 20)
|
||||
Reference in New Issue
Block a user