Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
456 changes: 456 additions & 0 deletions opensquilla-webui/e2e/compaction-live.spec.ts

Large diffs are not rendered by default.

272 changes: 272 additions & 0 deletions opensquilla-webui/e2e/compaction-pressure-live.spec.ts

Large diffs are not rendered by default.

302 changes: 302 additions & 0 deletions scripts/live_compaction_comparison.py

Large diffs are not rendered by default.

405 changes: 405 additions & 0 deletions scripts/live_compaction_gateway.py

Large diffs are not rendered by default.

756 changes: 671 additions & 85 deletions scripts/live_reasoning_replay_e2e.py

Large diffs are not rendered by default.

25 changes: 10 additions & 15 deletions src/opensquilla/context_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,6 @@
CHARS_PER_TOKEN = 4
LARGE_CONTEXT_MIN_TOKENS = 64_000
CONTEXT_RESERVE_FLOOR_TOKENS = 20_000
SMALL_CONTEXT_MIN_PROOF_CHARS = 4_000
SMALL_CONTEXT_MAX_PROOF_CHARS = 32_000
LARGE_CONTEXT_MIN_ARGUMENT_CHARS = 64_000
LARGE_CONTEXT_MAX_ARGUMENT_CHARS = 512_000
LARGE_CONTEXT_MIN_RESULT_CHARS = 128_000
Expand Down Expand Up @@ -135,26 +133,23 @@ def from_values(
thinking_budget = max(0, int(thinking_budget_tokens or 0))
threshold = _threshold(context_overflow_threshold)

max_reserve = max(1, context_tokens // 2)
output_reserve = min(max_output + thinking_budget, max_reserve)
# Callers without an adapter projection conservatively reserve both
# configured budgets. Final request admission passes the adapter's
# complete generation cap as max_output_tokens and zero thinking.
output_reserve = max_output + thinking_budget
context_reserve = (
CONTEXT_RESERVE_FLOOR_TOKENS
if context_tokens >= LARGE_CONTEXT_MIN_TOKENS
else max(512, context_tokens // 8)
)
reserved_tokens = min(
max(context_tokens - 1, 1),
output_reserve + context_reserve,
)
usable_tokens = max(1, context_tokens - reserved_tokens)
reserved_tokens = min(context_tokens, output_reserve + context_reserve)
usable_tokens = max(0, context_tokens - reserved_tokens)

explicit_proof = _positive_int(provider_request_proof_max_chars)
derived_provider_chars = int(usable_tokens * threshold * CHARS_PER_TOKEN)
if context_tokens < LARGE_CONTEXT_MIN_TOKENS:
derived_provider_chars = min(
SMALL_CONTEXT_MAX_PROOF_CHARS,
max(SMALL_CONTEXT_MIN_PROOF_CHARS, derived_provider_chars),
)
# The threshold is a soft compaction trigger, not another reduction
# of the provider's hard input budget. Character limits remain an
# independent guard, including when an operator supplies one.
derived_provider_chars = usable_tokens * CHARS_PER_TOKEN
provider_chars = explicit_proof or max(1, derived_provider_chars)

explicit_argument = _positive_int(tool_use_argument_provider_request_max_chars)
Expand Down
172 changes: 104 additions & 68 deletions src/opensquilla/engine/agent.py

Large diffs are not rendered by default.

14 changes: 7 additions & 7 deletions src/opensquilla/engine/capacity_admission.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,11 +2,12 @@

from __future__ import annotations

from opensquilla.context_budget import CHARS_PER_TOKEN, ContextBudgetGovernor
from opensquilla.context_budget import ContextBudgetGovernor
from opensquilla.provider.model_catalog import (
resolve_effective_context_window,
shared_catalog,
)
from opensquilla.provider.request_proof import effective_proof_token_budget

NON_MATERIAL_INPUT_HEADROOM_TOKENS = 8_192
MAX_THINKING_BUDGET_TOKENS = 50_000
Expand Down Expand Up @@ -91,13 +92,12 @@ def model_has_request_capacity(
max_output_tokens=max_output,
thinking_budget_tokens=max(0, int(thinking_budget_tokens)),
context_overflow_threshold=0.85,
provider_request_proof_max_chars=provider_request_proof_max_chars,
).snapshot()
safe_input_tokens = budget.provider_request_max_chars // CHARS_PER_TOKEN
if provider_request_proof_max_chars > 0:
safe_input_tokens = min(
safe_input_tokens,
int(provider_request_proof_max_chars) // CHARS_PER_TOKEN,
)
safe_input_tokens, _headroom = effective_proof_token_budget(budget.usable_tokens)
# This prefilter receives token estimates, not the serialized request.
# Its character cap is enforced separately by the final adapter proof;
# converting that cap to tokens would conflate two independent limits.
required_input_tokens = (
resolved_request_tokens
if resolved_request_tokens > 0
Expand Down
2 changes: 2 additions & 0 deletions src/opensquilla/engine/context_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,7 @@ def coordinate_provider_context_budget(
*,
projection_adapter: str,
proof_budget: int,
token_budget: int | None = None,
status_projection_mode: str = "native_or_none",
fallback_reason: str | None = None,
envelope_shape: ProviderRequestEnvelopeShape = CHAT_REQUEST_ENVELOPE,
Expand All @@ -50,6 +51,7 @@ def coordinate_provider_context_budget(
payload,
projection_adapter=projection_adapter,
proof_budget=proof_budget,
token_budget=token_budget,
status_projection_mode=status_projection_mode,
fallback_reason=fallback_reason,
envelope_shape=envelope_shape,
Expand Down
88 changes: 80 additions & 8 deletions src/opensquilla/engine/runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -284,6 +284,7 @@
count_provider_image_blocks,
project_provider_final_request,
project_provider_message_count,
provider_connection_config,
provider_metadata,
validate_provider_chat_admission,
)
Expand Down Expand Up @@ -2953,6 +2954,7 @@ async def prepare_image_continuation(
return None
identity = _fallback_deployment_identity(deployment)
window = max(1, int(catalog.context_window))
physical_window = window if getattr(catalog, "context_window_known", True) else 0
output_limit = min(max(1, int(config.max_tokens)), max(1, int(catalog.max_tokens)))
proof = ContextBudgetGovernor.from_values(
context_window_tokens=window,
Expand All @@ -2968,11 +2970,12 @@ async def prepare_image_continuation(
"model_capabilities": catalog.capabilities,
"model_vision_support": catalog.vision_support,
"provider_request_max_chars": proof,
"provider_context_window_tokens": physical_window,
"provider_request_max_chars_explicit_cap": explicit_proof,
})
self._selector = candidate_selector
self._provider = candidate_provider
self._fallback_deployment_limits[identity] = (window, output_limit)
self._fallback_deployment_limits[identity] = (physical_window, output_limit)
if catalog.capabilities is not None:
self._fallback_deployment_capabilities[identity] = catalog.capabilities
self._fallback_deployment_vision_support[identity] = catalog.vision_support
Expand Down Expand Up @@ -3167,6 +3170,7 @@ def _config_for_active_leg(self, config: Any) -> Any:
)

context_window, effective_max_tokens = self._active_fallback_limits()
updates["provider_context_window_tokens"] = context_window
try:
original_max_tokens = max(0, int(getattr(config, "max_tokens", 0) or 0))
except (TypeError, ValueError):
Expand Down Expand Up @@ -6986,7 +6990,9 @@ async def _load_turn_transcript() -> Sequence[Any]:
previous_deployment_identities=previous_deployment_identities,
fallback_provider_configs=selector_remaining_chain[1:],
compaction_config=configured_compaction,
context_window_tokens=compaction_context_window_tokens,
context_window_tokens=(
compaction_context_window_tokens if agent.config.context_window_known else 0
),
session_key=session_key,
credential_pool_acquirer=acquire_profile_credential,
credential_pool_failure_reporter=report_profile_credential_failure,
Expand All @@ -7013,8 +7019,35 @@ def _refresh_compaction_plan_for_operation() -> Any | None:
self._model_catalog,
fresh_model,
provider=fresh_provider,
global_override=(getattr(llm_cfg, "context_window_tokens", 0) or 0),
global_override=(
(getattr(llm_cfg, "context_window_tokens", 0) or 0)
if str(getattr(llm_cfg, "provider", "")).strip().lower()
== fresh_provider.strip().lower()
else 0
),
)
deployment_limits = getattr(
self._model_catalog, "resolve_deployment_limits", None,
)
if (
_fresh_window_source not in {"override", "config"}
and callable(deployment_limits)
):
limits = deployment_limits(
fresh_model,
provider=fresh_provider,
api_key=str(getattr(fresh_current, "api_key", "") or ""),
base_url=str(getattr(fresh_current, "base_url", "") or ""),
proxy=str(getattr(fresh_current, "proxy", "") or ""),
)
fresh_window = limits.context_window
_fresh_window_source = (
"catalog"
if getattr(limits, "context_window_known", True)
else "default"
)
if _fresh_window_source == "default":
fresh_window = 0
return resolve_compaction_execution_plan(
app_config=self._turn_config(),
active_provider=provider,
Expand All @@ -7030,6 +7063,7 @@ def _refresh_compaction_plan_for_operation() -> Any | None:
)

stable_consumer_window_tokens = compaction_context_window_tokens
stable_consumer_window_known = agent.config.context_window_known
stable_consumer_max_output_tokens = agent.config.max_tokens
stable_consumer_model_id = agent.config.model_id
stable_consumer_capabilities = agent.config.model_capabilities
Expand All @@ -7053,6 +7087,11 @@ def _refresh_compaction_plan_for_operation() -> Any | None:
if configured_llm_provider == base_provider.lower()
else 0
)
base_output_override = (
_non_negative_int(getattr(llm_cfg, "max_tokens", 0))
if configured_llm_provider == base_provider.lower()
else 0
)
stable_consumer_window_tokens, _stable_window_source = (
resolve_effective_context_window(
self._model_catalog,
Expand All @@ -7061,14 +7100,38 @@ def _refresh_compaction_plan_for_operation() -> Any | None:
global_override=base_global_window,
)
)
stable_consumer_window_known = _stable_window_source != "default"
stable_consumer_max_output_tokens = int(
self._model_catalog.resolve_max_tokens(
base_model,
user_override=0,
user_override=base_output_override,
provider=base_provider,
)
or agent.config.max_tokens
)
resolve_limits = getattr(
self._model_catalog, "resolve_deployment_limits", None,
)
if callable(resolve_limits):
connection = provider_connection_config(durable_base_consumer_provider)
limits = resolve_limits(
base_model, provider=base_provider,
api_key=connection.api_key, base_url=connection.base_url,
logical_max_tokens_override=base_output_override,
)
stable_consumer_max_output_tokens = limits.max_output_tokens
if _stable_window_source not in {"override", "config"}:
stable_consumer_window_tokens = limits.context_window
stable_consumer_window_known = bool(
getattr(limits, "context_window_known", True)
)
if base_output_override > 0:
# Match bootstrap's physical request configuration.
# Catalog output ceilings describe automatic defaults,
# not the output reserved by an explicit base config.
stable_consumer_max_output_tokens = min(
base_output_override, stable_consumer_window_tokens,
)
stable_consumer_model_id = base_model
stable_consumer_capabilities = self._model_catalog.get_capabilities(
base_model,
Expand Down Expand Up @@ -7109,6 +7172,7 @@ def _refresh_compaction_plan_for_operation() -> Any | None:
provider=durable_base_consumer_provider,
model_id=stable_consumer_model_id,
context_window_tokens=stable_consumer_window_tokens,
context_window_known=stable_consumer_window_known,
max_output_tokens=stable_consumer_max_output_tokens,
model_capabilities=stable_consumer_capabilities,
provider_request_proof_max_chars=(stable_consumer_proof_max_chars),
Expand Down Expand Up @@ -12332,14 +12396,15 @@ async def _maybe_compact_on_t3_upgrade(
transcript[:durable_prefix_end]
)
safety_margin = float(
getattr(compaction_config or CompactionConfig(), "safety_margin", 1.2) or 1.2
getattr(compaction_config or CompactionConfig(), "safety_margin", 1 / 0.85) or 1 / 0.85
)
trigger_ratio = self._preflight_compact_ratio()
durable_tokens_within_budget = bool(
durable_history_tokens * safety_margin <= history_window_tokens
durable_history_tokens < history_window_tokens * trigger_ratio
)
durable_chars_within_budget = bool(
history_capacity_chars is None
or durable_history_chars * safety_margin <= int(history_capacity_chars)
or durable_history_chars < int(history_capacity_chars) * trigger_ratio
)
if durable_tokens_within_budget and durable_chars_within_budget:
log.info(
Expand Down Expand Up @@ -12978,7 +13043,7 @@ async def _maybe_preflight_compact(
if active_user_index is not None
else 0
)
safety_margin = float(getattr(compaction_config, "safety_margin", 1.2) or 1.2)
safety_margin = float(getattr(compaction_config, "safety_margin", 1 / 0.85) or 1 / 0.85)
if (
protected_request_tokens > 0
and protected_request_tokens * safety_margin > history_window_tokens
Expand Down Expand Up @@ -13068,6 +13133,13 @@ async def _maybe_preflight_compact(
status="started",
tokens_before=total_tokens,
context_window_tokens=context_window_tokens,
history_capacity_tokens=history_window_tokens,
history_capacity_chars=history_capacity_chars,
durable_history_tokens=durable_history_tokens,
durable_history_chars=durable_history_chars,
threshold=threshold,
char_threshold=char_threshold,
ratio=ratio,
heartbeat_interval_seconds=compaction_config.heartbeat_interval_seconds,
**compaction_effect_payload(status="started"),
**compaction_lifecycle_payload(compaction_id, COMPACTION_TRIGGERED_EVENT),
Expand Down
30 changes: 28 additions & 2 deletions src/opensquilla/engine/subagent.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@ class SubagentExecutionTarget:
model_capabilities: Any = field(default=None, repr=False, compare=False)
compaction_plan: Any = field(default=None, repr=False, compare=False)
model_vision_support: Literal["supported", "unsupported", "unknown"] = "unknown"
context_window_known: bool = True


@dataclass
Expand Down Expand Up @@ -130,6 +131,7 @@ def resolve_subagent_execution_target(
provider_connection_config,
provider_metadata,
)
from opensquilla.provider.registry import LOCAL_RUNTIME_PROVIDERS
from opensquilla.session.compaction_deployment import (
build_compaction_execution_plan_from_provider,
)
Expand Down Expand Up @@ -232,6 +234,7 @@ def resolve_subagent_execution_target(
)
if deployment_matches_parent_config:
context_window = max(1, int(getattr(parent_config, "context_window_tokens", 0) or 0))
context_window_known = bool(getattr(parent_config, "context_window_known", True))
max_output = max(1, int(getattr(parent_config, "max_tokens", 0) or 0))
max_output = min(max_output, context_window)
capabilities = getattr(parent_config, "model_capabilities", None)
Expand All @@ -243,11 +246,33 @@ def resolve_subagent_execution_target(
catalog = shared_catalog()
entry = catalog.resolve_entry(model_id, provider=provider_id)
context_window = max(1, int(entry.context_window or 0))
resolve_with_source = getattr(catalog, "resolve_context_window_with_source", None)
context_window_known = True
if callable(resolve_with_source):
_resolved_window, context_window_source = resolve_with_source(
model_id, provider=provider_id,
)
context_window_known = (
context_window_source != "default" or provider_id in LOCAL_RUNTIME_PROVIDERS
)
deployment_output: int | None = None
resolve_limits = getattr(catalog, "resolve_deployment_limits", None)
if callable(resolve_limits):
connection = provider_connection_config(child_provider)
limits = resolve_limits(
model_id, provider=provider_id, api_key=connection.api_key,
base_url=connection.base_url,
)
context_window_known = bool(getattr(limits, "context_window_known", True))
if context_window_known:
context_window = max(1, int(limits.context_window))
deployment_output = int(limits.max_output_tokens)
max_output = max(
1,
min(
int(
catalog.resolve_max_tokens(
deployment_output if deployment_output is not None
else catalog.resolve_max_tokens(
model_id,
user_override=0,
provider=provider_id,
Expand Down Expand Up @@ -289,7 +314,7 @@ def resolve_subagent_execution_target(
else build_compaction_execution_plan_from_provider(
child_provider,
model=model_id or None,
context_window_tokens=context_window,
context_window_tokens=context_window if context_window_known else 0,
provider_request_max_chars=request_max_chars,
source="subagent_deployment",
)
Expand All @@ -309,6 +334,7 @@ def resolve_subagent_execution_target(
model_capabilities=capabilities,
compaction_plan=compaction_plan,
model_vision_support=model_vision_support,
context_window_known=context_window_known,
)


Expand Down
Loading
Loading