mirror of
https://github.com/usestrix/strix.git
synced 2026-10-01 02:03:55 +00:00
Merge 0e118a08e8 into ae38fe70cd
This commit is contained in:
commit
339d03545a
12 changed files with 512 additions and 8 deletions
|
|
@ -42,7 +42,26 @@ Configure Strix using environment variables or a config file.
|
|||
</ParamField>
|
||||
|
||||
<ParamField path="STRIX_REASONING_EFFORT" default="high" type="string">
|
||||
Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Defaults to `medium` for quick scan mode.
|
||||
Control thinking effort for reasoning models. Valid values: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Higher = more thinking tokens = higher cost and (usually) deeper analysis. The `--reasoning-effort` CLI flag overrides this per run.
|
||||
</ParamField>
|
||||
|
||||
### Cost / fan-out limits
|
||||
|
||||
Every spawned agent re-pays the full system prompt on each of its turns and (by
|
||||
default) inherits a full copy of its parent's history, so unbounded agent fan-out
|
||||
is a large token-cost driver on a single target. These knobs let operators bound
|
||||
it. **All default to `0` (disabled) — out-of-the-box behavior is unchanged.**
|
||||
|
||||
<ParamField path="STRIX_MAX_AGENTS" default="0" type="integer">
|
||||
Maximum total agents in the graph (root included). When reached, `create_agent` refuses to spawn and tells the agent to reuse an existing agent, wait for running ones, or do the work itself. `0` = unlimited.
|
||||
</ParamField>
|
||||
|
||||
<ParamField path="STRIX_MAX_AGENT_DEPTH" default="0" type="integer">
|
||||
Maximum spawn depth. The root is depth 1; a child of the root is depth 2. `0` = unlimited.
|
||||
</ParamField>
|
||||
|
||||
<ParamField path="STRIX_INHERIT_CONTEXT_MAX_TOKENS" default="0" type="integer">
|
||||
Token cap on the parent history copied into a child spawned with `inherit_context=true`. The most-recent tail is kept; older turns are dropped with a marker. `0` = copy the full parent history.
|
||||
</ParamField>
|
||||
|
||||
<ParamField path="STRIX_MEMORY_COMPRESSOR_TIMEOUT" default="30" type="integer">
|
||||
|
|
|
|||
|
|
@ -48,6 +48,14 @@ strix (--target <target> | --target-list <path>) [options]
|
|||
Scan depth: `quick`, `standard`, or `deep`.
|
||||
</ParamField>
|
||||
|
||||
<ParamField path="--reasoning-effort" type="string">
|
||||
Model reasoning effort for this run: `none`, `minimal`, `low`, `medium`,
|
||||
`high`, `xhigh`, or `max`. Higher = more thinking tokens = higher cost and
|
||||
(usually) deeper analysis. Overrides `STRIX_REASONING_EFFORT` and the config
|
||||
file for this run. Defaults to the configured value (`high`). Not written back
|
||||
to the config file, so it never changes what later runs do.
|
||||
</ParamField>
|
||||
|
||||
<ParamField path="--scope-mode" type="string" default="auto">
|
||||
Code scope mode: `auto` (enable PR diff-scope in CI/headless runs), `diff` (force changed-files scope), or `full` (disable diff-scope).
|
||||
</ParamField>
|
||||
|
|
|
|||
|
|
@ -9,11 +9,13 @@ Public surface:
|
|||
- :func:`load_settings` — memoized resolve (env > JSON file > defaults).
|
||||
- :func:`apply_config_override` — switch the JSON source to a custom path.
|
||||
- :func:`persist_current` — write currently-set env vars to the active file.
|
||||
- :func:`mark_run_scoped` — keep a per-run env var out of that file.
|
||||
"""
|
||||
|
||||
from strix.config.loader import (
|
||||
apply_config_override,
|
||||
load_settings,
|
||||
mark_run_scoped,
|
||||
persist_current,
|
||||
)
|
||||
from strix.config.settings import (
|
||||
|
|
@ -37,5 +39,6 @@ __all__ = [
|
|||
"TelemetrySettings",
|
||||
"apply_config_override",
|
||||
"load_settings",
|
||||
"mark_run_scoped",
|
||||
"persist_current",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -26,6 +26,10 @@ logger = logging.getLogger(__name__)
|
|||
_DEFAULT_PATH: Path = Path.home() / ".strix" / "cli-config.json"
|
||||
_override: Path | None = None
|
||||
_cached: Settings | None = None
|
||||
# Env vars set for this process only (e.g. from a per-run CLI flag). They still
|
||||
# win over the config file while the run lasts, but persist_current() must not
|
||||
# write them back, or a one-off flag would silently become the new default.
|
||||
_run_scoped: set[str] = set()
|
||||
|
||||
# Model, API key, and API base describe one provider connection. When the shell
|
||||
# changes any of them, the stored values of the others no longer belong together
|
||||
|
|
@ -60,6 +64,16 @@ def apply_config_override(path: Path) -> None:
|
|||
logger.info("config override applied: %s", path)
|
||||
|
||||
|
||||
def mark_run_scoped(*env_names: str) -> None:
|
||||
"""Exempt ``env_names`` from :func:`persist_current`.
|
||||
|
||||
For values that apply to the current run only. Without this, exporting a
|
||||
per-run CLI flag into the environment makes it indistinguishable from a
|
||||
setting the user chose to keep, and it lands in the config file.
|
||||
"""
|
||||
_run_scoped.update(name.upper() for name in env_names)
|
||||
|
||||
|
||||
def persist_current() -> None:
|
||||
"""Merge currently-set env vars into the active config file (0o600).
|
||||
|
||||
|
|
@ -67,6 +81,9 @@ def persist_current() -> None:
|
|||
run that gets its settings from the file does not erase them. An env
|
||||
var set to the empty string clears the field from the file. A change to
|
||||
any linked LLM connection var drops the whole stored connection first.
|
||||
|
||||
Run-scoped vars (see :func:`mark_run_scoped`) are left out entirely: a
|
||||
per-run override must neither be written nor clear what the file holds.
|
||||
"""
|
||||
s = load_settings()
|
||||
target = _override or _DEFAULT_PATH
|
||||
|
|
@ -80,7 +97,9 @@ def persist_current() -> None:
|
|||
for finfo in type(sub_model).model_fields.values():
|
||||
aliases = [alias.upper() for alias in _aliases_for(finfo)]
|
||||
active = next((alias for alias in aliases if alias in os.environ), None)
|
||||
if active is None:
|
||||
# A run-scoped value belongs to this run only, so it neither
|
||||
# overwrites nor clears the field the file already stores.
|
||||
if active is None or active in _run_scoped:
|
||||
continue
|
||||
for alias in aliases:
|
||||
env_block.pop(alias, None)
|
||||
|
|
|
|||
|
|
@ -94,6 +94,9 @@ class ContextSettings(BaseSettings):
|
|||
model_config = _BASE_CONFIG
|
||||
|
||||
auto_compact: bool = Field(default=True, alias="STRIX_CONTEXT_AUTO_COMPACT")
|
||||
# A larger buffer makes compaction fire sooner (smaller live window), trading
|
||||
# a few extra summary calls for a smaller per-turn context. Raise it to lower
|
||||
# token cost.
|
||||
compact_buffer_tokens: int = Field(default=20_000, gt=0, alias="STRIX_CONTEXT_BUFFER_TOKENS")
|
||||
keep_tokens: int = Field(default=8_000, gt=0, alias="STRIX_CONTEXT_KEEP_TOKENS")
|
||||
fallback_context_tokens: int = Field(
|
||||
|
|
@ -108,6 +111,30 @@ class ContextSettings(BaseSettings):
|
|||
)
|
||||
|
||||
|
||||
class AgentGraphSettings(BaseSettings):
|
||||
"""Multi-agent fan-out limits — optionally bound how many agents a scan spawns.
|
||||
|
||||
Every spawned agent re-pays the full system prompt on each of its turns and
|
||||
(by default) inherits a copy of its parent's history, so an unbounded fan-out
|
||||
is a large driver of token spend on one target. These knobs put a
|
||||
deterministic ceiling on it when set. All default to ``0`` (disabled), so the
|
||||
out-of-the-box behavior is unchanged; operators opt in to bound cost.
|
||||
"""
|
||||
|
||||
model_config = _BASE_CONFIG
|
||||
|
||||
# Max total agents in the graph (root included). 0 = unlimited (default).
|
||||
max_agents: int = Field(default=0, ge=0, alias="STRIX_MAX_AGENTS")
|
||||
# Max spawn depth. Root is depth 1; a child of root is depth 2. 0 = unlimited.
|
||||
max_agent_depth: int = Field(default=0, ge=0, alias="STRIX_MAX_AGENT_DEPTH")
|
||||
# Token cap on the parent history copied into a child spawned with
|
||||
# inherit_context=True. The tail (most recent turns) is kept; older turns are
|
||||
# dropped with a marker. 0 = copy the full parent history (default).
|
||||
inherit_context_max_tokens: int = Field(
|
||||
default=0, ge=0, alias="STRIX_INHERIT_CONTEXT_MAX_TOKENS"
|
||||
)
|
||||
|
||||
|
||||
class RuntimeSettings(BaseSettings):
|
||||
model_config = _BASE_CONFIG
|
||||
|
||||
|
|
@ -178,6 +205,7 @@ class Settings(BaseSettings):
|
|||
|
||||
llm: LlmSettings = Field(default_factory=LlmSettings)
|
||||
dedupe: DedupeSettings = Field(default_factory=DedupeSettings)
|
||||
agent_graph: AgentGraphSettings = Field(default_factory=AgentGraphSettings)
|
||||
runtime: RuntimeSettings = Field(default_factory=RuntimeSettings)
|
||||
context: ContextSettings = Field(default_factory=ContextSettings)
|
||||
telemetry: TelemetrySettings = Field(default_factory=TelemetrySettings)
|
||||
|
|
|
|||
|
|
@ -63,6 +63,9 @@ class AgentCoordinator:
|
|||
self.runtimes: dict[str, AgentRuntime] = {}
|
||||
self._parent_notified: set[str] = set()
|
||||
self._lock = asyncio.Lock()
|
||||
# Slots claimed by a spawn that has not registered its child yet. Counted
|
||||
# alongside live agents so a concurrent fan-out cannot overshoot the cap.
|
||||
self._reserved_slots = 0
|
||||
self._snapshot_path: Path | None = None
|
||||
self.is_shutting_down = False
|
||||
self._budget_stopped = False
|
||||
|
|
@ -465,6 +468,49 @@ class AgentCoordinator:
|
|||
if aid != agent_id and status in {"running", "waiting"}
|
||||
]
|
||||
|
||||
async def agent_count(self) -> int:
|
||||
"""Total agents in the graph (root included), across every status."""
|
||||
async with self._lock:
|
||||
return len(self.parent_of)
|
||||
|
||||
async def depth_of(self, agent_id: str) -> int:
|
||||
"""1-based spawn depth of ``agent_id`` (root = 1).
|
||||
|
||||
Walks parent links defensively: an unknown id or a cycle stops the walk
|
||||
rather than looping forever.
|
||||
"""
|
||||
async with self._lock:
|
||||
depth = 0
|
||||
seen: set[str] = set()
|
||||
current: str | None = agent_id
|
||||
while current is not None and current in self.parent_of and current not in seen:
|
||||
seen.add(current)
|
||||
depth += 1
|
||||
current = self.parent_of.get(current)
|
||||
return max(depth, 1)
|
||||
|
||||
async def try_reserve_agent_slot(self, max_agents: int) -> bool:
|
||||
"""Atomically claim one slot under ``max_agents``; ``0`` means unlimited.
|
||||
|
||||
Reading :meth:`agent_count` and then spawning is check-then-act: two
|
||||
parents asking for the last slot concurrently both see room before
|
||||
either child registers, and the graph overshoots the cap. Counting
|
||||
outstanding reservations with the live agents under the same lock
|
||||
closes that window. Release the slot with
|
||||
:meth:`release_agent_slot` once the child is registered or the spawn
|
||||
has failed.
|
||||
"""
|
||||
async with self._lock:
|
||||
if max_agents and len(self.parent_of) + self._reserved_slots >= max_agents:
|
||||
return False
|
||||
self._reserved_slots += 1
|
||||
return True
|
||||
|
||||
async def release_agent_slot(self) -> None:
|
||||
"""Give back a slot claimed by :meth:`try_reserve_agent_slot`."""
|
||||
async with self._lock:
|
||||
self._reserved_slots = max(self._reserved_slots - 1, 0)
|
||||
|
||||
async def graph_snapshot(
|
||||
self,
|
||||
) -> tuple[dict[str, str | None], dict[str, Status], dict[str, str], dict[str, str]]:
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from typing import TYPE_CHECKING, Any
|
|||
from agents.model_settings import ModelSettings
|
||||
from openai.types.shared import Reasoning
|
||||
|
||||
from strix.config import load_settings
|
||||
from strix.config.models import (
|
||||
DEFAULT_MODEL_RETRY,
|
||||
OPENROUTER_ATTRIBUTION_HEADERS,
|
||||
|
|
@ -23,6 +24,68 @@ from strix.config.models import (
|
|||
from strix.core.sessions import scrub_images_from_items
|
||||
|
||||
|
||||
# Rough tokens→chars factor for bounding inherited context without a model-bound
|
||||
# tokenizer. ~4 chars/token is the usual estimate; kept conservative so the cap
|
||||
# never lets more through than intended.
|
||||
_CHARS_PER_TOKEN = 4
|
||||
_HISTORY_TRUNCATED_MARKER = {
|
||||
"role": "user",
|
||||
"content": "[... older inherited context dropped to bound token cost ...]",
|
||||
}
|
||||
_OVERSIZED_ITEM_NOTE = "[... newest inherited item truncated to bound token cost ...]\n"
|
||||
|
||||
|
||||
def _fit_oversized_item(item: Any, char_budget: int) -> dict[str, str]:
|
||||
"""Render ``item`` as a text message cut down to ``char_budget``.
|
||||
|
||||
Used only when the single newest item is larger than the whole budget:
|
||||
keeping it whole would blow the cap on the child's very first request and
|
||||
can overflow the provider's context window. Nothing else survives the trim
|
||||
in that case, so rendering it as one text message cannot orphan a tool call
|
||||
from its output.
|
||||
"""
|
||||
body = json.dumps(item, ensure_ascii=False, default=str)
|
||||
while body:
|
||||
summary = {"role": "user", "content": f"{_OVERSIZED_ITEM_NOTE}{body}"}
|
||||
overshoot = len(json.dumps(summary, ensure_ascii=False)) - char_budget
|
||||
if overshoot <= 0:
|
||||
return summary
|
||||
body = body[: max(len(body) - overshoot, 0)]
|
||||
# Budget too small to hold even a snippet; the marker alone is the most that
|
||||
# can be said about the dropped item.
|
||||
return dict(_HISTORY_TRUNCATED_MARKER)
|
||||
|
||||
|
||||
def _trim_parent_history(parent_history: list[Any]) -> list[Any]:
|
||||
"""Keep the most-recent tail of ``parent_history`` within the configured cap.
|
||||
|
||||
A child inheriting its parent's whole history re-pays for it on every one of
|
||||
its own turns, so an unbounded copy multiplies token cost across the fan-out.
|
||||
``STRIX_INHERIT_CONTEXT_MAX_TOKENS`` bounds it; ``0`` keeps the full history.
|
||||
"""
|
||||
max_tokens = load_settings().agent_graph.inherit_context_max_tokens
|
||||
if max_tokens <= 0 or not parent_history:
|
||||
return parent_history
|
||||
|
||||
char_budget = max_tokens * _CHARS_PER_TOKEN
|
||||
kept: list[Any] = []
|
||||
used = 0
|
||||
for item in reversed(parent_history):
|
||||
size = len(json.dumps(item, ensure_ascii=False, default=str))
|
||||
if used + size > char_budget:
|
||||
# ``kept`` is empty only on the newest item, i.e. that one item is
|
||||
# over budget all by itself. Truncate it rather than keeping it
|
||||
# whole — otherwise the cap silently fails to bound anything.
|
||||
kept.append(
|
||||
_HISTORY_TRUNCATED_MARKER if kept else _fit_oversized_item(item, char_budget)
|
||||
)
|
||||
break
|
||||
kept.append(item)
|
||||
used += size
|
||||
kept.reverse()
|
||||
return kept
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from strix.config.settings import ReasoningEffort
|
||||
|
||||
|
|
@ -351,9 +414,13 @@ def child_initial_input(
|
|||
user messages.
|
||||
"""
|
||||
parts: list[str] = []
|
||||
# Scrub first, then measure: a screenshot's base64 block is replaced by a
|
||||
# short placeholder, so budgeting against the raw block would let one image
|
||||
# evict every useful text turn behind it for size the child never pays.
|
||||
parent_history = _trim_parent_history(scrub_images_from_items(parent_history))
|
||||
if parent_history:
|
||||
rendered = json.dumps(
|
||||
scrub_images_from_items(parent_history),
|
||||
parent_history,
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ import os
|
|||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from strix.config import apply_config_override
|
||||
from strix.config import apply_config_override, mark_run_scoped
|
||||
from strix.config.settings import DEFAULT_MAX_TURNS
|
||||
from strix.core.paths import run_dir_for, runtime_state_dir
|
||||
from strix.interface.scan_setup import attach_workspace_mount, build_targets_info
|
||||
|
|
@ -201,6 +201,19 @@ Strix Cloud:
|
|||
),
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--reasoning-effort",
|
||||
dest="reasoning_effort",
|
||||
type=str,
|
||||
choices=["none", "minimal", "low", "medium", "high", "xhigh", "max"],
|
||||
default=None,
|
||||
help=(
|
||||
"Model reasoning effort for this run. Higher = more thinking tokens = "
|
||||
"higher cost and (usually) deeper analysis. Overrides STRIX_REASONING_EFFORT "
|
||||
"and the config file for this run. Default: the configured value (high)."
|
||||
),
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--scope-mode",
|
||||
type=str,
|
||||
|
|
@ -315,6 +328,15 @@ Strix Cloud:
|
|||
if args.mcp_exclude:
|
||||
os.environ["STRIX_MCP_EXCLUDE"] = ",".join(args.mcp_exclude)
|
||||
|
||||
# Settings read STRIX_REASONING_EFFORT from the environment (env wins over the
|
||||
# config file), so exporting it here makes the flag win for this run. Mark it
|
||||
# run-scoped: persist_current() writes every settings env var it finds into
|
||||
# cli-config.json, which would turn this one-run flag into the new default
|
||||
# for every later scan.
|
||||
if args.reasoning_effort:
|
||||
os.environ["STRIX_REASONING_EFFORT"] = args.reasoning_effort
|
||||
mark_run_scoped("STRIX_REASONING_EFFORT")
|
||||
|
||||
if args.update:
|
||||
sys.exit(0 if self_update() else 1)
|
||||
|
||||
|
|
|
|||
|
|
@ -12,7 +12,8 @@ from typing import Any, Literal, get_args
|
|||
|
||||
from agents import RunContextWrapper, function_tool
|
||||
|
||||
from strix.core.agents import Status, coordinator_from_context
|
||||
from strix.config import load_settings
|
||||
from strix.core.agents import AgentCoordinator, Status, coordinator_from_context
|
||||
from strix.core.execution import notify_parent_on_terminal
|
||||
from strix.core.hooks import LLM_TURN_KEY
|
||||
from strix.report.state import get_global_report_state
|
||||
|
|
@ -25,6 +26,39 @@ _ACTIVE_STATUSES: frozenset[str] = frozenset({"running", "waiting"})
|
|||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _agent_limit_error(max_agents: int) -> str:
|
||||
"""Model-facing refusal for a spawn that would breach ``STRIX_MAX_AGENTS``."""
|
||||
return (
|
||||
f"Agent limit reached ({max_agents} agents). Cannot spawn another. "
|
||||
"Do this work yourself, reuse an existing agent via send_message_to_agent, "
|
||||
"or wait_for_agents to let running ones finish. The operator can raise "
|
||||
"STRIX_MAX_AGENTS if a larger fan-out is intended."
|
||||
)
|
||||
|
||||
|
||||
async def _depth_limit_error(coordinator: AgentCoordinator, parent_id: str) -> str | None:
|
||||
"""Return a model-facing error if a child of ``parent_id`` would sit too deep.
|
||||
|
||||
Bounds token spend: every extra agent re-pays the full system prompt on each
|
||||
of its turns, so an unbounded graph is the biggest single-target cost driver.
|
||||
``STRIX_MAX_AGENT_DEPTH`` caps the tree height; ``0`` disables the check. The
|
||||
parent's own depth is fixed for the life of the agent, so unlike the total
|
||||
agent cap this needs no reservation to stay correct under concurrent spawns.
|
||||
"""
|
||||
max_depth = load_settings().agent_graph.max_agent_depth
|
||||
if not max_depth:
|
||||
return None
|
||||
|
||||
child_depth = await coordinator.depth_of(parent_id) + 1
|
||||
if child_depth > max_depth:
|
||||
return (
|
||||
f"Agent depth limit reached (max {max_depth}). This agent is "
|
||||
"too deep in the tree to spawn a child. Run the subtask yourself or hand it "
|
||||
"back to a shallower agent. The operator can raise STRIX_MAX_AGENT_DEPTH."
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _ctx(ctx: RunContextWrapper) -> dict[str, Any]:
|
||||
return ctx.context if isinstance(ctx.context, dict) else {}
|
||||
|
||||
|
|
@ -569,10 +603,23 @@ async def create_agent(
|
|||
)
|
||||
|
||||
skill_list = list(skills or [])
|
||||
skill_error = validate_requested_skills(skill_list)
|
||||
if skill_error:
|
||||
spawn_error = validate_requested_skills(skill_list) or await _depth_limit_error(
|
||||
coordinator, parent_id
|
||||
)
|
||||
if spawn_error:
|
||||
return json.dumps(
|
||||
{"success": False, "error": skill_error, "agent_id": None},
|
||||
{"success": False, "error": spawn_error, "agent_id": None},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
||||
# Claim the slot before spawning, not just before checking: the child only
|
||||
# enters the graph once the spawner registers it, and two parents racing for
|
||||
# the last slot would otherwise both be waved through.
|
||||
max_agents = load_settings().agent_graph.max_agents
|
||||
if not await coordinator.try_reserve_agent_slot(max_agents):
|
||||
return json.dumps(
|
||||
{"success": False, "error": _agent_limit_error(max_agents), "agent_id": None},
|
||||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
|
|
@ -593,6 +640,11 @@ async def create_agent(
|
|||
ensure_ascii=False,
|
||||
default=str,
|
||||
)
|
||||
finally:
|
||||
# The spawner registers the child before it returns, so by now the slot
|
||||
# is accounted for by the graph itself (and on failure there is nothing
|
||||
# to account for).
|
||||
await coordinator.release_agent_slot()
|
||||
|
||||
logger.info(
|
||||
"create_agent: spawned %s (%s) parent=%s skills=%d task_len=%d",
|
||||
|
|
|
|||
102
tests/test_agent_fanout_limits.py
Normal file
102
tests/test_agent_fanout_limits.py
Normal file
|
|
@ -0,0 +1,102 @@
|
|||
"""Tests for multi-agent fan-out caps in strix.tools.agents_graph.tools."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from strix.config import loader
|
||||
from strix.core.agents import AgentCoordinator
|
||||
from strix.tools.agents_graph.tools import _depth_limit_error
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import pytest
|
||||
|
||||
|
||||
async def _graph(*edges: tuple[str, str | None]) -> AgentCoordinator:
|
||||
"""Build a coordinator from (agent_id, parent_id) edges, root first."""
|
||||
coordinator = AgentCoordinator()
|
||||
for agent_id, parent_id in edges:
|
||||
await coordinator.register(agent_id, agent_id, parent_id)
|
||||
return coordinator
|
||||
|
||||
|
||||
async def test_reserve_blocks_when_limit_reached() -> None:
|
||||
coordinator = await _graph(("root", None), ("child", "root"))
|
||||
|
||||
assert await coordinator.try_reserve_agent_slot(2) is False
|
||||
|
||||
|
||||
async def test_reserve_allows_below_limit() -> None:
|
||||
coordinator = await _graph(("root", None), ("child", "root"))
|
||||
|
||||
assert await coordinator.try_reserve_agent_slot(4) is True
|
||||
|
||||
|
||||
async def test_reserve_unlimited_when_zero() -> None:
|
||||
coordinator = await _graph(("root", None), ("a", "root"), ("b", "a"), ("c", "b"))
|
||||
|
||||
assert await coordinator.try_reserve_agent_slot(0) is True
|
||||
|
||||
|
||||
async def test_concurrent_reservations_cannot_overshoot_cap() -> None:
|
||||
# Two parents race for the last slot. Counting only registered agents would
|
||||
# wave both through, because neither child registers before the other checks.
|
||||
coordinator = await _graph(("root", None), ("a", "root"))
|
||||
|
||||
granted = await asyncio.gather(*(coordinator.try_reserve_agent_slot(3) for _ in range(2)))
|
||||
|
||||
assert sorted(granted) == [False, True]
|
||||
|
||||
|
||||
async def test_released_slot_is_reusable() -> None:
|
||||
coordinator = await _graph(("root", None), ("a", "root"))
|
||||
|
||||
assert await coordinator.try_reserve_agent_slot(3) is True
|
||||
assert await coordinator.try_reserve_agent_slot(3) is False
|
||||
await coordinator.release_agent_slot()
|
||||
assert await coordinator.try_reserve_agent_slot(3) is True
|
||||
|
||||
|
||||
async def test_release_does_not_go_negative() -> None:
|
||||
coordinator = await _graph(("root", None))
|
||||
|
||||
await coordinator.release_agent_slot()
|
||||
await coordinator.release_agent_slot()
|
||||
|
||||
# A stray release must not hand out a free slot beyond the cap.
|
||||
assert await coordinator.try_reserve_agent_slot(1) is False
|
||||
|
||||
|
||||
async def test_max_depth_blocks_grandchild(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("STRIX_MAX_AGENTS", "0")
|
||||
monkeypatch.setenv("STRIX_MAX_AGENT_DEPTH", "2")
|
||||
loader._cached = None
|
||||
try:
|
||||
coordinator = await _graph(("root", None), ("child", "root"))
|
||||
# Spawning from the child would create a depth-3 grandchild.
|
||||
child_error = await _depth_limit_error(coordinator, "child")
|
||||
# Spawning from the root creates a depth-2 child — allowed.
|
||||
root_error = await _depth_limit_error(coordinator, "root")
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
assert child_error is not None
|
||||
assert "depth limit reached" in child_error
|
||||
assert root_error is None
|
||||
|
||||
|
||||
async def test_depth_limit_disabled_when_zero(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("STRIX_MAX_AGENTS", "0")
|
||||
monkeypatch.setenv("STRIX_MAX_AGENT_DEPTH", "0")
|
||||
loader._cached = None
|
||||
try:
|
||||
coordinator = await _graph(
|
||||
("root", None), ("a", "root"), ("b", "a"), ("c", "b"), ("d", "c")
|
||||
)
|
||||
error = await _depth_limit_error(coordinator, "d")
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
assert error is None
|
||||
|
|
@ -406,6 +406,47 @@ def test_persist_current_replaces_corrupt_file(
|
|||
assert json.loads(target.read_text(encoding="utf-8")) == {"env": {"STRIX_LLM": "env-model"}}
|
||||
|
||||
|
||||
def test_persist_current_skips_run_scoped_env(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# A per-run flag exports its env var so it wins for this run; persisting it
|
||||
# would silently make it the default for every later run.
|
||||
monkeypatch.setenv("STRIX_LLM", "persisted-model")
|
||||
monkeypatch.setenv("STRIX_REASONING_EFFORT", "max")
|
||||
monkeypatch.setattr(loader, "_run_scoped", set())
|
||||
loader.mark_run_scoped("STRIX_REASONING_EFFORT")
|
||||
target = tmp_path / "cli-config.json"
|
||||
loader.apply_config_override(target)
|
||||
|
||||
loader.persist_current()
|
||||
|
||||
assert json.loads(target.read_text(encoding="utf-8")) == {
|
||||
"env": {"STRIX_LLM": "persisted-model"}
|
||||
}
|
||||
|
||||
|
||||
def test_run_scoped_env_does_not_clear_stored_value(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
# The per-run override must leave the user's stored preference alone, not
|
||||
# just avoid overwriting it with the run's value.
|
||||
target = tmp_path / "cli-config.json"
|
||||
target.write_text(
|
||||
json.dumps({"env": {"STRIX_REASONING_EFFORT": "low"}}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
loader.apply_config_override(target)
|
||||
monkeypatch.setenv("STRIX_REASONING_EFFORT", "max")
|
||||
monkeypatch.setattr(loader, "_run_scoped", set())
|
||||
loader.mark_run_scoped("STRIX_REASONING_EFFORT")
|
||||
|
||||
loader.persist_current()
|
||||
|
||||
assert json.loads(target.read_text(encoding="utf-8")) == {
|
||||
"env": {"STRIX_REASONING_EFFORT": "low"}
|
||||
}
|
||||
|
||||
|
||||
def test_persist_current_sets_0600_mode(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("STRIX_LLM", "persisted-model")
|
||||
target = tmp_path / "cli-config.json"
|
||||
|
|
|
|||
|
|
@ -2,13 +2,16 @@
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from itertools import pairwise
|
||||
from typing import Any
|
||||
|
||||
import litellm
|
||||
import pytest
|
||||
|
||||
from strix.config import loader
|
||||
from strix.core.inputs import (
|
||||
_trim_parent_history,
|
||||
build_root_task,
|
||||
build_scan_targets,
|
||||
build_scope_context,
|
||||
|
|
@ -62,6 +65,100 @@ def test_child_initial_input_no_consecutive_same_role(parent_history: list[Any])
|
|||
assert all(prev != nxt for prev, nxt in pairwise(roles))
|
||||
|
||||
|
||||
def test_child_initial_input_trims_inherited_history(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Cap (20 tokens ~= 80 chars) fits the newest item but not both.
|
||||
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "20")
|
||||
loader._cached = None
|
||||
try:
|
||||
history = [
|
||||
{"role": "assistant", "content": "oldest work item that should be dropped"},
|
||||
{"role": "assistant", "content": "newest work item that should be kept"},
|
||||
]
|
||||
result = child_initial_input(**_child_kwargs(history))
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
content = result[0]["content"]
|
||||
assert "newest work item that should be kept" in content
|
||||
assert "oldest work item that should be dropped" not in content
|
||||
assert "older inherited context dropped" in content
|
||||
|
||||
|
||||
def test_screenshot_does_not_evict_text_history(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# A screenshot is replaced by a short placeholder before the child ever sees
|
||||
# it, so budgeting against the raw base64 would drop useful text turns to
|
||||
# make room for bytes that are never sent.
|
||||
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "200")
|
||||
loader._cached = None
|
||||
try:
|
||||
history = [
|
||||
{"role": "assistant", "content": "earlier finding worth inheriting"},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_image", "image_url": "data:image/png;base64," + "A" * 20_000}
|
||||
],
|
||||
},
|
||||
]
|
||||
result = child_initial_input(**_child_kwargs(history))
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
content = result[0]["content"]
|
||||
assert "earlier finding worth inheriting" in content
|
||||
assert "screenshot omitted from inherited context" in content
|
||||
assert "older inherited context dropped" not in content
|
||||
|
||||
|
||||
def test_trim_truncates_oversized_newest_item(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# One item, larger than the whole budget: keeping it whole would mean the cap
|
||||
# bounds nothing at all on the child's first request.
|
||||
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "40")
|
||||
loader._cached = None
|
||||
try:
|
||||
history = [{"role": "assistant", "content": "x" * 5000}]
|
||||
trimmed = _trim_parent_history(history)
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
assert len(trimmed) == 1
|
||||
assert len(json.dumps(trimmed[0], ensure_ascii=False)) <= 40 * 4
|
||||
assert "truncated to bound token cost" in trimmed[0]["content"]
|
||||
|
||||
|
||||
def test_trim_falls_back_to_marker_when_budget_tiny(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "1")
|
||||
loader._cached = None
|
||||
try:
|
||||
history = [{"role": "assistant", "content": "x" * 5000}]
|
||||
trimmed = _trim_parent_history(history)
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
assert len(trimmed) == 1
|
||||
assert "x" * 100 not in trimmed[0]["content"]
|
||||
|
||||
|
||||
def test_child_initial_input_keeps_full_history_when_cap_disabled(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
monkeypatch.setenv("STRIX_INHERIT_CONTEXT_MAX_TOKENS", "0")
|
||||
loader._cached = None
|
||||
try:
|
||||
history = [
|
||||
{"role": "assistant", "content": "first item kept"},
|
||||
{"role": "assistant", "content": "second item kept"},
|
||||
]
|
||||
result = child_initial_input(**_child_kwargs(history))
|
||||
finally:
|
||||
loader._cached = None
|
||||
|
||||
content = result[0]["content"]
|
||||
assert "first item kept" in content
|
||||
assert "second item kept" in content
|
||||
assert "older inherited context dropped" not in content
|
||||
|
||||
|
||||
def _cache_points(model_name: str) -> Any:
|
||||
extra = make_model_settings(None, model_name=model_name).extra_args or {}
|
||||
return extra.get("cache_control_injection_points")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue