mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-11 03:40:03 +00:00
Some checks are pending
Pre-commit / run (ubuntu-latest) (push) Waiting to run
Tests ReMe / Unit Tests - py3.13 (push) Waiting to run
Tests ReMe / Unit Tests - py3.11 (push) Waiting to run
Tests ReMe / Unit Tests - py3.12 (push) Waiting to run
Windows Smoke / CLI smoke - py3.11 (push) Waiting to run
* feat: add ssh proxy * feat: add ssh proxy * feat: add ssh proxy * feat: add ssh proxy * feat: add prompt * feat: add agent wrapper * feat: add agent wrapper * feat: add agent wrapper * feat: add tushare skill * feat: add tushare skill * feat: add tushare skill * feat: add none stream * chore(deps): update dependency versions in pyproject.toml - Bump claude-agent-sdk from 0.2.123 to 0.2.126 - Upgrade pre-commit to version 4.6.1 or higher - Upgrade pytest to version 9.1.1 or higher * feat(agent_wrapper): add session compaction support and unify session commands - Introduce compact_session method to BaseAgentWrapper and implement it in AsAgentWrapper, CcAgentWrapper, and CodexAgentWrapper - Add session_command module with SessionCommandResult dataclass and handle_session_command function for /clear and /compact commands - Update __init__.py exports to include session_command handlers - Modify DingTalkWaitStep to handle session commands via handle_session_command function - Remove streaming mode from DingTalkWaitStep and simplify reply handling to final Markdown replies only - Add unit tests for session compaction methods and session command handling across wrappers and DingTalk integration - Clean up and remove obsolete streaming and card rendering code from DingTalk wait step - Adjust daily_cookbook.yaml to remove stream and card_update_interval config entries for DingTalk wait step * feat(auto_fin): add Auto Fin simulated portfolio cookbook workflow - Add comprehensive Auto Fin schema exports for multiple models and enums - Implement base class and helpers for Auto Fin analysis steps - Create file, state, and formatting utilities for Auto Fin with atomic file writes and locking - Define Auto Fin pipeline with four analysis agents: backtest, event, portfolio, and US correlation - Register Auto Fin package in cookbook workflows and schema initialization - Add detailed documentation in markdown describing the system design, workflow, and data contracts * feat(outbound_proxy): add application-scoped outbound HTTP proxy components - Introduce BaseOutboundProxy and OutboundProxyEndpoint as core contracts - Implement FixedHttpOutboundProxy for external HTTP proxy integration - Add SshHttpOutboundProxy providing SSH-backed local HTTP proxy tunnels - Register outbound proxy components in component registry and enumeration - Update components package to include outbound_proxy module - Add dependency on pproxy for SSH HTTP proxy bridging - Include comprehensive unit tests covering proxy lifecycle, validation, environment merging, error handling, readiness, and monitoring mechanisms * refactor(network): replace SSH proxy with explicit HTTP outbound proxy - Remove SSH proxy helper implementation and references in codebase - Add support for explicit HTTP proxy URL in arXiv and HuggingFace clients - Modify clients to use async context manager for consistent resource handling - Update daily paper steps to forward outbound proxy configuration explicitly - Change tests to cover new proxy usage model and remove SSH proxy mocks - Add outbound proxy component configuration in daily_cookbook.yaml - Ensure proxy URL usage disables environment trust in HTTP clients - Fix app context component enum access to be defensive against missing keys * feat(agent_wrapper): add managed proxy support for command environments - Introduce BaseOutboundProxy binding in BaseAgentWrapper for outbound proxy management - Add bash_environment and command_proxy_environment properties to apply proxy settings - Update WorkspaceBackend instantiation in AsAgentWrapper to use bash_environment - Inject managed proxy export commands into Claude Code Bash commands via hooks - Enhance CodexAgentWrapper to include managed proxy in shell environment policy - Modify daily_cookbook.yaml steps to specify outbound_proxy as default where needed - Add comprehensive unit tests verifying managed proxy injection and environment isolation - Ensure subprocess_environment remains unchanged while proxy is applied selectively to commands * refactor(memory): replace search job_tools with memory in daily cookbook config - Change workspace_dir default from .reme to reme_workspace - Replace search job_tools with memory across multiple components and jobs - Update descriptions to reflect long-term memory retrieval instead of search - Modify system prompts to instruct using memory for retrieving notes - Adjust unit tests to verify memory job_tools and job presence instead of search - Ensure consistency in configuration and tests for memory backend usage * refactor(config): rename memory to memory_search in daily cookbook config - Change all occurrences of "memory" to "memory_search" in job_tools and job definitions - Update related system prompts to reflect the new memory_search terminology - Modify unit tests to assert the presence of memory_search instead of memory - Ensure consistency across skills, job tools, and backend configurations in multiple components * feat(auto_fin): add deterministic quantitative research and ranking fusion - Introduce new schema models: EtfScore, RankingMetrics, ExtremeAnalysis, DimensionRanking, and FusionRanking to represent deterministic research outputs - Add ranking data to event, backtest, us_correlation, and portfolio analysis outputs - Implement ranking_section renderer to format Top20 scores and diagnostics in Markdown - Develop AutoFinQuantStep for deterministic ETF ranking using TuShare data, Polars, and a custom extremely randomized tree ensemble - Integrate quantitative rankings into backtest and portfolio analysis steps and reports - Extend auto_fin pipeline with new quant_enabled and quant_required config options - Enforce ranking constraints like unique codes, contiguous ranks, and normalized fusion weights - Update analysis YAMLs with rules limiting data freshness, universe, and ranking usage - Incorporate ranking outputs into all major markdown report bodies in Auto Fin pipeline - Add concurrency-limited asynchronous TuShare client to fetch required market data - Introduce cross-sectional rank correlation and NDCG metrics for ranking quality evaluation * feat(auto_fin): implement stage-wise notification and reporting for analysis pipeline - Refactor notification config in daily_cookbook.yaml to support dispatch steps - Update AutoFinNotificationStep to deduplicate notifications per run stage - Add _notify_stage method in pipeline to send notifications for each analysis stage - Implement persistence and notification for event, backtest, US correlation, and portfolio stages - Modify pipeline flow to persist reports and notify after each stage completion - Adjust metadata to track notifications and errors per stage - Update tests to verify stage-wise notification sending and deduplication - Remove older combined report persistence in favor of modular stage handling * feat(auto_fin): add outbound proxy support for Tushare API usage - Introduce BaseOutboundProxy reference in AutoFinPipelineStep and AutoFinQuantStep - Update TushareResearchClient and trade calendar fetch to accept and use proxy URL - Create _ProxiedTushareApi adapter to route Tushare requests via explicit HTTP proxy - Modify create_tushare_api utility to optionally return proxied API client - Add unit tests covering proxy forwarding and client behavior with managed proxies - Ensure proxy usage respects explicit proxy URL over environment fallback - Integrate outbound proxy into data fetching and quantitative research steps * feat(auto_fin): enforce checkpoint time validation and add state models - Introduce AnalysisState base class and specific states for event, backtest, and US correlation analyses - Replace analysis output types with corresponding state classes in run schemas - Add require_checkpoint_reached method to validate decision_at/data_cutoff against current time - Enforce checkpoint time checks before analysis steps in event, backtest, portfolio, and quant analyses - Refactor quant data loading to include adjustment factors and apply price adjustments without fallback - Update analysis YAML docs to require real-time checkpoint validation and forbid using future data - Improve portfolio run serialization by excluding redundant legacy fields and nested proposed actions - Add helper to extract readable sections from persisted checkpoint documents - Fix event analysis output validation to reject events and sources with future timestamps * feat(auto_fin): auto-select latest reached checkpoint if none specified - Extend checkpoint config to accept empty string for auto selection - Add static method to compute latest checkpoint reached by current time - Modify pipeline step to auto-select checkpoint based on trade calendar and time - Adjust force flag default depending on whether checkpoint is explicit or auto - Log details when checkpoint is auto-selected to improve observability - Add comprehensive tests for auto checkpoint selection logic and edge cases - Remove deprecated default and required constraints from force parameter in config * refactor(auto_fin): unify datetime comparison with compare_datetimes utility - Replace direct datetime comparisons with compare_datetimes function calls - Use cmp_to_key with compare_datetimes for sorting datetime tuples and lists - Update validation logic in backtest, event, analysis, and ledger modules for consistent datetime handling - Add unit tests to verify handling of naive and aware datetime comparisons in event and backtest validations - Ensure marked_at and interval_end timestamps are set and compared consistently using compare_datetimes - Improve correctness of ordering and conditional checks related to timestamps throughout auto_fin steps and ledger code * feat(auto_fin): add datetime comparison helper for mixed timezone data - Implement compare_datetimes function to handle naive and aware datetimes - Ensure naive datetime is interpreted in the known timezone of the counterpart - Facilitate comparisons between legacy and timezone-aware Auto Fin data - Add module docstring explaining purpose of the helpers * docs(auto_fin): enforce unique ETF representative per sub-theme in analysis rules - Update backtest.yaml to recommend or highlight only one ETF per sub-theme for ETF analyses - Modify event.yaml to map only one representative ETF per sub-theme, avoiding duplicate recommendations - Revise portfolio.yaml to restrict holdings/buys to a single ETF per sub-theme, preventing repeated buys of highly overlapping ETFs - Adjust us_correlation.yaml to retain only one representative A-share ETF per sub-theme for mapping or recommendation - Add test to verify presence of new sub-theme uniqueness guidance in step prompts * feat(auto_fin): separate draft model and include deterministic fusion ranking - Introduce _PortfolioProposalDraft pydantic model for agent-authored fields before ranking - Discard any "fusion_ranking" data from draft to prevent conflicts with canonical ranking - Modify AutoFinPortfolioStep to receive draft, enrich with fusion_ranking, and produce final output - Update tests to use _PortfolioProposalDraft and validate deterministic fusion ranking propagation - Add async test verifying fusion ranking is correctly set in portfolio output with no errors * refactor(auto_fin): rewrite and simplify Auto Fin schema and steps - Remove legacy Auto Fin analysis step modules and helpers - Replace complex ranking and portfolio models with simplified current-news models - Update schema to focus on news-case workflow with new domain models - Remove A-share decision checkpoints and backtest details from schema - Simplify recommendation and decision output structures - Clean up deprecated state and utility functions - Update Auto Fin steps initialization to new pipeline steps only - Improve uniqueness validation for themes and ETFs in research plan * feat(auto_fin): implement full local cache and analysis workflow for Auto Fin - Add AutoFinDataStep to prepare and cache daily TuShare data with lookback - Add AutoFinAnalysisStep to analyze cached data and generate Markdown report - Implement detailed time window, ETF filtering, and historical case validation - Introduce YAML prompts for planning and decision-making steps - Update .gitignore to include reme_workspace/ - Clean up config and import structure for auto_fin steps - Remove old pipeline.py and consolidate functionality into new modules - Use polars for efficient CSV reading and data processing - Ensure atomic writes and strict JSON serialization for cache files - Enforce rules on news timing, ETF universe, and historical case usage * fix(auto_fin): restrict news data source to '财联社' in analysis and cache - Update analysis templates to specify current news as from '财联社' only - Modify news fetching functions to filter by source '财联社' - Add validation method to check cached news source correctness - Update news caching logic to exclude non-'财联社' news - Enhance unit tests with multiple sources to ensure filtering works - Confirm news API calls include source filter parameter as '财联社' * refactor(auto_fin): convert I/O methods to asynchronous implementations - Change _news, _dataset, and _theme_data methods to async for improved concurrency - Move JSONL and CSV reading operations to asynchronous wrappers using asyncio.to_thread - Remove synchronous _read_jsonl and _read_csv functions, integrate them as static async class methods - Update cache validation methods to async, awaiting I/O operations accordingly - Adjust usage of dataset and news retrieval in analysis step to await asynchronous methods - Add async unit test to validate JSONL reading with unicode line separators - Preserve existing functionality while enabling non-blocking file and data access * fix(nx_file_graph): defer networkx import and improve dependency handling - Move networkx import inside NxFileGraph constructor for lazy loading - Raise ImportError with original exception context if networkx is missing - Remove module-level fallback assignment of nx to None - Expand test to block loading of multiple optional core dependencies eagerly - Change exception type in test from ModuleNotFoundError to AssertionError - Update test comments to reflect broader optional dependency checks * feat(embedding_store): add quota retry delay mechanism for embedding requests - Introduce quota_retry_delay parameter to configure wait time before retry on quota exhaustion - Implement detection of insufficient quota errors in LocalEmbeddingStore without external SDK - Add retry logic with custom delay when quota is insufficient during embedding requests - Update configuration to set max_retries and quota_retry_delay defaults for embedding store - Add unit tests covering quota exhaustion retry behavior with delay and opt-in control - Ensure existing retry behavior remains unchanged if quota_retry_delay is not set * feat(auto_fin): add detailed logging to analysis and data fetching steps - Add _preview static method for bounded diagnostic output in analysis.py - Log prompt start, completion, errors, and validation details in _reply method - Add info logs for major processing steps in execute method of analysis.py - Add debug and info logs for cache validation, data fetching, and pagination in data.py - Log conditions for skipping reports and cache plans in data.py execute method - Log download summaries and cache writes for news and ETF data - Improve error logging with exception details in cache validation functions - Ensure all logs include context such as record counts, paths, and parameters * refactor(auto_fin): overhaul Auto Fin workflow and schema contracts - Replace old Auto Fin schema models with comprehensive new data classes - Remove legacy Auto Fin analysis step in favor of modular agent-based steps - Introduce AutoFinAgentStep for validating structured agent replies - Simplify data cleaning and JSONL writing utilities for news cache - Remove synchronous and asynchronous dataset methods from analysis step - Redefine Auto Fin analysis configuration for 360-day news retention and multi-step pipeline - Remove embedded analysis prompt templates and replace with agent-driven logic - Update __init__.py exports to match new step implementations and remove deprecated classes - Improve error handling and validation in agent step reply processing - Clean up redundant imports and unused code in analysis and data preparation modules * feat(auto_fin): add detailed logging for analysis and data processing steps - Add timing logs to measure agent prompt processing duration in analysis.py - Log news cache hits and news write paths with record counts in data.py - Include detailed info logs for news download start and completion in data.py - Add start, progress, and completion logs with topic and event counts in history.py - Log start and completion of merge step including path and ETF count in merge.py - Add start and done logs with window and news counts in topic.py * feat(auto_fin): enhance schema and steps with detailed ETF and event modeling - Replace and add multiple AutoFin schema classes to support detailed ETF selection, historical research, market analysis, forecast models, and report output with validation - Implement Shanghai timezone normalization and strict validation in schema models - Remove deprecated AutoFin analysis agent step and consolidate reply handling in base step - Introduce AutoFinStep base class with shared helpers for prompt handling, data fetching, logging, and JSONL file operations - Add AutoFinDataStep to manage daily news data complete with schedule validation, caching, and source validation logic - Update cookbook configuration to customize auto_fin step parameters and simplify outbound proxy settings - Refactor imports and clean unused code for better maintainability * feat(auto_fin): introduce detailed historical event resolution and market similarity analysis - Add AutoFinHistoricalEventReference and AutoFinHistoricalSimilarity models for refined event referencing and similarity judgment - Implement validation to ensure non-empty critical fields and uniqueness of historical news IDs - Develop method to resolve Agent-selected historical event references from workspace files with strict path and existence checks - Enrich historical events with market entry and future returns data after resolution - Redesign market step to calculate similarity-weighted ETF forecasts based on matched historical event similarities - Enforce validation on matched historical events for uniqueness and proper weight summation - Simplify merge step output to final Markdown report without YAML frontmatter and redundant fields - Update user instructions for history search, market, and merge steps to reflect new data structures and responsibilities - Adjust test suite to cover new schema and step behavior changes, including enhanced validation and JSON output formats * feat(auto_fin): add new cron jobs and output analysis jsonl - Add new cron jobs auto_fin_1145_cron and auto_fin_1800_cron with auto_fin_steps - Change auto_fin_0930_cron schedule to run Monday to Sunday - Extend merge step to write analysis data to auto_fin_analysis.jsonl - Update unit tests to verify new cron jobs and their steps configuration * fix(auto_fin): improve atomic file write and refresh daily index - Change temporary file naming to include UUID for uniqueness and hidden prefix - Replace atomic write method from using Path.replace to os.replace with safe unlink - Add import and use os.replace for safer file replace operation - Refresh daily index after writing auto finance markdown and JSONL files - Import and call refresh_day_index in merge step to update file index asynchronously * docs(cookbook): add optional SSH proxy configuration in README files - Introduce optional SSH proxy setup in auto-fin and daily_paper cookbooks - Provide instructions to enable outbound proxy via `daily_cookbook.yaml` and environment variables - Add `REME_PROXY_IP` and `REME_PROXY_ACCOUNT` environment variables descriptions in multiple README files - Update English and Chinese README and README_ZH documents with proxy details - Maintain consistent formatting of environment variable tables across documents * fix(file_io): include schema_version in hidden metadata keys - Added "schema_version" to _INDEX_HIDDEN_METADATA_KEYS in _daily_index.py - Updated _render_notes_block to always include additional keys regardless of schema_version fix(deps): move pproxy dependency to later in pyproject.toml - Removed pproxy from early dependencies list - Added pproxy back near the end of dependency list for better ordering fix(outbound_proxy): require pproxy package for ssh_http proxy - Added importlib.util check for pproxy package presence - Raise RuntimeError if pproxy is not installed when using SSH HTTP outbound proxy - Improved error message suggests installing reme-ai with 'core' extra * docs(readme): update News section with new Cookbook workflows - Clarify introduction of optional Cookbooks with Daily Paper and Auto Fin workflows - Update English README to reflect both paper discovery and file-native ETF event research - Revise Chinese README to include financial news and historical market data research capability - Maintain announcement of paper acceptance at Findings of ACL 2026 * feat(auto_fin): add calculation results to final Markdown output - Implement _calculation_results to summarize forecast for each ETF analyzed - Include program-calculated results in the JSON input for the Markdown report - Update YAML template to incorporate calculation results and adjust recommendation rules - Refine recommendation logic to rely on event impact judgments combined with calculation outputs - Modify tests to verify presence of calculation results and updated report content and format * up prompt * fix(keyword_index): ignore non-indexable chunks during keyword sync - Add is_indexable method to base and BM25 keyword index classes to check text tokenizability - Update local file store to exclude non-indexable chunks from expected document IDs to prevent rebuild - Fix JSONL chunker to correctly handle Unicode line separator U+2028 inside JSON strings without splitting - Add test to ensure non-empty but non-indexable chunk does not trigger keyword index rebuild - Add test to verify U+2028 character does not cause incorrect JSONL record splitting
609 lines
22 KiB
Python
609 lines
22 KiB
Python
"""Tests for the Claude Code agent wrapper."""
|
|
|
|
from dataclasses import replace
|
|
from itertools import count
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
|
|
from reme.components.agent_wrapper.as_agent_wrapper import AsAgentWrapper
|
|
from reme.components.agent_wrapper.cc_agent_wrapper import CcAgentWrapper
|
|
from reme.components.agent_wrapper.cc_session_store import CcFileSessionStore
|
|
from reme.components.application_context import ApplicationContext
|
|
from reme.components.outbound_proxy import FixedHttpOutboundProxy
|
|
from reme.enumeration import ChunkEnum, ComponentEnum
|
|
|
|
# pylint: disable=protected-access
|
|
|
|
|
|
def _wrapper(tmp_path: Path) -> CcAgentWrapper:
|
|
return CcAgentWrapper(app_context=ApplicationContext(workspace_dir=str(tmp_path)))
|
|
|
|
|
|
def _skill_roots(tmp_path: Path) -> tuple[Path, Path]:
|
|
return (
|
|
tmp_path / ".claude" / "skills",
|
|
tmp_path / "mem_session" / "claude_config" / "skills",
|
|
)
|
|
|
|
|
|
def test_ensure_claude_skill_dir_adds_selected_skills_without_replacing_existing(
|
|
tmp_path,
|
|
):
|
|
"""Selected workspace skills are added while unrelated Claude skills remain."""
|
|
project_skills = tmp_path / "skills"
|
|
(project_skills / "one").mkdir(parents=True)
|
|
(project_skills / "two").mkdir()
|
|
for name in ("one", "two"):
|
|
(project_skills / name / "SKILL.md").write_text(f"# {name}", encoding="utf-8")
|
|
config_dir = tmp_path / "mem_session" / "claude_config"
|
|
|
|
for root in _skill_roots(tmp_path):
|
|
existing = root / "existing"
|
|
existing.mkdir(parents=True)
|
|
(existing / "SKILL.md").write_text("existing", encoding="utf-8")
|
|
|
|
_wrapper(tmp_path)._ensure_claude_skill_dir(config_dir, ["one"])
|
|
|
|
for root in _skill_roots(tmp_path):
|
|
assert (root / "one").is_symlink()
|
|
assert (root / "one").resolve() == (project_skills / "one").resolve()
|
|
assert not (root / "two").exists()
|
|
assert (root / "existing" / "SKILL.md").read_text(encoding="utf-8") == "existing"
|
|
|
|
|
|
def test_ensure_claude_skill_dir_all_adds_each_project_skill(tmp_path):
|
|
"""The all selector creates child links instead of replacing the skills root."""
|
|
project_skills = tmp_path / "skills"
|
|
(project_skills / "one").mkdir(parents=True)
|
|
(project_skills / "two").mkdir()
|
|
for name in ("one", "two"):
|
|
(project_skills / name / "SKILL.md").write_text(f"# {name}", encoding="utf-8")
|
|
config_dir = tmp_path / "mem_session" / "claude_config"
|
|
|
|
_wrapper(tmp_path)._ensure_claude_skill_dir(config_dir, "all")
|
|
|
|
for root in _skill_roots(tmp_path):
|
|
assert root.is_dir()
|
|
assert not root.is_symlink()
|
|
assert {path.name for path in root.iterdir()} == {"one", "two"}
|
|
|
|
|
|
def test_ensure_claude_skill_dir_preserves_existing_directory_link(tmp_path):
|
|
"""An existing skills link is user-owned and remains untouched."""
|
|
project_skills = tmp_path / "skills"
|
|
(project_skills / "one").mkdir(parents=True)
|
|
(project_skills / "one" / "SKILL.md").write_text("# one", encoding="utf-8")
|
|
config_dir = tmp_path / "mem_session" / "claude_config"
|
|
legacy_root = tmp_path / ".claude" / "skills"
|
|
legacy_root.parent.mkdir(parents=True)
|
|
legacy_root.symlink_to(project_skills, target_is_directory=True)
|
|
|
|
_wrapper(tmp_path)._ensure_claude_skill_dir(config_dir, ["one"])
|
|
|
|
assert legacy_root.is_dir()
|
|
assert legacy_root.is_symlink()
|
|
assert (legacy_root / "one").resolve() == (project_skills / "one").resolve()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_file_session_store_conforms_to_latest_sdk(tmp_path):
|
|
"""The file store follows the SDK project/session/subkey contract."""
|
|
from claude_agent_sdk.testing import run_session_store_conformance
|
|
|
|
sequence = count()
|
|
await run_session_store_conformance(lambda: CcFileSessionStore(tmp_path / str(next(sequence))))
|
|
|
|
|
|
def test_ensure_claude_skill_dir_rejects_paths_as_skill_names(tmp_path):
|
|
"""Skill selectors cannot escape the project skills directory."""
|
|
(tmp_path / "skills").mkdir()
|
|
|
|
with pytest.raises(ValueError, match="Invalid skill name"):
|
|
_wrapper(tmp_path)._ensure_claude_skill_dir(tmp_path / "config", ["../outside"])
|
|
|
|
|
|
def test_configured_skills_use_latest_sdk_allowlist(tmp_path):
|
|
"""Selected ReMe skills are passed through using the SDK's list semantics."""
|
|
project_skills = tmp_path / "skills"
|
|
(project_skills / "one").mkdir(parents=True)
|
|
(project_skills / "two").mkdir()
|
|
for name in ("one", "two"):
|
|
(project_skills / name / "SKILL.md").write_text(f"# {name}", encoding="utf-8")
|
|
|
|
opts = _wrapper(tmp_path)._build_options("hello", skills=["one"])
|
|
|
|
assert opts.skills == ["one"]
|
|
for root in _skill_roots(tmp_path):
|
|
assert (root / "one").resolve() == (project_skills / "one").resolve()
|
|
assert not (root / "two").exists()
|
|
|
|
|
|
def test_configured_project_path_sources_skills_outside_workspace(tmp_path):
|
|
"""Claude Code links project skills while keeping sessions in the workspace."""
|
|
project = tmp_path / "project"
|
|
workspace = project / ".reme"
|
|
skill = project / "skills" / "serper-search"
|
|
skill.mkdir(parents=True)
|
|
(skill / "SKILL.md").write_text("# Serper Search", encoding="utf-8")
|
|
wrapper = CcAgentWrapper(
|
|
app_context=ApplicationContext(workspace_dir=str(workspace)),
|
|
project_path="..",
|
|
)
|
|
|
|
opts = wrapper._build_options("hello", skills=["serper-search"])
|
|
|
|
assert opts.cwd == project
|
|
assert (project / ".claude" / "skills" / "serper-search").resolve() == skill
|
|
assert (workspace / "mem_session" / "claude_config" / "skills" / "serper-search").resolve() == skill
|
|
|
|
|
|
def test_sdk_native_system_prompt_preset_is_preserved(tmp_path):
|
|
"""System prompt dictionaries pass directly to the latest SDK."""
|
|
opts = _wrapper(tmp_path)._build_options(
|
|
"hello",
|
|
system_prompt={
|
|
"type": "preset",
|
|
"preset": "claude_code",
|
|
"append": "custom prompt",
|
|
},
|
|
)
|
|
|
|
assert opts.system_prompt == {
|
|
"type": "preset",
|
|
"preset": "claude_code",
|
|
"append": "custom prompt",
|
|
}
|
|
|
|
|
|
def test_web_search_is_disallowed_by_default(tmp_path):
|
|
"""Claude Code keeps its normal tools except for web search."""
|
|
opts = _wrapper(tmp_path)._build_options("hello")
|
|
|
|
assert opts.disallowed_tools == ["WebSearch"]
|
|
|
|
|
|
def test_api_credentials_use_only_wrapper_config(tmp_path, monkeypatch):
|
|
"""Claude Code credentials do not fall back to ambient or shared LLM configuration."""
|
|
wrapper = _wrapper(tmp_path)
|
|
wrapper.app_context.app_config.environment = {
|
|
"ANTHROPIC_AUTH_TOKEN": "application-key",
|
|
"ANTHROPIC_BASE_URL": "https://application.example.test",
|
|
"TOOL_ENV": "preserved",
|
|
}
|
|
wrapper.app_context.app_config.components[ComponentEnum.AS_LLM] = {
|
|
"default": SimpleNamespace(
|
|
credential={
|
|
"api_key": "default-key",
|
|
"base_url": "https://default.example.test",
|
|
},
|
|
),
|
|
}
|
|
for name in ("ANTHROPIC_AUTH_TOKEN", "CLAUDE_CODE_API_KEY", "LLM_API_KEY"):
|
|
monkeypatch.setenv(name, "ambient-key")
|
|
for name in ("ANTHROPIC_BASE_URL", "CLAUDE_CODE_BASE_URL", "LLM_BASE_URL"):
|
|
monkeypatch.setenv(name, "https://ambient.example.test")
|
|
configured = wrapper._build_options( # pylint: disable=protected-access
|
|
"hello",
|
|
api_key="configured-key",
|
|
base_url="https://configured.example.test",
|
|
credential={"api_key": "nested-key", "base_url": "https://nested.example.test"},
|
|
)
|
|
assert configured.env["ANTHROPIC_AUTH_TOKEN"] == "configured-key"
|
|
assert configured.env["ANTHROPIC_BASE_URL"] == "https://configured.example.test"
|
|
assert configured.env["TOOL_ENV"] == "preserved"
|
|
|
|
empty = wrapper._build_options( # pylint: disable=protected-access
|
|
"hello",
|
|
credential={"api_key": "nested-key", "base_url": "https://nested.example.test"},
|
|
)
|
|
assert empty.env["ANTHROPIC_AUTH_TOKEN"] == ""
|
|
assert empty.env["ANTHROPIC_BASE_URL"] == ""
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_managed_proxy_is_injected_only_into_claude_bash_commands(tmp_path):
|
|
"""The Claude CLI keeps its model environment while Bash receives proxy exports."""
|
|
from claude_agent_sdk import HookMatcher
|
|
|
|
wrapper = _wrapper(tmp_path)
|
|
proxy = FixedHttpOutboundProxy(url="http://127.0.0.1:18080")
|
|
await proxy.start()
|
|
wrapper.app_context.components = {ComponentEnum.OUTBOUND_PROXY: {"default": proxy}}
|
|
|
|
async def existing_hook(_hook_input, _tool_use_id, _context):
|
|
return {}
|
|
|
|
existing = HookMatcher(matcher="Read", hooks=[existing_hook])
|
|
await wrapper.start()
|
|
opts = wrapper._build_options(
|
|
"hello",
|
|
api_key="configured-key",
|
|
base_url="https://configured.example.test",
|
|
hooks={"PreToolUse": [existing]},
|
|
)
|
|
|
|
assert opts.env["ANTHROPIC_AUTH_TOKEN"] == "configured-key"
|
|
assert opts.env["ANTHROPIC_BASE_URL"] == "https://configured.example.test"
|
|
assert "HTTP_PROXY" not in opts.env
|
|
assert opts.hooks["PreToolUse"][0] is existing
|
|
proxy_hook = opts.hooks["PreToolUse"][1]
|
|
assert proxy_hook.matcher == "Bash"
|
|
|
|
result = await proxy_hook.hooks[0](
|
|
{
|
|
"hook_event_name": "PreToolUse",
|
|
"tool_name": "Bash",
|
|
"tool_input": {"command": "python tushare_analysis.py", "timeout": 30},
|
|
"tool_use_id": "tool-1",
|
|
},
|
|
"tool-1",
|
|
{"signal": None},
|
|
)
|
|
updated_input = result["hookSpecificOutput"]["updatedInput"]
|
|
assert updated_input["command"].endswith("; python tushare_analysis.py")
|
|
assert updated_input["timeout"] == 30
|
|
for key in ("HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "http_proxy", "https_proxy", "all_proxy"):
|
|
assert f"{key}={proxy.http_url}" in updated_input["command"]
|
|
|
|
await wrapper.close()
|
|
await proxy.close()
|
|
|
|
|
|
def test_build_options_accepts_empty_output_schema(tmp_path):
|
|
"""An empty schema remains a valid structured-output request."""
|
|
opts = _wrapper(tmp_path)._build_options("hello", output_schema={})
|
|
|
|
assert opts.output_format == {"type": "json_schema", "schema": {}}
|
|
|
|
|
|
def test_build_options_uses_native_sessions_and_allows_file_checkpointing(tmp_path):
|
|
"""Local Claude transcripts remain the default and do not conflict with checkpoints."""
|
|
opts = _wrapper(tmp_path)._build_options("hello", enable_file_checkpointing=True)
|
|
|
|
assert opts.enable_file_checkpointing is True
|
|
assert opts.session_store is None
|
|
assert opts.env["CLAUDE_CONFIG_DIR"] == str(tmp_path / "mem_session" / "claude_config")
|
|
|
|
|
|
def test_build_options_preserves_explicit_session_store(tmp_path):
|
|
"""Callers can still opt into the SDK's external transcript mirror."""
|
|
store = CcFileSessionStore(tmp_path / "mirror")
|
|
|
|
opts = _wrapper(tmp_path)._build_options("hello", session_store=store)
|
|
|
|
assert opts.session_store is store
|
|
|
|
|
|
def test_job_tools_reject_non_mapping_mcp_config_instead_of_discarding_it(tmp_path, monkeypatch):
|
|
"""Adding ReMe tools never silently replaces an SDK MCP config path."""
|
|
wrapper = _wrapper(tmp_path)
|
|
job = SimpleNamespace(name="remember", description="Remember", parameters={})
|
|
monkeypatch.setattr(wrapper, "_resolve_job_tools", lambda _names: [job])
|
|
|
|
with pytest.raises(ValueError, match="mcp_servers to be a mapping"):
|
|
wrapper._build_options("hello", job_tools=["remember"], mcp_servers=tmp_path / "mcp.json")
|
|
|
|
|
|
def test_job_tools_merge_with_mapping_mcp_config(tmp_path, monkeypatch):
|
|
"""Existing MCP configuration remains reusable beside ReMe tools."""
|
|
wrapper = _wrapper(tmp_path)
|
|
job = SimpleNamespace(name="remember", description="Remember", parameters={})
|
|
monkeypatch.setattr(wrapper, "_resolve_job_tools", lambda _names: [job])
|
|
external = {"type": "http", "url": "https://mcp.example.test"}
|
|
mcp_servers = {"external": external}
|
|
allowed_tools = ["Read"]
|
|
|
|
first = wrapper._build_options(
|
|
"hello",
|
|
job_tools=["remember"],
|
|
mcp_servers=mcp_servers,
|
|
allowed_tools=allowed_tools,
|
|
)
|
|
second = wrapper._build_options(
|
|
"hello",
|
|
job_tools=["remember"],
|
|
mcp_servers=mcp_servers,
|
|
allowed_tools=allowed_tools,
|
|
)
|
|
|
|
assert mcp_servers == {"external": external}
|
|
assert allowed_tools == ["Read"]
|
|
for opts in (first, second):
|
|
assert opts.mcp_servers["external"] is external
|
|
assert opts.mcp_servers[wrapper.MCP_SERVER_NAME]["type"] == "sdk"
|
|
assert opts.allowed_tools == ["Read", "remember"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_preserves_falsy_structured_output(tmp_path, monkeypatch):
|
|
"""Falsy structured output is returned instead of being discarded."""
|
|
from claude_agent_sdk import ResultMessage
|
|
|
|
message = ResultMessage(
|
|
subtype="success",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=False,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
result="{}",
|
|
structured_output={"placeholder": True},
|
|
)
|
|
|
|
async def query(**_kwargs):
|
|
yield replace(message, structured_output={})
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
|
|
result = await _wrapper(tmp_path).reply("hello", output_schema={})
|
|
|
|
assert "structured_output" in result
|
|
assert result["structured_output"] == {}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_compact_session_uses_claude_command(tmp_path, monkeypatch):
|
|
"""Claude Code compaction uses its native slash command."""
|
|
wrapper = _wrapper(tmp_path)
|
|
calls = []
|
|
|
|
async def reply(inputs, **kwargs):
|
|
calls.append((inputs, kwargs))
|
|
return {"last_message": {"is_error": False}}
|
|
|
|
monkeypatch.setattr(wrapper, "reply", reply)
|
|
|
|
await wrapper.compact_session("session-1")
|
|
|
|
assert calls == [("/compact", {"resume": "session-1"})]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_agentscope_compact_session_forces_and_persists_compression(tmp_path, monkeypatch):
|
|
"""AgentScope compaction forces the threshold and persists new state."""
|
|
wrapper = AsAgentWrapper(as_llm="", app_context=ApplicationContext(workspace_dir=str(tmp_path)))
|
|
observed = {}
|
|
|
|
class FakeAgent:
|
|
"""Minimal AgentScope agent double."""
|
|
|
|
state = SimpleNamespace()
|
|
|
|
async def compress_context(self, config):
|
|
"""Capture the forced context configuration."""
|
|
observed["config"] = config
|
|
|
|
async def build_agent(inputs, **kwargs):
|
|
observed["build"] = (inputs, kwargs)
|
|
return FakeAgent(), inputs
|
|
|
|
async def dump_state(state):
|
|
observed["state"] = state
|
|
|
|
monkeypatch.setattr(wrapper, "_build_agent", build_agent)
|
|
monkeypatch.setattr(wrapper, "_dump_state", dump_state)
|
|
|
|
await wrapper.compact_session("session-1")
|
|
|
|
assert observed["build"] == (None, {"resume": "session-1"})
|
|
assert observed["config"].trigger_ratio == 1e-9
|
|
assert observed["state"] is FakeAgent.state
|
|
|
|
|
|
def test_error_result_with_success_subtype_is_not_suppressed():
|
|
"""Latest SDK can report API failures with subtype=success and is_error=True."""
|
|
from claude_agent_sdk import ResultMessage
|
|
|
|
message = ResultMessage(
|
|
subtype="success",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=True,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
errors=["upstream unavailable"],
|
|
api_error_status=529,
|
|
)
|
|
|
|
chunks = CcAgentWrapper._result_message_to_chunks(message)
|
|
|
|
error = next(chunk for chunk in chunks if chunk.chunk_type.value == "error")
|
|
assert error.chunk == "upstream unavailable"
|
|
assert error.metadata == {"api_error_status": 529}
|
|
|
|
|
|
def test_latest_sdk_server_tool_blocks_are_converted():
|
|
"""Server-side tools use the same unified call/result lifecycle."""
|
|
from claude_agent_sdk import AssistantMessage, ServerToolResultBlock
|
|
|
|
call = CcAgentWrapper._raw_event_to_chunk(
|
|
{
|
|
"type": "content_block_start",
|
|
"index": 0,
|
|
"content_block": {
|
|
"type": "server_tool_use",
|
|
"id": "tool-1",
|
|
"name": "web_search",
|
|
},
|
|
},
|
|
)
|
|
message = AssistantMessage(
|
|
content=[ServerToolResultBlock(tool_use_id="tool-1", content={"type": "web_search_result"})],
|
|
model="claude",
|
|
)
|
|
results = CcAgentWrapper._message_content_to_chunks(message, visible_tool_call_ids={"tool-1"})
|
|
|
|
assert call is not None and call.chunk_type.value == "tool_call"
|
|
assert len(results) == 1
|
|
assert results[0].chunk_type.value == "tool_result"
|
|
assert results[0].chunk == {
|
|
"tool_use_id": "tool-1",
|
|
"content": {"type": "web_search_result"},
|
|
}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_stream_emits_one_reply_end_for_normal_sdk_lifecycle(tmp_path, monkeypatch):
|
|
"""Message delta reports usage while message stop is the sole lifecycle end."""
|
|
from claude_agent_sdk import ResultMessage, StreamEvent
|
|
|
|
async def query(**_kwargs):
|
|
for event in (
|
|
{
|
|
"type": "message_start",
|
|
"message": {"id": "message-1", "role": "assistant"},
|
|
},
|
|
{
|
|
"type": "message_delta",
|
|
"delta": {"stop_reason": "end_turn"},
|
|
"usage": {"output_tokens": 2},
|
|
},
|
|
{"type": "message_stop"},
|
|
):
|
|
yield StreamEvent(uuid="event-1", session_id="session-1", event=event)
|
|
yield ResultMessage(
|
|
subtype="success",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=False,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
usage={"input_tokens": 1, "output_tokens": 2},
|
|
)
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
chunks = [chunk async for chunk in _wrapper(tmp_path).reply_stream("hello")]
|
|
|
|
assert sum(chunk.chunk_type == ChunkEnum.REPLY_END for chunk in chunks) == 1
|
|
delta_usage = next(chunk for chunk in chunks if chunk.metadata.get("stop_reason") == "end_turn")
|
|
assert delta_usage.chunk_type == ChunkEnum.USAGE
|
|
assert delta_usage.output_tokens == 2
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_stream_only_reports_rejected_rate_limit(tmp_path, monkeypatch):
|
|
"""Rate-limit warnings are informational; only rejected is an error."""
|
|
from claude_agent_sdk import RateLimitEvent, RateLimitInfo, ResultMessage
|
|
|
|
async def query(**_kwargs):
|
|
yield RateLimitEvent(
|
|
rate_limit_info=RateLimitInfo(status="allowed_warning"),
|
|
uuid="warning",
|
|
session_id="session-1",
|
|
)
|
|
yield RateLimitEvent(
|
|
rate_limit_info=RateLimitInfo(status="rejected"),
|
|
uuid="rejected",
|
|
session_id="session-1",
|
|
)
|
|
yield ResultMessage(
|
|
subtype="success",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=False,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
)
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
chunks = [chunk async for chunk in _wrapper(tmp_path).reply_stream("hello")]
|
|
|
|
errors = [chunk for chunk in chunks if chunk.chunk_type.value == "error"]
|
|
assert [chunk.chunk for chunk in errors] == ["Rate limit exceeded"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_stream_uses_error_result_as_terminal_error(tmp_path, monkeypatch):
|
|
"""The SDK's non-zero process exit does not duplicate an emitted error result."""
|
|
from claude_agent_sdk import ResultMessage
|
|
|
|
async def query(**_kwargs):
|
|
yield ResultMessage(
|
|
subtype="error_during_execution",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=True,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
errors=["tool failed"],
|
|
)
|
|
raise RuntimeError("Claude Code returned an error result: tool failed")
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
chunks = [chunk async for chunk in _wrapper(tmp_path).reply_stream("hello")]
|
|
|
|
errors = [chunk for chunk in chunks if chunk.chunk_type.value == "error"]
|
|
assert [chunk.chunk for chunk in errors] == ["tool failed"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_stream_does_not_hide_unrelated_error_after_error_result(tmp_path, monkeypatch):
|
|
"""Only the SDK's exact trailing process error is suppressed."""
|
|
from claude_agent_sdk import ResultMessage
|
|
|
|
async def query(**_kwargs):
|
|
yield ResultMessage(
|
|
subtype="error_during_execution",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=True,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
errors=["tool failed"],
|
|
)
|
|
raise RuntimeError("unrelated store failure")
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
|
|
with pytest.raises(RuntimeError, match="unrelated store failure"):
|
|
_ = [chunk async for chunk in _wrapper(tmp_path).reply_stream("hello")]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_reply_stream_reports_session_mirror_errors(tmp_path, monkeypatch):
|
|
"""The latest SDK's non-fatal mirror failures remain visible to callers."""
|
|
from claude_agent_sdk import MirrorErrorMessage, ResultMessage
|
|
|
|
async def query(**_kwargs):
|
|
yield MirrorErrorMessage(
|
|
subtype="mirror_error",
|
|
data={},
|
|
key={"project_key": "project", "session_id": "session-1"},
|
|
error="disk full",
|
|
)
|
|
yield ResultMessage(
|
|
subtype="success",
|
|
duration_ms=1,
|
|
duration_api_ms=1,
|
|
is_error=False,
|
|
num_turns=1,
|
|
session_id="session-1",
|
|
)
|
|
|
|
monkeypatch.setattr("claude_agent_sdk.query", query)
|
|
chunks = [chunk async for chunk in _wrapper(tmp_path).reply_stream("hello")]
|
|
|
|
diagnostics = [chunk for chunk in chunks if chunk.metadata.get("event") == "session_mirror_error"]
|
|
assert len(diagnostics) == 1
|
|
assert diagnostics[0].chunk_type == ChunkEnum.DATA
|
|
assert diagnostics[0].chunk == "Session mirror failed: disk full"
|
|
assert not any(chunk.chunk_type == ChunkEnum.ERROR for chunk in chunks)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
"wrapper_factory",
|
|
[
|
|
_wrapper,
|
|
lambda tmp_path: AsAgentWrapper(as_llm="", app_context=ApplicationContext(workspace_dir=str(tmp_path))),
|
|
],
|
|
)
|
|
@pytest.mark.parametrize("schema", [{}, {"type": "object"}])
|
|
async def test_reply_stream_rejects_output_schema(tmp_path, wrapper_factory, schema):
|
|
"""Streaming wrappers reject structured-output schemas consistently."""
|
|
wrapper = wrapper_factory(tmp_path)
|
|
|
|
with pytest.raises(NotImplementedError, match="Structured output is not supported"):
|
|
await anext(wrapper.reply_stream("hello", output_schema=schema))
|