Refine sidecar bootstrap guidance and retry helpers

This commit is contained in:
Admin 2026-04-09 02:35:54 +08:00
parent 5b799551b0
commit 659d2da39e
2 changed files with 83 additions and 38 deletions

View file

@ -45,3 +45,24 @@ Recommended user-facing invocation:
```text
对当前这轮工作做一次 sidecar 自进化。不要改代码,不要接管任务。请调用 openspace_evolution.evolve_from_context基于当前对话、git diff 和关键改动,自动提炼 task/summary最多生成 1 个高复用 skill并告诉我 skill 名称、路径、为什么值得保留。
```
## New Project Bootstrap
If the user wants this sidecar workflow in a new repository, treat it as a project bootstrap task first.
Bootstrap order:
- Add or update a project launcher before relying on sidecar evolution.
- Point `OPENSPACE_WORKSPACE` at the new repository root.
- Keep the user's main Codex Desktop workflow unchanged.
- Do not modify global `~/.codex` defaults unless the user explicitly asks.
Expected bootstrap outputs:
- a project-level launcher such as `scripts/codex-desktop-evolution`
- a project-level `AGENTS.md` section documenting the sidecar trigger phrases
- sidecar skill output routed to `~/.codex-openspace-desktop/projects/<project-name>/skills`
When a user asks to initialize a new project for this workflow, default to:
- creating the launcher first
- wiring `OPENSPACE_WORKSPACE` to the repository root
- preserving the normal Codex Desktop login path
- only then enabling phrases like `sidecar 自进化一下`

View file

@ -179,6 +179,63 @@ async def _openai_compat_stream_completion(
)
def _classify_retryable_error(error_text: str) -> tuple[bool, bool, bool]:
is_rate_limit = any(
keyword in error_text
for keyword in ['rate limit', 'rate_limit', 'too many requests', '429']
)
is_overloaded = any(
keyword in error_text
for keyword in ['overloaded', '500', '502', '503', '504', 'internal server error', 'service unavailable']
)
is_connection_error = any(
keyword in error_text
for keyword in ['cannot connect', 'connection refused', 'connection reset',
'connectionerror', 'timeout', 'name resolution',
'temporary failure', 'network unreachable']
)
return is_rate_limit, is_overloaded, is_connection_error
def _retry_backoff(error_text: str, attempt: int) -> tuple[str, int]:
is_rate_limit, _is_overloaded, is_connection_error = _classify_retryable_error(error_text)
if is_rate_limit:
return "Rate limit", 60 + (attempt * 30)
if is_connection_error:
return "Connection", min(10 * (2 ** attempt), 60)
return "Server overload", min(5 * (2 ** attempt), 60)
def _should_retry_completion_error(exc: Exception) -> bool:
error_text = str(exc).lower()
return any(
keyword in error_text
for keyword in [
'rate limit', 'rate_limit', 'too many requests', '429',
'overloaded', '500', '502', '503', '504', 'internal server error',
'service unavailable', 'cannot connect', 'connection refused',
'connection reset', 'connectionerror', 'timeout', 'name resolution',
'temporary failure', 'network unreachable'
]
)
def _should_use_nonstream_compat_only(completion_kwargs: Optional[Dict] = None) -> bool:
return False
async def _execute_openai_compat_completion(completion_kwargs: Dict, timeout: float):
return await _openai_compat_stream_completion(
model=completion_kwargs["model"],
messages=completion_kwargs["messages"],
timeout=timeout,
litellm_kwargs=completion_kwargs,
tools=completion_kwargs.get("tools"),
tool_choice=completion_kwargs.get("tool_choice"),
reasoning_effort=completion_kwargs.get("reasoning_effort"),
)
def _sanitize_schema(params: Dict) -> Dict:
"""Sanitize tool parameter schema to comply with Claude API requirements.
@ -724,15 +781,7 @@ class LLMClient:
# Add timeout to the completion call
if _should_use_openai_stream_compat(completion_kwargs):
response = await asyncio.wait_for(
_openai_compat_stream_completion(
model=completion_kwargs["model"],
messages=completion_kwargs["messages"],
timeout=self.timeout,
litellm_kwargs=completion_kwargs,
tools=completion_kwargs.get("tools"),
tool_choice=completion_kwargs.get("tool_choice"),
reasoning_effort=completion_kwargs.get("reasoning_effort"),
),
_execute_openai_compat_completion(completion_kwargs, self.timeout),
timeout=self.timeout,
)
else:
@ -757,35 +806,10 @@ class LLMClient:
last_exception = e
error_str = str(e).lower()
# Check if it's a retryable error
is_rate_limit = any(
keyword in error_str
for keyword in ['rate limit', 'rate_limit', 'too many requests', '429']
)
is_overloaded = any(
keyword in error_str
for keyword in ['overloaded', '500', '502', '503', '504', 'internal server error', 'service unavailable']
)
is_connection_error = any(
keyword in error_str
for keyword in ['cannot connect', 'connection refused', 'connection reset',
'connectionerror', 'timeout', 'name resolution',
'temporary failure', 'network unreachable']
)
if attempt < self.max_retries - 1 and (is_rate_limit or is_overloaded or is_connection_error):
if is_rate_limit:
backoff_delay = 60 + (attempt * 30) # 60s, 90s, 120s
error_type = "Rate limit"
elif is_connection_error:
backoff_delay = min(10 * (2 ** attempt), 60) # 10s, 20s, 40s, max 60s
error_type = "Connection"
else:
backoff_delay = min(5 * (2 ** attempt), 60) # 5s, 10s, 20s, max 60s
error_type = "Server overload"
is_rate_limit, is_overloaded, is_connection_error = _classify_retryable_error(error_str)
if attempt < self.max_retries - 1 and _should_retry_completion_error(e):
error_type, backoff_delay = _retry_backoff(error_str, attempt)
self._logger.warning(
f"{error_type} error (attempt {attempt + 1}/{self.max_retries}), "
f"waiting {backoff_delay}s before retry..."