mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-29 01:41:38 +00:00
* refactor(components): extract shared component state into mixin - Introduce ComponentMixin class with shared state for components and steps - Move identity, config, and vault path functionality to ComponentMixin - Update BaseComponent to inherit from ComponentMixin - Update BaseStep to inherit from ComponentMixin - Consolidate vault path helper methods in ComponentMixin - Remove duplicate vault path implementations from BaseComponent and BaseStep - Add ComponentMixin to components module exports * refactor(file_io): implement path locks cache eviction mechanism - Add _PATH_LOCKS_MAX constant set to 1024 for cache size limit - Implement cache eviction logic when locks exceed maximum capacity - Remove half of unlocked entries when cache limit is reached - Use list comprehension to identify unlocked locks for removal - Maintain existing path normalization and locking behavior fix(edit): correct method call from public to private fail method - Change self.fail to self._fail for internal error handling - Maintain consistent private method usage within class fix(mcp_client): change pop to get for optional command and args - Replace kwargs.pop with kwargs.get to avoid removing keys - Preserve original kwargs dictionary contents - Maintain default empty string and list values feat(reme): add client backend validation with error raising - Check if client_cls is None before instantiation - Raise ValueError with descriptive message for unknown backends - Provide clear error feedback for invalid backend configurations * fix(components): move directory creation to start method - Moved component_metadata_path.mkdir call from __init__ to _start in base_keyword_index - Moved component_metadata_path.mkdir call from __init__ to _start in local_file_graph - Moved component_metadata_path.mkdir call from __init__ to _start in local_file_store - Ensures directory creation happens after component initialization - Prevents potential issues with path creation during object construction * fix(steps): replace assertions with runtime errors for app_context validation - Replace assert statements with explicit RuntimeError exceptions when app_context is None - Add descriptive error messages for better debugging when resolving components - Replace assert in resolve_component method with proper exception handling - Replace assert in get_file_parser method with proper exception handling - Maintain same functionality while improving error reporting clarity * refactor(file_io): split file IO utilities into modular components - Move daily note helpers to separate _daily_index module - Extract path validation and resolution to new _path module - Remove unused code and imports from _file_io module - Update import statements across affected modules - Introduce WikilinkHandler utility for link parsing - Replace regex-based link extraction with WikilinkHandler - Add integration JSONL files to gitignore - Consolidate file locking mechanism in _file_io module * style(formatter): fix spacing issues in file IO and chunked file parser - Fixed whitespace around colon in slice notation in file_io.py - Corrected spacing around colon in slice notation in chunked_file_parser.py - Applied consistent formatting for array slicing operations - Improved code readability by standardizing space placement in ranges * refactor(steps): replace property-based component resolution with Ref descriptor - Introduce Ref descriptor class for lazy component dependency resolution - Replace _resolve method and individual properties with Ref descriptors - Add as_llm, as_llm_formatter, as_token_counter, file_store, and embedding Ref attributes - Remove legacy property methods and resolve logic from BaseStep - Add cache clearing mechanism for Ref values during step calls - Update UpdateCatalogStep to use Ref instead of property-based resolution
190 lines
6.1 KiB
Python
190 lines
6.1 KiB
Python
"""Encoding-aware file IO, output truncation, and per-path write locks."""
|
|
|
|
import asyncio
|
|
from pathlib import Path
|
|
from typing import Iterable
|
|
|
|
import aiofiles
|
|
import aiofiles.os
|
|
|
|
from ...constants import DEFAULT_MAX_BYTES, MAX_FILE_READ_BYTES, TRUNCATION_NOTICE_MARKER
|
|
from ...utils import get_logger
|
|
|
|
logger = get_logger()
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# In-process per-path write lock.
|
|
# ---------------------------------------------------------------------------
|
|
_PATH_LOCKS_MAX = 1024
|
|
_PATH_LOCKS: dict[str, asyncio.Lock] = {}
|
|
_PATH_LOCKS_REGISTRY = asyncio.Lock()
|
|
|
|
|
|
async def get_path_lock(target: Path) -> asyncio.Lock:
|
|
"""Return the asyncio.Lock for ``target``; created lazily on first request."""
|
|
key = str(target)
|
|
async with _PATH_LOCKS_REGISTRY:
|
|
lock = _PATH_LOCKS.get(key)
|
|
if lock is None:
|
|
if len(_PATH_LOCKS) >= _PATH_LOCKS_MAX:
|
|
to_remove = [k for k, v in _PATH_LOCKS.items() if not v.locked()]
|
|
for k in to_remove[: len(_PATH_LOCKS) // 2]:
|
|
del _PATH_LOCKS[k]
|
|
lock = asyncio.Lock()
|
|
_PATH_LOCKS[key] = lock
|
|
return lock
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Encoding detection
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_STANDARD_TEXT_EXTS = {
|
|
".md",
|
|
".py",
|
|
".js",
|
|
".ts",
|
|
".json",
|
|
".yaml",
|
|
".yml",
|
|
".html",
|
|
".css",
|
|
".xml",
|
|
".log",
|
|
".conf",
|
|
".ini",
|
|
".txt",
|
|
".sh",
|
|
}
|
|
_NON_STANDARD_EXTS = {".csv", ".bat", ".cmd", ".reg"}
|
|
|
|
|
|
def _try_decode(data: bytes, encodings: Iterable[str]) -> tuple[str, str] | None:
|
|
"""Return ``(text, encoding)`` for the first encoding that decodes ``data`` cleanly."""
|
|
for enc in encodings:
|
|
try:
|
|
return data.decode(enc), enc
|
|
except (UnicodeDecodeError, LookupError):
|
|
continue
|
|
return None
|
|
|
|
|
|
def _decode_known_file(data: bytes, file_extension: str) -> tuple[str, str]:
|
|
"""Decode file bytes using the extension as a hint. Returns ``(text, encoding)``."""
|
|
if data.startswith(b"\xef\xbb\xbf"):
|
|
return data.decode("utf-8-sig"), "utf-8-sig"
|
|
if data.startswith((b"\xff\xfe", b"\xfe\xff")):
|
|
try:
|
|
return data.decode("utf-16"), "utf-16"
|
|
except UnicodeDecodeError:
|
|
pass
|
|
|
|
ext = (file_extension or "").lower()
|
|
|
|
if ext in _STANDARD_TEXT_EXTS:
|
|
try:
|
|
return data.decode("utf-8-sig"), "utf-8"
|
|
except UnicodeDecodeError:
|
|
pass
|
|
|
|
if ext in _NON_STANDARD_EXTS:
|
|
result = _try_decode(data, ("utf-8-sig", "gbk"))
|
|
if result is not None:
|
|
text, enc = result
|
|
return text, "utf-8" if enc == "utf-8-sig" else enc
|
|
|
|
return data.decode("utf-8", errors="replace"), "utf-8"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# File read / write
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
async def read_file_safe(file_path, max_bytes: int = MAX_FILE_READ_BYTES) -> tuple[str, str]:
|
|
"""Read file in byte mode and decode using extension-aware strategy.
|
|
|
|
Returns ``(text, encoding)``.
|
|
"""
|
|
stat = await aiofiles.os.stat(str(file_path))
|
|
read_size = min(stat.st_size, max_bytes)
|
|
async with aiofiles.open(str(file_path), "rb") as f:
|
|
data = await f.read(read_size)
|
|
return _decode_known_file(data, Path(file_path).suffix)
|
|
|
|
|
|
async def detect_file_encoding(file_path, sniff_bytes: int = 8192) -> str:
|
|
"""Detect the encoding of an existing file so writes can preserve it."""
|
|
try:
|
|
async with aiofiles.open(str(file_path), "rb") as f:
|
|
data = await f.read(sniff_bytes)
|
|
except Exception:
|
|
return "utf-8"
|
|
_, enc = _decode_known_file(data, Path(file_path).suffix)
|
|
return enc
|
|
|
|
|
|
async def write_file_safe(file_path: Path, content: str | bytes, encoding: str = "utf-8") -> None:
|
|
"""Write ``content`` to ``file_path`` in binary mode; creates parent dirs."""
|
|
file_path.parent.mkdir(parents=True, exist_ok=True)
|
|
if isinstance(content, str):
|
|
try:
|
|
payload = content.encode(encoding)
|
|
except (UnicodeEncodeError, LookupError):
|
|
logger.warning(
|
|
"write_file_safe: %r cannot encode all chars, falling back to utf-8",
|
|
encoding,
|
|
)
|
|
payload = content.encode("utf-8")
|
|
else:
|
|
payload = content
|
|
async with aiofiles.open(str(file_path), "wb") as f:
|
|
await f.write(payload)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Output truncation
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def truncate_text_output(
|
|
text: str,
|
|
*,
|
|
start_line: int = 1,
|
|
total_lines: int = 0,
|
|
max_bytes: int = DEFAULT_MAX_BYTES,
|
|
file_path: str | None = None,
|
|
encoding: str = "utf-8",
|
|
) -> str:
|
|
"""Truncate text by bytes preserving line integrity; append a continuation notice."""
|
|
if not text or max_bytes <= 0:
|
|
return text
|
|
|
|
try:
|
|
text_bytes = text.encode(encoding)
|
|
if len(text_bytes) <= max_bytes:
|
|
return text
|
|
|
|
truncated = text_bytes[:max_bytes]
|
|
result = truncated.decode(encoding, errors="ignore")
|
|
newline_count = result.count("\n")
|
|
next_line = start_line + max(1, newline_count)
|
|
|
|
if next_line <= total_lines:
|
|
read_from = next_line
|
|
elif start_line < total_lines:
|
|
read_from = total_lines
|
|
else:
|
|
return result
|
|
|
|
notice = (
|
|
TRUNCATION_NOTICE_MARKER + f"\nThe output above was truncated."
|
|
f"\nThe full content is saved to the file and contains {total_lines} lines in total."
|
|
f"\nThis excerpt starts at line {start_line} and covers the next {max_bytes} bytes."
|
|
f"\nIf the current content is not enough, call `read` with file={file_path or ''} "
|
|
f"start_line={read_from} to read more."
|
|
)
|
|
return result + notice
|
|
except Exception:
|
|
logger.warning("truncate_text_output failed, returning original text", exc_info=True)
|
|
return text
|