ReMe/reme/steps/evolve/dream/schema.py
jinliyl a3bd81bde2
Update version to 0.4.0.2 and improve tokenizer index handling (#290)
* fix(core): update version number to 0.4.0.1

- Incremented version from 0.4.0.0 to 0.4.0.1 in __init__.py

* fix(index): remove stopwords path from tokenizer config and add keyword index repair

- Remove stopwords_path from tokenizer config to prevent index forking by install path
- Add _sync_keyword_index_from_chunks method to repair keyword index when persisted state mismatches
- Implement test for keyword index repair from persisted chunks when missing
- Add test to verify tokenizer fingerprint ignores stopwords absolute path
- Update version from 0.4.0.1 to 0.4.0.2

* feat(dream): add scan_days parameter to dream extraction process

- Add scan_days configuration option to default.yaml with default value of 2
- Implement recent_dates utility function to calculate date ranges for scanning
- Modify DreamExtractStep to scan multiple days based on scan_days parameter
- Update dream extraction to process files across multiple dates instead of single day
- Extend DreamState schema to include dates and scan_days fields
- Update DreamTopicsStep to handle multi-day topic processing
- Modify finish step to checkpoint files from all scanned dates
- Add comprehensive tests for multi-day scanning functionality
- Update prompt templates to include scan dates information
- Refactor topics writing logic to target specific date rather than current date
2026-06-24 15:01:07 +08:00

95 lines
3.5 KiB
Python

"""Shared auto-dream schemas."""
from typing import Literal
from pydantic import BaseModel, Field
BUCKETS: tuple[str, ...] = ("procedure", "personal", "wiki")
Bucket = Literal["procedure", "personal", "wiki"]
class DreamUnit(BaseModel):
"""One cross-file memory unit emitted by global extract."""
name: str = Field(description="Short kebab-case handle for the abstraction.")
bucket: str = Field(description="procedure, personal, or wiki; unknown values route to wiki.")
summary: str = Field(description="Grounded abstraction summary with evidence pointers.")
paths: list[str] = Field(default_factory=list, description="Workspace-relative source paths.")
class DreamTopic(BaseModel):
"""One topic candidate emitted by global extract."""
title: str = Field(description="Specific user-interest topic title.")
reason: str = Field(description="Why this topic may interest the user.")
evidence: str = Field(description="Grounded evidence pointer.")
keywords: list[str] = Field(default_factory=list, description="Keywords for de-duplication.")
paths: list[str] = Field(default_factory=list, description="Workspace-relative source paths.")
class DreamExtractOutput(BaseModel):
"""Structured output for ``dream_extract_step``."""
units: list[DreamUnit] = Field(default_factory=list)
topics: list[DreamTopic] = Field(default_factory=list)
class IntegrateOutcome(BaseModel):
"""Structured output for one unit integration."""
action: Literal["CREATE", "CORROBORATE", "REFINE", "CORRECT"] = Field(description="Write decision.")
target_path: str = Field(description="Digest path written or edited.")
note: str = Field(default="", description="Short summary of what landed.")
class TopicSelectionOutput(BaseModel):
"""Structured output for daily topic selection."""
topics: list[DreamTopic] = Field(default_factory=list)
class ProactiveResult(BaseModel):
"""Result of reading daily interest topics."""
date: str = ""
path: str = ""
topics: list[dict] = Field(default_factory=list)
content: str = ""
skipped: bool = False
error: str = ""
summary: str = ""
class DreamState(BaseModel):
"""Shared state passed across the four dream steps."""
date: str = ""
dates: list[str] = Field(default_factory=list)
scan_days: int = 2
hint: str = ""
daily_dir: str = ""
workspace: str = ""
files_scanned: int = 0
files_unchanged: int = 0
files_changed: int = 0
files_deleted: int = 0
changed_paths: list[str] = Field(default_factory=list)
unchanged_paths: list[str] = Field(default_factory=list)
deleted_paths: list[str] = Field(default_factory=list)
existing: dict[str, float] = Field(default_factory=dict)
indexed: dict[str, float] = Field(default_factory=dict)
units: list[dict] = Field(default_factory=list)
topics: list[dict] = Field(default_factory=list)
extract_summary: str = ""
integrate_results: list[dict] = Field(default_factory=list)
nodes_created: list[str] = Field(default_factory=list)
nodes_updated: list[str] = Field(default_factory=list)
failed_units: list[dict] = Field(default_factory=list)
failed_paths: list[str] = Field(default_factory=list)
interests_path: str = ""
interests_paths: list[str] = Field(default_factory=list)
topics_written: int = 0
topic_error: str = ""
checkpoint_paths: list[str] = Field(default_factory=list)
errors: list[str] = Field(default_factory=list)
summary: str = ""