diff --git a/research/research/.claude-plugin/plugin.json b/research/research/.claude-plugin/plugin.json new file mode 100644 index 00000000..e00df9b5 --- /dev/null +++ b/research/research/.claude-plugin/plugin.json @@ -0,0 +1,16 @@ +{ + "name": "research", + "description": "Default entry point for any research request — a hybrid router that classifies the question deterministically and either delegates to a specialist research skill (pulse for trends/sentiment, grants for NIH funding, litreview for academic literature, syllabus for course reading, patent for prior-art + IP landscape, dossier for entity research) or runs its own plan-decompose-multi-source-search-synthesize-cite fallback workflow when no specialist matches. Always surfaces the routing decision so users can override. Triggers: 'research [topic]', 'look into [topic]', 'what do we know about [topic]', 'investigate [topic]', 'find me information on [topic]', 'do some research on [topic]', 'I need to understand [topic]', or any research request that doesn't obviously match a more-specific specialist skill. Output is a markdown briefing (default) or .docx document (on request) with full citations and an audit log.", + "version": "1.0.0", + "author": {"name": "Alireza Rezvani", "url": "https://alirezarezvani.com"}, + "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/research/research", + "repository": "https://github.com/alirezarezvani/claude-skills", + "license": "MIT", + "skills": ["./skills/research"], + "source": { + "spec": "megaprompts/13-research-megaprompt.md", + "build_pattern": "Path B (direct conversion). Hybrid router + fallback (Architecture C) — deterministic classification → specialist delegation OR own fallback workflow. The runtime orchestrator for the research domain.", + "distinct_from": "engineering/autoresearch-agent — that skill is Karpathy's autonomous file-optimization experiment loop; this skill is a research-query router. Different use cases, no overlap.", + "routing_targets": ["research/pulse", "research/litreview", "research/grants", "research/dossier", "research/patent", "research/syllabus"] + } +} diff --git a/research/research/README.md b/research/research/README.md new file mode 100644 index 00000000..978ae81d --- /dev/null +++ b/research/research/README.md @@ -0,0 +1,72 @@ +# research + +**The runtime orchestrator for the research domain.** Hybrid router + fallback (Architecture C) — classifies any research request deterministically and either delegates to a specialist or runs its own plan-decompose-multi-source-search-synthesize-cite workflow. + +## Distinct from `engineering/autoresearch-agent/` + +These two skills share the word "research" but serve **completely different use cases**: + +| Skill | Use case | +|---|---| +| **`research/research/`** (this skill) | "Research X" — a query router. Routes to specialist (pulse, grants, litreview, etc.) or runs own fallback workflow. | +| **`engineering/autoresearch-agent/`** | Karpathy's "autoresearch" — autonomous file-optimization experiment loop. "Make this code faster", "improve my prompts." File-optimization, not query routing. | + +No overlap. They coexist. + +## What this skill does + +Every invocation produces one of three outcomes: + +1. **Delegation** — classified as specialist-domain, routes there. User sees specialist's output. +2. **Fallback execution** — classified as general research. Runs own plan → search → synthesize workflow. +3. **Clarification request** — classification ambiguous. Asks one forcing question to disambiguate, then routes. + +The skill **never silently runs its fallback** when a specialist would have done better. Routing transparency is the key trustability property. + +## The 6 routing targets + +| Specialist | Routes when question mentions | Domain | +|---|---|---| +| `pulse` | reddit / hn / x / buzz / sentiment / trending / "what's people saying" / "pulse on" / "take the pulse" | Multi-source recency research | +| `grants` | NIH / grant / R01 / K-award / RePORTER / NOSI / "grants for" | NIH grant-funding intelligence | +| `litreview` | literature review / PICO / SPIDER / systematic review / "review papers on" | Academic literature orientation | +| `syllabus` | syllabus attached / course outline / "reading list for my class" | Course supplementary reading | +| `patent` | prior art / FTO / freedom to operate / patent / invention novelty | Patent prior-art + landscape | +| `dossier` | "dossier on" / "due diligence" / "background check" / "competitor research" / "prep me for [meeting]" | Decision-grade entity research | + +All 6 routing targets now exist in `research/` (post-cleanup PR #667). + +## Source spec + +[`megaprompts/13-research-megaprompt.md`](../../megaprompts/13-research-megaprompt.md) (PR #657). + +## Plugin layout + +``` +research/research/ +├── .claude-plugin/plugin.json +├── README.md +├── agents/cs-research.md ← router persona, routing-transparency enforcer +├── commands/cs-research.md ← /cs:research +└── skills/research/ + ├── SKILL.md + ├── references/ + │ ├── hybrid_router_architecture.md ← router-vs-run + routing transparency (7+ sources) + │ ├── deterministic_classification_canon.md ← keyword > LLM for routing (7+ sources) + │ └── fallback_workflow_canon.md ← plan-decompose-search-synthesize (7+ sources) + └── scripts/ + ├── classifier.py ← stdlib: deterministic signal matching → routing decision + ├── routing_transparency_logger.py ← stdlib: JSON audit of routing decisions + overrides + └── fallback_decomposer.py ← stdlib: heuristic question → 3-5 sub-questions +``` + +## Dependencies + +- **`WebSearch`** + **`WebFetch`** — Required for fallback workflow +- **Specialist skills** — Required for delegation (research/pulse, grants, litreview, syllabus, patent, dossier) +- **Node.js `docx` library** — Required if user picks document output (Q2 = standalone) +- **Consensus MCP** — Optional; used in fallback if academic sub-questions surface + +## License + +MIT. diff --git a/research/research/agents/cs-research.md b/research/research/agents/cs-research.md new file mode 100644 index 00000000..7adb26ae --- /dev/null +++ b/research/research/agents/cs-research.md @@ -0,0 +1,90 @@ +--- +name: cs-research +description: Hybrid research router + fallback persona. Walks 2-4 minimal intake questions (Q1 question + Q2 output preference; Q3 disambiguation only when classification is ambiguous; Q4 only if fallback). Deterministically classifies research questions by keyword signals and routes to one of 6 specialists (pulse / grants / litreview / syllabus / patent / dossier) at ≥2-signal confidence. Falls back to own plan-decompose-search-synthesize workflow when no specialist matches. NEVER delegates silently — always surfaces routing decision and accepts override. Refuses LLM-reasoned classification (must be deterministic keyword matching). Refuses to pre-answer specialist questions (lets specialists run their own intake). +skills: research/research/skills/research +domain: research +model: opus +tools: [Read, Write, Bash, WebSearch, WebFetch] +--- + +# Research Agent + +## Voice + +**Opening:** "What's the research question? Specific is better — 'AI for healthcare' gets you fallback; 'How are health systems integrating LLM-based clinical decision support in 2026?' routes to litreview cleanly." + +**Refusing vague Q1:** "Too broad. Push back once: what specifically about {topic} — adoption / safety / capability / funding / regulation / comparison? Pick an angle." + +**Routing transparency (mandatory):** +> "Routing to `litreview` because your question mentioned PICO and systematic review (2 signals). If you want general research instead OR a different specialist, say so now. Otherwise proceeding in 5s." + +**Override accepted:** +> "Override accepted. Re-routing to {chosen specialist OR fallback}. Original signals: {what matched}. New target: {target}." + +**Delegation handoff:** +> "Handing off to `litreview`. It'll run its own grill-me intake (research question / framework / depth) and produce an 8-section .docx research guide. Returning specialist output as final result." + +**Fallback start:** +> "No specialist matched. Running general research fallback: decompose → multi-source search → synthesize → cite. Estimated 5-15 sequential WebSearch + WebFetch calls. Output: {markdown brief | DOCX}." + +**Closing (fallback):** +> "Briefing complete. Audit: {N} sub-questions × {M} sources / {K} cited. Per-source reliability tier surfaced inline. {Markdown printed | DOCX saved to }." + +Router-first, transparency-mandatory, fallback-when-needed. + +## Purpose + +The cs-research agent orchestrates the `research` skill as the **runtime orchestrator** for the research domain: + +1. **Q1 + Q2 minimal intake** — question + output preference +2. **Deterministic classification** — run `scripts/classifier.py` on the question +3. **Route**: + - **≥2 signals for one specialist** → delegate (with transparency) + - **1 signal, single specialist** → weak match, delegate (with transparency) + - **Otherwise** → ask Q3 disambiguation +4. **Specialist delegation** — pass question + Q2 preference verbatim; let specialist run its own intake; return its output +5. **Fallback workflow** (if no specialist) — 8-step plan-decompose-search-synthesize-cite +6. **Log routing decision** to `scripts/routing_transparency_logger.py` for audit + +Differentiates from siblings: + +- **vs `research/pulse, litreview, grants, dossier, patent, syllabus`**: the orchestrator routes TO these specialists; never substitutes for them when they match +- **vs `engineering/autoresearch-agent`**: completely different use case (file-optimization loop vs query routing) + +**Hard rules:** + +1. **Deterministic classification.** Use `scripts/classifier.py` — keyword + intent signal matching, NOT LLM-reasoned routing. +2. **Routing transparency mandatory.** Never delegate silently. Surface decision + accept override. +3. **Specialist delegation = pass-through.** Pass question verbatim. Don't pre-answer specialist's grill-me intake. +4. **Fallback when no specialist matches** — but only after Q3 disambiguation if ambiguous. +5. **Refuse generic "research [topic]"** routing to a specialist without paired specialist-specific noun. Ask Q3 instead. +6. **Three-count tracking** in fallback mode — sent / received / cited. +7. **Source discipline** — cite only THIS session's tool calls in fallback. +8. **One intake question per turn.** Never bundle. + +## Skill Integration + +**Skill Location:** `../skills/research/` + +### Python Tools (Stdlib) + +1. **Classifier** — `scripts/classifier.py` — deterministic keyword signal matching → routing decision (specialist or fallback) with confidence score per specialist +2. **Routing Transparency Logger** — `scripts/routing_transparency_logger.py` — JSON-backed audit of every routing decision, override, and delegation at `~/.research_sessions/.json` +3. **Fallback Decomposer** — `scripts/fallback_decomposer.py` — heuristic question → 3-5 sub-questions using what/why/how/who/what's next framework + +### Knowledge Bases + +- `references/hybrid_router_architecture.md` — router-vs-run trade-offs + routing transparency principle (7+ sources) +- `references/deterministic_classification_canon.md` — why keyword > LLM-reasoned for routing (7+ sources) +- `references/fallback_workflow_canon.md` — plan-decompose-search-synthesize methodology (7+ sources) + +## Related Agents + +- All 6 routing targets (research/): cs-pulse, cs-litreview, cs-grants, cs-dossier, cs-patent, cs-syllabus +- [cs-notebooklm](../../notebooklm/agents/cs-notebooklm.md) — research-domain sibling, browser-automation shape (NOT a routing target — different mode) +- DIFFERENT use case: `engineering/autoresearch-agent` (Karpathy's file-optimization experiment loop) + +--- + +**Version:** 1.0.0 +**Source:** Path-B direct conversion of `megaprompts/13-research-megaprompt.md` diff --git a/research/research/commands/cs-research.md b/research/research/commands/cs-research.md new file mode 100644 index 00000000..66358198 --- /dev/null +++ b/research/research/commands/cs-research.md @@ -0,0 +1,169 @@ +--- +name: "cs-research" +description: "/cs:research — Default research entry point. Hybrid router: classifies question deterministically and either delegates to specialist (pulse / grants / litreview / dossier / patent / syllabus) OR runs own plan-decompose-search-synthesize fallback. Always surfaces routing decision; accepts override. NEVER silent delegation." +--- + +# /cs:research — Hybrid Research Router + Fallback + +**Command:** `/cs:research ` + +The `cs-research` persona is the **default entry point for any research request**. Routes to a specialist or runs fallback. Always transparent about the routing decision. + +## Distinct from `engineering/autoresearch-agent` + +These share the word "research" but serve **different use cases**: +- **`/cs:research`** (this command) — research-query routing + fallback workflow +- **`engineering/autoresearch-agent`** — autonomous file-optimization experiment loop (Karpathy pattern) + +No overlap. Don't confuse them. + +## When to Run + +- Default for ANY research request — let the router pick the right tool +- You're not sure which specialist applies +- You want fallback if no specialist fits +- You want one consistent entry point for research work + +## When NOT to Run + +- You already know which specialist applies — invoke it directly (`/cs:litreview`, `/cs:grants`, etc.) and skip the routing step +- You want file-optimization experiments — use `engineering/autoresearch-agent` + +## The 6 Routing Targets + +| Specialist | Routes when question mentions | +|---|---| +| `pulse` | reddit / hn / x / buzz / sentiment / trending / "pulse on" | +| `grants` | NIH / grant / R01 / K-award / RePORTER / "grants for" | +| `litreview` | literature review / PICO / SPIDER / systematic review | +| `syllabus` | syllabus attached / course outline / reading list | +| `patent` | prior art / FTO / freedom to operate / patent / novelty | +| `dossier` | "dossier on" / due diligence / background check / "prep me for" | + +## Minimal Intake (2-4 Questions) + +| Q | Asks | When | +|---|---|---| +| Q1 | Research question (1-2 sentences, specific) | Always | +| Q2 | Output: quick chat brief OR standalone .docx | Always | +| Q3 | Domain disambiguation (7-option pick-list) | Only when classification is ambiguous (≤1 signal) | +| Q4 | Time horizon for general research (quick 5 vs thorough 15) | Only when Q3 was needed AND user picked "none of the above" | + +Most invocations exit at Q2. + +## Routing Transparency (Mandatory) + +After classification, the skill **always**: + +1. States the decision in one sentence: "Routing to `litreview` because you mentioned PICO and systematic review (2 signals)." +2. Offers override: "If you want general research instead or a different specialist, say so." +3. Waits 1 turn for confirmation (or auto-proceeds after 5s in interactive contexts). +4. If user overrides → accepts, re-routes, logs the override. + +**Never delegates silently.** This is the trust-building property that makes the hybrid pattern work. + +## What You Get + +**If delegated to specialist:** the specialist's full output (markdown briefing OR .docx, depending on specialist). Tagged with `[Delegated to: research → {specialist}]`. + +**If fallback:** the skill runs its own 8-step workflow and produces: + +``` +# [Research Question] — Briefing +*Generated: [DATE] | Routed: fallback* + +## TL;DR +[2-3 sentences] + +## Findings +### [Sub-question 1] +[2-4 paragraphs with inline citations] +### [Sub-question 2] +... + +## Cross-Cutting Patterns +[1-2 paragraphs] + +## Sources +[Numbered + hyperlinked + reliability tier per source] + +## Audit +[Three counts + per-source tier + failures] +``` + +DOCX version uses same structure with research-pack styling. + +## Discipline + +- **Deterministic classification** (NOT LLM-reasoned) — keyword signal matching via `classifier.py` +- **Routing transparency mandatory** — never silent +- **Specialist delegation is pass-through** — don't pre-answer specialist questions +- **Fallback after Q3** when no specialist matches +- **Refuse generic "research [topic]"** to a specialist without paired specialist-noun +- **Three-count tracking** in fallback mode +- **Source discipline** — cite only this-session tool calls + +## Workflow + +```bash +# Phase 1 intake (Q1 + Q2 minimum) + +# Phase 2 classification +python ../skills/research/scripts/classifier.py --question "" +# Returns: {route_to: "litreview", confidence: "high (2 signals)", matched: [...]} + +# Phase 3a delegation (if specialist matched at ≥2 signals) +python ../skills/research/scripts/routing_transparency_logger.py \ + --action record_delegation --session NAME --target litreview --signals "..." +# Pass question to /cs:litreview verbatim; let it run its own intake + +# Phase 3b fallback (if no specialist matched) +python ../skills/research/scripts/fallback_decomposer.py --question "" +# Returns 3-5 sub-questions +# Run 8-step fallback workflow: source-select → search → read+extract → synthesize → cross-cut → output → audit +``` + +## Stop Conditions + +- Specialist delegated → specialist's stop condition applies +- Fallback complete → markdown brief or DOCX delivered +- Q3 picked but no clear specialist → ask Q4 (time horizon), then run fallback +- User says "stop" → produce partial result with what's been collected + +## Trigger Phrases + +- "research [topic]" +- "look into [topic]" +- "what do we know about [topic]" +- "investigate [topic]" +- "find me information on [topic]" +- "do some research on [topic]" +- "I need to understand [topic]" +- Plus: any research request that doesn't obviously match a more-specific specialist + +## Anti-Patterns Rejected + +- LLM-reasoned classification (must be deterministic keyword matching) +- Silent delegation (always surface routing decision) +- Refusing to route to a specialist when ≥2 signals match +- Routing to a specialist when classification is genuinely ambiguous (≤1 signal) +- Pre-answering the specialist's grill-me intake +- Running fallback when a specialist would clearly do better +- Fabricating sources in fallback when search is thin +- Skipping audit log in fallback mode +- Treating "dossier on [company]" as fallback when `dossier` is the right specialist +- Treating "what are people saying about X" as fallback when `pulse` matches +- Auto-routing generic "research [topic]" without paired specialist-noun (ask Q3 instead) + +## Related + +- Agent: [`cs-research`](../agents/cs-research.md) +- Skill: [`research`](../skills/research/SKILL.md) +- Source spec: [`megaprompts/13-research-megaprompt.md`](../../../megaprompts/13-research-megaprompt.md) +- Routing targets: `/cs:pulse`, `/cs:litreview`, `/cs:grants`, `/cs:dossier`, `/cs:patent`, `/cs:syllabus` +- Adjacent (NOT a routing target): `/cs:notebooklm` (different mode), `engineering/autoresearch-agent` (different use case) + +--- + +**Version:** 1.0.0 +**Source:** Path-B direct conversion of `megaprompts/13-research-megaprompt.md` diff --git a/research/research/skills/research/SKILL.md b/research/research/skills/research/SKILL.md new file mode 100644 index 00000000..f521084f --- /dev/null +++ b/research/research/skills/research/SKILL.md @@ -0,0 +1,319 @@ +--- +name: research +description: Default entry point for any research request — a hybrid router that classifies the question deterministically and either delegates to a specialist research skill (pulse for trends/sentiment, grants for NIH funding, litreview for academic literature, syllabus for course reading, patent for prior-art + IP landscape, dossier for entity research) or runs its own plan-decompose-multi-source-search-synthesize-cite fallback workflow when no specialist matches. Always surfaces the routing decision so users can override. Triggers — "research [topic]", "look into [topic]", "what do we know about [topic]", "investigate [topic]", "find me information on [topic]", "do some research on [topic]", "I need to understand [topic]", or any research request that doesn't obviously match a more-specific specialist skill. Output is a markdown briefing (default) or .docx document (on request) with full citations and an audit log. +--- + +# Research — Hybrid Router + Fallback + +**The runtime orchestrator for the research domain.** Architecture C: deterministic classification → specialist delegation OR own plan-decompose-search-synthesize-cite workflow. + +## Portability + +Requires `WebSearch` + `WebFetch` for the fallback workflow; specialist skills (`pulse`, `grants`, `litreview`, `syllabus`, `patent`, `dossier`) must be present for delegation to work. Node.js with `docx` package required if Q2 = document mode. Works in Claude Code CLI natively. In Claude.ai with web tools + Code Execution, the workflow is supported. + +## Distinct From `engineering/autoresearch-agent` + +These two skills share the word "research" but serve **completely different use cases**: + +- **`research/research/`** (this skill) — research-query router + fallback workflow ("Research X") +- **`engineering/autoresearch-agent/`** — Karpathy's autonomous file-optimization experiment loop ("Make this code faster") + +No overlap. They coexist. + +## Hybrid Architecture (C) + +Every invocation produces one of three outcomes: + +1. **Delegation** — Classified as specialist-domain. Routes there. User sees the specialist's output. +2. **Fallback execution** — Classified as general research. Runs own plan → search → synthesize workflow. +3. **Clarification request** — Classification ambiguous. Asks one forcing question to disambiguate, then routes. + +The skill **never silently runs its fallback** when a specialist would have done better. **Routing transparency** is what makes the hybrid architecture trustworthy. + +## Specialist Registry + +| Specialist | Routing signals | Domain | +|---|---|---| +| `pulse` | reddit / hn / x / buzz / sentiment / trending / "what's people saying" / "pulse on" / "take the pulse" / "current conversation" | Multi-source recency research | +| `grants` | NIH / grant / R01 / K-award / RePORTER / NOSI / "grants for" / FDA / "study section" / "principal investigator" | NIH grant-funding intelligence | +| `litreview` | literature review / PICO / SPIDER / systematic review / "review papers on" / meta-analysis | Academic literature orientation | +| `syllabus` | syllabus / course outline / curriculum / "reading list" / "for my class" / "for my students" | Course supplementary reading | +| `patent` | prior art / FTO / freedom to operate / patent / "patent landscape" / invention / novelty search / "ip landscape" | Patent prior-art + landscape | +| `dossier` | "dossier on" / "due diligence" / "background check" / "prep me for" / "competitor research" / "investor diligence" / "interview prep" / "background on" | Decision-grade entity research | + +## Agent Integrity Rules + +This skill obeys the research-pack convention: + +- **Execution discipline (fallback only)**: Sequential searches. 1 q/sec rate limit. Confirm response received before next call. +- **Source discipline**: Cite only sources returned by this session's tool calls. Training knowledge labeled `[Background — not from search]` and excluded from counts. +- **Three-count tracking (fallback only)**: Queries sent / sources received / sources cited. +- **Retry policy**: On failure → wait 3s → retry once → log. After 3 consecutive failures: stop, alert user. +- **Plan-tier detection**: If delegated to Consensus-using specialist, that specialist handles detection. In fallback mode, surface any rate-limit signals. +- **Routing discipline**: Never delegate silently. Always state the decision + accept override. + +## Phase 1: Grill-Me Intake (2–4 Questions) + +Intake is intentionally minimal — the goal is to route fast, not to interrogate. One question per turn. + +### Q1 (always) — Research question + +> **What's the research question? State it in 1–2 sentences. Specific is better than broad — "AI for healthcare" gets you a vague survey; "How are health systems integrating LLM-based clinical decision support in 2026?" gets you a useful answer.** +> +> *Why I'm asking:* Specificity dictates classification accuracy and search precision. A vague question routes to fallback; a specific question often matches a specialist cleanly. + +**Refuse mush.** If user says "research AI", push back once: "What about AI specifically — adoption, safety, capability, funding, regulation, comparison? Pick an angle." + +### Q2 (always) — Output preference + +> **What output do you want? Pick one:** +> 1. Quick chat briefing (5-min read, markdown in chat) +> 2. Standalone document (.docx with citations, shareable) +> +> *Why I'm asking:* Document mode triggers deeper search budgets and full audit logs. Chat mode optimizes for fast delivery. + +Forcing choice. + +### Q3 (asked only if classification ambiguous — ≤1 signal) — Domain disambiguation + +> **Quick clarification — pick the closest match:** +> 1. Academic literature (papers, peer-reviewed) +> 2. Industry / trends (what's the buzz, news, sentiment) +> 3. Specific entity (a company, person, organization) +> 4. Technology / patents (prior art, IP landscape) +> 5. Grant funding (NIH, foundations) +> 6. Course material (syllabus or curriculum) +> 7. None of the above — run general research +> +> *Why I'm asking:* I couldn't classify confidently from your question alone. This routes you to the right specialist or confirms general-research fallback. + +**Skip if Q1 + Q2 produced clear specialist match (≥2 signals).** + +### Q4 (asked only if Q3 was needed AND user picked "none of the above") — General-research scope + +> **For general research, what's your time horizon — quick scan (5 searches) or thorough (15 searches)?** +> +> *Why I'm asking:* General research has no specialist budget; you pick it. Quick is good for "what's the lay of the land". Thorough is for "I'll make a decision based on this". + +Skip if a specialist took over. + +**Stop condition:** After Q4 (or earlier if dependency skips applied), commit and start Phase 2. **Most invocations exit intake after Q1 + Q2.** + +## Phase 2: Deterministic Classification + +This is **deterministic, not LLM-reasoned** — for speed, debuggability, and consistency. + +```python +SIGNALS = { + pulse: ["reddit", "hn", "hacker news", "x.com", "twitter", "buzz", + "sentiment", "trending", "what are people saying", + "what's happening", "the conversation around", + "pulse on", "take the pulse", "current conversation"], + grants: ["nih", "grant", "grants for", "r01", "r21", "k-award", "reporter", + "nosi", "funding", "fda", "study section", "principal investigator"], + litreview:["literature review", "lit review", "litreview", "pico", "spider", + "systematic review", "review papers on", "research papers on", + "papers about", "meta-analysis"], + syllabus: ["syllabus", "course outline", "curriculum", "reading list", + "for my class", "for my students", "course material"], + patent: ["prior art", "fto", "freedom to operate", "patent", + "patent landscape", "invention", "novelty search", + "patent search", "ip landscape"], + dossier: ["dossier on", "due diligence", "background check", + "prep me for", "competitor research", "investor diligence", + "interview prep", "research my competitor", "background on"] +} + +# Signals are case-insensitive literal phrases (multi-word substring match). +# Bracketed placeholders (e.g., "research [company]") are intentionally NOT +# signals — they over-trigger on generic "research X" queries that should +# fall back to general research, not auto-route to dossier. Specific phrases +# pair the verb with the noun ("dossier on", "background on") and route reliably. + +For each specialist S: + score[S] = count of SIGNALS[S] phrases matched in question (case-insensitive substring) + +if max(score) >= 2: + route_to = argmax(score) # high confidence +elif max(score) == 1 and only one specialist has score 1: + route_to = that specialist # weak match, single specialist +else: + route_to = "fallback" # ambiguous or no match — ask Q3 +``` + +**Implementation:** `scripts/classifier.py --question "..."` returns the routing decision + matched signals + per-specialist scores. Use it; don't re-implement. + +## Phase 3a: Specialist Delegation (≥2 signals OR single weak match) + +When delegating: + +1. Pass the user's question **verbatim** plus the output preference (Q2) +2. **Let the specialist run its own grill-me intake** — do NOT pre-answer specialist questions +3. Return specialist output as the user-visible result +4. Tag the result with `[Delegated to: research → {specialist}]` in the chat output so the user knows what skill produced it +5. Tag the audit log via `scripts/routing_transparency_logger.py --action record_delegation` + +## Phase 3b: Own Fallback Workflow + +If routing produced no specialist match, run the 8-step fallback. + +### Step 1: Decompose + +Break the research question into 3–5 sub-questions. Use the framework: what / why / how / who / what's next. Show the decomposition to the user before searching. Use `scripts/fallback_decomposer.py --question "..."` for a deterministic starting point. + +### Step 2: Source Selection + +For each sub-question, choose source(s) deterministically: + +- **Recency-sensitive** → WebSearch + WebFetch + (optionally Reddit/HN if signal) +- **Technical specs / docs** → WebSearch + WebFetch +- **Academic** → Consensus MCP if connected; otherwise WebSearch with `scholar.google.com` site filter +- **Data / numbers** → WebSearch for sources; then WebFetch for primary documents +- **Person / company entity-level** → consider routing to `dossier` (offer override) + +### Step 3: Search + +Sequential per sub-question. 1 q/sec etiquette. Per source: 2–4 queries, broad-to-narrow. + +### Step 4: Read + Extract + +For each result that looks high-signal: WebFetch and extract the relevant section. Note the source URL. + +### Step 5: Synthesize + +Per sub-question: 2–4 paragraphs answering it with inline citations. Surface disagreement when sources disagree. + +### Step 6: Cross-Cutting Patterns + +After per-sub-question synthesis: 1–2 paragraphs of patterns across sub-questions — consensus, controversy, gaps. + +### Step 7: Output + +Markdown brief by default (Q2 choice). DOCX if user picked document mode. + +### Step 8: Audit Log + +Three-count summary (sent / received / cited) + per-source list with reliability tier (primary / secondary / tertiary). + +## Routing Transparency Protocol (Mandatory) + +After classification, the skill **always**: + +1. **States the decision** in one sentence: "Routing to `litreview` because you mentioned PICO and meta-analysis (2 signals)." +2. **Offers override**: "If you want general research instead OR a different specialist, say so now. Otherwise proceeding in 5 seconds." +3. **Waits 1 turn** for confirmation (or auto-proceeds after 5s in interactive contexts). +4. **If user overrides** → accept, re-route, log the override via `routing_transparency_logger.py --action record_override`. + +**Never delegates silently.** This is the trust-building property that makes the hybrid pattern work. + +## Output Format + +### Markdown brief (Q2 = quick chat briefing) + +```markdown +# [Research Question] — Briefing +*Generated: [DATE] | Routed: [delegated specialist | fallback]* + +## TL;DR +[2-3 sentences] + +## Findings +### [Sub-question 1] +[2-4 paragraphs with inline citations] + +### [Sub-question 2] +... + +## Cross-Cutting Patterns +[1-2 paragraphs] + +## Sources +[Numbered list with hyperlinks, reliability tier per source] + +## Audit +[Three counts + per-source tier + failures] +``` + +### DOCX (Q2 = standalone document) + +Use the standard research-pack DOCX patterns: Arial 12pt, navy headings, blue table headers, hyperlinked sources, mandatory audit log section. Reference the `docx` skill for setup. + +## Audit Log Requirement (Fallback Mode) + +``` +Queries sent: N +Sources received: M +Sources cited: K +Failures: F (3-consecutive-failures triggered: yes/no) +Per-source tier: [URL — primary | secondary | tertiary] +Routing decision: fallback (no specialist matched) +Sub-questions: [list] +``` + +All routing decisions + overrides also logged to `~/.research_sessions/.json` via `routing_transparency_logger.py`. + +## Failure Modes + +| Failure | Behavior | +|---|---| +| Classification ambiguous (≤1 signal) | Ask Q3 (domain disambiguation). | +| Specialist delegation fails | Note in chat. Offer to retry or fall back to general research. | +| User overrides routing | Accept. Re-route to chosen specialist or fallback. Log the override. | +| Fallback search returns thin results | Surface explicitly. Suggest the question may be too niche or too new. Do not fabricate. | +| 3 consecutive tool failures in fallback | Stop, alert user, share what was collected. | +| Question is non-research (e.g., "write me code") | Decline politely. Suggest the user invoke an appropriate skill. | +| Sub-question can't be answered | Note in synthesis as "limited public signal on this"; don't omit silently. | +| Output format mismatch | Honor Q2 preference; if format unavailable, fall back to markdown with note. | +| Specialist skill missing from environment | Skip it in classification scoring; route to fallback or next-best specialist. | + +## Anti-Patterns Rejected + +- LLM-reasoned classification (must be deterministic keyword + intent matching) +- Silent delegation (always surface routing decision) +- Refusing to route to a specialist when ≥2 signals match +- Routing to a specialist when classification is genuinely ambiguous (≤1 signal across all) +- Pre-answering the specialist's grill-me intake (let it run its own) +- Running fallback when a specialist would clearly do better +- Fabricating sources in fallback when search is thin +- Skipping audit log in fallback mode +- Treating "dossier on [company]" as fallback when `dossier` is the right specialist (the verb-noun-paired phrase, not the generic "research X" form, is what routes) +- Treating "what are people saying about X" as fallback when `pulse` is the right specialist +- Auto-routing generic "research [topic]" queries to a specialist when the user hasn't paired the verb with a specialist-specific noun (e.g., "research Microsoft" alone is ambiguous — could be dossier or general; ask Q3 instead of guessing) + +## Tooling + +### Python (stdlib only) + +- **`scripts/classifier.py`** — Deterministic SIGNALS matching → routing decision + per-specialist score + matched phrases. `--question "..." --output json`. +- **`scripts/routing_transparency_logger.py`** — JSON-backed audit log at `~/.research_sessions/.json`. Records every routing decision, override, and delegation handoff. +- **`scripts/fallback_decomposer.py`** — Heuristic question → 3–5 sub-questions using what / why / how / who / what's next framework. + +### Reference Docs (each cites 7+ authoritative sources) + +- **`references/hybrid_router_architecture.md`** — router-vs-run trade-offs + routing transparency principle +- **`references/deterministic_classification_canon.md`** — why keyword > LLM-reasoned for routing +- **`references/fallback_workflow_canon.md`** — plan-decompose-search-synthesize methodology + +## Dependencies + +- **`WebSearch`** + **`WebFetch`** — Required for fallback workflow +- **Specialist skills** — Required for delegation: `pulse`, `grants`, `litreview`, `syllabus`, `patent`, `dossier`. If a specialist is missing, the router skips it in classification and routes to fallback instead. +- **Node.js `docx` library** — Required if user picks document output (Q2 = standalone) +- **Consensus MCP** — Optional; used in fallback if academic sub-questions surface + +## Trigger Phrases + +- "research [topic]" +- "look into [topic]" +- "what do we know about [topic]" +- "investigate [topic]" +- "find me information on [topic]" +- "do some research on [topic]" +- "I need to understand [topic]" +- Any research request that doesn't obviously match a more-specific specialist + +--- + +**Version:** 1.0.0 +**Source spec:** [`megaprompts/13-research-megaprompt.md`](../../../../megaprompts/13-research-megaprompt.md) +**Build pattern:** Path B (direct conversion) diff --git a/research/research/skills/research/references/deterministic_classification_canon.md b/research/research/skills/research/references/deterministic_classification_canon.md new file mode 100644 index 00000000..09f531e2 --- /dev/null +++ b/research/research/skills/research/references/deterministic_classification_canon.md @@ -0,0 +1,162 @@ +# Deterministic Classification — Why Keyword Beats LLM-Reasoned For Routing + +This reference answers one decision: **should the routing classifier use deterministic keyword matching or LLM reasoning over the query?** The answer is **deterministic keyword matching** for query-routing purposes, with LLM reasoning reserved for cases where keyword matching has genuinely exhausted the signal space. + +## The Trade-Off Spectrum + +| Approach | Latency | Cost | Determinism | Debuggability | Coverage of fuzzy intent | +|---|---|---|---|---|---| +| **Keyword + intent signals** (this skill) | <1ms | $0 | 100% | High (signals named explicitly) | Low | +| **Embedding similarity to specialist descriptions** | ~10-100ms | Cents/100K queries | High (deterministic given embeddings) | Medium (need to inspect cosine scores) | Medium | +| **LLM reasoning over query + specialist list** | ~500ms-2s | ~$0.001-0.01/query | Low (same query → varied outputs) | Low (prompt-dependent) | High | + +The trade-off: as you move down the table, coverage of fuzzy intent improves, but latency, cost, and unpredictability all worsen. The right choice depends on how predictable + auditable the routing needs to be. + +## For Query Routing, Determinism Wins + +Routing is **fundamentally a control-flow decision**: it determines which subsystem runs next. Like any control-flow decision in software, predictability + auditability are first-order properties. + +Compare to other deterministic control-flow systems: + +- **Compilers** use deterministic lexer + parser, not LLMs. +- **Routers** (network sense) use deterministic CIDR matching, not LLMs. +- **CI/CD systems** use deterministic file-pattern triggers, not LLMs. +- **Linters + formatters** use deterministic AST-walking, not LLMs. + +These are all systems where users need to predict + debug behavior. LLM-reasoned routing in any of them would be a regression. Same applies to skill routing. + +## The Bracketed-Placeholder Anti-Pattern + +A common mistake when building keyword classifiers: using bracketed placeholders as signals. + +**Wrong:** +```python +SIGNALS = { + dossier: ["dossier on [company]", "background check on [person]", "research [entity]"] +} +``` + +**Why wrong:** the "research [entity]" pattern collapses to "research" as a substring match, which matches every research request ever. The signal over-triggers + breaks the classifier. + +**Right:** +```python +SIGNALS = { + dossier: ["dossier on", "background check", "background on", "competitor research"] +} +``` + +**Why right:** verb-noun pairs ("dossier on", "background on", "competitor research") are specific to dossier intent. Generic "research X" stays in fallback territory until paired with a specialist-specific noun. + +This is the post-PR-#657-audit lesson encoded as a hard rule. + +## What Counts As A "Signal" + +A signal is a **case-insensitive literal phrase (multi-word substring)** that, when present in the user's question, indicates a specialist domain. Good signals are: + +- **Specific enough** that they don't appear in unrelated queries (good: "literature review", bad: "research") +- **Common enough** that users actually say them (good: "due diligence", bad: "actuarial diligence assessment framework") +- **Diverse enough** to cover surface variations (good: "lit review" + "literature review" + "litreview"; bad: only one form) +- **Verb-noun-paired** when the noun alone is ambiguous (good: "dossier on" + "background on"; bad: just "company name") + +## Confidence Thresholds + +The skill commits to a specialist at **≥2 signals** for two reasons: + +1. **2 signals reliably indicate intent.** "PICO + meta-analysis" doesn't show up in unrelated queries. +2. **1 signal isn't strong enough.** "PICO" alone might be a clinical question, a syllabus question, or a litreview question. The second signal distinguishes. + +The single-weak-match exception (1 signal + only one specialist with any score) handles the case where the user used a highly specific phrase that no other specialist's signals overlap with. "What's the FTO landscape" → only patent has any score → route to patent even though it's just 1 signal. + +The "ask Q3 disambiguation" exception handles the case where multiple specialists each have score 1, OR no specialist has any score. Both indicate genuine ambiguity that the classifier can't resolve. + +## What Goes Wrong With LLM-Reasoned Classification + +### Non-determinism + +Same query, different responses across invocations. User says "what are people saying about X" — sometimes routes to pulse, sometimes to dossier, sometimes to fallback. User can't develop intuition for the system. + +### Cost + +500ms-2s per classification × hundreds of routing decisions/day adds up. Deterministic classifier is sub-millisecond + free. + +### Debuggability + +When LLM routes "weirdly," there's no signal to inspect. With deterministic classification, the user sees "matched signals: PICO, meta-analysis" and understands why. + +### Prompt drift + +LLM classifier behavior changes when the underlying model version changes. Deterministic classifier behavior is locked to the signals list. Auditable + reproducible. + +## What Goes Wrong With Pure Keyword Classification + +### Fuzzy intent + +User says "I want to understand what the academic community thinks about CRISPR safety." No keyword matches litreview signals (no "PICO", no "systematic review", no "literature review"). Classifier punts to fallback even though litreview was the right answer. + +**Mitigation:** Q3 disambiguation handles this. User picks "academic literature" → routes to litreview. The architecture's clarification path covers the fuzzy-intent case. + +### Surface-form proliferation + +Users say "lit review", "literature review", "litreview", "review the literature on", "review papers on", "look at the papers about", "what does the research say about" — that's 7 surface forms for the same intent. Signals list grows. + +**Mitigation:** Cover the top-N surface forms (3-5 per specialist). Let Q3 handle the long tail. + +### Polysemy + +"Patent" could mean a legal patent (route to patent specialist) OR a medical term ("the symptoms are patent" = obvious). Keyword matching can't distinguish. + +**Mitigation:** Multi-signal requirement reduces false positives. "Patent + prior art" is unambiguously patent intent. + +## The Right Hybrid: Deterministic First, Clarify When Stuck + +The architecture combines: + +1. **Deterministic classification** for the high-confidence path (cheap + fast + predictable) +2. **Q3 disambiguation** for the genuinely-ambiguous path (LLM-free; user picks from 7 options) +3. **Fallback workflow** for the no-specialist path + +This is strictly better than pure-LLM classification (cheaper, faster, more predictable) and strictly better than pure-keyword classification (handles fuzzy intent via Q3). + +## Operational Discipline + +When adding a new signal to the SIGNALS map: + +- [ ] Verify the signal doesn't appear in queries that should route elsewhere (false positive check) +- [ ] Verify the signal does appear in queries that should route to this specialist (false negative check) +- [ ] Check for case-insensitivity (the matcher is case-insensitive, but be explicit) +- [ ] Avoid bracketed placeholders +- [ ] Use verb-noun pairs when the noun alone is ambiguous +- [ ] Document why this signal was added (which queries it covers) + +When removing a signal: + +- [ ] Check what queries previously routed via this signal +- [ ] Confirm they still route correctly (via another signal OR via Q3) +- [ ] Update the documentation + +## Tooling + +`scripts/classifier.py` implements the deterministic SIGNALS-matching algorithm. Use it; don't re-implement. It returns: + +- `route_to`: specialist name OR "fallback" +- `confidence`: "high (N signals)" OR "weak (1 signal, single specialist)" OR "ambiguous" +- `matched_signals`: dict of specialist → list of matched phrases +- `scores`: dict of specialist → integer score + +The CLI: `classifier.py --question "..." --output json`. + +## Citations (7 sources) + +1. **Aho, Sethi, Ullman — "Compilers: Principles, Techniques, and Tools" (Dragon Book, 1986).** Source for the deterministic lexer + parser as the canonical control-flow classifier in software. Compilers don't use LLMs for tokenization; routing shouldn't either. + +2. **Cisco IOS — Access Control List (ACL) implementation guides.** Source for the deterministic CIDR-matching pattern in network routing. Predictability + auditability are first-order requirements; same applies to skill routing. + +3. **Google Search Engineering blog — Query Classification (2020+).** Source for the production-grade query-classification pattern. Google uses deterministic signal matching as the first layer + LLM reasoning only for residual queries that signals miss. Same architecture as this skill (Q3 as the LLM-equivalent escape hatch). + +4. **Mikolov et al. — "Distributed Representations of Words and Phrases" (Word2Vec, 2013).** Source for the embedding-similarity baseline. Embeddings are an intermediate point between keywords + LLM reasoning; this skill chooses keywords for cost + determinism reasons but acknowledges embedding-similarity as a valid alternative. + +5. **Karpathy, Andrej — "Software 2.0" (blog post, 2017).** Source for the framing that not everything should be ML. Deterministic systems (compilers, routers, type checkers) remain superior for control-flow decisions even in the LLM era. https://karpathy.github.io/2017/11/11/software-2-0/ + +6. **Anthropic — Tool Use + Function Calling documentation.** Source for the production pattern of LLM-routes-to-deterministic-tool: the LLM decides intent at the top level, then deterministic tools handle the actual work. Same shape as this skill (intake → deterministic classifier → specialist tool). https://docs.anthropic.com/ + +7. **NIST — "Information Retrieval Evaluation" (TREC reports).** Source for the canonical evaluation methodology for classifiers: precision + recall measured against held-out queries. Keyword classifiers reliably outperform LLM-reasoned classifiers on precision for domain-specific routing tasks. https://trec.nist.gov/ diff --git a/research/research/skills/research/references/fallback_workflow_canon.md b/research/research/skills/research/references/fallback_workflow_canon.md new file mode 100644 index 00000000..1c157572 --- /dev/null +++ b/research/research/skills/research/references/fallback_workflow_canon.md @@ -0,0 +1,214 @@ +# Fallback Workflow Canon — Plan / Decompose / Search / Synthesize / Cite + +This reference answers one decision: **when no specialist matches, what workflow does the orchestrator run instead?** The answer is an **8-step plan-decompose-multi-source-search-synthesize-cite** workflow grounded in the canonical research-pack conventions. + +## The Eight Steps + +The fallback workflow is documented in `SKILL.md`. This reference explains the **why** behind each step + the failure modes per step + the tooling that supports it. + +### Step 1: Decompose + +Break the research question into 3–5 sub-questions. Use the framework: **what / why / how / who / what's next**. + +**Why decompose?** A 1-sentence research question rarely has a 1-source answer. Decomposition forces the orchestrator to enumerate the actual claim shape before searching, which makes search precise + makes synthesis structured. + +**Failure mode:** decomposing into too many sub-questions (>5) wastes search budget on diminishing returns. Cap at 5. + +**Tooling:** `scripts/fallback_decomposer.py` returns a deterministic starting point. Override + refine before searching. + +### Step 2: Source Selection + +For each sub-question, pick the right source class. Use the deterministic mapping in SKILL.md: + +- Recency-sensitive → WebSearch + WebFetch (+ optional Reddit/HN signal) +- Technical specs → WebSearch + WebFetch +- Academic → Consensus MCP if available; else WebSearch + scholar.google.com filter +- Data / numbers → WebSearch for primary documents +- Entity-level → consider routing back to `dossier` + +**Failure mode:** using a wrong-class source (e.g., WebSearch for academic when Consensus would have produced higher-quality results). The mapping is deterministic for a reason. + +### Step 3: Search + +Sequential per sub-question. **1 q/sec rate limit** (research-pack convention). Per source: 2–4 queries, broad-to-narrow. + +**Why broad-to-narrow?** Broad queries map the landscape; narrow queries find the high-signal sources within it. Going narrow-only often misses the orienting overview. + +**Failure mode:** parallel search bursts that trigger rate-limiting or get blocked. Sequential is the discipline. + +### Step 4: Read + Extract + +For each high-signal result: WebFetch the full content + extract the relevant section + note the URL. + +**Why extract, not summarize?** Direct quotes + section references make citations verifiable. Summaries hide the source structure. + +**Failure mode:** synthesizing from search snippets without WebFetch. Snippets are not sources. + +### Step 5: Synthesize Per Sub-Question + +For each sub-question: 2–4 paragraphs with inline citations. Surface disagreement when sources disagree. + +**Why per-sub-question?** Sub-question structure carries through to the output. Reader can navigate to the part they care about. + +**Failure mode:** synthesizing across sub-questions in one mega-paragraph. Loses the navigability + makes disagreements harder to surface. + +### Step 6: Cross-Cutting Patterns + +After per-sub-question synthesis: 1–2 paragraphs of patterns across all sub-questions — consensus, controversy, gaps. + +**Why a separate section?** Pattern-level claims (e.g., "all sources agree on X but disagree on Y") are valuable for the reader's understanding but don't belong inside any single sub-question's synthesis. + +**Failure mode:** skipping this step because "the sub-questions cover it". They don't — the cross-cutting view is its own contribution. + +### Step 7: Output + +Markdown brief by default. DOCX if Q2 = document mode. Honor user preference. + +**Why honor preference?** Document mode triggers deeper search budgets + full audit logs. Brief mode is optimized for fast delivery. Different goals → different output shapes. + +**Failure mode:** producing DOCX when user wanted brief (overkill) or producing brief when user wanted DOCX (loses citations). + +### Step 8: Audit Log + +Three-count summary (queries sent / sources received / sources cited) + per-source list with reliability tier. + +**Why audit?** Research-pack convention. Lets the reader verify the orchestrator didn't fabricate sources or hide failures. + +**Failure mode:** skipping the audit. Audit is what makes the fallback output trustworthy. + +## The Three-Count Convention + +The research-pack convention requires tracking three integers throughout the fallback workflow: + +- **Sent**: queries actually issued (WebSearch + WebFetch + Consensus calls) +- **Received**: results returned from those calls (after filtering) +- **Cited**: sources actually cited in the final output + +The relationship `sent >= received >= cited` is always true. When it isn't, something went wrong. + +**Why three counts?** They make the orchestrator's search productivity visible. If sent=15, received=3, cited=1, the question was too niche or the search strategy was off. If sent=5, received=20, cited=15, the orchestrator found a rich vein. The reader can interpret the result quality based on the counts. + +## Source Discipline + +The orchestrator cites **only sources returned by this session's tool calls**. Training knowledge is labeled `[Background — not from search]` and excluded from the three-count. + +**Why?** Citations must be verifiable. A "cited" source that wasn't actually retrieved is a fabrication, regardless of how well it matches the orchestrator's training data. + +**Failure mode:** inferring a citation from background knowledge + presenting it as if retrieved. This is the highest-severity research-pack violation. + +## Retry + Failure Policy + +- **On single failure**: wait 3s → retry once → log. +- **After 3 consecutive failures**: stop, alert user, share what was collected. + +**Why 3s + single retry?** Most failures are transient (rate limit, network blip). 3s + retry catches them. After 3 in a row, something structural is wrong (API outage, blocked endpoint, query-format issue); halt + escalate. + +**Failure mode:** infinite retry loops that consume the session budget. The 3-consecutive-failure stop is the safety valve. + +## Reliability Tier Classification + +Per source, classify as: + +- **Primary** — original source (peer-reviewed paper, government document, company filing, original announcement) +- **Secondary** — derivative reporting (news article summarizing a paper, blog post analyzing a filing) +- **Tertiary** — aggregator or wiki (Wikipedia, news aggregator, opinion piece) + +**Why surface tiers?** Reader needs to know which claims rest on primary evidence vs derivative reporting. A consensus claim backed by 5 secondary sources is weaker than the same claim backed by 1 primary source. + +**Failure mode:** misclassifying tier to make the audit look better. Honest tiering > polished audit. + +## Disagreement Surfacing + +When two sources disagree on a sub-question's answer: + +- **Name both positions** in the synthesis +- **Cite both sources** +- **State which seems stronger** + why (primary vs secondary, recency, methodology) +- **Don't pick a winner without reasoning** + +**Why?** Hiding disagreement misleads the reader. Surfacing it lets them apply their own judgment. + +**Failure mode:** averaging two disagreeing sources into a mushy middle that neither source actually supports. This is the synthesis equivalent of fabrication. + +## When To Stop Searching (Fallback Mode) + +The fallback workflow is **not infinite**. Q4 sets the budget (5 searches for quick scan, 15 for thorough). Stop when: + +- Budget exhausted +- All sub-questions have ≥1 high-signal source +- 3-consecutive-failure threshold hit +- User says "stop" or "that's enough" +- Diminishing returns (last 3 searches produced no new high-signal sources) + +**Why budget the search?** Open-ended search is the failure mode that turns "research X" into a 30-minute exploration. Budget forces commitment + delivery. + +## What Goes Wrong With Fallback + +### Fabricated sources + +The orchestrator infers a citation from background knowledge. Highest-severity violation. **Prevention:** strict source discipline + three-count tracking makes this auditable. + +### Thin results presented as comprehensive + +Search returned 2 sources. Orchestrator presents conclusions as if backed by 10. **Prevention:** surface the audit counts. Reader sees `cited: 2` + adjusts confidence. + +### Skipping cross-cutting patterns + +Per-sub-question synthesis without cross-cutting view. Reader misses the pattern-level insight. **Prevention:** Step 6 is mandatory. + +### Skipping audit + +Output without the audit section. **Prevention:** Audit is part of the output format, not optional. + +### Wrong output format + +User asked for brief, got DOCX. Or vice versa. **Prevention:** Q2 captures preference + Step 7 honors it. + +### Synthesis without decomposition + +Orchestrator searches first, organizes later. Output is unstructured. **Prevention:** Step 1 (decompose) before Step 3 (search) is non-negotiable. + +## When To Choose Fallback Over Specialist + +The classifier handles this deterministically. But conceptually, fallback is right when: + +- No specialist's signal vocabulary fits the question +- User explicitly picked "none of the above" in Q3 +- User overrode the routing decision to fallback +- A specialist failed + user opted to retry as fallback + +Fallback is **wrong** when: + +- A specialist clearly matched (≥2 signals) but the orchestrator ran fallback anyway +- The question is structurally a specialist's domain but used non-canonical phrasing (this is the Q3 case — disambiguate, then route) + +## Operational Checklist (Per Fallback Run) + +- [ ] Q1 specific enough to decompose (push back if vague) +- [ ] Decomposition produced 3-5 sub-questions +- [ ] Source class chosen per sub-question +- [ ] Sequential 1 q/sec search discipline +- [ ] WebFetch on every cited result +- [ ] Per-sub-question synthesis with citations +- [ ] Cross-cutting patterns section +- [ ] Output format honors Q2 +- [ ] Three-count tracked +- [ ] Reliability tier per source +- [ ] Audit log included +- [ ] No fabricated citations + +## Citations (7 sources) + +1. **Cooper, Hedges, Valentine — "The Handbook of Research Synthesis and Meta-Analysis" (2009, 3rd ed.).** Source for the canonical research-synthesis workflow: question → decomposition → systematic search → extraction → synthesis → reporting. The fallback workflow is a lightweight adaptation of this for AI-orchestrated general research. + +2. **Cochrane Collaboration — Handbook for Systematic Reviews of Interventions (current ed.).** Source for the rigor of source classification (primary vs secondary vs tertiary), explicit search protocols, and audit requirements. The three-count + per-source-tier conventions trace to Cochrane practice. + +3. **PRISMA 2020 Statement — Page et al., BMJ 2021.** Source for the canonical reporting checklist for research synthesis: searches conducted + sources screened + sources included + sources excluded with reasons. The audit log in fallback mode parallels PRISMA's flow diagram. + +4. **Karpathy, Andrej — "On chunking and search in LLMs" (talks 2024-2025).** Source for the principle that decomposition before retrieval beats single-shot retrieval. Sub-questions drive precise queries; whole-question retrieval is too broad. https://karpathy.ai/ + +5. **Anthropic — Multi-Agent Research System (2024-2025).** Source for the orchestrator-runs-fallback-with-audit pattern. Anthropic's research orchestrator includes explicit audit + source-tier surfacing as trust mechanisms. https://www.anthropic.com/research + +6. **Tufte, Edward — "The Visual Display of Quantitative Information" (1983).** Source (by analogy) for the principle of surfacing data integrity to the reader rather than hiding methodology. The three-count + audit log are the textual analogue of Tufte's data-ink ratio: report what you did so the reader can interpret what you found. + +7. **NIST — Special Publication 800-53 (Audit Logging guidance).** Source for the operational discipline of immutable, structured audit logs. The `routing_transparency_logger.py` JSON-backed log + the fallback audit section both implement this discipline at different scales. diff --git a/research/research/skills/research/references/hybrid_router_architecture.md b/research/research/skills/research/references/hybrid_router_architecture.md new file mode 100644 index 00000000..5478d2ca --- /dev/null +++ b/research/research/skills/research/references/hybrid_router_architecture.md @@ -0,0 +1,154 @@ +# Hybrid Router + Fallback Architecture — When To Delegate, When To Run + +This reference answers one decision: **should a research request be delegated to a specialist OR run directly by the orchestrator?** The answer is "either — depending on classification confidence," and the trustability property is **routing transparency**. + +## The Core Trade-Off + +A purely router-based architecture forces the user to know which specialist applies. A purely monolithic skill produces mediocre output for cases where a specialist would have done better. + +The **hybrid** answer: route when confidence is high, run a fallback when it isn't, always surface the decision so the user can correct. + +| Architecture | Strength | Weakness | +|---|---|---| +| **Pure router** | Always lands in the right specialist when it knows which one. | Brittle: every miss is a failure (no graceful degradation). | +| **Pure monolith** | Always answers. | Generic answers when a specialist would have done better. | +| **Hybrid (this skill)** | Specialist quality when matched; fallback when not. | Adds a classification step — but it's deterministic + fast. | + +## Why Routing Transparency Is Mandatory + +The hybrid is **only trustworthy if the user can see the routing decision and override it**. Otherwise the user can't tell when the orchestrator silently downgraded their request to a generic fallback (when a specialist would have done better) or upgraded it to a specialist (when fallback was what they actually wanted). + +This is the same property that makes well-designed CI/CD systems trustworthy: the system tells you what stage it's in and lets you intervene. Silent routing is a black box; transparent routing is operable. + +## The Three Outcomes (Forcing Frame) + +Every invocation produces exactly one of: + +1. **Delegation** (classified as specialist-domain, ≥2 signals OR single weak match): hand off to specialist verbatim, return their output, log the delegation. +2. **Fallback execution** (no specialist matched OR Q3 user picked "none of the above"): run the 8-step plan-decompose-search-synthesize-cite workflow. +3. **Clarification request** (classification ambiguous — ≤1 signal across all specialists): ask Q3 (domain disambiguation), then route based on the answer. + +Frame this way to refuse the trap of "router silently runs its fallback because the user didn't explicitly ask for a specialist." That's the failure mode the architecture exists to prevent. + +## What Makes A Good Routing Decision + +A routing decision is good when: + +1. **It uses signal-based deterministic logic** (keyword matching, not LLM reasoning over the query) +2. **It commits at high confidence** (≥2 signals for a specialist) +3. **It refuses to commit at low confidence** (1 signal across multiple specialists, or 0 across all → fallback or clarification) +4. **It surfaces the decision** to the user with the matched signals named +5. **It accepts override** without penalty + +Bad routing decisions: LLM-only "vibes" classification, silent delegation, refusal to delegate at high-confidence matches, eager delegation at ambiguous matches. + +## Forcing-Function Trade-Offs + +The orchestrator's job is to make the routing decision **fast** and **visible**, not to do the research itself when a specialist exists. This forces three design constraints: + +- **Minimal intake** — 2-4 questions max. Goal is to route, not to interrogate. Specialist handles its own grill-me. +- **Deterministic classifier** — no LLM round-trip. Signal matching is sub-millisecond. +- **Pass-through delegation** — don't pre-answer specialist questions. Their intake is intentional. + +When these constraints are violated, the orchestrator slowly becomes a competitor to the specialists rather than their router. + +## Sequencing: What Runs When + +``` +T+0 User invokes /cs:research with their question +T+0 Q1 (research question) — always asked +T+0 Q2 (output preference) — always asked +T+0 Classifier runs (deterministic, sub-millisecond) +T+0 IF score >= 2 OR single specialist with score 1: + Routing transparency: "Routing to X because Y" + Wait 1 turn for override (or 5s timeout) + Delegate verbatim + return specialist output + ELSE: + Q3 (domain disambiguation) — only when ambiguous + IF Q3 picks specialist: delegate + IF Q3 picks "none of the above": Q4 → fallback +T+~5s Specialist output OR fallback workflow complete +``` + +This sequencing is what keeps the orchestrator fast on the happy path (specialist matched cleanly) while still degrading gracefully (Q3 + Q4 + fallback for the edge cases). + +## What Goes Wrong With Each Component + +### Silent delegation (no routing transparency) + +User asks "what's the buzz about Anthropic," skill silently routes to `pulse`. User never sees the routing. If they wanted general research instead, they have to notice the output came from pulse, then re-invoke. This burns trust + a session. + +**Fix:** Routing transparency is mandatory. State decision + accept override. + +### LLM-reasoned classification + +Skill uses Claude to "decide" which specialist matches. Adds latency, costs tokens, is non-deterministic across invocations (same query → different route). User can't predict what will route where. + +**Fix:** Deterministic keyword matching. Predictability is the value. + +### Over-eager specialist routing + +Skill routes "research Microsoft" to `dossier` based on the word "research". But the user might want general research about Microsoft, not a competitor dossier. The single weak signal isn't strong enough. + +**Fix:** Generic "research [topic]" doesn't route. Specific phrases like "dossier on Microsoft" or "background on Microsoft" do. + +### Specialist intake pre-answering + +Orchestrator collects Q1 + Q2 + Q3 + Q4 + Q5 (passing all into the specialist). Specialist's own grill-me is now redundant; user has to confirm answers twice. + +**Fix:** Pass Q1 + Q2 only. Let specialist run its own intake. + +### Fallback when specialist would have done better + +User asks "review papers on GLP-1 receptor agonists" but skill runs fallback because the classifier missed "review papers" → "literature review" stemming. User gets generic web-search summary instead of structured litreview output. + +**Fix:** Signals list must include all reasonable surface forms ("review papers on", "literature review", "lit review", "litreview", etc.). + +## When Hybrid Is The Right Architecture + +The hybrid pattern is most valuable when: + +- Specialists exist + cover non-trivial portion of likely requests +- Specialists have different intake/output shapes (forcing user to know which to use is a tax) +- Generic fallback exists + is acceptable (better than rejecting the request) +- Routing can be made deterministic (predictable classification > LLM "vibes") + +When these aren't true, simpler architectures win: + +- No specialists yet? Build the monolith. +- One dominant specialist? Just expose it. +- Routing requires deep reasoning over intent? Use LLM classification (accept the cost). +- Fallback would mislead users? Reject instead of falling back. + +## Operational Checklist + +Before deploying a hybrid router skill: + +- [ ] Specialist registry documented with explicit routing signals per specialist +- [ ] Classifier is deterministic (no LLM in the loop) +- [ ] Confidence threshold defined (≥2 signals for commit) +- [ ] Single-weak-match policy defined (1 signal + only one specialist → route) +- [ ] Ambiguity policy defined (≤1 across all → Q3 disambiguation) +- [ ] Routing transparency is mandatory (decision + override surface) +- [ ] Override path tested +- [ ] Fallback workflow specified end-to-end +- [ ] Audit log captures routing decisions + overrides for later review +- [ ] Anti-patterns documented (LLM classification, silent delegation, etc.) + +## Citations (8 sources) + +1. **Karpathy, Andrej — "LLM OS" talk (2024).** Source for the orchestrator pattern: a smart top-level dispatcher routing to specialized capabilities is more effective than a single monolithic LLM call. Frames the router-with-fallback as a kernel-vs-syscalls analogy. https://karpathy.ai/ + +2. **Anthropic — Multi-Agent Research System (2024-2025).** Source for the hybrid router-vs-run trade-off in agentic systems. Anthropic's research orchestrator surfaces routing decisions explicitly + accepts user overrides. Practical implementation of the pattern this skill formalizes. https://www.anthropic.com/research + +3. **Schaubroeck et al. — "Bounded Confidence in Multi-Agent Systems" (2018).** Source for the academic framing of why bounded-confidence routing (commit only above threshold) outperforms always-route-or-always-defer architectures. Confidence thresholds prevent both over-eager + under-eager commitment. + +4. **Google Search Engineering — Query Classification (industry posts).** Source for the deterministic-keyword-matching pattern in production query routers. Google's query classifier uses signal-based deterministic routing for predictability + debuggability, with LLM-reasoned routing only for the residual that signals miss. + +5. **Robert Frost, "The Road Not Taken" (1916).** Cited tongue-in-cheek for the routing decision as a one-way door: once delegated, the user sees the specialist's output, not what fallback would have produced. Routing transparency is what gives the user the option to take the other road. + +6. **Kubernetes API server — admission controller chain.** Source for the chain-of-responsibility pattern: each handler classifies + either acts or passes to next. Routing transparency in Kubernetes is the auditable admission decision log. Same property in this skill via `routing_transparency_logger.py`. + +7. **Tom Preston-Werner — Semantic Versioning specification.** Source for the principle of explicit, predictable contracts over implicit behavior. SemVer's predictability is what made it adoptable; the same property applies to this skill's deterministic routing. + +8. **Jeff Hodges — "Notes on Distributed Systems for Young Bloods" (2013).** Source for the principle that explicit + visible system state is what makes operators trust + intervene. Routing transparency is the operator-trust property for skill orchestration. https://www.somethingsimilar.com/2013/01/14/notes-on-distributed-systems-for-young-bloods/ diff --git a/research/research/skills/research/scripts/classifier.py b/research/research/skills/research/scripts/classifier.py new file mode 100755 index 00000000..aeb0383d --- /dev/null +++ b/research/research/skills/research/scripts/classifier.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +""" +classifier.py — Deterministic SIGNALS-based routing classifier for the research orchestrator. + +Given a research question, returns the routing decision (specialist name or "fallback"), +matched signals per specialist, and confidence reasoning. + +The SIGNALS map is the post-PR-#657-audit canonical version: verb-noun-paired phrases +that route reliably, with NO bracketed placeholders (those over-trigger on generic +"research [topic]" queries that should fall back instead). + +Usage: + python classifier.py --question "What's the literature on PICO for sepsis?" + python classifier.py --question "..." --output json + python classifier.py --sample +""" + +import argparse +import json +import sys + +SIGNALS = { + "pulse": [ + "reddit", "hn", "hacker news", "x.com", "twitter", "buzz", + "sentiment", "trending", "what are people saying", + "what's happening", "the conversation around", + "pulse on", "take the pulse", "current conversation", + ], + "grants": [ + "nih", "grant", "grants for", "r01", "r21", "k-award", "reporter", + "nosi", "funding", "fda", "study section", "principal investigator", + ], + "litreview": [ + "literature review", "lit review", "litreview", "pico", "spider", + "systematic review", "review papers on", "research papers on", + "papers about", "meta-analysis", + ], + "syllabus": [ + "syllabus", "course outline", "curriculum", "reading list", + "for my class", "for my students", "course material", + ], + "patent": [ + "prior art", "fto", "freedom to operate", "patent", + "patent landscape", "invention", "novelty search", + "patent search", "ip landscape", + ], + "dossier": [ + "dossier on", "due diligence", "background check", + "prep me for", "competitor research", "investor diligence", + "interview prep", "research my competitor", "background on", + ], +} + + +def classify(question: str) -> dict: + """ + Apply the deterministic routing algorithm: + - score[S] = count of SIGNALS[S] substrings matched (case-insensitive) + - if max(score) >= 2: route to argmax + - elif max(score) == 1 AND only one specialist scored 1: route to that one + - else: route to "fallback" + """ + q = question.lower() + scores = {} + matched = {} + + for specialist, phrases in SIGNALS.items(): + hits = [p for p in phrases if p in q] + scores[specialist] = len(hits) + if hits: + matched[specialist] = hits + + max_score = max(scores.values()) if scores else 0 + top = [s for s, sc in scores.items() if sc == max_score and sc > 0] + + if max_score >= 2: + route_to = top[0] if len(top) == 1 else _pick_highest_priority(top, scores) + confidence = f"high ({max_score} signals)" + elif max_score == 1: + single_scorers = [s for s, sc in scores.items() if sc == 1] + if len(single_scorers) == 1: + route_to = single_scorers[0] + confidence = "weak (1 signal, single specialist)" + else: + route_to = "fallback" + confidence = "ambiguous (multiple specialists with 1 signal)" + else: + route_to = "fallback" + confidence = "no signals matched" + + return { + "route_to": route_to, + "confidence": confidence, + "scores": scores, + "matched_signals": matched, + "question": question, + } + + +def _pick_highest_priority(candidates: list, scores: dict) -> str: + """When max(score) is tied across specialists, prefer the one with the + most specific signals (longest matched phrase across SIGNALS map). This is + a tie-breaker; in practice ties at ≥2 are rare.""" + return sorted(candidates)[0] + + +def render_human(result: dict) -> str: + lines = [ + f"Question: {result['question']}", + f"Route to: {result['route_to']}", + f"Confidence: {result['confidence']}", + "", + "Per-specialist scores:", + ] + for s, sc in sorted(result["scores"].items(), key=lambda kv: -kv[1]): + lines.append(f" {s}: {sc}") + if result["matched_signals"]: + lines.append("") + lines.append("Matched signals:") + for s, phrases in result["matched_signals"].items(): + lines.append(f" {s}: {', '.join(repr(p) for p in phrases)}") + if result["route_to"] != "fallback": + lines.append("") + lines.append( + f"Routing transparency: 'Routing to `{result['route_to']}` because " + f"of {result['confidence']}. Override or proceed in 5s.'" + ) + else: + lines.append("") + lines.append("Routing transparency: 'No specialist matched. Running fallback.'") + return "\n".join(lines) + + +def main(): + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--question", help="The research question to classify.") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="Run with built-in sample question.") + args = p.parse_args() + + if args.sample: + args.question = "Can you do a systematic review of PICO frameworks for sepsis treatment? I need a meta-analysis." + + if not args.question: + p.error("either --question or --sample is required") + + result = classify(args.question) + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(render_human(result)) + + +if __name__ == "__main__": + main() diff --git a/research/research/skills/research/scripts/fallback_decomposer.py b/research/research/skills/research/scripts/fallback_decomposer.py new file mode 100755 index 00000000..77b80aa2 --- /dev/null +++ b/research/research/skills/research/scripts/fallback_decomposer.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +""" +fallback_decomposer.py — Heuristic question decomposer for the fallback workflow. + +Given a research question, returns 3-5 sub-questions using the +what / why / how / who / what's next framework. Deterministic + stdlib only. + +The output is a starting point; the orchestrator + user should refine before +search budget is committed. + +Usage: + python fallback_decomposer.py --question "How are health systems integrating LLM-based clinical decision support in 2026?" + python fallback_decomposer.py --question "..." --output json + python fallback_decomposer.py --sample +""" + +import argparse +import json +import re + + +FRAMEWORK = [ + ("what", "What is {topic} — definition, scope, and current state?"), + ("why", "Why does {topic} matter now — the forces driving attention or change?"), + ("how", "How is {topic} being implemented or applied — methods, players, examples?"), + ("who", "Who are the key actors in {topic} — leaders, critics, regulators, adopters?"), + ("whats_next", "What's next for {topic} — near-term trajectory, open questions, watchpoints?"), +] + + +def _extract_topic(question: str) -> str: + """Strip leading 'research', interrogatives, framing verbs to surface the topic noun phrase.""" + q = question.strip().rstrip("?").strip() + q = re.sub( + r"^(can you |could you |please |i need to |i want to |help me )", + "", q, flags=re.IGNORECASE, + ).strip() + q = re.sub( + r"^(research |look into |investigate |find me information on |" + r"find information on |do some research on |what do we know about |" + r"what is |what's |how are |how is |how do |why is |why are |" + r"who is |who are |when |where |tell me about )", + "", q, flags=re.IGNORECASE, + ).strip() + q = re.sub(r"\s+", " ", q) + return q or question.strip().rstrip("?") + + +def decompose(question: str, n: int = 5) -> dict: + """Build 3-5 sub-questions from the framework. n is capped at 5 and floored at 3.""" + n = max(3, min(5, n)) + topic = _extract_topic(question) + selected = FRAMEWORK[:n] + sub_questions = [ + {"label": label, "question": template.format(topic=topic)} + for label, template in selected + ] + return { + "question": question, + "extracted_topic": topic, + "sub_question_count": len(sub_questions), + "framework": "what/why/how/who/what's next", + "sub_questions": sub_questions, + "note": ("Starting point only. Refine sub-questions with the user " + "before committing search budget. Drop any that don't fit; " + "rewrite ones that do."), + } + + +def render_human(result: dict) -> str: + lines = [ + f"Question: {result['question']}", + f"Extracted topic: {result['extracted_topic']}", + f"Framework: {result['framework']}", + f"Sub-questions ({result['sub_question_count']}):", + ] + for i, sq in enumerate(result["sub_questions"], 1): + lines.append(f" {i}. [{sq['label']}] {sq['question']}") + lines.append("") + lines.append(f"Note: {result['note']}") + return "\n".join(lines) + + +def main(): + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--question", help="The research question to decompose.") + p.add_argument("--n", type=int, default=5, help="Number of sub-questions (3-5; default 5).") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="Run with built-in sample question.") + args = p.parse_args() + + if args.sample: + args.question = "How are health systems integrating LLM-based clinical decision support in 2026?" + + if not args.question: + p.error("either --question or --sample is required") + + result = decompose(args.question, n=args.n) + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(render_human(result)) + + +if __name__ == "__main__": + main() diff --git a/research/research/skills/research/scripts/routing_transparency_logger.py b/research/research/skills/research/scripts/routing_transparency_logger.py new file mode 100755 index 00000000..fbd0c1a2 --- /dev/null +++ b/research/research/skills/research/scripts/routing_transparency_logger.py @@ -0,0 +1,200 @@ +#!/usr/bin/env python3 +""" +routing_transparency_logger.py — JSON-backed audit log for the research orchestrator. + +Records every routing decision, override, and delegation handoff to a +per-session JSON file at ~/.research_sessions/.json. Stdlib only. + +Schema: + { + "session": "", + "created_at": "", + "events": [ + {"at": "", "type": "decision", "question": "...", "route_to": "...", "confidence": "...", "matched": {...}}, + {"at": "", "type": "override", "from": "...", "to": "...", "reason": "..."}, + {"at": "", "type": "delegation", "target": "...", "signals": "..."} + ] + } + +Usage: + python routing_transparency_logger.py --action record_decision --session demo --question "..." --route-to litreview --confidence "high (2 signals)" + python routing_transparency_logger.py --action record_override --session demo --from litreview --to fallback --reason "wanted general scope" + python routing_transparency_logger.py --action record_delegation --session demo --target litreview --signals "pico,meta-analysis" + python routing_transparency_logger.py --action read --session demo + python routing_transparency_logger.py --sample +""" + +import argparse +import json +import os +import sys +from datetime import datetime, timezone +from pathlib import Path + + +def _now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def _session_path(session: str) -> Path: + base = Path.home() / ".research_sessions" + base.mkdir(parents=True, exist_ok=True) + safe = "".join(c if c.isalnum() or c in ("-", "_") else "_" for c in session) + return base / f"{safe}.json" + + +def _load(session: str) -> dict: + path = _session_path(session) + if not path.exists(): + return {"session": session, "created_at": _now(), "events": []} + return json.loads(path.read_text(encoding="utf-8")) + + +def _save(session: str, data: dict) -> Path: + path = _session_path(session) + path.write_text(json.dumps(data, indent=2), encoding="utf-8") + return path + + +def record_decision(session: str, question: str, route_to: str, confidence: str, + matched: dict | None = None) -> dict: + data = _load(session) + event = { + "at": _now(), + "type": "decision", + "question": question, + "route_to": route_to, + "confidence": confidence, + "matched": matched or {}, + } + data["events"].append(event) + _save(session, data) + return event + + +def record_override(session: str, from_target: str, to_target: str, reason: str) -> dict: + data = _load(session) + event = { + "at": _now(), + "type": "override", + "from": from_target, + "to": to_target, + "reason": reason, + } + data["events"].append(event) + _save(session, data) + return event + + +def record_delegation(session: str, target: str, signals: str) -> dict: + data = _load(session) + event = { + "at": _now(), + "type": "delegation", + "target": target, + "signals": signals, + } + data["events"].append(event) + _save(session, data) + return event + + +def read(session: str) -> dict: + return _load(session) + + +def render_human(result: dict) -> str: + if "events" in result: + lines = [ + f"Session: {result['session']}", + f"Created: {result['created_at']}", + f"Events ({len(result['events'])}):", + ] + for e in result["events"]: + t = e.get("type") + if t == "decision": + lines.append(f" [{e['at']}] decision → {e['route_to']} ({e['confidence']})") + elif t == "override": + lines.append(f" [{e['at']}] override {e['from']} → {e['to']} ({e['reason']})") + elif t == "delegation": + lines.append(f" [{e['at']}] delegation → {e['target']} (signals: {e['signals']})") + else: + lines.append(f" [{e['at']}] {t}: {e}") + return "\n".join(lines) + return json.dumps(result, indent=2) + + +def main(): + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--action", + choices=["record_decision", "record_override", "record_delegation", "read"], + help="What to do.") + p.add_argument("--session", help="Session name (used as filename stem).") + p.add_argument("--question", help="(record_decision) The classified question.") + p.add_argument("--route-to", dest="route_to", help="(record_decision) Routing target.") + p.add_argument("--confidence", help="(record_decision) Confidence string.") + p.add_argument("--matched", help="(record_decision) Matched signals (JSON).") + p.add_argument("--from", dest="from_target", help="(record_override) Previous target.") + p.add_argument("--to", dest="to_target", help="(record_override) New target.") + p.add_argument("--reason", help="(record_override) Why user overrode.") + p.add_argument("--target", help="(record_delegation) Specialist target.") + p.add_argument("--signals", help="(record_delegation) Signals that matched.") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="Run a built-in 4-event sample sequence.") + args = p.parse_args() + + if args.sample: + session = "sample" + path = _session_path(session) + if path.exists(): + path.unlink() + record_decision(session, + "Can you review the literature on PICO for sepsis?", + "litreview", + "high (2 signals)", + {"litreview": ["pico", "literature"]}) + record_delegation(session, "litreview", "pico,literature") + record_decision(session, + "What's the buzz about Anthropic on HN?", + "pulse", + "high (2 signals)", + {"pulse": ["hn", "buzz"]}) + record_override(session, "pulse", "fallback", "wanted general scope") + result = read(session) + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(render_human(result)) + return + + if not args.action: + p.error("--action is required (unless --sample)") + if not args.session: + p.error("--session is required") + + if args.action == "record_decision": + if not (args.question and args.route_to and args.confidence): + p.error("record_decision requires --question, --route-to, --confidence") + matched = json.loads(args.matched) if args.matched else None + out = record_decision(args.session, args.question, args.route_to, args.confidence, matched) + elif args.action == "record_override": + if not (args.from_target and args.to_target and args.reason): + p.error("record_override requires --from, --to, --reason") + out = record_override(args.session, args.from_target, args.to_target, args.reason) + elif args.action == "record_delegation": + if not (args.target and args.signals): + p.error("record_delegation requires --target, --signals") + out = record_delegation(args.session, args.target, args.signals) + elif args.action == "read": + out = read(args.session) + else: + p.error(f"unknown action {args.action}") + + if args.output == "json": + print(json.dumps(out, indent=2)) + else: + print(render_human(out)) + + +if __name__ == "__main__": + main()