Merge branch 'main' into feat/shared-sibling-index-store

This commit is contained in:
Gergő Magyar 2026-09-24 17:17:23 +01:00 • committed by GitHub
commit 0b83561143
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 205 additions and 15 deletions

View file

@ -30,7 +30,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"langchain": "^1.5.11",
"lru-cache": "^11.5.3",
"lucide-react": "^1.44.0",
"lucide-react": "^1.46.0",
"mermaid": "^11.17.2",
"mnemonist": "^0.40.4",
"pandemonium": "^2.4.0",
@ -5865,9 +5865,9 @@
}
},
"node_modules/lucide-react": {
"version": "1.44.0",
"resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.44.0.tgz",
"integrity": "sha512-2egNApH4hX4j/qdCgRublh88+9u3mEhz9iSlW5ckm4kaQEqZbXbMr0l5u5JZLy8nmWRx2dbHGQkEDYz6C9aCgw==",
"version": "1.46.0",
"resolved": "https://registry.npmjs.org/lucide-react/-/lucide-react-1.46.0.tgz",
"integrity": "sha512-Bv+FZXgZPrxc/NCl1e7JJVQFLdiCxYgxNVhqoV7X0p6I8ADJo8DxBnK1auH0fZz4AmqOJ3jgneL4f1i8LJQRAA==",
"license": "ISC",
"peerDependencies": {
"react": "^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0"

View file

@ -40,7 +40,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"langchain": "^1.5.11",
"lru-cache": "^11.5.3",
"lucide-react": "^1.44.0",
"lucide-react": "^1.46.0",
"mermaid": "^11.17.2",
"mnemonist": "^0.40.4",
"pandemonium": "^2.4.0",

View file

@ -5,6 +5,8 @@
* Aggregation already pushes a hub onto every owning process. This helper
* is the last shaping step: slice each process to `max_symbols`, keep one
* row per `(id, process_id)`, then set `symbol_count` from those rows.
* `content` stays only on the first row for each symbol id so a hub in
* many flows does not repeat its source text.
*/
export type QueryProcessAttach = {
@ -55,6 +57,7 @@ export function shapeQueryProcessAttaches<S extends QueryProcessAttach>(
): { processes: QueryProcessCard[]; process_symbols: S[] } {
const { maxSymbolsPerProcess, chainByProcessId } = options;
const seen = new Set<string>();
const contentSeen = new Set<string>();
const process_symbols: S[] = [];
const countByProcess = new Map<string, number>();
@ -63,8 +66,16 @@ export function shapeQueryProcessAttaches<S extends QueryProcessAttach>(
const key = attachPairKey(s.id, s.process_id);
if (seen.has(key)) continue;
seen.add(key);
const row =
p.entryPointId && s.id === p.entryPointId ? ({ ...s, is_entry_point: true } as S) : s;
let row: S = s;
if (s.content !== undefined) {
if (contentSeen.has(s.id)) {
const { content: _content, ...rest } = s;
row = rest as S;
} else {
contentSeen.add(s.id);
}
}
if (p.entryPointId && s.id === p.entryPointId) row = { ...row, is_entry_point: true };
process_symbols.push(row);
countByProcess.set(s.process_id, (countByProcess.get(s.process_id) ?? 0) + 1);
}

View file

@ -63,7 +63,7 @@ function getNextStepHint(toolName: string, args: Record<string, any> | undefined
return `\n\n---\n**Next:** READ gitnexus://repo/{name}/context for any repo above to get its overview and check staleness. If pagination.hasMore is true, call list_repos again with offset set to pagination.nextOffset to fetch the rest.`;
case 'query':
return `\n\n---\n**Next:** To understand a specific symbol in depth, use context({name: "<symbol_name>"${repoParam}}) to see categorized refs and process participation.`;
return `\n\n---\n**Next:** To understand a specific symbol in depth, use context({name: "<symbol_name>"${repoParam}}) to see categorized refs and process participation.${args?.include_content === true ? ` For source of a process_symbols row without content, use context({uid: "<id>", include_content: true${repoParam}}).` : ''}`;
case 'context':
return `\n\n---\n**Next:** If planning changes, use impact({target: "${args?.name || '<name>'}", direction: "upstream"${repoParam}}) to check blast radius. To see execution flows, READ gitnexus://repo/${repoPath}/processes.`;

View file

@ -144,11 +144,11 @@ specify the "repo" parameter explicitly.`,
Returns ranked processes plus a flat process_symbols list. Join processes[].id to process_symbols[].process_id.
WHEN TO USE: Understanding how code works together. Use this when you need execution flows and relationships, not just file matches. Complements grep/IDE search.
AFTER THIS: Use context() on a specific symbol for 360-degree view (callers, callees, categorized refs).
AFTER THIS: Use context() on a specific symbol for 360-degree view (callers, callees, categorized refs). With include_content, context() also returns that symbol's source.
Returns results grouped by process (execution flow):
- processes: ranked execution flows with relevance priority. When a process has an HTTP endpoint, each item includes route and method string aliases plus routes: [{ url, method? }] (same shape as context). When chain_depth > 0, each item also includes chain — layered upstream callers + downstream callees from the process entry symbol (same BFS as context({chain_depth})).
- process_symbols: search-hit symbols in those flows with file locations and module (functional area). On the single-repo envelope { processes, process_symbols, definitions }: One row per (id, process_id) — the same symbol id may appear under more than one process. Join a process to its rows by process_id; symbol_count is the number of those rows. When the process entry is among those hits, it is marked is_entry_point: true. A repo of "@<group>" returns { group, query, results, per_repo } and does not include process_symbols. results[].symbol_count is the member's post-slice attach count; when service is set, it counts only attaches under that prefix. To get process_symbols for one member, query again with repo "@<group>/<memberPath>" (member path from group.yaml, or results[]._repo).
- process_symbols: search-hit symbols in those flows with file locations and module (functional area). On the single-repo envelope { processes, process_symbols, definitions }: One row per (id, process_id) — the same symbol id may appear under more than one process. Join a process to its rows by process_id; symbol_count is the number of those rows. When the process entry is among those hits, it is marked is_entry_point: true. With include_content, content appears only on the first row for each symbol id across the whole process_symbols array, not per process; later rows for that id omit it, even under a different process_id. To get content for a row without it, find the earlier row with the same id, or call context({uid: "<id>", include_content: true}). A repo of "@<group>" returns { group, query, results, per_repo } and does not include process_symbols. results[].symbol_count is the member's post-slice attach count; when service is set, it counts only attaches under that prefix. To get process_symbols for one member, query again with repo "@<group>/<memberPath>" (member path from group.yaml, or results[]._repo).
- definitions: standalone types/interfaces not in any process. Keyword hits on Route URLs (route_fts) are bridged to their handler via HANDLES_ROUTE (handlerSymbolId, routes) when the edge exists; use route_map({route}) for the full HTTP surface.
Hybrid ranking: BM25 keyword + semantic vector search, ranked by Reciprocal Rank Fusion.
@ -196,7 +196,7 @@ ${HOT_READ_STALENESS_NOTE}`,
include_content: {
type: 'boolean',
description:
'Include source text retained for matching symbols (default: false). The response reports contentAvailability; indexes built with content retention "none" explicitly report unavailable content.',
'Include source text retained for matching symbols (default: false). The response reports contentAvailability; indexes built with content retention "none" explicitly report unavailable content. Content is sent once per symbol id, on its first process_symbols row; context({uid: "<id>", include_content: true}) returns it for any row.',
default: false,
},
chain_depth: {

View file

@ -207,6 +207,19 @@ withTestLbugDB(
// leaked some other node's community onto it. It must have none.
expect(validate.module).toBeUndefined();
expect(validate.content).toBe('function validate() {}');
// validate is a step in two processes, so it has two process_symbols
// rows; content is emitted once per symbol id — only the first row
// carries it, the sibling row omits the key entirely.
const validateRows = (validateRes.process_symbols ?? []).filter(
(s: { id: string }) => s.id === 'func:validate',
);
expect(validateRows).toHaveLength(2);
const withContent = validateRows.filter(
(s: { content?: string }) => s.content === 'function validate() {}',
);
expect(withContent).toHaveLength(1);
const withoutContent = validateRows.filter((s: object) => !('content' in s));
expect(withoutContent).toHaveLength(1);
});
it('reports content capability for the default full profile', async () => {

View file

@ -102,7 +102,7 @@ describe('shapeQueryProcessAttaches', () => {
expect(processes[0]?.symbol_count).toBe(1);
});
it('keeps include_content on every emitted hub attach (KTD5)', () => {
it('keeps include_content only on the first row for a hub id', () => {
const { process_symbols } = shapeQueryProcessAttaches(
[
ranked('proc:login-flow', [
@ -115,10 +115,128 @@ describe('shapeQueryProcessAttaches', () => {
{ maxSymbolsPerProcess: 25 },
);
expect(process_symbols.map((s) => s.content)).toEqual([
'function validate() {}',
'function validate() {}',
expect(process_symbols.map((s) => s.process_id)).toEqual(['proc:login-flow', 'proc:beta-flow']);
expect(process_symbols[0]?.content).toBe('function validate() {}');
expect(process_symbols[1]).not.toHaveProperty('content');
});
it('strips content from a later entry-point row while flagging it', () => {
const { process_symbols } = shapeQueryProcessAttaches(
[
ranked('proc:login-flow', [
attach('func:validate', 'proc:login-flow', { content: 'function validate() {}' }),
]),
ranked(
'proc:beta-flow',
[attach('func:validate', 'proc:beta-flow', { content: 'function validate() {}' })],
{ entryPointId: 'func:validate' },
),
],
{ maxSymbolsPerProcess: 25 },
);
expect(process_symbols.map((s) => s.process_id)).toEqual(['proc:login-flow', 'proc:beta-flow']);
expect(process_symbols[0]?.content).toBe('function validate() {}');
expect(process_symbols[0]).not.toHaveProperty('is_entry_point');
expect(process_symbols[1]?.is_entry_point).toBe(true);
expect(process_symbols[1]).not.toHaveProperty('content');
});
it('keeps content only on the first of three rows for a hub id', () => {
const { processes, process_symbols } = shapeQueryProcessAttaches(
[
ranked('proc:login-flow', [
attach('func:validate', 'proc:login-flow', { content: 'function validate() {}' }),
]),
ranked('proc:beta-flow', [
attach('func:validate', 'proc:beta-flow', { content: 'function validate() {}' }),
]),
ranked('proc:gamma-flow', [
attach('func:validate', 'proc:gamma-flow', { content: 'function validate() {}' }),
]),
],
{ maxSymbolsPerProcess: 25 },
);
expect(process_symbols.map((s) => s.process_id)).toEqual([
'proc:login-flow',
'proc:beta-flow',
'proc:gamma-flow',
]);
expect(process_symbols.map((s) => Object.hasOwn(s, 'content'))).toEqual([true, false, false]);
expect(process_symbols[0]?.content).toBe('function validate() {}');
expect(processes.map((p) => [p.id, p.symbol_count])).toEqual([
['proc:login-flow', 1],
['proc:beta-flow', 1],
['proc:gamma-flow', 1],
]);
});
it('moves content to the first emitted row when max_symbols drops an earlier hub row', () => {
const { processes, process_symbols } = shapeQueryProcessAttaches(
[
ranked('proc:alpha-flow', [
attach('func:other', 'proc:alpha-flow', { content: 'function other() {}' }),
attach('func:validate', 'proc:alpha-flow', { content: 'function validate() {}' }),
]),
ranked('proc:beta-flow', [
attach('func:validate', 'proc:beta-flow', { content: 'function validate() {}' }),
]),
],
{ maxSymbolsPerProcess: 1 },
);
expect(process_symbols.map((s) => [s.id, s.process_id])).toEqual([
['func:other', 'proc:alpha-flow'],
['func:validate', 'proc:beta-flow'],
]);
const hubRows = process_symbols.filter((s) => s.id === 'func:validate');
expect(hubRows).toHaveLength(1);
expect(hubRows[0]?.content).toBe('function validate() {}');
expect(processes.map((p) => [p.id, p.symbol_count])).toEqual([
['proc:alpha-flow', 1],
['proc:beta-flow', 1],
]);
});
it('keeps content on a first row that is also the entry point', () => {
const { process_symbols } = shapeQueryProcessAttaches(
[
ranked(
'proc:login-flow',
[attach('func:validate', 'proc:login-flow', { content: 'function validate() {}' })],
{ entryPointId: 'func:validate' },
),
ranked('proc:beta-flow', [
attach('func:validate', 'proc:beta-flow', { content: 'function validate() {}' }),
]),
],
{ maxSymbolsPerProcess: 25 },
);
expect(process_symbols.map((s) => s.process_id)).toEqual(['proc:login-flow', 'proc:beta-flow']);
expect(process_symbols[0]?.content).toBe('function validate() {}');
expect(process_symbols[0]?.is_entry_point).toBe(true);
expect(process_symbols[1]).not.toHaveProperty('content');
expect(process_symbols[1]).not.toHaveProperty('is_entry_point');
});
it('does not mutate the input attach objects', () => {
const first = attach('func:validate', 'proc:login-flow', { content: 'function validate() {}' });
const second = attach('func:validate', 'proc:beta-flow', { content: 'function validate() {}' });
const before = structuredClone([first, second]);
shapeQueryProcessAttaches(
[
ranked('proc:login-flow', [first]),
ranked('proc:beta-flow', [second], { entryPointId: 'func:validate' }),
],
{ maxSymbolsPerProcess: 25 },
);
expect([first, second]).toEqual(before);
expect(second.content).toBe('function validate() {}');
expect(second).not.toHaveProperty('is_entry_point');
});
it('marks the entry-point hit and preserves process card extras', () => {

View file

@ -252,6 +252,37 @@ describe('getNextStepHint (via tool call response)', () => {
// The actual hint logic is tested via the integration path.
expect(backend.callTool).not.toHaveBeenCalled(); // not called until request
});
const QUERY_HINT_BASE =
'\n\n---\n**Next:** To understand a specific symbol in depth, use context({name: "<symbol_name>", repo: "demo"}) to see categorized refs and process participation.';
it('query hint is unchanged when include_content is not set', async () => {
const backend = createMockBackend({
callTool: vi.fn().mockResolvedValue({ processes: [] }),
});
const { text } = await callToolThroughServer(backend, 'query', {
search_query: 'auth',
repo: 'demo',
});
expect(text.endsWith(QUERY_HINT_BASE)).toBe(true);
expect(text).not.toContain('context({uid:');
});
it('query hint names the context({uid, include_content}) fallback when include_content is true', async () => {
const backend = createMockBackend({
callTool: vi.fn().mockResolvedValue({ processes: [] }),
});
const { text } = await callToolThroughServer(backend, 'query', {
search_query: 'auth',
repo: 'demo',
include_content: true,
});
expect(
text.endsWith(
`${QUERY_HINT_BASE} For source of a process_symbols row without content, use context({uid: "<id>", include_content: true, repo: "demo"}).`,
),
).toBe(true);
});
});
describe('MCP output budgets', () => {

View file

@ -136,6 +136,23 @@ describe('GITNEXUS_TOOLS', () => {
'when service is set, it counts only attaches under that prefix',
);
expect(queryTool.description).toContain('query again with repo "@<group>/<memberPath>"');
expect(queryTool.description).toContain(
'content appears only on the first row for each symbol id across the whole process_symbols array, not per process',
);
expect(queryTool.description).toContain('even under a different process_id');
expect(queryTool.description).toContain('context({uid: "<id>", include_content: true})');
expect(queryTool.description).toContain(
"With include_content, context() also returns that symbol's source.",
);
});
it('query include_content property states the once-per-id rule and the context() fallback', () => {
const queryTool = GITNEXUS_TOOLS.find((t) => t.name === 'query')!;
const description = queryTool.inputSchema.properties.include_content.description;
expect(description).toContain('Include source text retained for matching symbols');
expect(description).toContain(
'Content is sent once per symbol id, on its first process_symbols row; context({uid: "<id>", include_content: true}) returns it for any row.',
);
});
it('query tool requires "search_query" parameter (renamed from "query" for #2175)', () => {