diff --git a/gitnexus-web/package-lock.json b/gitnexus-web/package-lock.json index 6b31e660a..392117c8e 100644 --- a/gitnexus-web/package-lock.json +++ b/gitnexus-web/package-lock.json @@ -132,6 +132,7 @@ "integrity": "sha512-H3mcG6ZDLTlYfaSNi0iOKkigqMFvkTKlGUYlD8GW7nNOYRrevuA46iTypPyv+06V3fEmvvazfntkBU34L0azAw==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@babel/code-frame": "^7.28.6", "@babel/generator": "^7.28.6", @@ -1710,6 +1711,7 @@ "resolved": "https://registry.npmjs.org/@langchain/core/-/core-1.1.15.tgz", "integrity": "sha512-b8RN5DkWAmDAlMu/UpTZEluYwCLpm63PPWniRKlE8ie3KkkE7IuMQ38pf4kV1iaiI+d99BEQa2vafQHfCujsRA==", "license": "MIT", + "peer": true, "dependencies": { "@cfworker/json-schema": "^4.0.2", "ansi-styles": "^5.0.0", @@ -3692,6 +3694,7 @@ "resolved": "https://registry.npmjs.org/@types/node/-/node-24.10.9.tgz", "integrity": "sha512-ne4A0IpG3+2ETuREInjPNhUGis1SFjv1d5asp8MzEAGtOZeTeHVDOYqOgqfhvseqg/iXty2hjBf1zAOb7RNiNw==", "license": "MIT", + "peer": true, "dependencies": { "undici-types": "~7.16.0" } @@ -3713,6 +3716,7 @@ "resolved": "https://registry.npmjs.org/@types/react/-/react-18.3.27.tgz", "integrity": "sha512-cisd7gxkzjBKU2GgdYrTdtQx1SORymWyaAFhaxQPK9bYO9ot3Y5OikQRvY0VYQtvwjeQnizCINJAenh/V7MK2w==", "license": "MIT", + "peer": true, "dependencies": { "@types/prop-types": "*", "csstype": "^3.2.2" @@ -4014,6 +4018,7 @@ "resolved": "https://registry.npmjs.org/acorn/-/acorn-8.15.0.tgz", "integrity": "sha512-NZyJarBfL7nWwIq+FDL6Zp/yHEhePMNnnJ0y3qfieCrmNvYct8uvtiV41UvlSe6apAfk0fY1FbWx+NwfmpvtTg==", "license": "MIT", + "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -4303,6 +4308,7 @@ } ], "license": "MIT", + "peer": true, "dependencies": { "baseline-browser-mapping": "^2.9.0", "caniuse-lite": "^1.0.30001759", @@ -4756,6 +4762,7 @@ "resolved": "https://registry.npmjs.org/cytoscape/-/cytoscape-3.33.1.tgz", "integrity": "sha512-iJc4TwyANnOGR1OmWhsS9ayRS3s+XQ185FmuHObThD+5AeJCakAAbWv8KimMTt08xCCLNgneQwFp+JRJOr9qGQ==", "license": "MIT", + "peer": true, "engines": { "node": ">=0.10" } @@ -5156,6 +5163,7 @@ "resolved": "https://registry.npmjs.org/d3-selection/-/d3-selection-3.0.0.tgz", "integrity": "sha512-fmTRWbNMmsmWq6xJV8D19U/gw/bwrHfNXxrIN+HfZgnzqTHp9jOmKMhsTUjXOJnZOdZY9Q28y4yebKzqDKlxlQ==", "license": "ISC", + "peer": true, "engines": { "node": ">=12" } @@ -8834,6 +8842,7 @@ "resolved": "https://registry.npmjs.org/react/-/react-18.3.1.tgz", "integrity": "sha512-wS+hAgJShR0KhEvPJArfuPVN1+Hz1t0Y6n5jLrGQbkb4urgPE/0Rve+1kMB1v/oWgHgm4WIcV+i7F2pTVj+2iQ==", "license": "MIT", + "peer": true, "dependencies": { "loose-envify": "^1.1.0" }, @@ -8846,6 +8855,7 @@ "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-18.3.1.tgz", "integrity": "sha512-5m4nQKp+rZRb09LNH59GM4BxTh9251/ylbKIbpe7TpGxfJ+9kv6BLkLBXIjjspbgbnIBNqlI23tRnTWT0snUIw==", "license": "MIT", + "peer": true, "dependencies": { "loose-envify": "^1.1.0", "scheduler": "^0.23.2" @@ -9149,6 +9159,7 @@ "resolved": "https://registry.npmjs.org/rollup/-/rollup-4.55.1.tgz", "integrity": "sha512-wDv/Ht1BNHB4upNbK74s9usvl7hObDnvVzknxqY/E/O3X6rW1U1rV1aENEfJ54eFZDTNo7zv1f5N4edCluH7+A==", "license": "MIT", + "peer": true, "dependencies": { "@types/estree": "1.0.8" }, @@ -9397,6 +9408,7 @@ "resolved": "https://registry.npmjs.org/sigma/-/sigma-3.0.2.tgz", "integrity": "sha512-/BUbeOwPGruiBOm0YQQ6ZMcLIZ6tf/W+Jcm7dxZyAX0tK3WP9/sq7/NAWBxPIxVahdGjCJoGwej0Gdrv0DxlQQ==", "license": "MIT", + "peer": true, "dependencies": { "events": "^3.3.0", "graphology-utils": "^2.5.2" @@ -9865,6 +9877,7 @@ "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", "dev": true, "license": "Apache-2.0", + "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -10100,6 +10113,7 @@ "resolved": "https://registry.npmjs.org/vite/-/vite-5.4.21.tgz", "integrity": "sha512-o5a9xKjbtuhY6Bi5S3+HvbRERmouabWbyUcpXXUA1u+GNUKoROi9byOJ8M0nHbHYHkYICiMlqxkg1KkYmm25Sw==", "license": "MIT", + "peer": true, "dependencies": { "esbuild": "^0.21.3", "postcss": "^8.4.43", @@ -11240,6 +11254,7 @@ "resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz", "integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==", "license": "MIT", + "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } diff --git a/gitnexus/.env.example b/gitnexus/.env.example new file mode 100644 index 000000000..0c4cd297d --- /dev/null +++ b/gitnexus/.env.example @@ -0,0 +1,12 @@ +# GitNexus HTTP Embedding Configuration +# Copy to .env and uncomment to use a remote OpenAI-compatible endpoint +# instead of the local snowflake-arctic-embed-xs model. +# When unset, local embeddings are used unchanged. + +# GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1 +# GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5 +# GITNEXUS_EMBEDDING_DIMS=1024 +# GITNEXUS_EMBEDDING_API_KEY=your-key + +# Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. +# See README for details. diff --git a/gitnexus/README.md b/gitnexus/README.md index e4fe5b623..43c75db17 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -158,6 +158,20 @@ gitnexus wiki [path] # Generate LLM-powered docs from knowledge grap gitnexus wiki --model # Wiki with custom LLM model (default: gpt-4o-mini) ``` +## Remote Embeddings + +Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint instead of the local model: + +```bash +export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1 +export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5 +export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384 +export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused" +gitnexus analyze . --embeddings +``` + +Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. When unset, local embeddings are used unchanged. + ## Multi-Repo Support GitNexus supports indexing multiple repositories. Each `gitnexus analyze` registers the repo in a global registry (`~/.gitnexus/registry.json`). The MCP server serves all indexed repos automatically. diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index ff28a06d8..9e814e1a8 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -3129,6 +3129,7 @@ "resolved": "https://registry.npmjs.org/express/-/express-4.22.1.tgz", "integrity": "sha512-F2X8g9P1X7uCPZMA3MVf9wcTqlyNp7IhH5qPCI0izhaOIYXaW9L535tGA3qmjRzpH+bZczqq7hVKxTR4NWnu+g==", "license": "MIT", + "peer": true, "dependencies": { "accepts": "~1.3.8", "array-flatten": "1.1.1", @@ -4167,6 +4168,7 @@ "integrity": "sha512-5gTmgEY/sqK6gFXLIsQNH19lWb4ebPDLA4SdLP7dsWkIXHWlG66oPuVvXSGFPppYZz8ZDZq0dYYrbHfBCVUb1Q==", "dev": true, "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -4946,6 +4948,7 @@ "integrity": "sha512-7dxoA6kYvtgWw80265MyqJlkRl4yawIjO7S5MigytjELkX43fV2WsAXzsNfO7sBpPPCF5Gp0+XzHk0DwLCq3xQ==", "hasInstallScript": true, "license": "MIT", + "peer": true, "dependencies": { "node-addon-api": "^8.0.0", "node-gyp-build": "^4.8.0" @@ -5348,6 +5351,7 @@ "integrity": "sha512-5C1sg4USs1lfG0GFb2RLXsdpXqBSEhAaA/0kPL01wxzpMqLILNxIxIOKiILz+cdg/pLnOUxFYOR5yhHU666wbw==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "esbuild": "~0.27.0", "get-tsconfig": "^4.7.5" @@ -5468,6 +5472,7 @@ "integrity": "sha512-w+N7Hifpc3gRjZ63vYBXA56dvvRlNWRczTdmCBBa+CotUzAPf5b7YMdMR/8CQoeYE5LX3W4wj6RYTgonm1b9DA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "esbuild": "^0.27.0", "fdir": "^6.5.0", @@ -5543,6 +5548,7 @@ "integrity": "sha512-hOQuK7h0FGKgBAas7v0mSAsnvrIgAvWmRFjmzpJ7SwFHH3g1k2u37JtYwOwmEKhK6ZO3v9ggDBBm0La1LCK4uQ==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@vitest/expect": "4.0.18", "@vitest/mocker": "4.0.18", @@ -5826,6 +5832,7 @@ "resolved": "https://registry.npmjs.org/zod/-/zod-4.3.6.tgz", "integrity": "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg==", "license": "MIT", + "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 50bc7ef8d..77855fbb9 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -260,17 +260,27 @@ export const analyzeCommand = async ( // ── Phase 3.5: Re-insert cached embeddings ──────────────────────── if (cachedEmbeddings.length > 0) { - updateBar(88, `Restoring ${cachedEmbeddings.length} cached embeddings...`); - const EMBED_BATCH = 200; - for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) { - const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH); - const paramsList = batch.map(e => ({ nodeId: e.nodeId, embedding: e.embedding })); - try { - await executeWithReusedStatement( - `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`, - paramsList, - ); - } catch { /* some may fail if node was removed, that's fine */ } + // Check if cached embedding dimensions match current schema + const cachedDims = cachedEmbeddings[0].embedding.length; + const { EMBEDDING_DIMS } = await import('../core/lbug/schema.js'); + if (cachedDims !== EMBEDDING_DIMS) { + // Dimensions changed (e.g. switched embedding model) — discard cache and re-embed all + console.error(`⚠️ Embedding dimensions changed (${cachedDims}d → ${EMBEDDING_DIMS}d), discarding cache`); + cachedEmbeddings = []; + cachedEmbeddingNodeIds = new Set(); + } else { + updateBar(88, `Restoring ${cachedEmbeddings.length} cached embeddings...`); + const EMBED_BATCH = 200; + for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) { + const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH); + const paramsList = batch.map(e => ({ nodeId: e.nodeId, embedding: e.embedding })); + try { + await executeWithReusedStatement( + `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`, + paramsList, + ); + } catch { /* some may fail if node was removed, that's fine */ } + } } } @@ -289,7 +299,9 @@ export const analyzeCommand = async ( } if (!embeddingSkipped) { - updateBar(90, 'Loading embedding model...'); + const { isHttpMode } = await import('../core/embeddings/http-client.js'); + const httpMode = isHttpMode(); + updateBar(90, httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...'); const t0Emb = Date.now(); const { runEmbeddingPipeline } = await import('../core/embeddings/embedding-pipeline.js'); await runEmbeddingPipeline( @@ -297,7 +309,9 @@ export const analyzeCommand = async ( executeWithReusedStatement, (progress) => { const scaled = 90 + Math.round((progress.percent / 100) * 8); - const label = progress.phase === 'loading-model' ? 'Loading embedding model...' : `Embedding ${progress.nodesProcessed || 0}/${progress.totalNodes || '?'}`; + const label = progress.phase === 'loading-model' + ? (httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...') + : `Embedding ${progress.nodesProcessed || 0}/${progress.totalNodes || '?'}`; updateBar(scaled, label); }, {}, diff --git a/gitnexus/src/core/embeddings/embedder.ts b/gitnexus/src/core/embeddings/embedder.ts index 807123b0e..13ad35400 100644 --- a/gitnexus/src/core/embeddings/embedder.ts +++ b/gitnexus/src/core/embeddings/embedder.ts @@ -20,6 +20,7 @@ import { execFileSync } from 'child_process'; import { join, dirname } from 'path'; import { createRequire } from 'module'; import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js'; +import { isHttpMode, getHttpDimensions, httpEmbed } from './http-client.js'; /** * Check whether the onnxruntime-node package that @huggingface/transformers @@ -118,6 +119,13 @@ export const initEmbedder = async ( config: Partial = {}, forceDevice?: 'dml' | 'cuda' | 'cpu' | 'wasm' ): Promise => { + if (isHttpMode()) { + throw new Error( + 'initEmbedder() should not be called in HTTP mode. ' + + 'Use embedText()/embedBatch() which handle HTTP transparently.' + ); + } + // Return existing instance if available if (embedderInstance) { return embedderInstance; @@ -230,13 +238,27 @@ export const initEmbedder = async ( * Check if the embedder is initialized and ready */ export const isEmbedderReady = (): boolean => { - return embedderInstance !== null; + return isHttpMode() || embedderInstance !== null; +}; + +/** + * Get the effective embedding dimensions. + * In HTTP mode, uses GITNEXUS_EMBEDDING_DIMS if set, otherwise the default. + */ +export const getEmbeddingDimensions = (): number => { + if (isHttpMode()) { + return getHttpDimensions() ?? DEFAULT_EMBEDDING_CONFIG.dimensions; + } + return DEFAULT_EMBEDDING_CONFIG.dimensions; }; /** * Get the embedder instance (throws if not initialized) */ export const getEmbedder = (): FeatureExtractionPipeline => { + if (isHttpMode()) { + throw new Error('getEmbedder() is not available in HTTP embedding mode. Use embedText()/embedBatch() instead.'); + } if (!embedderInstance) { throw new Error('Embedder not initialized. Call initEmbedder() first.'); } @@ -247,9 +269,14 @@ export const getEmbedder = (): FeatureExtractionPipeline => { * Embed a single text string * * @param text - Text to embed - * @returns Float32Array of embedding vector (384 dimensions) + * @returns Float32Array of embedding vector */ export const embedText = async (text: string): Promise => { + if (isHttpMode()) { + const [vec] = await httpEmbed([text]); + return vec; + } + const embedder = getEmbedder(); const result = await embedder(text, { @@ -273,6 +300,10 @@ export const embedBatch = async (texts: string[]): Promise => { return []; } + if (isHttpMode()) { + return httpEmbed(texts); + } + const embedder = getEmbedder(); // Process batch diff --git a/gitnexus/src/core/embeddings/embedding-pipeline.ts b/gitnexus/src/core/embeddings/embedding-pipeline.ts index 5fbd5cd0b..bb99a6ee2 100644 --- a/gitnexus/src/core/embeddings/embedding-pipeline.ts +++ b/gitnexus/src/core/embeddings/embedding-pipeline.ts @@ -161,14 +161,16 @@ export const runEmbeddingPipeline = async ( modelDownloadPercent: 0, }); - await initEmbedder((modelProgress: ModelProgress) => { - const downloadPercent = modelProgress.progress ?? 0; - onProgress({ - phase: 'loading-model', - percent: Math.round(downloadPercent * 0.2), - modelDownloadPercent: downloadPercent, - }); - }, finalConfig); + if (!isEmbedderReady()) { + await initEmbedder((modelProgress: ModelProgress) => { + const downloadPercent = modelProgress.progress ?? 0; + onProgress({ + phase: 'loading-model', + percent: Math.round(downloadPercent * 0.2), + modelDownloadPercent: downloadPercent, + }); + }, finalConfig); + } onProgress({ phase: 'loading-model', @@ -326,7 +328,7 @@ export const semanticSearch = async ( // Query the vector index on CodeEmbedding to get nodeIds and distances const vectorQuery = ` CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', - CAST(${queryVecStr} AS FLOAT[384]), ${k}) + CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${k}) YIELD node AS emb, distance WITH emb, distance WHERE distance < ${maxDistance} diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts new file mode 100644 index 000000000..b16496440 --- /dev/null +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -0,0 +1,226 @@ +/** + * HTTP Embedding Client + * + * Shared fetch+retry logic for OpenAI-compatible /v1/embeddings endpoints. + * Imported by both the core embedder (batch) and MCP embedder (query). + */ + +const HTTP_TIMEOUT_MS = 30_000; +const HTTP_MAX_RETRIES = 2; +const HTTP_RETRY_BACKOFF_MS = 1_000; +const HTTP_BATCH_SIZE = 64; +const DEFAULT_DIMS = 384; + +interface HttpConfig { + baseUrl: string; + model: string; + apiKey: string; + dimensions?: number; +} + +/** + * Build config from the current process.env snapshot. + * Returns null when GITNEXUS_EMBEDDING_URL + GITNEXUS_EMBEDDING_MODEL are unset. + * Not cached — env vars are read fresh so late configuration takes effect. + */ +const readConfig = (): HttpConfig | null => { + const baseUrl = process.env.GITNEXUS_EMBEDDING_URL; + const model = process.env.GITNEXUS_EMBEDDING_MODEL; + if (!baseUrl || !model) return null; + + const rawDims = process.env.GITNEXUS_EMBEDDING_DIMS; + let dimensions: number | undefined; + if (rawDims !== undefined) { + const parsed = parseInt(rawDims, 10); + if (Number.isNaN(parsed) || parsed <= 0) { + throw new Error( + `GITNEXUS_EMBEDDING_DIMS must be a positive integer, got "${rawDims}"`, + ); + } + dimensions = parsed; + } + + return { + baseUrl: baseUrl.replace(/\/+$/, ''), + model, + apiKey: process.env.GITNEXUS_EMBEDDING_API_KEY ?? 'unused', + dimensions, + }; +}; + +/** + * Check whether HTTP embedding mode is active (env vars are set). + */ +export const isHttpMode = (): boolean => readConfig() !== null; + +/** + * Return the configured embedding dimensions for HTTP mode, or undefined + * if HTTP mode is not active or no explicit dimensions are set. + */ +export const getHttpDimensions = (): number | undefined => readConfig()?.dimensions; + +/** + * Return a safe representation of a URL for error messages. + * Strips query string (may contain tokens) and userinfo. + */ +const safeUrl = (url: string): string => { + try { + const u = new URL(url); + return `${u.protocol}//${u.host}${u.pathname}`; + } catch { + return ''; + } +}; + +interface EmbeddingItem { + embedding: number[]; +} + +/** + * Send a single batch of texts to the embedding endpoint with retry. + * + * @param url - Full endpoint URL (e.g. https://host/v1/embeddings) + * @param batch - Texts to embed + * @param model - Model name for the request body + * @param apiKey - Bearer token (only used in Authorization header) + * @param batchIndex - Logical batch number (for error context) + * @param attempt - Current retry attempt (internal) + */ +const httpEmbedBatch = async ( + url: string, + batch: string[], + model: string, + apiKey: string, + batchIndex = 0, + attempt = 0, +): Promise => { + let resp: Response; + try { + resp = await fetch(url, { + method: 'POST', + signal: AbortSignal.timeout(HTTP_TIMEOUT_MS), + headers: { + 'Content-Type': 'application/json', + 'Authorization': `Bearer ${apiKey}`, + }, + body: JSON.stringify({ input: batch, model }), + }); + } catch (err) { + // Timeouts should not be retried — the server is unresponsive. + // AbortSignal.timeout() throws DOMException with name 'TimeoutError'. + const isTimeout = err instanceof DOMException && err.name === 'TimeoutError'; + if (isTimeout) { + throw new Error( + `Embedding request timed out after ${HTTP_TIMEOUT_MS}ms (${safeUrl(url)}, batch ${batchIndex})`, + ); + } + // DNS, connection errors — retry with backoff + if (attempt < HTTP_MAX_RETRIES) { + const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1); + await new Promise(r => setTimeout(r, delay)); + return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1); + } + const reason = err instanceof Error ? err.message : String(err); + throw new Error( + `Embedding request failed (${safeUrl(url)}, batch ${batchIndex}): ${reason}`, + ); + } + + if (!resp.ok) { + const status = resp.status; + if ((status === 429 || status >= 500) && attempt < HTTP_MAX_RETRIES) { + const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1); + await new Promise(r => setTimeout(r, delay)); + return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1); + } + throw new Error( + `Embedding endpoint returned ${status} (${safeUrl(url)}, batch ${batchIndex})`, + ); + } + + const data = (await resp.json()) as { data: EmbeddingItem[] }; + return data.data; +}; + +/** + * Embed texts via the HTTP backend, splitting into batches. + * Reads config from env vars on every call. + * + * @param texts - Array of texts to embed + * @returns Array of Float32Array embedding vectors + */ +export const httpEmbed = async (texts: string[]): Promise => { + if (texts.length === 0) return []; + + const config = readConfig(); + if (!config) throw new Error('HTTP embedding not configured'); + + const url = `${config.baseUrl}/embeddings`; + const allVectors: Float32Array[] = []; + + for (let i = 0; i < texts.length; i += HTTP_BATCH_SIZE) { + const batch = texts.slice(i, i + HTTP_BATCH_SIZE); + const batchIndex = Math.floor(i / HTTP_BATCH_SIZE); + const items = await httpEmbedBatch(url, batch, config.model, config.apiKey, batchIndex); + + if (items.length !== batch.length) { + throw new Error( + `Embedding endpoint returned ${items.length} vectors for ${batch.length} texts ` + + `(${safeUrl(url)}, batch ${batchIndex})`, + ); + } + + for (const item of items) { + const vec = new Float32Array(item.embedding); + // Fail fast on dimension mismatch rather than inserting bad vectors + // into the FLOAT[N] column which would cause a cryptic Kuzu error. + const expected = config.dimensions ?? DEFAULT_DIMS; + if (vec.length !== expected) { + const hint = config.dimensions + ? 'Update GITNEXUS_EMBEDDING_DIMS to match your model output.' + : `Set GITNEXUS_EMBEDDING_DIMS=${vec.length} to match your model output.`; + throw new Error( + `Embedding dimension mismatch: endpoint returned ${vec.length}d vector, ` + + `but expected ${expected}d. ${hint}`, + ); + } + + allVectors.push(vec); + } + } + + return allVectors; +}; + +/** + * Embed a single query text via the HTTP backend. + * Convenience for MCP search where only one vector is needed. + * + * @param text - Query text to embed + * @returns Embedding vector as number array + */ +export const httpEmbedQuery = async (text: string): Promise => { + const config = readConfig(); + if (!config) throw new Error('HTTP embedding not configured'); + + const url = `${config.baseUrl}/embeddings`; + const items = await httpEmbedBatch(url, [text], config.model, config.apiKey); + if (!items.length) { + throw new Error(`Embedding endpoint returned empty response (${safeUrl(url)})`); + } + + const embedding = items[0].embedding; + // Same dimension checks as httpEmbed — catch mismatches before they + // reach the Kuzu FLOAT[N] cast in search queries. + const expected = config.dimensions ?? DEFAULT_DIMS; + if (embedding.length !== expected) { + const hint = config.dimensions + ? 'Update GITNEXUS_EMBEDDING_DIMS to match your model output.' + : `Set GITNEXUS_EMBEDDING_DIMS=${embedding.length} to match your model output.`; + throw new Error( + `Embedding dimension mismatch: endpoint returned ${embedding.length}d vector, ` + + `but expected ${expected}d. ${hint}`, + ); + } + return embedding; +}; diff --git a/gitnexus/src/core/embeddings/index.ts b/gitnexus/src/core/embeddings/index.ts index 4b4f10bb5..19b326187 100644 --- a/gitnexus/src/core/embeddings/index.ts +++ b/gitnexus/src/core/embeddings/index.ts @@ -5,6 +5,7 @@ */ export * from './types.js'; +export * from './http-client.js'; export * from './embedder.js'; export * from './text-generator.js'; export * from './embedding-pipeline.js'; diff --git a/gitnexus/src/core/embeddings/types.ts b/gitnexus/src/core/embeddings/types.ts index 7978bf4c0..25af985c8 100644 --- a/gitnexus/src/core/embeddings/types.ts +++ b/gitnexus/src/core/embeddings/types.ts @@ -53,7 +53,7 @@ export interface EmbeddingProgress { * Configuration for the embedding pipeline */ export interface EmbeddingConfig { - /** Model identifier for transformers.js */ + /** Model identifier for transformers.js (local) or the HTTP endpoint model name */ modelId: string; /** Number of nodes to embed in each batch */ batchSize: number; @@ -65,6 +65,7 @@ export interface EmbeddingConfig { maxSnippetLength: number; } + /** * Default embedding configuration * Uses snowflake-arctic-embed-xs for browser efficiency diff --git a/gitnexus/src/core/lbug/schema.ts b/gitnexus/src/core/lbug/schema.ts index a47aa9674..0e4b80b07 100644 --- a/gitnexus/src/core/lbug/schema.ts +++ b/gitnexus/src/core/lbug/schema.ts @@ -414,10 +414,19 @@ CREATE REL TABLE ${REL_TABLE_NAME} ( // Separate table for vector storage to avoid copy-on-write overhead // ============================================================================ +/** Embedding vector dimensions. Default 384 (snowflake-arctic-embed-xs). */ +const _rawDims = parseInt(process.env.GITNEXUS_EMBEDDING_DIMS ?? '384', 10); +if (Number.isNaN(_rawDims) || _rawDims <= 0) { + throw new Error( + `GITNEXUS_EMBEDDING_DIMS must be a positive integer, got "${process.env.GITNEXUS_EMBEDDING_DIMS}"`, + ); +} +export const EMBEDDING_DIMS = _rawDims; + export const EMBEDDING_SCHEMA = ` CREATE NODE TABLE ${EMBEDDING_TABLE_NAME} ( nodeId STRING, - embedding FLOAT[384], + embedding FLOAT[${EMBEDDING_DIMS}], PRIMARY KEY (nodeId) )`; diff --git a/gitnexus/src/mcp/core/embedder.ts b/gitnexus/src/mcp/core/embedder.ts index ee480a6a9..11261ff36 100644 --- a/gitnexus/src/mcp/core/embedder.ts +++ b/gitnexus/src/mcp/core/embedder.ts @@ -6,10 +6,10 @@ */ import { pipeline, env, type FeatureExtractionPipeline } from '@huggingface/transformers'; +import { isHttpMode, getHttpDimensions, httpEmbedQuery } from '../../core/embeddings/http-client.js'; // Model config const MODEL_ID = 'Snowflake/snowflake-arctic-embed-xs'; -const EMBEDDING_DIMS = 384; // Module-level state for singleton pattern let embedderInstance: FeatureExtractionPipeline | null = null; @@ -20,6 +20,10 @@ let initPromise: Promise | null = null; * Initialize the embedding model (lazy, on first search) */ export const initEmbedder = async (): Promise => { + if (isHttpMode()) { + throw new Error('initEmbedder() should not be called in HTTP mode.'); + } + if (embedderInstance) { return embedderInstance; } @@ -87,12 +91,16 @@ export const initEmbedder = async (): Promise => { /** * Check if embedder is ready */ -export const isEmbedderReady = (): boolean => embedderInstance !== null; +export const isEmbedderReady = (): boolean => isHttpMode() || embedderInstance !== null; /** * Embed a query text for semantic search */ export const embedQuery = async (query: string): Promise => { + if (isHttpMode()) { + return httpEmbedQuery(query); + } + const embedder = await initEmbedder(); const result = await embedder(query, { @@ -106,7 +114,9 @@ export const embedQuery = async (query: string): Promise => { /** * Get embedding dimensions */ -export const getEmbeddingDims = (): number => EMBEDDING_DIMS; +export const getEmbeddingDims = (): number => { + return getHttpDimensions() ?? 384; +}; /** * Cleanup embedder diff --git a/gitnexus/test/unit/http-embedder.test.ts b/gitnexus/test/unit/http-embedder.test.ts new file mode 100644 index 000000000..e0e9150be --- /dev/null +++ b/gitnexus/test/unit/http-embedder.test.ts @@ -0,0 +1,327 @@ +import { describe, it, expect, vi, afterEach } from 'vitest'; +import { getEmbeddingDims, isEmbedderReady } from '../../src/mcp/core/embedder.js'; + +const ENV_KEYS = [ + 'GITNEXUS_EMBEDDING_URL', + 'GITNEXUS_EMBEDDING_MODEL', + 'GITNEXUS_EMBEDDING_API_KEY', + 'GITNEXUS_EMBEDDING_DIMS', +] as const; + +/** 384d mock vector matching the default schema dimensions. */ +const mockVec = Array.from({ length: 384 }, (_, i) => i / 384); + +describe('HTTP embedding backend', () => { + // Save original env state before any test mutates it + const savedEnv = Object.fromEntries( + ENV_KEYS.map(k => [k, process.env[k]]), + ); + + afterEach(() => { + vi.unstubAllGlobals(); + vi.resetModules(); + // Restore env vars to pre-test state so a mid-test throw can't leak + for (const key of ENV_KEYS) { + if (savedEnv[key] === undefined) { + delete process.env[key]; + } else { + process.env[key] = savedEnv[key]; + } + } + }); + + describe('MCP embedder', () => { + it('returns 384 dimensions by default', () => { + expect(getEmbeddingDims()).toBe(384); + }); + + it('returns false before initialization', () => { + expect(isEmbedderReady()).toBe(false); + }); + + it('returns true when HTTP environment variables are set', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://localhost:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + const mod = await import('../../src/mcp/core/embedder.js'); + expect(mod.isEmbedderReady()).toBe(true); + }); + + it('reads custom dimensions from environment', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://localhost:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + const mod = await import('../../src/mcp/core/embedder.js'); + expect(mod.getEmbeddingDims()).toBe(1024); + }); + + it('retries query on transient server error', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const ok = { ok: true, json: async () => ({ data: [{ embedding: mockVec }] }) }; + vi.stubGlobal('fetch', vi.fn() + .mockResolvedValueOnce({ ok: false, status: 503 }) + .mockResolvedValueOnce(ok)); + + const mod = await import('../../src/mcp/core/embedder.js'); + const result = await mod.embedQuery('test query'); + + expect(fetch).toHaveBeenCalledTimes(2); + expect(result).toEqual(mockVec); + + }); + }); + + describe('core embedder HTTP path', () => { + it('sends correct request payload', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_API_KEY = 'test-key'; + + const mockEmbedding = Array.from({ length: 384 }, (_, i) => i * 0.001); + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: mockEmbedding }] }), + })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + expect(fetch).toHaveBeenCalledOnce(); + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect(body.model).toBe('test-model'); + expect(body.input).toEqual(['test text']); + expect(result).toBeInstanceOf(Float32Array); + expect(result.length).toBe(384); + + }); + + it('retries on server error', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const ok = { ok: true, json: async () => ({ data: [{ embedding: mockVec }] }) }; + vi.stubGlobal('fetch', vi.fn() + .mockResolvedValueOnce({ ok: false, status: 503 }) + .mockResolvedValueOnce(ok)); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await embedText('test'); + expect(fetch).toHaveBeenCalledTimes(2); + + }); + + it('retries on rate limit', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const ok = { ok: true, json: async () => ({ data: [{ embedding: mockVec }] }) }; + vi.stubGlobal('fetch', vi.fn() + .mockResolvedValueOnce({ ok: false, status: 429 }) + .mockResolvedValueOnce(ok)); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await embedText('test'); + expect(fetch).toHaveBeenCalledTimes(2); + + }); + + it('throws when all retries are exhausted', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ ok: false, status: 500 })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await expect(embedText('test')).rejects.toThrow('500'); + + }); + + it('excludes API key from error messages', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_API_KEY = 'secret-key-12345'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ ok: false, status: 500 })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + try { + await embedText('test'); + } catch (e: any) { + expect(e.message).not.toContain('secret-key-12345'); + expect(e.message).not.toContain('Authorization'); + } + + }); + + it('includes abort signal for timeout', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: mockVec }] }), + })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await embedText('test'); + + const opts = (fetch as any).mock.calls[0][1]; + expect(opts.signal).toBeDefined(); + + }); + + it('splits large inputs into batches', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const makeResp = (n: number) => ({ + ok: true, + json: async () => ({ data: Array.from({ length: n }, () => ({ embedding: mockVec })) }), + }); + vi.stubGlobal('fetch', vi.fn() + .mockResolvedValueOnce(makeResp(64)) + .mockResolvedValueOnce(makeResp(6))); + + const { embedBatch } = await import('../../src/core/embeddings/embedder.js'); + const results = await embedBatch(Array.from({ length: 70 }, (_, i) => `text ${i}`)); + + expect(fetch).toHaveBeenCalledTimes(2); + expect(results).toHaveLength(70); + + }); + + it('rejects initEmbedder when using HTTP backend', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const { initEmbedder } = await import('../../src/core/embeddings/embedder.js'); + await expect(initEmbedder()).rejects.toThrow('HTTP mode'); + + }); + + it('rejects getEmbedder when using HTTP backend', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const { getEmbedder } = await import('../../src/core/embeddings/embedder.js'); + expect(() => getEmbedder()).toThrow('HTTP embedding mode'); + + }); + + it('throws on empty response from endpoint', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [] }), + })); + + const mod = await import('../../src/mcp/core/embedder.js'); + await expect(mod.embedQuery('test')).rejects.toThrow('empty response'); + + }); + + it('throws when endpoint returns fewer embeddings than texts', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: mockVec }] }), + })); + + const { embedBatch } = await import('../../src/core/embeddings/embedder.js'); + await expect(embedBatch(['text1', 'text2', 'text3'])).rejects.toThrow('1 vectors for 3 texts'); + + }); + + it('throws on dimension mismatch when GITNEXUS_EMBEDDING_DIMS is set', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_DIMS = '512'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: [0.1, 0.2, 0.3] }] }), + })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await expect(embedText('test')).rejects.toThrow('Embedding dimension mismatch'); + + }); + }); + + describe('schema dimensions', () => { + it('defaults to 384 dimensions', async () => { + const { EMBEDDING_DIMS } = await import('../../src/core/lbug/schema.js'); + expect(EMBEDDING_DIMS).toBe(384); + }); + + it('reads dimensions from environment variable', async () => { + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + const { EMBEDDING_DIMS } = await import('../../src/core/lbug/schema.js'); + expect(EMBEDDING_DIMS).toBe(1024); + }); + }); + + describe('timeout and network error handling', () => { + it('does not retry on timeout', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const timeoutErr = new DOMException('The operation was aborted due to timeout', 'TimeoutError'); + vi.stubGlobal('fetch', vi.fn().mockRejectedValue(timeoutErr)); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await expect(embedText('test')).rejects.toThrow('timed out'); + expect(fetch).toHaveBeenCalledTimes(1); + }); + + it('retries on network error then succeeds', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const ok = { ok: true, json: async () => ({ data: [{ embedding: mockVec }] }) }; + vi.stubGlobal('fetch', vi.fn() + .mockRejectedValueOnce(new TypeError('fetch failed')) + .mockResolvedValueOnce(ok)); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test'); + expect(fetch).toHaveBeenCalledTimes(2); + expect(result).toBeInstanceOf(Float32Array); + }); + }); + + describe('dimension mismatch on query path', () => { + it('throws on explicit dim mismatch in embedQuery', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_DIMS = '512'; + + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: mockVec }] }), + })); + + const mod = await import('../../src/mcp/core/embedder.js'); + await expect(mod.embedQuery('test')).rejects.toThrow('dimension mismatch'); + }); + + it('throws with Set hint when GITNEXUS_EMBEDDING_DIMS is unset', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + + const vec768 = Array.from({ length: 768 }, (_, i) => i / 768); + vi.stubGlobal('fetch', vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec768 }] }), + })); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + await expect(embedText('test')).rejects.toThrow('Set GITNEXUS_EMBEDDING_DIMS=768'); + }); + }); +});