diff --git a/.env.example b/.env.example index 5839167..d5ee937 100644 --- a/.env.example +++ b/.env.example @@ -633,6 +633,34 @@ HEADROOM_LLMLINGUA=false # Values: auto | cpu | cuda | mps HEADROOM_LLMLINGUA_DEVICE=auto +# DESCRIPTION: TencentDB-Agent-Memory sidecar (team memory hub: L0-L3 chat +# memory, Skills, Wiki, CodeGraph). When enabled, `lynkr start` launches the +# memory-core + memory-hub containers from Docker Hub (agentmemory/*) — same +# pattern as the Headroom sidecar. Panel UI: http://localhost:8125 +# Requires Docker (skips gracefully if unavailable). Containers persist +# across Lynkr restarts (remove with: docker rm -f tdai-memory-core tdai-memory-hub). +# Values: true | false +TENCENTDB_MEMORY_ENABLED=true +# DESCRIPTION: Let Lynkr manage the containers. Set false if you run the +# stack yourself via the project's deploy scripts. +# Values: true | false +# TENCENTDB_MEMORY_DOCKER_ENABLED=true +# DESCRIPTION: LLM endpoint the memory services use for extraction and wiki +# ingest. Defaults to Lynkr's own OpenAI-compatible endpoint (tier routing +# picks the model), so no extra API key is needed. Override to point at a +# provider directly. +# TENCENTDB_MEMORY_LLM_BASE_URL=http://host.docker.internal:8081/v1 +# TENCENTDB_MEMORY_LLM_API_KEY=lynkr-local +# TENCENTDB_MEMORY_LLM_MODEL=auto +# DESCRIPTION: Memory extraction style. `code` extracts changes/issues/tool +# usage (coding agents); `chat` extracts general conversational facts. +# Values: code | chat +# TENCENTDB_MEMORY_PROMPT_MODE=code +# DESCRIPTION: Host port overrides (defaults shown). +# TENCENTDB_MEMORY_CORE_PORT=8420 +# TENCENTDB_MEMORY_PANEL_PORT=8125 +# TENCENTDB_MEMORY_KNOWLEDGE_PORT=8424 + # DESCRIPTION: Master switch for the long-term Titans-inspired memory system. # Values: true | false MEMORY_ENABLED=true diff --git a/bin/lynkr-init.js b/bin/lynkr-init.js index b246a76..59ec9bd 100644 --- a/bin/lynkr-init.js +++ b/bin/lynkr-init.js @@ -236,6 +236,13 @@ const BASELINE_ENV = { HEADROOM_CCR: 'true', HEADROOM_CCR_TTL: '300', + // ── TencentDB-Agent-Memory sidecar (team memory hub) ────────────────── + // Launches agentmemory/memory-core + memory-hub containers on + // `lynkr start` (Panel UI :8125). First start pulls two Docker Hub + // images; skips gracefully when Docker isn't running. See .env.example + // for all knobs. + TENCENTDB_MEMORY_ENABLED: 'true', + // ── Memory + token tracking ─────────────────────────────────────────── MEMORY_ENABLED: 'true', MEMORY_RETRIEVAL_LIMIT: '5', diff --git a/package.json b/package.json index e16393b..f82fd67 100644 --- a/package.json +++ b/package.json @@ -36,8 +36,8 @@ "dev": "nodemon index.js", "lint": "eslint src index.js", "test": "npm run test:unit && npm run test:performance", - "test:unit": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com LOG_FILE_ENABLED=false node --test test/routing.test.js test/hybrid-routing-integration.test.js test/retry-logic.test.js test/sse-transformer.test.js test/passthrough-stream.test.js test/passthrough-mode.test.js test/openrouter-error-resilience.test.js test/format-conversion.test.js test/azure-openai-config.test.js test/azure-openai-format-conversion.test.js test/azure-openai-routing.test.js test/azure-openai-streaming.test.js test/azure-openai-error-resilience.test.js test/azure-openai-integration.test.js test/openai-integration.test.js test/toon-compression.test.js test/gcf-compression.test.js test/llamacpp-integration.test.js test/resilience.test.js test/telemetry-routing.test.js test/memory/store.test.js test/memory/surprise.test.js test/memory/extractor.test.js test/memory/search.test.js test/memory/retriever.test.js test/distill.test.js test/large-payload.test.js test/prompt-cache-injection.test.js test/risk-analyzer.test.js test/interaction-block.test.js test/preflight.test.js test/token-reduction.test.js test/session-affinity.test.js test/model-registry-cost.test.js test/output-format-guard.test.js test/tier-fallback.test.js test/wrap.test.js test/init.test.js test/tool-call-response-metadata.test.js test/degradation.test.js test/routing-telemetry-columns.test.js test/sticky-routing.test.js test/knn-ambiguous-escalate.test.js test/deescalator.test.js test/client-profiles.test.js test/strip-internal-fields.test.js test/complexity-tool-subtraction.test.js test/bandit.test.js test/routing-propensity.test.js test/reward-pipeline.test.js test/knn-cold-start.test.js test/calibration.test.js test/feedback-loop.test.js test/session-fingerprint.test.js test/side-request-guards.test.js test/verifier.test.js test/intent-score.test.js test/difficulty-classifier.test.js test/classifier-setup.test.js test/usage-stats.test.js test/loop-guard.test.js test/moonshot-model-mapping.test.js", - "test:memory": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node --test test/memory/store.test.js test/memory/surprise.test.js test/memory/extractor.test.js test/memory/search.test.js test/memory/retriever.test.js", + "test:unit": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com LOG_FILE_ENABLED=false node --test test/routing.test.js test/hybrid-routing-integration.test.js test/retry-logic.test.js test/sse-transformer.test.js test/passthrough-stream.test.js test/passthrough-mode.test.js test/openrouter-error-resilience.test.js test/format-conversion.test.js test/azure-openai-config.test.js test/azure-openai-format-conversion.test.js test/azure-openai-routing.test.js test/azure-openai-streaming.test.js test/azure-openai-error-resilience.test.js test/azure-openai-integration.test.js test/openai-integration.test.js test/toon-compression.test.js test/gcf-compression.test.js test/llamacpp-integration.test.js test/resilience.test.js test/telemetry-routing.test.js test/memory/store.test.js test/memory/surprise.test.js test/memory/extractor.test.js test/memory/search.test.js test/memory/retriever.test.js test/memory/distiller.test.js test/memory/distiller-freeze.test.js test/memory/wiki.test.js test/memory/skills-cache.test.js test/memory/tencentdb-launcher.test.js test/distill.test.js test/large-payload.test.js test/prompt-cache-injection.test.js test/risk-analyzer.test.js test/interaction-block.test.js test/preflight.test.js test/token-reduction.test.js test/session-affinity.test.js test/cache-state.test.js test/cache-switch-cost.test.js test/model-registry-cost.test.js test/output-format-guard.test.js test/tier-fallback.test.js test/wrap.test.js test/init.test.js test/tool-call-response-metadata.test.js test/degradation.test.js test/routing-telemetry-columns.test.js test/sticky-routing.test.js test/knn-ambiguous-escalate.test.js test/deescalator.test.js test/client-profiles.test.js test/strip-internal-fields.test.js test/complexity-tool-subtraction.test.js test/bandit.test.js test/routing-propensity.test.js test/reward-pipeline.test.js test/knn-cold-start.test.js test/calibration.test.js test/feedback-loop.test.js test/session-fingerprint.test.js test/side-request-guards.test.js test/verifier.test.js test/intent-score.test.js test/difficulty-classifier.test.js test/classifier-setup.test.js test/usage-stats.test.js test/loop-guard.test.js test/moonshot-model-mapping.test.js", + "test:memory": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node --test test/memory/store.test.js test/memory/surprise.test.js test/memory/extractor.test.js test/memory/search.test.js test/memory/retriever.test.js test/memory/distiller.test.js test/memory/distiller-freeze.test.js test/memory/wiki.test.js test/memory/skills-cache.test.js test/memory/tencentdb-launcher.test.js", "test:new-features": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node --test test/passthrough-mode.test.js test/openrouter-error-resilience.test.js test/format-conversion.test.js", "test:performance": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node test/hybrid-routing-performance.test.js && DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node test/performance-tests.js", "test:benchmark": "DATABRICKS_API_KEY=test-key DATABRICKS_API_BASE=http://test.com node test/performance-benchmark.js", diff --git a/public/dashboard.html b/public/dashboard.html index 5cd6ea8..d4a1e3b 100644 --- a/public/dashboard.html +++ b/public/dashboard.html @@ -493,6 +493,35 @@

Routing Accuracy (last ${d ` : emptyState('No routing data for last 24h')} `, 'mb-6')} + + ${card(` +

Cache Economics (last ${d.window || '24h'})

+ ${d.cacheEconomics && d.cacheEconomics.decisions > 0 ? ` +
+
+

${fmt.usd(d.cacheEconomics.totalDollarsSaved)}

+

Cache $ saved by routing

+

(holds + cleared switches)

+
+
+

${fmt.num(d.cacheEconomics.holds)}

+

Switches held

+

(warm prefix protected)

+
+
+

${fmt.num(d.cacheEconomics.switches)}

+

Switches cleared

+

(break-even ≤ remaining turns)

+
+
+

${fmt.usd(d.cacheEconomics.dollarsSavedByHolds)}

+

Saved by holding pins

+

(cache re-writes avoided)

+
+
+ ` : emptyState('No gated switch decisions yet — appears once a pinned session hits a downgrade decision')} + `, 'mb-6')} + ${card(`

Provider Stats (last ${d.window || '24h'})

diff --git a/src/clients/databricks.js b/src/clients/databricks.js index e7f8293..1f5b3fc 100644 --- a/src/clients/databricks.js +++ b/src/clients/databricks.js @@ -2999,6 +2999,7 @@ async function invokeModel(body, options = {}) { candidates: routingResult.candidates ?? null, pinned: routingResult.pinned ? 1 : 0, switch_reason: routingResult.switch_reason ?? null, + cache_decision: routingResult._cacheDecision ?? null, }); // WS5.4 — feedback loop (success path). @@ -3230,6 +3231,7 @@ async function invokeModel(body, options = {}) { candidates: routingResult.candidates ?? null, pinned: routingResult.pinned ? 1 : 0, switch_reason: routingResult.switch_reason ?? null, + cache_decision: routingResult._cacheDecision ?? null, }); // WS5.4 — feedback loop (primary-failed, no fallback). Low quality @@ -3348,6 +3350,7 @@ async function invokeModel(body, options = {}) { candidates: routingResult.candidates ?? null, pinned: routingResult.pinned ? 1 : 0, switch_reason: routingResult.switch_reason ?? null, + cache_decision: routingResult._cacheDecision ?? null, }); // WS5.4 — feedback loop (fallback success). The served provider @@ -3415,6 +3418,7 @@ async function invokeModel(body, options = {}) { candidates: routingResult.candidates ?? null, pinned: routingResult.pinned ? 1 : 0, switch_reason: routingResult.switch_reason ?? null, + cache_decision: routingResult._cacheDecision ?? null, }); // WS5.4 — feedback loop (double failure). quality=0 is a hard diff --git a/src/clients/prompt-cache-injection.js b/src/clients/prompt-cache-injection.js index cb0615a..d865b66 100644 --- a/src/clients/prompt-cache-injection.js +++ b/src/clients/prompt-cache-injection.js @@ -4,9 +4,24 @@ * Injects `cache_control` breakpoints into requests for providers * that support explicit prompt caching (Anthropic, Bedrock, Vertex/Gemini). * - * Strategy: "system_and_3" — places up to 4 breakpoints: - * 1. System prompt (stable across turns — highest cache hit rate) - * 2-4. Last 3 non-system messages (rolling window) + * Strategy: "stable_hierarchy" — up to 4 breakpoints ordered by stability + * (Phase 5, cache-aware routing): + * 1. Tools block — never moves (tools render before system in the + * provider's prefix, so this read point survives system edits) + * 2. System prompt — never moves + * 3. Frozen history boundary — advances only every K user turns + * (K = config.memory.distillation.refreshEveryTurns, default 5; shared + * with the distiller's freeze window). Deterministic from the message + * list, so consecutive requests inside a bucket mark the same bytes. + * 4. Rolling marker on the newest message — pays the 1.25x write on the + * per-turn delta once so the next turn reads it at 0.1x. Kept + * deliberately: dropping it would re-pay full input price on + * everything after the boundary every turn until the next refresh. + * + * The previous "system_and_3" strategy rolled breakpoints 2-4 across the + * last three messages; markers moved every turn, and history-rewriting + * layers (distiller) invalidated the prefix wholesale. Stability of the + * marked bytes is what compounds hits. * * Providers with automatic caching (OpenAI, DeepSeek) need no injection. * @@ -17,12 +32,64 @@ const logger = require('../logger'); const CACHE_MARKER = { type: 'ephemeral' }; const MAX_BREAKPOINTS = 4; +const DEFAULT_BOUNDARY_EVERY_TURNS = 5; + +function _boundaryEveryTurns() { + try { + const config = require('../config'); + const k = config.memory?.distillation?.refreshEveryTurns; + return Number.isFinite(k) && k > 0 ? k : DEFAULT_BOUNDARY_EVERY_TURNS; + } catch { + return DEFAULT_BOUNDARY_EVERY_TURNS; + } +} + +/** Mark the last content block of a message; converts string content. */ +function _markMessage(msg) { + if (!msg) return false; + if (typeof msg.content === 'string') { + msg.content = [{ + type: 'text', + text: msg.content, + cache_control: CACHE_MARKER, + }]; + return true; + } + if (Array.isArray(msg.content) && msg.content.length > 0) { + const lastBlock = msg.content[msg.content.length - 1]; + if (lastBlock && typeof lastBlock === 'object' && !lastBlock.cache_control) { + lastBlock.cache_control = CACHE_MARKER; + return true; + } + } + return false; +} + +/** + * Index of the frozen-boundary message: the bucket-th user-role message, + * where bucket = floor(userTurns / K) * K. Deterministic in the message + * list, so every request inside a K-turn bucket marks the same message — + * the marked prefix bytes stay identical until the bucket advances. + * + * @returns {number} message index, or -1 when the conversation is too + * young (bucket < K) or the boundary can't be placed. + */ +function _frozenBoundaryIndex(messages, everyTurns) { + if (!Array.isArray(messages) || messages.length === 0) return -1; + const userIdx = []; + for (let i = 0; i < messages.length; i++) { + if (messages[i]?.role === 'user') userIdx.push(i); + } + const bucket = Math.floor(userIdx.length / everyTurns) * everyTurns; + if (bucket < everyTurns) return -1; + return userIdx[bucket - 1]; +} /** * Inject cache_control breakpoints into an Anthropic-format request body. * Mutates the body in-place for zero-copy performance. * - * @param {Object} body - Request body with system and messages + * @param {Object} body - Request body with system, tools, and messages * @returns {number} Number of breakpoints injected */ function injectAnthropicCacheBreakpoints(body) { @@ -30,7 +97,16 @@ function injectAnthropicCacheBreakpoints(body) { let injected = 0; - // Breakpoint 1: System prompt + // Breakpoint 1: tools block — most stable prefix region. + if (Array.isArray(body.tools) && body.tools.length > 0) { + const lastTool = body.tools[body.tools.length - 1]; + if (lastTool && typeof lastTool === 'object' && !lastTool.cache_control) { + lastTool.cache_control = CACHE_MARKER; + injected++; + } + } + + // Breakpoint 2: system prompt. if (body.system) { if (typeof body.system === 'string') { // Convert string system to array format for cache_control support @@ -50,32 +126,19 @@ function injectAnthropicCacheBreakpoints(body) { } } - // Breakpoints 2-4: Last 3 non-system messages if (Array.isArray(body.messages) && body.messages.length > 0) { - const remaining = MAX_BREAKPOINTS - injected; - const messagesToMark = Math.min(remaining, 3, body.messages.length); - - for (let i = 0; i < messagesToMark; i++) { - const msgIdx = body.messages.length - 1 - i; - const msg = body.messages[msgIdx]; - if (!msg) continue; - - if (typeof msg.content === 'string') { - // Convert string content to array for cache_control - msg.content = [{ - type: 'text', - text: msg.content, - cache_control: CACHE_MARKER, - }]; - injected++; - } else if (Array.isArray(msg.content) && msg.content.length > 0) { - // Mark the last content block in this message - const lastBlock = msg.content[msg.content.length - 1]; - if (lastBlock && typeof lastBlock === 'object' && !lastBlock.cache_control) { - lastBlock.cache_control = CACHE_MARKER; - injected++; - } - } + const lastIdx = body.messages.length - 1; + + // Breakpoint 3: frozen history boundary (advances every K user turns). + const boundaryIdx = _frozenBoundaryIndex(body.messages, _boundaryEveryTurns()); + if (boundaryIdx >= 0 && boundaryIdx < lastIdx && injected < MAX_BREAKPOINTS) { + if (_markMessage(body.messages[boundaryIdx])) injected++; + } + + // Breakpoint 4: rolling marker on the newest message — caches this + // turn's delta so the next turn reads it instead of re-paying input. + if (injected < MAX_BREAKPOINTS) { + if (_markMessage(body.messages[lastIdx])) injected++; } } diff --git a/src/config/index.js b/src/config/index.js index a30fcdf..586a417 100644 --- a/src/config/index.js +++ b/src/config/index.js @@ -809,6 +809,33 @@ var config = { extraction: { enabled: memoryExtractionEnabled, }, + // TencentDB-Agent-Memory-inspired token optimization (L0-L3 pipeline). + // Fixed defaults by design — tune here, not via env. + distillation: { + enabled: true, + turnThreshold: 10, // distill once conversation reaches 10 user turns + keepRecentTurns: 3, // last 3 user turns stay verbatim + // Size-based rescue trigger: fires regardless of turn count once the + // history alone hits this many estimated tokens, so small-context + // local models (4k-8k) don't overflow before the turn threshold. + tokenThreshold: 3000, + // Cache-aware routing (Phase 5): once emitted, the distilled block is + // frozen and only re-distilled every this-many user turns, so the + // block stays byte-identical between refreshes and compounds provider + // prompt-cache hits instead of rewriting history on every request. + // The stable-breakpoint hierarchy in prompt-cache-injection.js shares + // this K for its frozen history boundary. + refreshEveryTurns: 5, + }, + skills: { + enabled: true, + minSavings: 0.6, // compression outcome must save 60%+ to be cached + }, + wiki: { + enabled: true, + minTokens: 500, // register blocks larger than ~500 tokens + similarityThreshold: 0.85, + }, decay: { enabled: memoryDecayEnabled, halfLifeDays: Number.isNaN(memoryDecayHalfLifeDays) ? 30 : memoryDecayHalfLifeDays, @@ -888,6 +915,41 @@ var config = { provider: headroomProvider, logLevel: headroomLogLevel, }, + // TencentDB-Agent-Memory sidecar (memory-core + memory-hub, Docker Hub + // `agentmemory/*` images). Opt-in like Headroom; when enabled, `lynkr start` + // launches both containers. Their internal LLM calls default to routing + // back through Lynkr itself, so no extra API key is needed. + tencentdbMemory: { + enabled: process.env.TENCENTDB_MEMORY_ENABLED === "true", + docker: { + enabled: process.env.TENCENTDB_MEMORY_DOCKER_ENABLED !== "false", // default true when sidecar enabled + network: "tdai-memory-stack", + core: { + image: process.env.TENCENTDB_MEMORY_CORE_IMAGE ?? "agentmemory/memory-core:latest", + containerName: "tdai-memory-core", + port: Number.parseInt(process.env.TENCENTDB_MEMORY_CORE_PORT ?? "8420", 10), + volume: "tdai-memory-core-data", + }, + hub: { + image: process.env.TENCENTDB_MEMORY_HUB_IMAGE ?? "agentmemory/memory-hub:latest", + containerName: "tdai-memory-hub", + panelPort: Number.parseInt(process.env.TENCENTDB_MEMORY_PANEL_PORT ?? "8125", 10), + knowledgePort: Number.parseInt(process.env.TENCENTDB_MEMORY_KNOWLEDGE_PORT ?? "8424", 10), + volume: "tdai-panel-data", + }, + }, + // LLM the memory services use for extraction/summarization/wiki ingest. + // Defaults route through Lynkr's own OpenAI-compatible endpoint (tier + // routing decides the actual model), so local Ollama setups run free. + llm: { + baseUrl: process.env.TENCENTDB_MEMORY_LLM_BASE_URL + ?? `http://host.docker.internal:${Number.isNaN(port) ? 8080 : port}/v1`, + apiKey: process.env.TENCENTDB_MEMORY_LLM_API_KEY ?? "lynkr-local", + model: process.env.TENCENTDB_MEMORY_LLM_MODEL ?? "auto", + protocol: process.env.TENCENTDB_MEMORY_LLM_PROTOCOL ?? "openai", + }, + promptMode: process.env.TENCENTDB_MEMORY_PROMPT_MODE ?? "code", // code | chat + }, security: { // Content filtering contentFilterEnabled: process.env.SECURITY_CONTENT_FILTER_ENABLED !== "false", // default true @@ -949,6 +1011,21 @@ var config = { // If all exit 0, short-circuit the request with zero LLM cost. preflightEnabled: process.env.LYNKR_PREFLIGHT_ENABLED === 'true', preflightTimeoutMs: Number(process.env.LYNKR_PREFLIGHT_TIMEOUT_MS) || 120000, + // Cache-aware routing (Phases 1-3). Fixed defaults by design — tune + // here, not via env (same policy as memory.distillation). + cacheAware: { + enabled: true, + // Median remaining turns assumed when telemetry is too sparse to + // estimate — conservative so a warm pin isn't dropped for a switch + // that only pays off over a long horizon that may not happen. + defaultRemainingTurns: 10, + // Per-turn size assumptions when the payload gives no better signal. + newTokensPerTurn: 2000, + outputTokensPerTurn: 800, + // Local models: switching costs prefill latency, not dollars. Hold + // the pin while the warm prefix exceeds this many tokens. + localMaxSwitchPrefixTokens: 16000, + }, }, // Model Tier Configuration (REQUIRED) @@ -973,6 +1050,8 @@ var config = { command: process.env.CODE_GRAPH_COMMAND || 'graphify', workspace: process.env.CODE_GRAPH_WORKSPACE || process.cwd(), timeout: parseInt(process.env.CODE_GRAPH_TIMEOUT, 10) || 10000, + maxDepth: 2, // symbol dependency depth for graph queries + maxFiles: 5, // max files returned as relevant context }, // Large payload optimization (skip cloning media blocks that get discarded) diff --git a/src/context/compression.js b/src/context/compression.js index b53605d..73aa9aa 100644 --- a/src/context/compression.js +++ b/src/context/compression.js @@ -13,6 +13,13 @@ const logger = require('../logger'); const config = require('../config'); const distill = require('./distill'); +// Lazy-loaded so requiring this module doesn't pull in the SQLite store +let skillsCache = null; +function getSkillsCache() { + if (!skillsCache) skillsCache = require('../memory/skills-cache'); + return skillsCache; +} + /** * Compress conversation history to fit within token budget * @@ -277,14 +284,26 @@ function compressToolResultBlock(block, options = {}) { tool_use_id: block.tool_use_id, }; - // Compress content using Distill when content is large enough to benefit + // Compress content using Distill when content is large enough to benefit. + // The skills cache remembers how each output shape compressed before: + // known-futile shapes skip dedup work, proven-compressible shapes get a + // tighter length budget. + const compressWithSkills = (text, previousResult) => { + const skills = getSkillsCache(); + const hints = skills.getCompressionHints(text); + const result = distill.compressToolResult(text, { + previousResult, + maxLength: Math.floor(500 * hints.maxLengthFactor), + skipDedup: hints.skipDedup, + }); + const savings = text.length > 0 ? 1 - result.text.length / text.length : 0; + skills.record(hints.signature, result.method, savings); + return result.text; + }; + if (typeof block.content === 'string') { if (block.content.length > 500) { - const result = distill.compressToolResult(block.content, { - previousResult: options.previousResult, - maxLength: 500, - }); - compressed.content = result.text; + compressed.content = compressWithSkills(block.content, options.previousResult); } else { compressed.content = block.content; } @@ -292,14 +311,14 @@ function compressToolResultBlock(block, options = {}) { compressed.content = block.content.map(item => { if (typeof item === 'string') { if (item.length > 500) { - return distill.compressToolResult(item, { maxLength: 500 }).text; + return compressWithSkills(item); } return item; } else if (item.type === 'text') { if (item.text && item.text.length > 500) { return { type: 'text', - text: distill.compressToolResult(item.text, { maxLength: 500 }).text, + text: compressWithSkills(item.text), }; } return item; diff --git a/src/context/distill.js b/src/context/distill.js index 3440b5a..1c6b8d2 100644 --- a/src/context/distill.js +++ b/src/context/distill.js @@ -342,6 +342,7 @@ function deduplicateBlocks(blocks, options = {}) { * @param {Object} options * @param {string} options.previousResult - Previous tool result for delta rendering * @param {number} options.maxLength - Max output length (default 1000) + * @param {boolean} options.skipDedup - Skip section dedup (known-futile shapes) * @returns {Object} { text, method, stats } */ function compressToolResult(text, options = {}) { @@ -380,7 +381,7 @@ function compressToolResult(text, options = {}) { } // Step 3: Internal dedup — split into logical sections and dedup - const sections = result.split(/\n{2,}/); + const sections = options.skipDedup ? [] : result.split(/\n{2,}/); if (sections.length > 3) { const { compressed, stats } = deduplicateBlocks(sections); if (stats.duplicatesRemoved > 0) { diff --git a/src/dashboard/api.js b/src/dashboard/api.js index 30aae7c..45104cf 100644 --- a/src/dashboard/api.js +++ b/src/dashboard/api.js @@ -205,7 +205,11 @@ function routing(req, res) { if (s) providerStats[p] = s; } - res.json({ tierDefinitions: TIER_DEFINITIONS, accuracy, stats, providerStats, circuitBreakers: cbStates, window: win.label }); + // Cache-aware routing (Phase 6): per-decision switch/hold economics, + // aggregated into "cache dollars saved by routing". + const cacheEconomics = telemetry.getCacheEconomics({ since }); + + res.json({ tierDefinitions: TIER_DEFINITIONS, accuracy, stats, providerStats, circuitBreakers: cbStates, cacheEconomics, window: win.label }); } catch (e) { res.status(500).json({ error: 'routing_api_error', detail: e.message }); } diff --git a/src/memory/distiller.js b/src/memory/distiller.js new file mode 100644 index 0000000..ff17e74 --- /dev/null +++ b/src/memory/distiller.js @@ -0,0 +1,350 @@ +/** + * Conversation Distiller — L0-L3 Pipeline (TencentDB-Agent-Memory inspired) + * + * Long conversations resend every turn on every request. Once a + * conversation reaches the turn threshold, this module replaces the older + * turns with one compact distilled block and keeps only the most recent + * turns verbatim: + * + * L0 — raw turns (the messages themselves, dropped after distillation) + * L1 — facts/decisions extracted from the dropped turns (heuristic) + * L2 — scenario summary: what was asked, done, and decided + * L3 — persona: durable user preferences pulled from the memory store + * + * Large repeated blocks inside the dropped turns are dereferenced through + * the wiki registry before summarization. All processing is local and + * synchronous — no LLM calls. + */ + +const store = require("./store"); +const extractor = require("./extractor"); +const wiki = require("./wiki"); +const config = require("../config"); +const logger = require("../logger"); + +const MAX_SCENARIO_POINTS = 12; +const MAX_PERSONA_ITEMS = 5; +const MAX_POINT_CHARS = 140; +const DEFAULT_REFRESH_EVERY_TURNS = 5; +const MAX_FROZEN_SESSIONS = 500; + +function distillConfig() { + return config.memory?.distillation ?? {}; +} + +// --------------------------------------------------------------------------- +// Frozen distilled blocks (Phase 5, cache-aware routing). +// +// Re-distilling on every request rewrites the front of the conversation each +// turn, which invalidates the provider's prompt cache wholesale — on exactly +// the long sessions where caching matters most. Once emitted, a session's +// distilled block is frozen: the same block bytes and the same split point +// are served verbatim until K more user turns have accumulated +// (config.memory.distillation.refreshEveryTurns, default 5). Each refresh is +// then ONE deliberate, scheduled cache write instead of one per request. +// --------------------------------------------------------------------------- + +/** @type {Map} */ +const _frozen = new Map(); + +function _refreshEveryTurns() { + const k = distillConfig().refreshEveryTurns; + return Number.isFinite(k) && k > 0 ? k : DEFAULT_REFRESH_EVERY_TURNS; +} + +/** + * Cheap byte-stability fingerprint for the frozen prefix: split position + * plus samples of the first and boundary messages. Detects client-side + * history rewrites (compaction, edits) that make the frozen block stale. + */ +function _boundaryFingerprint(messages, splitIdx) { + const sample = (m) => { + if (!m) return "?"; + const c = typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? ""); + return `${m.role}:${c.slice(0, 80)}`; + }; + return `${splitIdx}|${sample(messages[0])}|${sample(messages[splitIdx - 1])}`; +} + +function _rememberFrozen(sessionId, entry) { + if (!sessionId) return; + _frozen.delete(sessionId); + _frozen.set(sessionId, entry); + if (_frozen.size > MAX_FROZEN_SESSIONS) { + const oldest = _frozen.keys().next().value; + if (oldest !== undefined) _frozen.delete(oldest); + } +} + +/** Test helper — drop all frozen blocks. */ +function _clearFrozen() { + _frozen.clear(); +} + +/** + * A "real" user turn carries user-authored text — tool_result-only + * user-role messages are plumbing, not turns. + */ +function isRealUserTurn(msg) { + if (msg?.role !== "user") return false; + if (typeof msg.content === "string") return msg.content.trim().length > 0; + if (Array.isArray(msg.content)) { + return msg.content.some(b => b?.type === "text" && b.text?.trim()); + } + return false; +} + +function realUserTurnIndices(messages) { + const indices = []; + for (let i = 0; i < messages.length; i++) { + if (isRealUserTurn(messages[i])) indices.push(i); + } + return indices; +} + +const CHARS_PER_TOKEN = 4; + +/** + * Rough token estimate for the message history (text + tool content). + */ +function estimateHistoryTokens(messages) { + let chars = 0; + for (const msg of messages) { + if (typeof msg.content === "string") { + chars += msg.content.length; + } else if (Array.isArray(msg.content)) { + for (const block of msg.content) { + if (block?.text) chars += block.text.length; + else if (typeof block?.content === "string") chars += block.content.length; + else if (Array.isArray(block?.content)) { + for (const item of block.content) { + chars += typeof item === "string" ? item.length : (item?.text?.length ?? 0); + } + } + if (block?.input) chars += JSON.stringify(block.input).length; + } + } + } + return Math.ceil(chars / CHARS_PER_TOKEN); +} + +/** + * Whether the conversation should be distilled. Two triggers: + * - turn count: conversation reached turnThreshold user turns, OR + * - size rescue: history alone exceeds tokenThreshold estimated tokens + * (protects small-context models that overflow long before the turn + * threshold — observed with 4k-context Ollama models dying at turn 8). + * Either way there must be at least one turn older than the keep window, + * or there is nothing to distill. + */ +function needsDistillation(messages) { + const cfg = distillConfig(); + if (cfg.enabled === false) return false; + if (!messages?.length) return false; + + const turns = realUserTurnIndices(messages).length; + const keepRecent = cfg.keepRecentTurns ?? 3; + if (turns <= keepRecent) return false; + + if (turns >= (cfg.turnThreshold ?? 10)) return true; + + const tokenThreshold = cfg.tokenThreshold ?? 3000; + return estimateHistoryTokens(messages) >= tokenThreshold; +} + +function extractText(msg) { + if (typeof msg?.content === "string") return msg.content; + if (Array.isArray(msg?.content)) { + return msg.content + .filter(b => b?.type === "text" && b.text) + .map(b => b.text) + .join(" "); + } + return ""; +} + +function truncate(text, max = MAX_POINT_CHARS) { + const clean = text.replace(/\s+/g, " ").trim(); + return clean.length > max ? `${clean.slice(0, max)}…` : clean; +} + +/** + * L2 — scenario summary of the dropped turns: user asks, decisions and + * facts surfaced by the assistant, and tool usage counts. + */ +function buildScenario(oldMessages) { + const asks = []; + const toolCounts = new Map(); + let assistantText = ""; + + for (const msg of oldMessages) { + if (isRealUserTurn(msg)) { + asks.push(truncate(extractText(msg), 100)); + } else if (msg.role === "assistant") { + assistantText += `${extractText(msg)}\n`; + if (Array.isArray(msg.content)) { + for (const block of msg.content) { + if (block?.type === "tool_use" && block.name) { + toolCounts.set(block.name, (toolCounts.get(block.name) ?? 0) + 1); + } + } + } + } + } + + // L1 — reuse the extraction patterns on the dropped assistant output + const decisions = extractor.extractByType(assistantText, "decision").slice(0, 4); + const facts = extractor.extractByType(assistantText, "fact").slice(0, 4); + + const parts = []; + if (asks.length) { + const shown = asks.slice(-MAX_SCENARIO_POINTS); + const omitted = asks.length - shown.length; + parts.push(`Requests${omitted > 0 ? ` (${omitted} earlier omitted)` : ""}: ${shown.join(" → ")}`); + } + if (decisions.length) parts.push(`Decisions: ${decisions.map(d => truncate(d)).join("; ")}`); + if (facts.length) parts.push(`Facts: ${facts.map(f => truncate(f)).join("; ")}`); + if (toolCounts.size) { + const tools = Array.from(toolCounts.entries()) + .sort((a, b) => b[1] - a[1]) + .slice(0, 8) + .map(([name, count]) => (count > 1 ? `${name}×${count}` : name)) + .join(", "); + parts.push(`Tools used: ${tools}`); + } + + return parts.join("\n"); +} + +/** + * L3 — persona line from durable preference memories (global + session). + */ +function buildPersona(sessionId) { + try { + const prefs = store + .getMemoriesByType("preference", 50) + .filter(m => m.sessionId === null || m.sessionId === sessionId) + .sort((a, b) => (b.importance ?? 0) - (a.importance ?? 0)) + .slice(0, MAX_PERSONA_ITEMS); + + if (!prefs.length) return ""; + return prefs.map(m => truncate(m.content, 80)).join("; "); + } catch (err) { + logger.warn({ err, sessionId }, "[distiller] Persona build failed"); + return ""; + } +} + +/** + * Distill a long conversation: dereference repeated large blocks via wiki, + * summarize everything before the last keepRecentTurns user turns into one + * block, and keep the recent turns verbatim. + * + * The split lands on a real user turn boundary, so assistant tool_use / + * user tool_result pairs are never separated. + * + * @param {Array} messages - Full conversation + * @param {Object} options + * @param {string} options.sessionId + * @returns {{messages: Array, applied: boolean, stats: Object}} + */ +function distillMessages(messages, options = {}) { + const { sessionId = null } = options; + + if (!needsDistillation(messages)) { + return { messages, applied: false, stats: {} }; + } + + const keepRecent = distillConfig().keepRecentTurns ?? 3; + const turnIndices = realUserTurnIndices(messages); + + // Phase 5 freeze: serve the session's frozen block verbatim while it is + // still fresh (fewer than K user turns since it was built) and the + // history prefix it covers is byte-stable. The block and split point are + // identical across requests, so the provider prompt cache keeps hitting. + if (sessionId) { + const cached = _frozen.get(sessionId); + if ( + cached + && cached.splitIdx < messages.length + && turnIndices.length - cached.frozenAtTurns < _refreshEveryTurns() + && _boundaryFingerprint(messages, cached.splitIdx) === cached.boundaryFp + ) { + return { + messages: [ + { role: "user", content: cached.distilledContent }, + ...messages.slice(cached.splitIdx), + ], + applied: true, + stats: { ...cached.stats, fromFrozenCache: true }, + }; + } + } + + const splitIdx = turnIndices[Math.max(0, turnIndices.length - keepRecent)]; + + if (!splitIdx) { + return { messages, applied: false, stats: {} }; + } + + const oldMessages = messages.slice(0, splitIdx); + const recentMessages = messages.slice(splitIdx); + + // Wiki pass over dropped turns: registers large blocks for cross-request + // dedup and shrinks repeats before the scenario is built + const { messages: dereferenced, stats: wikiStats } = wiki.dereferenceMessages(oldMessages); + + const scenario = buildScenario(dereferenced); + const persona = buildPersona(sessionId); + + const sections = [`[Distilled context — earlier ${turnIndices.length - keepRecent} turns compressed]`]; + if (persona) sections.push(`User profile: ${persona}`); + if (scenario) sections.push(scenario); + + const distilledBlock = { + role: "user", + content: sections.join("\n"), + }; + + const originalChars = JSON.stringify(oldMessages).length; + const distilledChars = distilledBlock.content.length; + + const stats = { + droppedMessages: oldMessages.length, + keptMessages: recentMessages.length, + originalChars, + distilledChars, + savingsPct: originalChars > 0 + ? (((originalChars - distilledChars) / originalChars) * 100).toFixed(1) + : "0.0", + wiki: wikiStats, + }; + + logger.debug({ sessionId, ...stats }, "[distiller] Conversation distilled"); + + // Freeze the block for the next K user turns (Phase 5). Keyed by the + // split fingerprint so a client-side history rewrite invalidates it. + _rememberFrozen(sessionId, { + frozenAtTurns: turnIndices.length, + splitIdx, + boundaryFp: _boundaryFingerprint(messages, splitIdx), + distilledContent: distilledBlock.content, + stats, + }); + + return { + messages: [distilledBlock, ...recentMessages], + applied: true, + stats, + }; +} + +module.exports = { + needsDistillation, + distillMessages, + buildScenario, + buildPersona, + isRealUserTurn, + estimateHistoryTokens, + _clearFrozen, +}; diff --git a/src/memory/extractor.js b/src/memory/extractor.js index fd131a7..ecf5049 100644 --- a/src/memory/extractor.js +++ b/src/memory/extractor.js @@ -358,19 +358,11 @@ function calculateInitialImportance(type, surpriseScore) { return Math.min(1.0, base + (surpriseScore * 0.3)); } -/** - * Parse entities from content - */ -function parseEntities(content) { - return extractEntities(content); -} - module.exports = { extractMemories, extractContent, extractByType, extractEntities, extractRelationships, - parseEntities, classifyCategory, }; diff --git a/src/memory/format.js b/src/memory/format.js index 00dbc81..601914b 100644 --- a/src/memory/format.js +++ b/src/memory/format.js @@ -20,11 +20,26 @@ function formatMemoriesForContext(memories, format = 'compact') { return formatCompact(memories); } +/** + * Deterministic render order (cache-aware routing, Phase 5): retrieval + * returns equal-relevance memories in unstable order, and the rendered + * bullets sit in the system prompt — the very front of the provider's + * prompt-cache prefix. A reordered bullet list is a byte change that + * silently invalidates the entire cached prefix (observed live: identical + * memory sets flipping order between consecutive requests). Sorting by + * content pins the bytes; presentation order carries no meaning here. + */ +function stableOrder(memories) { + return [...memories].sort((a, b) => + String(a.content) < String(b.content) ? -1 : String(a.content) > String(b.content) ? 1 : 0 + ); +} + /** * Compact memory format - 75% fewer tokens */ function formatCompact(memories) { - const items = memories + const items = stableOrder(memories) .map(mem => `- ${mem.content}`) .join('\n'); @@ -35,7 +50,7 @@ function formatCompact(memories) { * Verbose XML format (original) */ function formatVerbose(memories) { - const items = memories.map((mem, idx) => { + const items = stableOrder(memories).map((mem, idx) => { const age = formatAge(mem.createdAt); const type = mem.type ? `[${mem.type}] ` : ''; return `${idx + 1}. ${type}${mem.content} (${age})`; diff --git a/src/memory/index.js b/src/memory/index.js index e3ac254..307a665 100644 --- a/src/memory/index.js +++ b/src/memory/index.js @@ -15,6 +15,9 @@ const retriever = require("./retriever"); const extractor = require("./extractor"); const surprise = require("./surprise"); const tools = require("./tools"); +const distiller = require("./distiller"); +const wiki = require("./wiki"); +const skillsCache = require("./skills-cache"); module.exports = { // Store operations @@ -52,4 +55,11 @@ module.exports = { // Tools tools, MEMORY_TOOLS: tools.MEMORY_TOOLS, + + // TencentDB-inspired L0-L3 token optimization + distiller, + needsDistillation: distiller.needsDistillation, + distillMessages: distiller.distillMessages, + wiki, + skillsCache, }; diff --git a/src/memory/retriever.js b/src/memory/retriever.js index c4b494b..89a361e 100644 --- a/src/memory/retriever.js +++ b/src/memory/retriever.js @@ -3,6 +3,10 @@ const search = require("./search"); const logger = require("../logger"); const format = require("./format"); +// Infrastructure memory types (wiki registry, compression skills) are used +// by the distillation pipeline, never injected as conversational context +const INTERNAL_MEMORY_TYPES = new Set(["wiki", "skill"]); + /** * Retrieve relevant memories using multi-signal ranking * @@ -41,8 +45,9 @@ function retrieveRelevantMemories(query, options = {}) { sessionId: includeGlobal ? null : sessionId, }); - // 4. Merge and deduplicate - const candidates = mergeUnique([ftsResults, recentMemories, importantMemories]); + // 4. Merge and deduplicate, dropping infrastructure entries + const candidates = mergeUnique([ftsResults, recentMemories, importantMemories]) + .filter(m => !INTERNAL_MEMORY_TYPES.has(m.type)); // 5. Score and rank const scored = candidates.map(memory => ({ diff --git a/src/memory/skills-cache.js b/src/memory/skills-cache.js new file mode 100644 index 0000000..6def7f9 --- /dev/null +++ b/src/memory/skills-cache.js @@ -0,0 +1,180 @@ +/** + * Skills Cache — Compression Strategy Memory (TencentDB-Agent-Memory inspired) + * + * Remembers which compression method worked (and how well) for each + * structural shape of tool output. Validated skills (savings >= minSavings) + * persist in the memories table (type='skill') and let future compressions: + * - skip section-dedup for shapes where it never helps (latency win) + * - compress proven-compressible shapes more aggressively (token win) + */ + +const store = require("./store"); +const distill = require("../context/distill"); +const config = require("../config"); +const logger = require("../logger"); + +const MAX_CACHED_SKILLS = 1000; +const SIGNATURE_LINES = 30; + +// signature -> { method, avgSavings, hits, persistedId } +let skillMap = null; + +function skillsConfig() { + return config.memory?.skills ?? {}; +} + +/** + * Structural shape signature for a tool output: leading token of each of + * the first N normalized lines. Two grep outputs, two test runs, or two + * JSON blobs with the same shape produce the same signature even when the + * values differ. + */ +function shapeSignature(text) { + if (!text) return "empty"; + const lines = distill.normalizeText(text).split("\n").slice(0, SIGNATURE_LINES); + const shape = lines + .map(l => { + const trimmed = l.trim(); + if (!trimmed) return ""; + // Leading structural token: punctuation kept, words folded + const lead = trimmed.match(/^[{}[\]"'\-+*#>|]|^[A-Za-z_$]+|^\d+/); + return lead ? (/^\d+$/.test(lead[0]) ? "N" : lead[0]) : trimmed[0]; + }) + .join(","); + // djb2 hash keeps keys short + let hash = 5381; + for (let i = 0; i < shape.length; i++) { + hash = ((hash << 5) + hash + shape.charCodeAt(i)) | 0; + } + return `s${lines.length}_${(hash >>> 0).toString(36)}`; +} + +function loadCache() { + if (skillMap) return skillMap; + + skillMap = new Map(); + try { + const rows = store.getMemoriesByType("skill", MAX_CACHED_SKILLS); + for (const row of rows) { + const meta = typeof row.metadata === "string" ? JSON.parse(row.metadata) : row.metadata; + if (!meta?.signature) continue; + skillMap.set(meta.signature, { + method: meta.method, + avgSavings: meta.avgSavings ?? 0, + hits: meta.hits ?? 1, + persistedId: row.id, + }); + } + } catch (err) { + logger.warn({ err }, "[skills] Failed to load skills, starting empty"); + } + return skillMap; +} + +/** + * Look up a known compression skill for a shape signature. + * @returns {{method: string, avgSavings: number, hits: number}|null} + */ +function lookup(signature) { + if (skillsConfig().enabled === false) return null; + return loadCache().get(signature) ?? null; +} + +/** + * Record a compression outcome. Kept in memory always (so known-futile + * shapes can skip dedup work); persisted only when the running average + * savings clears minSavings — the "validated skill" bar. + * + * @param {string} signature - shapeSignature() of the original text + * @param {string} method - compression method that ran ('delta'|'distill'|'passthrough') + * @param {number} savings - fraction saved, 0..1 + */ +function record(signature, method, savings) { + if (skillsConfig().enabled === false || !signature) return; + + const cache = loadCache(); + const existing = cache.get(signature); + const entry = existing + ? { + ...existing, + method, + hits: existing.hits + 1, + avgSavings: (existing.avgSavings * existing.hits + savings) / (existing.hits + 1), + } + : { method, avgSavings: savings, hits: 1, persistedId: null }; + + cache.set(signature, entry); + if (cache.size > MAX_CACHED_SKILLS) { + cache.delete(cache.keys().next().value); + } + + const minSavings = skillsConfig().minSavings ?? 0.6; + if (entry.avgSavings < minSavings) return; + + // Persist validated skill off the hot path + setImmediate(() => { + try { + const metadata = { + signature, + method: entry.method, + avgSavings: entry.avgSavings, + hits: entry.hits, + }; + if (entry.persistedId) { + const current = store.getMemory(entry.persistedId); + if (current) { + store.updateMemory(entry.persistedId, { metadata }); + return; + } + } + const memory = store.createMemory({ + sessionId: null, + content: `Compression skill: ${entry.method} saves ${(entry.avgSavings * 100).toFixed(0)}% on shape ${signature}`, + type: "skill", + category: "optimization", + importance: 0.3, + metadata, + }); + entry.persistedId = memory.id; + } catch (err) { + logger.warn({ err, signature }, "[skills] Failed to persist skill"); + } + }); +} + +/** + * Compression hints for a given text, derived from validated skills. + * + * @param {string} text - Tool result text about to be compressed + * @returns {{signature: string, skipDedup: boolean, maxLengthFactor: number}} + */ +function getCompressionHints(text) { + const signature = shapeSignature(text); + const skill = lookup(signature); + const minSavings = skillsConfig().minSavings ?? 0.6; + + if (!skill || skill.hits < 2) { + return { signature, skipDedup: false, maxLengthFactor: 1 }; + } + + return { + signature, + // Dedup/delta repeatedly achieved almost nothing — don't burn CPU on it + skipDedup: skill.method === "passthrough" || skill.avgSavings < 0.05, + // Proven highly-compressible shape — compress harder + maxLengthFactor: skill.avgSavings >= minSavings ? 0.8 : 1, + }; +} + +/** Reset the in-memory cache (test support). */ +function resetCache() { + skillMap = null; +} + +module.exports = { + shapeSignature, + lookup, + record, + getCompressionHints, + resetCache, +}; diff --git a/src/memory/tencentdb-launcher.js b/src/memory/tencentdb-launcher.js new file mode 100644 index 0000000..16bf436 --- /dev/null +++ b/src/memory/tencentdb-launcher.js @@ -0,0 +1,459 @@ +/** + * TencentDB-Agent-Memory Sidecar Launcher + * + * Launches the TencentDB-Agent-Memory stack (memory-core gateway + + * memory-hub panel/knowledge services) via dockerode when + * TENCENTDB_MEMORY_ENABLED=true — the same in-process pattern as the + * Headroom sidecar launcher, extended to a two-container stack. + * + * Mirrors the project's own deploy/global-images/start-*.sh scripts: + * - shared docker network `tdai-memory-stack` with service aliases + * - memory-core config generated locally and mounted read-only + * - first-boot admin user init, key persisted to data/tencentdb-memory/ + * + * The memory services need an LLM for extraction/summarization; by default + * that points back at Lynkr's own OpenAI-compatible endpoint, so the whole + * stack runs on whatever providers Lynkr already has. + * + * Containers use restart=unless-stopped and are intentionally NOT stopped + * when Lynkr exits — a team memory hub outlives any one proxy process. + * Remove with: docker rm -f tdai-memory-core tdai-memory-hub + */ + +let Docker; +try { + Docker = require("dockerode"); +} catch { + Docker = null; +} +const path = require("path"); +const fs = require("fs"); +const crypto = require("crypto"); +const logger = require("../logger"); +const config = require("../config"); + +const docker = Docker ? new Docker() : null; + +const DATA_DIR = path.join(process.cwd(), "data", "tencentdb-memory"); +const ADMIN_KEY_FILE = path.join(DATA_DIR, ".admin-key"); +const CORE_CONFIG_FILE = path.join(DATA_DIR, "tdai-gateway.yaml"); + +let isStarting = false; + +function sidecarConfig() { + return config.tencentdbMemory ?? {}; +} + +/** + * Gateway config for memory-core, matching the template the project's + * start-memory-core.sh generates. Mounted read-only into the container. + */ +function buildCoreConfigYaml() { + const { llm, promptMode } = sidecarConfig(); + return `# Generated by Lynkr's TencentDB memory launcher — overwritten on every start. +deployMode: standalone +stateBackend: local + +server: + port: 8420 + host: 0.0.0.0 + +data: + baseDir: /data/tdai-memory + +llm: + baseUrl: "${llm.baseUrl}" + apiKey: "${llm.apiKey}" + model: "${llm.model}" + maxTokens: 32000 + timeoutMs: 300000 + +memory: + promptMode: ${promptMode} + capture: { enabled: true } + extraction: + enabled: true + enableDedup: true + maxMemoriesPerSession: 20 + persona: + triggerEveryN: 50 + maxScenes: 15 + pipeline: + everyNConversations: 5 + enableWarmup: true + l1IdleTimeoutSeconds: 600 + l2DelayAfterL1Seconds: 90 + l2MinIntervalSeconds: 900 + l2MaxIntervalSeconds: 3600 + recall: + enabled: true + maxResults: 5 + scoreThreshold: 0.3 + strategy: hybrid + timeoutMs: 5000 + storeBackend: sqlite + embedding: + provider: none + +skill: + enabled: true + routing: + mode: bm25 + searchTopK: 20 + extraction: + enabled: true + maxIterations: 16 + queue: + backend: local + keyPrefix: tdai + resultTtlSeconds: 86400 + lockTtlMs: 600000 + maxRetries: 2 + retryBackoffsMs: [5000, 15000] + resources: + maxResourceSizeBytes: 5000000 +`; +} + +/** + * sk-mem-<32 alphanumeric chars> — the user-key format memory-core expects. + */ +function generateUserKey() { + let raw = ""; + while (raw.length < 32) { + raw += crypto.randomBytes(48).toString("base64").replace(/[^A-Za-z0-9]/g, ""); + } + return `sk-mem-${raw.slice(0, 32)}`; +} + +function maskKey(key) { + if (!key || key.length < 16) return "****"; + return `${key.slice(0, 11)}****${key.slice(-4)}`; +} + +async function ensureNetwork(name) { + try { + await docker.getNetwork(name).inspect(); + } catch (err) { + if (err.statusCode !== 404) throw err; + logger.info({ network: name }, "[tdai] Creating docker network"); + await docker.createNetwork({ Name: name, CheckDuplicate: true }); + } +} + +async function getExistingContainer(containerName) { + const containers = await docker.listContainers({ + all: true, + filters: { name: [containerName] }, + }); + const match = containers.find( + (c) => c.Names.includes(`/${containerName}`) || c.Names.includes(containerName) + ); + return match ? docker.getContainer(match.Id) : null; +} + +async function imageExists(imageName) { + try { + await docker.getImage(imageName).inspect(); + return true; + } catch (err) { + if (err.statusCode === 404) return false; + throw err; + } +} + +function pullImage(imageName) { + logger.info({ image: imageName }, "[tdai] Pulling image (first start may take a few minutes)"); + return new Promise((resolve, reject) => { + docker.pull(imageName, (err, stream) => { + if (err) return reject(err); + docker.modem.followProgress(stream, (progressErr, output) => { + if (progressErr) return reject(progressErr); + logger.info({ image: imageName }, "[tdai] Image pull complete"); + resolve(output); + }); + }); + }); +} + +/** + * Poll an HTTP endpoint until it responds (any status < 500). + */ +async function waitForHttp(url, { maxRetries = 90, intervalMs = 1000, label } = {}) { + for (let i = 0; i < maxRetries; i++) { + try { + const response = await fetch(url, { signal: AbortSignal.timeout(2000) }); + if (response.status < 500) return true; + } catch { + // not up yet + } + if (i % 15 === 14) { + logger.debug({ url, attempt: i + 1, maxRetries }, `[tdai] Waiting for ${label ?? url}`); + } + await new Promise((resolve) => setTimeout(resolve, intervalMs)); + } + throw new Error(`${label ?? url} did not become healthy after ${maxRetries}s`); +} + +/** + * Create+start one container, or start it if it exists but is stopped. + * Returns the action taken. + */ +async function ensureContainer(spec) { + const existing = await getExistingContainer(spec.name); + + if (existing) { + const info = await existing.inspect(); + if (info.State.Running) return "existing_running"; + logger.info({ container: spec.name }, "[tdai] Starting existing container"); + await existing.start(); + return "started_existing"; + } + + if (!(await imageExists(spec.image))) { + await pullImage(spec.image); + } + + logger.info({ container: spec.name, image: spec.image }, "[tdai] Creating container"); + const container = await docker.createContainer({ + Image: spec.image, + name: spec.name, + Env: spec.env, + ExposedPorts: Object.fromEntries( + Object.values(spec.ports).map((p) => [`${p.container}/tcp`, {}]) + ), + HostConfig: { + PortBindings: Object.fromEntries( + Object.values(spec.ports).map((p) => [ + `${p.container}/tcp`, + [{ HostPort: String(p.host) }], + ]) + ), + Binds: spec.binds, + NetworkMode: spec.network, + ExtraHosts: ["host.docker.internal:host-gateway"], + RestartPolicy: { Name: "unless-stopped" }, + }, + NetworkingConfig: { + EndpointsConfig: { + [spec.network]: { Aliases: [spec.alias] }, + }, + }, + }); + await container.start(); + return "created_new"; +} + +/** + * First-boot admin bootstrap, mirroring start-memory-core.sh: + * generate an sk-mem- key (or reuse the persisted one), init-admin with it, + * verify, and persist to .admin-key with tight permissions. + */ +async function ensureAdminUser(gatewayUrl) { + let adminKey; + if (fs.existsSync(ADMIN_KEY_FILE)) { + adminKey = fs.readFileSync(ADMIN_KEY_FILE, "utf8").trim(); + } else { + adminKey = generateUserKey(); + } + + const response = await fetch(`${gatewayUrl}/v3/internal/meta/user/init-admin`, { + method: "POST", + headers: { + "Content-Type": "application/json", + "x-tdai-service-id": "default", + }, + body: JSON.stringify({ username: "admin", user_key: adminKey }), + signal: AbortSignal.timeout(10000), + }); + + if (response.status === 200) { + fs.writeFileSync(ADMIN_KEY_FILE, adminKey, { mode: 0o600 }); + logger.info({ key: maskKey(adminKey), keyFile: ADMIN_KEY_FILE }, "[tdai] Admin user created"); + } else if (response.status === 409) { + if (!fs.existsSync(ADMIN_KEY_FILE)) { + logger.warn( + "[tdai] Admin user already exists but .admin-key is missing — the key cannot be recovered. " + + "Reset with: docker rm -f tdai-memory-core && docker volume rm tdai-memory-core-data" + ); + return null; + } + } else { + logger.warn({ status: response.status }, "[tdai] init-admin returned unexpected status"); + return null; + } + + // Verify the key actually authenticates + try { + const verify = await fetch(`${gatewayUrl}/v3/meta/auth/verify`, { + method: "POST", + headers: { + "Content-Type": "application/json", + "x-tdai-service-id": "default", + }, + body: JSON.stringify({ user_key: adminKey }), + signal: AbortSignal.timeout(5000), + }); + if (verify.status !== 200) { + logger.warn({ status: verify.status }, "[tdai] Admin key failed verification — volume and .admin-key may be out of sync"); + return null; + } + } catch (err) { + logger.debug({ err: err.message }, "[tdai] Admin key verification skipped"); + } + + return adminKey; +} + +/** + * Ensure the full memory stack (core + hub) is running. + * Idempotent: containers started by the project's own scripts are detected + * by name and reused. + */ +async function ensureRunning() { + const cfg = sidecarConfig(); + + if (!cfg.enabled) return { started: false, reason: "disabled" }; + if (!cfg.docker?.enabled) return { started: false, reason: "docker_disabled" }; + if (!docker) return { started: false, reason: "dockerode_unavailable" }; + if (isStarting) return { started: false, reason: "already_starting" }; + + isStarting = true; + try { + const { network, core, hub } = cfg.docker; + const gatewayUrl = `http://localhost:${core.port}`; + + await ensureNetwork(network); + + // ── memory-core ── + fs.mkdirSync(DATA_DIR, { recursive: true }); + fs.writeFileSync(CORE_CONFIG_FILE, buildCoreConfigYaml()); + + const coreAction = await ensureContainer({ + name: core.containerName, + image: core.image, + alias: "memory-core", + network, + ports: { gateway: { host: core.port, container: 8420 } }, + binds: [ + `${core.volume}:/data/tdai-memory`, + `${CORE_CONFIG_FILE}:/data/config/tdai-gateway.yaml:ro`, + ], + env: [ + "TDAI_GATEWAY_PORT=8420", + "TDAI_GATEWAY_HOST=0.0.0.0", + "TDAI_GATEWAY_API_KEY=", // empty = no bearer gate (local deployment) + "TDAI_DATA_DIR=/data/tdai-memory", + ], + }); + await waitForHttp(`${gatewayUrl}/`, { label: "memory-core" }); + logger.info({ action: coreAction, url: gatewayUrl }, "[tdai] memory-core ready"); + + const adminKey = await ensureAdminUser(gatewayUrl); + + // ── memory-hub (panel + knowledge) ── + const { llm } = cfg; + const hubAction = await ensureContainer({ + name: hub.containerName, + image: hub.image, + alias: "memory-hub", + network, + ports: { + panel: { host: hub.panelPort, container: 8125 }, + knowledge: { host: hub.knowledgePort, container: 8424 }, + }, + binds: [`${hub.volume}:/data/knowledge`], + env: [ + "PANEL_PORT=8125", + "KNOWLEDGE_PORT=8424", + `KNOWLEDGE_PUBLIC_BASE_URL=http://host.docker.internal:${hub.knowledgePort}/v3`, + "REMOTE_INSTANCE_ID=default", + "REMOTE_INSTANCE_NAME=default", + "REMOTE_INSTANCE_URL=http://memory-core:8420", + "REMOTE_INSTANCE_KEY=local", + // Panel's "client base URL" card points at Lynkr — Lynkr is the proxy here + `REMOTE_INSTANCE_PROXY_URL=http://localhost:${config.port}`, + "LLM_MODE=custom", + `LLM_PROTOCOL=${llm.protocol}`, + `LLM_API_KEY=${llm.apiKey}`, + `LLM_BASE_URL=${llm.baseUrl}`, + `LLM_MODEL=${llm.model}`, + "KNOWLEDGE_LLM_BINDING_SYNC=0", + ], + }); + await waitForHttp(`http://localhost:${hub.knowledgePort}/health`, { label: "memory-hub", maxRetries: 120 }); + logger.info({ action: hubAction }, "[tdai] memory-hub ready"); + + return { + started: true, + actions: { core: coreAction, hub: hubAction }, + adminKey: adminKey ? maskKey(adminKey) : null, + endpoints: { + gateway: gatewayUrl, + panel: `http://localhost:${hub.panelPort}/`, + knowledge: `http://localhost:${hub.knowledgePort}/v3`, + }, + }; + } finally { + isStarting = false; + } +} + +/** + * Stop both containers (kept on disk; volumes untouched). + * Not called automatically on Lynkr shutdown — see module header. + */ +async function stop() { + if (!docker) return; + const { core, hub } = sidecarConfig().docker ?? {}; + for (const name of [hub?.containerName, core?.containerName].filter(Boolean)) { + try { + const container = await getExistingContainer(name); + if (container) { + const info = await container.inspect(); + if (info.State.Running) { + logger.info({ container: name }, "[tdai] Stopping container"); + await container.stop({ t: 10 }); + } + } + } catch (err) { + logger.warn({ err, container: name }, "[tdai] Failed to stop container"); + } + } +} + +async function getStatus() { + if (!docker) return { available: false }; + const { core, hub } = sidecarConfig().docker ?? {}; + const status = { available: true, containers: {} }; + for (const [key, name] of [["core", core?.containerName], ["hub", hub?.containerName]]) { + if (!name) continue; + try { + const container = await getExistingContainer(name); + if (!container) { + status.containers[key] = { exists: false, running: false }; + continue; + } + const info = await container.inspect(); + status.containers[key] = { + exists: true, + running: info.State.Running, + status: info.State.Status, + image: info.Config.Image, + }; + } catch (err) { + status.containers[key] = { exists: false, running: false, error: err.message }; + } + } + return status; +} + +module.exports = { + ensureRunning, + stop, + getStatus, + // exported for tests + buildCoreConfigYaml, + generateUserKey, + maskKey, +}; diff --git a/src/memory/wiki.js b/src/memory/wiki.js new file mode 100644 index 0000000..1ac5a89 --- /dev/null +++ b/src/memory/wiki.js @@ -0,0 +1,171 @@ +/** + * Wiki — Large-Block Content Registry (TencentDB-Agent-Memory inspired) + * + * Registers large content blocks (docs, configs, big tool outputs) the + * first time they appear in old conversation history, then replaces later + * near-duplicates with a short reference + summary instead of resending + * the full block. Entries persist in the memories table (type='wiki') so + * dedup works across sessions. + */ + +const store = require("./store"); +const distill = require("../context/distill"); +const config = require("../config"); +const logger = require("../logger"); + +const CHARS_PER_TOKEN = 4; +const MAX_CACHED_ENTRIES = 500; +const MAX_SIGNATURE_LINES = 200; + +// In-memory entry cache: [{ id, summary, signature: Set }] +let entryCache = null; + +function wikiConfig() { + return config.memory?.wiki ?? {}; +} + +function minChars() { + return (wikiConfig().minTokens ?? 500) * CHARS_PER_TOKEN; +} + +/** + * Load persisted wiki entries into the in-memory cache (lazy, once). + */ +function loadCache() { + if (entryCache) return entryCache; + + entryCache = []; + try { + const rows = store.getMemoriesByType("wiki", MAX_CACHED_ENTRIES); + for (const row of rows) { + const meta = typeof row.metadata === "string" ? JSON.parse(row.metadata) : row.metadata; + if (!meta?.signatureLines?.length) continue; + entryCache.push({ + id: row.id, + summary: row.content, + signature: new Set(meta.signatureLines), + }); + } + } catch (err) { + logger.warn({ err }, "[wiki] Failed to load wiki entries, starting empty"); + } + return entryCache; +} + +/** + * Build a one-line summary for a content block: first heading or first + * meaningful line, plus a size note. + */ +function makeSummary(text) { + const lines = distill.normalizeText(text).split("\n").filter(Boolean); + const heading = lines.find(l => /^#{1,4}\s|^[A-Z][^a-z]*$/.test(l.trim())); + const first = (heading || lines[0] || "").trim().slice(0, 120); + return `${first} (${lines.length} lines, ~${Math.ceil(text.length / CHARS_PER_TOKEN)} tokens)`; +} + +/** + * Register a large block, or return a compact reference if a similar + * block is already known. + * + * @param {string} text - Content block from old history + * @returns {{ref: string, id: number, saved: number}|null} Reference when a + * similar entry exists; null when the block was registered or is too small + */ +function registerOrDereference(text) { + if (wikiConfig().enabled === false) return null; + if (!text || text.length < minChars()) return null; + + const threshold = wikiConfig().similarityThreshold ?? 0.85; + const signature = distill.extractSignature(text); + const cache = loadCache(); + + for (const entry of cache) { + const sim = distill.jaccardSimilarity(signature, entry.signature); + if (sim >= threshold) { + const ref = `[wiki:${entry.id} — ${(sim * 100).toFixed(0)}% match] ${entry.summary}`; + return { ref, id: entry.id, saved: text.length - ref.length }; + } + } + + // No match — register for future dedup + try { + const summary = makeSummary(text); + const memory = store.createMemory({ + sessionId: null, // wiki entries are global + content: summary, + type: "wiki", + category: "reference", + importance: 0.3, + metadata: { + signatureLines: Array.from(signature).slice(0, MAX_SIGNATURE_LINES), + originalChars: text.length, + }, + }); + cache.push({ id: memory.id, summary, signature }); + if (cache.length > MAX_CACHED_ENTRIES) cache.shift(); + } catch (err) { + logger.warn({ err }, "[wiki] Failed to register wiki entry"); + } + + return null; +} + +/** + * Replace large repeated blocks inside old-history messages with wiki + * references. Only text and tool_result content is touched. + * + * @param {Array} messages - Old (about-to-be-summarized) messages + * @returns {{messages: Array, stats: {registered: number, dereferenced: number, charsSaved: number}}} + */ +function dereferenceMessages(messages) { + const stats = { registered: 0, dereferenced: 0, charsSaved: 0 }; + if (wikiConfig().enabled === false || !messages?.length) { + return { messages: messages || [], stats }; + } + + const processText = (text) => { + const before = loadCache().length; + const result = registerOrDereference(text); + if (result) { + stats.dereferenced++; + stats.charsSaved += result.saved; + return result.ref; + } + if (loadCache().length > before) stats.registered++; + return text; + }; + + const processed = messages.map(msg => { + if (typeof msg.content === "string") { + if (msg.content.length < minChars()) return msg; + return { ...msg, content: processText(msg.content) }; + } + if (!Array.isArray(msg.content)) return msg; + + const newContent = msg.content.map(block => { + if (block.type === "text" && block.text?.length >= minChars()) { + return { ...block, text: processText(block.text) }; + } + if (block.type === "tool_result" && typeof block.content === "string" && + block.content.length >= minChars()) { + return { ...block, content: processText(block.content) }; + } + return block; + }); + return { ...msg, content: newContent }; + }); + + return { messages: processed, stats }; +} + +/** Reset the in-memory cache (test support). */ +function resetCache() { + entryCache = null; +} + +module.exports = { + registerOrDereference, + dereferenceMessages, + makeSummary, + resetCache, +}; diff --git a/src/orchestrator/index.js b/src/orchestrator/index.js index 465f98b..58bd77c 100644 --- a/src/orchestrator/index.js +++ b/src/orchestrator/index.js @@ -21,6 +21,7 @@ const crypto = require("crypto"); const { getSemanticCache, isSemanticCacheEnabled } = require("../cache/semantic"); const { areSimilarToolCalls } = require("../clients/gpt-utils"); const { getModelRegistrySync } = require("../routing/model-registry"); +const sessionAffinity = require("../routing/session-affinity"); /** * Get destination URL for audit logging based on provider type @@ -1457,6 +1458,30 @@ async function runAgentLoop({ if (steps === 1 && agentTimer) agentTimer.mark("preCompression"); + + // === CONVERSATION DISTILLATION (TencentDB-inspired L0-L3 pipeline) === + // Long conversations collapse older turns into one distilled block + // (persona + scenario summary) before history compression runs. + if (steps === 1 && config.memory?.enabled !== false && config.memory?.distillation?.enabled !== false) { + try { + const distiller = require('../memory/distiller'); + if (distiller.needsDistillation(cleanPayload.messages)) { + const result = distiller.distillMessages(cleanPayload.messages, { + sessionId: session?.id, + }); + if (result.applied) { + cleanPayload.messages = result.messages; + logger.debug({ + sessionId: session?.id ?? null, + ...result.stats, + }, '[distiller] Conversation distillation applied'); + } + } + } catch (err) { + logger.warn({ err, sessionId: session?.id }, 'Distillation failed, continuing with full history'); + } + } + if (steps === 1 && config.historyCompression?.enabled !== false) { try { if (historyCompression.needsCompression(cleanPayload.messages)) { @@ -1919,6 +1944,28 @@ IMPORTANT TOOL USAGE RULES: } } + // Cache-aware routing (Phase 1): persist the session's warm-prefix state + // from the response's cache counters so the router can price a mid-session + // model switch against the live cache clock. Best-effort — never blocks + // the response path. + if (session?.id && actualUsage) { + try { + // Prefer the ROUTED provider/model over the request-level default: + // the tier router inside invokeModel may have diverged from + // providerType, and the cache lives with whoever actually served + // (verified live: databricks default label on ollama-served turns). + const served = databricksResponse.routingDecision || {}; + sessionAffinity.recordCacheUsage(session.id, { + provider: served.provider || providerType, + model: served.model || cleanPayload.model, + cacheReadTokens: actualUsage.cacheReadTokens, + cacheCreationTokens: actualUsage.cacheCreationTokens, + }); + } catch (err) { + logger.debug({ err: err.message }, "[Orchestrator] cache-state update failed"); + } + } + if (auditLogger.enabled) { const latencyMs = Date.now() - start; diff --git a/src/routing/affinity-store.js b/src/routing/affinity-store.js index bde7ac8..349d0e5 100644 --- a/src/routing/affinity-store.js +++ b/src/routing/affinity-store.js @@ -56,6 +56,12 @@ function _db() { if (!cols.has("has_tool_history")) { db.exec("ALTER TABLE session_pins ADD COLUMN has_tool_history INTEGER DEFAULT 0"); } + // Additive migration for cache-aware routing (Phase 1): JSON blob + // holding {warmPrefixTokens, provider, model, lastRequestAt, ttlMs}, + // updated after every upstream response that reports cache usage. + if (!cols.has("cache_state")) { + db.exec("ALTER TABLE session_pins ADD COLUMN cache_state TEXT"); + } schemaEnsured = true; } catch (err) { degradation.record("feedback", err); @@ -160,6 +166,67 @@ function save(sessionId, pin) { } } +/** + * Persist per-session prompt-cache state (Phase 1, cache-aware routing). + * Piggybacks on the session_pins row; creates a minimal row when the session + * has no pin yet (possible when sticky sessions are disabled but tracking + * is on). Best-effort like everything else in this module. + * + * @param {string} sessionId + * @param {{warmPrefixTokens:number, provider:string, model:string|null, lastRequestAt:number, ttlMs:number}} state + */ +function saveCacheState(sessionId, state) { + if (!sessionId || !state?.provider) return; + const db = _db(); + if (!db) return; + try { + const json = JSON.stringify(state); + const res = _stmt( + db, + "cache_state_update", + "UPDATE session_pins SET cache_state = ? WHERE session_id = ?" + ).run(json, sessionId); + if (res.changes === 0) { + _stmt( + db, + "cache_state_insert", + `INSERT INTO session_pins (session_id, provider, model, ts, cache_state) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(session_id) DO UPDATE SET cache_state = excluded.cache_state` + ).run(sessionId, state.provider, state.model ?? null, Date.now(), json); + } + } catch (err) { + degradation.record("feedback", err); + } +} + +/** + * Load per-session cache state. Returns null when absent, unparsable, or the + * DB is unavailable — callers treat null as "no cache signal for this + * session" (provider doesn't report cache usage, or no response seen yet). + * + * @param {string} sessionId + * @returns {{warmPrefixTokens:number, provider:string, model:string|null, lastRequestAt:number, ttlMs:number}|null} + */ +function loadCacheState(sessionId) { + if (!sessionId) return null; + const db = _db(); + if (!db) return null; + try { + const row = _stmt( + db, + "cache_state_load", + "SELECT cache_state FROM session_pins WHERE session_id = ?" + ).get(sessionId); + if (!row?.cache_state) return null; + const parsed = JSON.parse(row.cache_state); + return parsed && typeof parsed === "object" ? parsed : null; + } catch (err) { + degradation.record("feedback", err); + return null; + } +} + /** * Remove a pin. * @param {string} sessionId @@ -205,4 +272,4 @@ function _clear() { } catch { /* best-effort */ } } -module.exports = { load, save, remove, cleanup, _clear }; +module.exports = { load, save, remove, cleanup, saveCacheState, loadCacheState, _clear }; diff --git a/src/routing/cache-economics.js b/src/routing/cache-economics.js new file mode 100644 index 0000000..a76eb35 --- /dev/null +++ b/src/routing/cache-economics.js @@ -0,0 +1,123 @@ +/** + * Per-provider prompt-cache economics. + * + * Single resolution point for "what does the prompt cache cost on this + * provider/model": read price, write price, TTL, and mechanism. The + * switch-cost math in `cache-switch-cost.js` and the cache-state tracker in + * `session-affinity.js` read from here — decision code never hardcodes + * per-model numbers. + * + * Resolution order per field: + * 1. model-registry entry (models.dev carries absolute cacheRead/cacheWrite + * $/1M for many models) + * 2. provider-type fallback table below (multipliers of the model's input + * price), researched Aug 2026. + * + * Mechanisms: + * - 'explicit' — cache_control breakpoints, paid writes (Anthropic-style). + * TTL refreshes on every read, so lastRequestAt + ttlMs is a live clock. + * - 'automatic' — provider caches transparently, writes are free + * (OpenAI/DeepSeek/Gemini-style). TTL is best-effort. + * - 'local' — no dollar cost at all; the "cost" of a cold prefix is + * prefill latency, handled separately in cache-switch-cost.js. + * + * @module routing/cache-economics + */ + +const logger = require('../logger'); + +// Fallback economics keyed by Lynkr provider type. `readMult`/`writeMult` +// are multipliers of the model's per-1M input price. +const PROVIDER_CACHE_DEFAULTS = { + // Anthropic-hosted (explicit cache_control, 1.25x write for the 5-min TTL, + // TTL refreshed on every read). + 'azure-anthropic': { readMult: 0.1, writeMult: 1.25, ttlMs: 5 * 60 * 1000, mechanism: 'explicit' }, + bedrock: { readMult: 0.1, writeMult: 1.25, ttlMs: 5 * 60 * 1000, mechanism: 'explicit' }, + databricks: { readMult: 0.1, writeMult: 1.25, ttlMs: 5 * 60 * 1000, mechanism: 'explicit' }, + + // OpenAI automatic prefix caching: writes free, ~5-10 min idle eviction. + openai: { readMult: 0.1, writeMult: 0, ttlMs: 10 * 60 * 1000, mechanism: 'automatic' }, + 'azure-openai': { readMult: 0.1, writeMult: 0, ttlMs: 10 * 60 * 1000, mechanism: 'automatic' }, + + // Gemini implicit caching / GLM / Kimi / aggregators: free automatic + // caching, best-effort TTL. Aggregators (openrouter/edenai) depend on the + // underlying model; models.dev per-model data wins when present. + vertex: { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + zai: { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + moonshot: { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + openrouter: { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + edenai: { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + // DeepSeek direct (via aggregators today, kept for model-level matches): + deepseek: { readMult: 0.02, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }, + + // Local runtimes: no dollars; KV cache lives as long as the model stays + // loaded (ollama keep_alive default ~5 min). + ollama: { readMult: 0, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'local' }, + llamacpp: { readMult: 0, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'local' }, + lmstudio: { readMult: 0, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'local' }, +}; + +const GENERIC_DEFAULT = { readMult: 0.1, writeMult: 0, ttlMs: 5 * 60 * 1000, mechanism: 'automatic' }; + +/** + * Fallback cache economics for a provider type (no model lookup). + * @param {string|null} providerType + * @returns {{readMult:number, writeMult:number, ttlMs:number, mechanism:string}} + */ +function getProviderCacheDefaults(providerType) { + return PROVIDER_CACHE_DEFAULTS[providerType] || GENERIC_DEFAULT; +} + +/** + * Resolve full cache economics for a provider/model pair. + * + * @param {string|null} providerType - Lynkr provider type (e.g. 'databricks') + * @param {string|null} model - model name for registry lookup + * @returns {{ + * inputPerM:number, outputPerM:number, + * cacheReadPerM:number, cacheWritePerM:number, + * ttlMs:number, mechanism:string, unknownPricing:boolean + * }} All prices are USD per 1M tokens. + */ +function resolveCacheEconomics(providerType, model) { + const defaults = getProviderCacheDefaults(providerType); + + let cost = null; + try { + const { getModelRegistrySync } = require('./model-registry'); + cost = model ? getModelRegistrySync().getCost(model) : null; + } catch (err) { + logger.debug({ err: err.message }, '[CacheEconomics] registry lookup failed'); + } + + const inputPerM = typeof cost?.input === 'number' ? cost.input : 0; + const outputPerM = typeof cost?.output === 'number' ? cost.output : 0; + + // Registry cacheTtlMs/cacheMechanism (Phase 4 entries) win over the + // provider fallback; models.dev absolute cache prices win over multipliers. + const mechanism = cost?.cacheMechanism || defaults.mechanism; + const ttlMs = typeof cost?.cacheTtlMs === 'number' ? cost.cacheTtlMs : defaults.ttlMs; + + const cacheReadPerM = mechanism === 'local' + ? 0 + : (typeof cost?.cacheRead === 'number' ? cost.cacheRead : inputPerM * defaults.readMult); + const cacheWritePerM = mechanism === 'local' + ? 0 + : (typeof cost?.cacheWrite === 'number' ? cost.cacheWrite : inputPerM * defaults.writeMult); + + return { + inputPerM, + outputPerM, + cacheReadPerM, + cacheWritePerM, + ttlMs, + mechanism, + unknownPricing: !!cost?.unknown, + }; +} + +module.exports = { + PROVIDER_CACHE_DEFAULTS, + getProviderCacheDefaults, + resolveCacheEconomics, +}; diff --git a/src/routing/cache-switch-cost.js b/src/routing/cache-switch-cost.js new file mode 100644 index 0000000..5c57720 --- /dev/null +++ b/src/routing/cache-switch-cost.js @@ -0,0 +1,194 @@ +/** + * Switch-cost break-even model (Phase 3, cache-aware routing). + * + * Answers one question for a session that holds a warm prompt-cache prefix + * on its pinned model: does switching to a cheaper target model pay for the + * cache it breaks, within the turns this session is expected to have left? + * + * stayPerTurn = warm*cacheRead(cur) + new*input(cur) + out*output(cur) + * switchOnce = (warm + new) * cacheWrite(target) // one-time + * switchPerTurn = warm*cacheRead(tgt) + new*input(tgt) + out*output(tgt) + * breakEvenTurns = switchOnce / (stayPerTurn - switchPerTurn) + * + * Switch iff breakEvenTurns <= expectedRemainingTurns. + * + * The output-token term is deliberately included: for de-escalation the + * dominant per-turn saving is often the output-price spread (e.g. Opus $25/M + * vs Haiku $5/M out), and omitting it makes the router hold expensive pins + * long past the point the economics justify. + * + * Scope: this module prices COST-MOTIVATED switches (de-escalator, bandit + * exploration, economic downgrades). Hard triggers — risk keywords, force + * phrases, context overflow, vision, upward drift — are never gated here; + * correctness beats cost and the math can never favor a pricier model anyway + * (savings would be negative). + * + * Local models have no dollar cost; the price of a cold prefix is prefill + * latency, so a warm local prefix above a size threshold holds the pin. + * + * All pricing comes from cache-economics.js (registry-first, provider + * fallback) — no per-model numbers live in this file. + * + * @module routing/cache-switch-cost + */ + +const config = require('../config'); +const logger = require('../logger'); + +const PER_TOKEN = 1 / 1_000_000; // registry prices are USD per 1M tokens + +function _cfg() { + const c = config.routing?.cacheAware || {}; + return { + enabled: c.enabled !== false, + defaultRemainingTurns: c.defaultRemainingTurns ?? 10, + newTokensPerTurn: c.newTokensPerTurn ?? 2000, + outputTokensPerTurn: c.outputTokensPerTurn ?? 800, + localMaxSwitchPrefixTokens: c.localMaxSwitchPrefixTokens ?? 16000, + }; +} + +/** + * Evaluate whether a cost-motivated switch away from the session's warm + * prefix should be allowed. + * + * @param {Object} args + * @param {Object|null} args.cacheState - from sessionAffinity.getCacheState: + * {warmPrefixTokens, provider, model, lastRequestAt, ttlMs, cold} or null. + * @param {{provider:string, model:string|null}} args.current - pinned target. + * @param {{provider:string, model:string|null}} args.target - proposed target. + * @param {number} [args.newTokensPerTurn] - est. fresh input tokens per turn. + * @param {number} [args.outputTokensPerTurn] - est. output tokens per turn. + * @param {number|null} [args.expectedRemainingTurns] - median remaining + * turns; falls back to the conservative config default when null. + * @param {Object} [args.deps] - test injection: {resolveCacheEconomics}. + * @returns {{ + * switchAllowed: boolean, + * reason: string, + * breakEvenTurns: number|null, + * expectedRemainingTurns: number, + * warmPrefixTokens: number, + * switchOnceUsd: number|null, + * stayPerTurnUsd: number|null, + * switchPerTurnUsd: number|null, + * projectedStaySavingsUsd: number|null, + * }} + */ +function evaluateSwitch({ + cacheState, + current, + target, + newTokensPerTurn, + outputTokensPerTurn, + expectedRemainingTurns, + deps = {}, +} = {}) { + const cfg = _cfg(); + const remaining = Number.isFinite(expectedRemainingTurns) && expectedRemainingTurns > 0 + ? expectedRemainingTurns + : cfg.defaultRemainingTurns; + + const base = { + switchAllowed: true, + reason: 'no_cache_state', + breakEvenTurns: null, + expectedRemainingTurns: remaining, + warmPrefixTokens: cacheState?.warmPrefixTokens ?? 0, + switchOnceUsd: null, + stayPerTurnUsd: null, + switchPerTurnUsd: null, + projectedStaySavingsUsd: null, + }; + + if (!cfg.enabled) return { ...base, reason: 'feature_disabled' }; + if (!current?.provider || !target?.provider) return base; + if (current.provider === target.provider && current.model === target.model) { + return { ...base, reason: 'same_model' }; + } + + // No tracked state → nothing warm to protect (provider reports no cache + // usage, or no response seen yet). Switching is cache-free. + if (!cacheState || !(cacheState.warmPrefixTokens > 0)) return base; + + // State recorded for a different model than the pin we're defending — + // stale after a prior switch; treat as no protection. + if (cacheState.model && current.model && cacheState.model !== current.model) { + return { ...base, reason: 'stale_cache_state' }; + } + + // Phase 2 — TTL clock. Anthropic-style caches refresh on every read, so + // lastRequestAt + ttlMs going stale means the prefix is already cold and + // the switch costs nothing extra. + if (cacheState.cold) return { ...base, reason: 'cache_cold' }; + + let resolve = deps.resolveCacheEconomics; + if (typeof resolve !== 'function') { + ({ resolveCacheEconomics: resolve } = require('./cache-economics')); + } + + const econCur = resolve(current.provider, current.model); + const econTgt = resolve(target.provider, target.model); + + const W = cacheState.warmPrefixTokens; + const N = Number.isFinite(newTokensPerTurn) && newTokensPerTurn >= 0 + ? newTokensPerTurn : cfg.newTokensPerTurn; + const O = Number.isFinite(outputTokensPerTurn) && outputTokensPerTurn >= 0 + ? outputTokensPerTurn : cfg.outputTokensPerTurn; + + // Local pin: no dollars in either column — the cost of abandoning the + // warm prefix is prefill latency on whatever serves the next turn. + // Hold while the prefix is large; small prefixes re-fill fast enough. + if (econCur.mechanism === 'local' && econTgt.mechanism === 'local') { + const allowed = W <= cfg.localMaxSwitchPrefixTokens; + return { + ...base, + switchAllowed: allowed, + reason: allowed ? 'local_prefix_small' : 'local_prefill_hold', + }; + } + + const stayPerTurn = (W * econCur.cacheReadPerM + N * econCur.inputPerM + O * econCur.outputPerM) * PER_TOKEN; + const switchOnce = (W + N) * econTgt.cacheWritePerM * PER_TOKEN; + const switchPerTurn = (W * econTgt.cacheReadPerM + N * econTgt.inputPerM + O * econTgt.outputPerM) * PER_TOKEN; + const savingsPerTurn = stayPerTurn - switchPerTurn; + + const priced = { + ...base, + switchOnceUsd: switchOnce, + stayPerTurnUsd: stayPerTurn, + switchPerTurnUsd: switchPerTurn, + projectedStaySavingsUsd: savingsPerTurn, + }; + + // Switching to a same-or-more-expensive target never pays on economics. + // (Escalations don't route through this gate, so blocking is safe.) + if (savingsPerTurn <= 0) { + return { + ...priced, + switchAllowed: false, + reason: 'never_profitable', + breakEvenTurns: Infinity, + }; + } + + const breakEvenTurns = switchOnce / savingsPerTurn; + const allowed = breakEvenTurns <= remaining; + + logger.debug({ + current: `${current.provider}:${current.model}`, + target: `${target.provider}:${target.model}`, + warmPrefixTokens: W, + breakEvenTurns: Number(breakEvenTurns.toFixed(2)), + expectedRemainingTurns: remaining, + allowed, + }, '[CacheSwitchCost] break-even evaluated'); + + return { + ...priced, + switchAllowed: allowed, + reason: allowed ? 'break_even_cleared' : 'break_even_blocked', + breakEvenTurns, + }; +} + +module.exports = { evaluateSwitch }; diff --git a/src/routing/complexity-analyzer.js b/src/routing/complexity-analyzer.js index d8de3dc..b1814b8 100644 --- a/src/routing/complexity-analyzer.js +++ b/src/routing/complexity-analyzer.js @@ -795,7 +795,13 @@ async function analyzeComplexity(payload, options = {}) { threshold, mode: 'weighted', recommendation, - breakdown: weighted.dimensions, + // taskType rides alongside the numeric dimensions. The legacy path + // has always carried breakdown.taskType, and both the deescalator + // (tier + request_type evidence key) and the telemetry request_type + // column derive from it — without it here, weighted-mode requests + // (the default) recorded request_type NULL and could never + // accumulate demotion evidence. + breakdown: { ...weighted.dimensions, taskType: taskTypeResult }, weights: weighted.weights, meta: weighted.meta, forceReason: taskTypeResult.reason?.startsWith('force_') ? taskTypeResult.reason : null, diff --git a/src/routing/context-compressor.js b/src/routing/context-compressor.js new file mode 100644 index 0000000..e6a173c --- /dev/null +++ b/src/routing/context-compressor.js @@ -0,0 +1,170 @@ +/** + * Context Compression Module + * + * Inspired by TencentDB Agent Memory's short-term memory compression. + * Reduces conversation context size before complexity scoring to enable + * more accurate tier routing. + * + * Key insight: Long conversations bloat context → force expensive routing + * even for simple follow-ups. By compressing tool outputs and previous + * turns, we keep more requests in SIMPLE/MEDIUM tiers. + * + * @module routing/context-compressor + */ + +const logger = require('../logger'); + +/** + * Compress conversation messages by: + * 1. Offloading verbose tool_result content + * 2. Summarizing repetitive patterns + * 3. Keeping only essential context for scoring + * + * Does NOT modify the actual request sent to LLM — only used for complexity scoring. + * + * @param {Array} messages - Conversation messages + * @returns {{ compressed: Array, stats: Object }} - Compressed messages + stats + */ +function compressMessages(messages) { + if (!Array.isArray(messages) || messages.length === 0) { + return { compressed: messages, stats: { original: 0, compressed: 0, ratio: 1.0 } }; + } + + const compressed = []; + let originalSize = 0; + let compressedSize = 0; + let toolResultsOffloaded = 0; + let messagesKept = 0; + + // Keep system message always + const systemMsg = messages.find(m => m.role === 'system'); + if (systemMsg) { + compressed.push(systemMsg); + const size = estimateMessageSize(systemMsg); + originalSize += size; + compressedSize += size; + } + + // Keep last N user/assistant turns (sliding window) + const WINDOW_SIZE = 5; + const recentTurns = messages + .filter(m => m.role === 'user' || m.role === 'assistant') + .slice(-WINDOW_SIZE); + + for (const msg of recentTurns) { + const original = estimateMessageSize(msg); + originalSize += original; + + // Compress tool_result blocks (biggest token consumers) + if (Array.isArray(msg.content)) { + const compressedContent = msg.content.map(block => { + if (block?.type === 'tool_result') { + toolResultsOffloaded++; + // Keep only metadata, offload actual content + return { + type: 'tool_result', + tool_use_id: block.tool_use_id, + content: '[offloaded]', // Replaced with placeholder + _compressed: true, + }; + } + return block; + }); + + const compressedMsg = { ...msg, content: compressedContent }; + compressed.push(compressedMsg); + compressedSize += estimateMessageSize(compressedMsg); + messagesKept++; + } else { + // No compression needed for string content + compressed.push(msg); + compressedSize += original; + messagesKept++; + } + } + + const ratio = originalSize > 0 ? compressedSize / originalSize : 1.0; + + const stats = { + original: originalSize, + compressed: compressedSize, + ratio, + reduction: Math.round((1 - ratio) * 100), + toolResultsOffloaded, + messagesKept, + messagesDropped: messages.length - messagesKept - (systemMsg ? 1 : 0), + }; + + logger.debug({ + ...stats, + originalMsgs: messages.length, + compressedMsgs: compressed.length, + }, '[context-compressor] Compression complete'); + + return { compressed, stats }; +} + +/** + * Estimate message size in tokens (rough heuristic: chars / 4) + */ +function estimateMessageSize(msg) { + if (!msg) return 0; + + let size = 0; + + // Role + metadata overhead + size += 10; + + // Content + if (typeof msg.content === 'string') { + size += Math.ceil(msg.content.length / 4); + } else if (Array.isArray(msg.content)) { + for (const block of msg.content) { + if (block?.type === 'text' && block.text) { + size += Math.ceil(block.text.length / 4); + } else if (block?.type === 'tool_result') { + // Tool results are verbose (code, logs, errors) + const content = Array.isArray(block.content) + ? block.content.map(c => c?.text || '').join('') + : (block.content || ''); + size += Math.ceil(content.length / 4); + } else if (block?.type === 'tool_use') { + // Tool use is compact (function name + args) + size += 50; + } + } + } + + return size; +} + +/** + * Check if compression is beneficial for this payload. + * Only compress if conversation is long enough to matter. + */ +function shouldCompress(payload) { + if (!payload?.messages || !Array.isArray(payload.messages)) { + return false; + } + + const msgCount = payload.messages.length; + + // Count total tool_result blocks (not just messages with tool results) + let toolResultCount = 0; + for (const msg of payload.messages) { + if (Array.isArray(msg.content)) { + toolResultCount += msg.content.filter(c => c?.type === 'tool_result').length; + } + } + + // Compress if: + // 1. More than 10 messages (long conversation), OR + // 2. More than 3 tool_result blocks (verbose outputs) + return msgCount > 10 || toolResultCount > 3; +} + +module.exports = { + compressMessages, + shouldCompress, + estimateMessageSize, +}; diff --git a/src/routing/difficulty-classifier.js b/src/routing/difficulty-classifier.js index 4fbf317..5d07752 100644 --- a/src/routing/difficulty-classifier.js +++ b/src/routing/difficulty-classifier.js @@ -7,14 +7,16 @@ * hard problems. An LLM reading the actual sentence knows better. * * DESIGN: - * - Uses whatever model is configured for the SIMPLE tier (fetched at call - * time via getModelTierSelector). When the user later swaps in a - * fine-tuned classifier as SIMPLE, this picks it up automatically. + * - Dedicated classifier model (CLASSIFIER_MODEL below), deliberately + * decoupled from tier serving — swapping TIER_* env vars does not change + * what classifies prompts. The tier DEPLOYMENT (which models the tiers + * route to) is surfaced to the classifier as tie-breaker context; see + * _buildTierContext. * - Structured JSON output; parse failure → null → caller falls back. - * - Hard 2500ms timeout; on timeout → null → caller falls back. + * - Hard timeout (TIMEOUT_MS); on timeout → null → caller falls back. * - LRU cache keyed by sha256(text.trim().toLowerCase()); capacity 500. * - Skip conditions surface via classifyDifficulty returning null with - * reason=skipped: text.length<15, force-pattern matched, risk=high, + * reason=skipped: empty text, force-pattern matched, risk=high, * cache hit is transparent (returns cached). * - Hardcoded kill-switch CLASSIFIER_ENABLED — no env var per user policy. * @@ -31,7 +33,6 @@ const CLASSIFIER_ENABLED = true; // LRU cache to keep amortized latency low. const TIMEOUT_MS = 10000; const CACHE_CAPACITY = 500; -const MIN_TEXT_LENGTH = 15; // Classifier model — decoupled from tier serving so SIMPLE tier can run a // more capable model for real traffic while the classifier stays fast and @@ -44,7 +45,7 @@ const CLASSIFIER_MODEL = 'qwen2.5:3b'; const VALID_TIERS = ['SIMPLE', 'MEDIUM', 'COMPLEX', 'REASONING']; -// One-shot classification prompt (v2). Kept in a const so drift is diffable. +// One-shot classification prompt (v3). Kept in a const so drift is diffable. // Difficulty framing (not intent) — matches config B routing goals. // // v2 (2026-07-21): added the follow-up rule, negative examples under @@ -54,11 +55,16 @@ const VALID_TIERS = ['SIMPLE', 'MEDIUM', 'COMPLEX', 'REASONING']; // flavored and nothing said surface vocabulary isn't the signal. Baseline // on data/difficulty-eval-followups.jsonl: 60% overall, 33% on SIMPLE, // 3 SIMPLE→REASONING criticals. +// +// v3 (2026-08-09): added trivial-arithmetic SIMPLE examples after a live +// over-route: "12+21" scored anchor 25 (MEDIUM band) and qwen ALSO said +// MEDIUM conf 1.0 — bare arithmetic read as "a specific mechanical task". +// Any tiny model adds two numbers; it belongs in SIMPLE. const CLASSIFY_PROMPT = `You are a classifier for an LLM routing proxy. Classify the difficulty of the CURRENT user prompt into exactly one of four tiers. Reply with ONLY valid JSON on a single line, no other text. Tiers: - SIMPLE: casual acknowledgments, greetings, one-word answers, trivial factual lookups, and short conversational follow-up questions about people, stories, events, or everyday facts. Any tiny model handles. - examples: "hi", "ok thanks", "yes continue", "what time is it", "who is doctor doom?", "who kills him?", "why did he do that?", "and then what happened?", "does bleach kill mold?" + examples: "hi", "ok thanks", "yes continue", "what time is it", "who is doctor doom?", "who kills him?", "why did he do that?", "and then what happened?", "does bleach kill mold?", "12+21", "what is 15% of 80?", "convert 3km to miles" - MEDIUM: one specific mechanical task or a focused explanation. Mid-size local model suffices. examples: "list the exports from this file", "run the unit tests", "fix the linter warnings", "explain this regex", "add error handling to this block", "verify the file exists before reading it" - COMPLEX: multi-file design, systemic refactor, architecture review, debugging that requires broad code understanding. Needs a strong general model. @@ -81,15 +87,101 @@ Reply format (strict): {"tier":"SIMPLE|MEDIUM|COMPLEX|REASONING","confidence":0. const CONTEXT_MAX_TEXT_LENGTH = 40; const CONTEXT_MAX_CHARS = 300; +// --- Tier deployment context ------------------------------------------------- +// +// Tells the classifier WHAT each tier routes to (model + coarse traits from +// the registry), so borderline verdicts can weigh the actual fleet — e.g. +// prefer MEDIUM when its configured model is a large cloud model, or lean +// away from a tier whose model lacks the needed capability. +// +// Constraints that shape the design: +// - TIE-BREAKER ONLY. The difficulty rubric in CLASSIFY_PROMPT governs; +// the deployment block must not redefine what the tiers mean, or eval +// labels (data/difficulty-eval*.jsonl) stop being comparable. +// - BYTE-STABLE per process. Tier config is env-driven and fixed for the +// process lifetime, so the block is built once and memoized. It also +// participates in the LRU cache key (fingerprint) so a reconfigured +// deployment never serves verdicts cached under the old one. +// - FAIL-SAFE. Registry misses or config errors degrade to provider-type +// traits or to no block at all — classification proceeds regardless. +// MEASURED OFF (2026-08-08): A/B on data/difficulty-eval-followups.jsonl — +// baseline 84.0% overall / 0 SIMPLE→REASONING criticals; with the block +// 76.0% / 1 critical ('prove me wrong lol' → REASONING conf 1.0, the exact +// failure class prompt v2 was built to kill). qwen2.5:3b can't use fleet +// info — it dilutes the rubric. Machinery kept dark for a future stronger +// classifier: flip this flag, then re-run +// scripts/validate-difficulty-classifier.js on BOTH eval files and hold +// the ship bar (>=85% overall, zero MEDIUM→REASONING false positives). +const TIER_CONTEXT_ENABLED = false; +const LOCAL_PROVIDERS = new Set(['ollama', 'llamacpp', 'lmstudio']); + +/** Pure builder — exported for tests. @param {Object} tiers {TIER: 'provider:model'} */ +function _buildTierContext(tiers) { + const lines = []; + for (const tier of VALID_TIERS) { + const spec = tiers?.[tier]; + if (typeof spec !== 'string' || !spec.includes(':')) continue; + const sep = spec.indexOf(':'); + const provider = spec.slice(0, sep); + const model = spec.slice(sep + 1); + const traits = [LOCAL_PROVIDERS.has(provider) ? 'local' : 'cloud']; + try { + const { getModelRegistrySync } = require('./model-registry'); + const cost = getModelRegistrySync().getCost(model); + if (cost && !cost.unknown) { + if (Number.isFinite(cost.context) && cost.context >= 1000) { + traits.push(`${Math.round(cost.context / 1000)}k ctx`); + } + if (cost.reasoning) traits.push('reasoning'); + if (cost.vision) traits.push('vision'); + } + } catch { /* registry unavailable — provider trait only */ } + lines.push(`- ${tier} routes to ${model} (${traits.join(', ')})`); + } + if (!lines.length) return null; + const block = `Deployment (what each tier currently routes to — use only as a tie-breaker between adjacent tiers; the tier definitions above still govern): +${lines.join('\n')}`; + return { + block, + fingerprint: crypto.createHash('sha256').update(block).digest('hex').slice(0, 16), + }; +} + +/** @type {{block:string, fingerprint:string}|null|undefined} undefined = not built yet */ +let _tierContext; + +function _tierDeploymentContext() { + // Test override (set via _setTierContextForTests) wins over the flag so + // the machinery stays exercised while shipping dark. + if (_tierContext !== undefined) return _tierContext; + if (!TIER_CONTEXT_ENABLED) { _tierContext = null; return null; } + try { + const config = require('../config'); + _tierContext = _buildTierContext(config.modelTiers); + } catch { + _tierContext = null; + } + return _tierContext; +} + +/** Test helper — override or reset (pass undefined) the memoized block. */ +function _setTierContextForTests(tiers) { + _tierContext = tiers === undefined ? undefined : _buildTierContext(tiers); +} + function _buildPrompt(text, context) { + const tierCtx = _tierDeploymentContext(); + const preamble = tierCtx ? `${CLASSIFY_PROMPT} +${tierCtx.block} +` : CLASSIFY_PROMPT; if (context) { - return `${CLASSIFY_PROMPT} + return `${preamble} Conversation so far (context only — classify the CURRENT prompt, inheriting topic difficulty per the rules): ${context} CURRENT user prompt: """${text}"""`; } - return `${CLASSIFY_PROMPT} + return `${preamble} User prompt: """${text}"""`; } @@ -212,7 +304,12 @@ async function classifyDifficulty(text, opts = {}) { if (!CLASSIFIER_ENABLED) return null; if (typeof text !== 'string') return null; const trimmed = text.trim(); - if (trimmed.length < MIN_TEXT_LENGTH) return null; + // No minimum length (removed 2026-08-09): short prompts are exactly what + // the v2 context feature was built for, and the 15-char gate meant the + // classifier could never rescue anchor over-reads like '12+21' -> MEDIUM. + // Greetings still skip upstream via force patterns (opts.forceMatched), + // and the LRU absorbs repeats, so the added model calls are bounded. + if (!trimmed) return null; if (opts.forceMatched) return null; if (opts.riskLevel === 'high') return null; @@ -222,8 +319,11 @@ async function classifyDifficulty(text, opts = {}) { : null; // Context participates in the cache key: the same follow-up text means - // different things in different conversations. - const key = _cacheKey(context ? `${trimmed}${context}` : trimmed); + // different things in different conversations. The tier-deployment + // fingerprint participates too (NUL separators are collision-proof), so + // verdicts cached under one deployment never serve a reconfigured one. + const tierFp = _tierDeploymentContext()?.fingerprint ?? ""; + const key = _cacheKey(`${tierFp}${context ? `${trimmed}${context}` : trimmed}`); const cached = _cache.get(key); if (cached) return { ...cached, source: 'cache' }; @@ -256,6 +356,8 @@ module.exports = { _parseResult, _cacheKey, _buildPrompt, + _buildTierContext, + _setTierContextForTests, _clearCacheForTests, _getCacheStats, }; diff --git a/src/routing/index.js b/src/routing/index.js index 9e4fc0d..d451353 100644 --- a/src/routing/index.js +++ b/src/routing/index.js @@ -156,6 +156,7 @@ function getBestLocalProvider() { * @returns {Object} Routing decision with provider and metadata */ const sessionAffinity = require('./session-affinity'); +const { evaluateSwitch: evaluateCacheSwitch } = require('./cache-switch-cost'); // --------------------------------------------------------------------------- // WS1 — sticky sessions @@ -272,12 +273,25 @@ function buildDecision(fields = {}) { // Fail-open (routing must never throw on shape problems) but loud. logger.error({ keys: Object.keys(fields) }, '[Routing] buildDecision called without provider/method'); } + // WS2.3 repair: the telemetry record sites read `analysis.requestType`, + // but nothing ever set that field — every routing_telemetry row recorded + // request_type NULL (verified live: 4052/4052 rows), so the evidence-based + // deescalator (keyed on tier + request_type) could never accumulate + // demotion evidence. Derive it here at the canonical constructor, using + // the SAME derivation the deescalator applies, so telemetry rows and + // demotion queries group by identical values. + let analysis = fields.analysis ?? null; + if (analysis && typeof analysis === 'object' && analysis.requestType == null) { + const requestType = analysis.breakdown?.taskType?.reason ?? analysis.taskType ?? null; + if (requestType != null) analysis = { ...analysis, requestType }; + } return { model: null, tier: null, reason: null, score: null, - analysis: null, + // `analysis` is appended after the ...fields spread (requestType + // derivation above); it defaults to null there. embeddingsResult: null, agenticResult: null, knnResult: null, @@ -291,6 +305,61 @@ function buildDecision(fields = {}) { propensity: 1.0, candidates: [{ provider: fields.provider, model: fields.model ?? null }], ...fields, + analysis, + }; +} + +/** + * Cache-aware switch gate (Phases 2+3). + * + * Prices a DOWNWARD (cost-motivated) model change against the session's warm + * prompt-cache prefix: inside the provider's cache TTL, the pin holds unless + * the break-even math clears within the session's expected remaining turns. + * A cold prefix (TTL elapsed — Phase 2) or absent cache state makes the + * switch cache-free and it passes. + * + * Upward/lateral moves (guard escalations, risk, drift to a higher tier) + * MUST NOT route through here — correctness beats cost, and the math can + * never favor a pricier model anyway. + * + * @returns {object|null} evaluation result, or null when gating doesn't + * apply (same model, missing data, or evaluator failure — fail-open). + */ +function _cacheAwareSwitchGate(sessionId, pin, fresh, payload) { + if (!sessionId || !pin?.model || !fresh?.model) return null; + if (pin.provider === fresh.provider && pin.model === fresh.model) return null; + try { + const cacheState = sessionAffinity.getCacheState(sessionId); + const currentTurns = Array.isArray(payload?.messages) ? payload.messages.length : 0; + let expectedRemainingTurns = null; + try { + expectedRemainingTurns = telemetry.getExpectedRemainingTurns(currentTurns); + } catch { /* sparse-data default applies */ } + return evaluateCacheSwitch({ + cacheState, + current: { provider: pin.provider, model: pin.model }, + target: { provider: fresh.provider, model: fresh.model }, + expectedRemainingTurns, + }); + } catch (err) { + degradation.record('cache_switch_cost', err); + return null; // fail-open: gate must never block routing on an internal error + } +} + +/** Shape the gate result into the telemetry receipt (Phase 6). */ +function _cacheDecisionReceipt(evaluation, decision) { + if (!evaluation) return null; + return { + decision, + reason: evaluation.reason, + warmPrefixTokens: evaluation.warmPrefixTokens, + breakEvenTurns: Number.isFinite(evaluation.breakEvenTurns) + ? Number(evaluation.breakEvenTurns.toFixed(2)) + : evaluation.breakEvenTurns === Infinity ? -1 : null, + projectedSwitchCostUsd: evaluation.switchOnceUsd, + projectedStaySavingsUsd: evaluation.projectedStaySavingsUsd, + expectedRemainingTurns: evaluation.expectedRemainingTurns, }; } @@ -349,7 +418,28 @@ async function determineProviderSmart(payload, options = {}) { fresh.switch_reason = 'score_drift'; if (_tierPriority(fresh.tier) >= _tierPriority(pinCheck.pin.tier)) { writeSessionPin(pinCheck.sessionId, fresh, payload); + return fresh; + } + // Fresh decision came back BELOW the pin on a drift re-decide — + // that's a cost-motivated downgrade (deescalator/bandit inside the + // fresh decision), so it must clear the cache break-even gate + // (Phases 2+3) before abandoning the warm prefix. + const gate = _cacheAwareSwitchGate(pinCheck.sessionId, pinCheck.pin, fresh, payload); + if (gate && !gate.switchAllowed) { + logger.info({ + sessionId: pinCheck.sessionId, + pinModel: pinCheck.pin.model, + freshModel: fresh.model, + reason: gate.reason, + breakEvenTurns: gate.breakEvenTurns, + warmPrefixTokens: gate.warmPrefixTokens, + }, '[Routing] Cache break-even holds pin — downgrade suppressed'); + const served = _pinToDecision(pinCheck.pin, { reason: 'cache_hold', risk: null }); + served._cacheDecision = _cacheDecisionReceipt(gate, 'hold'); + writeSessionPin(pinCheck.sessionId, pinCheck.pin, payload); + return served; } + if (gate) fresh._cacheDecision = _cacheDecisionReceipt(gate, 'switch'); return fresh; } } @@ -376,6 +466,18 @@ async function determineProviderSmart(payload, options = {}) { // fires on the compaction path — guard escalations are mandatory. if (pinCheck.reason === 'compaction') { fresh.switch_reason = 'compaction'; + // Compaction reset the provider-side prefix upstream (Phase 2: the + // cache is cold by construction), so the break-even gate doesn't apply; + // record the receipt so the dashboard can attribute the free switch. + fresh._cacheDecision = { + decision: 'switch', + reason: 'compaction_cache_reset', + warmPrefixTokens: 0, + breakEvenTurns: 0, + projectedSwitchCostUsd: null, + projectedStaySavingsUsd: null, + expectedRemainingTurns: null, + }; const promptTokensEst = _tryCountTokens(payload, pin.model || fresh.model); if (!_economicDowngradeAllowed(promptTokensEst, pin.model, fresh.model)) { logger.debug({ diff --git a/src/routing/model-registry.js b/src/routing/model-registry.js index 047ce29..03ecbd7 100644 --- a/src/routing/model-registry.js +++ b/src/routing/model-registry.js @@ -17,16 +17,19 @@ const CACHE_FILE = path.join(__dirname, '../../data/model-prices-cache.json'); const CACHE_TTL_MS = 24 * 60 * 60 * 1000; // 24 hours const REFRESH_RETRY_MS = 5 * 60 * 1000; // backoff between failed refresh attempts -// Databricks fallback pricing (based on Anthropic direct API prices) +// Databricks fallback pricing (based on Anthropic direct API prices). +// cacheRead/cacheWrite are absolute USD/1M (0.1x / 1.25x input — Anthropic +// explicit caching, 5-min TTL); non-Claude rows omit them and inherit the +// provider-type multipliers in cache-economics.js. const DATABRICKS_FALLBACK = { // Claude models - 'databricks-claude-opus-4-6': { input: 5.0, output: 25.0, context: 1000000 }, - 'databricks-claude-opus-4-5': { input: 5.0, output: 25.0, context: 200000 }, - 'databricks-claude-opus-4-1': { input: 15.0, output: 75.0, context: 200000 }, - 'databricks-claude-sonnet-4-5': { input: 3.0, output: 15.0, context: 200000 }, - 'databricks-claude-sonnet-4': { input: 3.0, output: 15.0, context: 200000 }, - 'databricks-claude-3-7-sonnet': { input: 3.0, output: 15.0, context: 200000 }, - 'databricks-claude-haiku-4-5': { input: 1.0, output: 5.0, context: 200000 }, + 'databricks-claude-opus-4-6': { input: 5.0, output: 25.0, cacheRead: 0.5, cacheWrite: 6.25, context: 1000000 }, + 'databricks-claude-opus-4-5': { input: 5.0, output: 25.0, cacheRead: 0.5, cacheWrite: 6.25, context: 200000 }, + 'databricks-claude-opus-4-1': { input: 15.0, output: 75.0, cacheRead: 1.5, cacheWrite: 18.75, context: 200000 }, + 'databricks-claude-sonnet-4-5': { input: 3.0, output: 15.0, cacheRead: 0.3, cacheWrite: 3.75, context: 200000 }, + 'databricks-claude-sonnet-4': { input: 3.0, output: 15.0, cacheRead: 0.3, cacheWrite: 3.75, context: 200000 }, + 'databricks-claude-3-7-sonnet': { input: 3.0, output: 15.0, cacheRead: 0.3, cacheWrite: 3.75, context: 200000 }, + 'databricks-claude-haiku-4-5': { input: 1.0, output: 5.0, cacheRead: 0.1, cacheWrite: 1.25, context: 200000 }, // Llama models 'databricks-llama-4-maverick': { input: 1.0, output: 1.0, context: 128000 }, @@ -208,6 +211,16 @@ class ModelRegistry { prices[modelId.toLowerCase()] = { input: inputCost, output: outputCost, + // Cache economics (Phase 4) — LiteLLM publishes per-token cache + // read/write costs for providers that price them. Absent fields + // stay undefined so cache-economics.js falls back to its + // provider-type multiplier table. + ...(info.cache_read_input_token_cost != null + ? { cacheRead: info.cache_read_input_token_cost * 1_000_000 } + : {}), + ...(info.cache_creation_input_token_cost != null + ? { cacheWrite: info.cache_creation_input_token_cost * 1_000_000 } + : {}), context: info.max_input_tokens || info.max_tokens || 128000, maxOutput: info.max_output_tokens || 4096, toolCall: info.supports_function_calling ?? true, @@ -296,9 +309,21 @@ class ModelRegistry { this.modelIndex.set(modelId, info); } - // Add LiteLLM (highest priority) + // Add LiteLLM (highest priority) — but merge cache economics from the + // entry it shadows: LiteLLM wins on base prices, while cacheRead/ + // cacheWrite survive from models.dev when LiteLLM doesn't carry them. + // Without this, a LiteLLM entry missing cache prices hides the + // models.dev cache data and the switch-cost math degrades to + // provider-multiplier approximations (observed live: 0 LiteLLM entries + // with cacheRead shadowing 4.9k models.dev entries that had it). for (const [modelId, info] of Object.entries(this.litellmPrices)) { - this.modelIndex.set(modelId, info); + const prev = this.modelIndex.get(modelId); + const merged = { ...info }; + if (prev) { + if (merged.cacheRead == null && typeof prev.cacheRead === 'number') merged.cacheRead = prev.cacheRead; + if (merged.cacheWrite == null && typeof prev.cacheWrite === 'number') merged.cacheWrite = prev.cacheWrite; + } + this.modelIndex.set(modelId, merged); } } diff --git a/src/routing/session-affinity.js b/src/routing/session-affinity.js index abd6e97..d417784 100644 --- a/src/routing/session-affinity.js +++ b/src/routing/session-affinity.js @@ -172,6 +172,64 @@ function shouldRepin(pin, payload) { return { repin: false, reason: null }; } +// --------------------------------------------------------------------------- +// Cache-aware routing (Phase 1) — per-session prompt-cache state. +// +// After every upstream response we record how much of the session's prefix +// is warm on the provider side: cache_read + cache_creation tokens together +// describe the cached prefix as of that response. Anthropic-style explicit +// caches refresh their TTL on every read, so `lastRequestAt + ttlMs` is a +// live cache clock the switch-timing logic (Phase 2) can consult. +// --------------------------------------------------------------------------- + +/** + * Record prompt-cache usage from an upstream response. No-ops when the + * response carried no cache signal (both counters zero/absent) so providers + * without cache reporting simply never accrue state. + * + * Overwrites unconditionally on signal: if the session switched models, the + * new response's numbers ARE the new model's cache state (caches are + * model-scoped — never carry a warm-prefix figure across a switch). + * + * @param {string} sessionId + * @param {{provider:string, model?:string|null, cacheReadTokens?:number, cacheCreationTokens?:number}} usage + */ +function recordCacheUsage(sessionId, usage) { + if (!sessionId || !usage?.provider) return; + const read = Number(usage.cacheReadTokens) || 0; + const created = Number(usage.cacheCreationTokens) || 0; + const warm = read + created; + if (warm <= 0) return; // no cache signal — state stays absent + + let ttlMs = 5 * 60 * 1000; + try { + const { resolveCacheEconomics } = require("./cache-economics"); + ttlMs = resolveCacheEconomics(usage.provider, usage.model ?? null).ttlMs; + } catch { /* keep default */ } + + store.saveCacheState(sessionId, { + warmPrefixTokens: warm, + provider: usage.provider, + model: usage.model ?? null, + lastRequestAt: Date.now(), + ttlMs, + }); +} + +/** + * Load the session's cache state, annotated with liveness. Returns null when + * no state exists (provider reports no cache usage, or no response yet). + * + * @param {string} sessionId + * @returns {({warmPrefixTokens:number, provider:string, model:string|null, lastRequestAt:number, ttlMs:number, cold:boolean})|null} + */ +function getCacheState(sessionId) { + const state = store.loadCacheState(sessionId); + if (!state || typeof state.lastRequestAt !== "number") return null; + const ttl = typeof state.ttlMs === "number" ? state.ttlMs : 5 * 60 * 1000; + return { ...state, cold: Date.now() - state.lastRequestAt > ttl }; +} + /** Test/maintenance helper — clear the in-memory Map only. */ function _clear() { pins.clear(); @@ -223,6 +281,8 @@ module.exports = { setPin, removePin, shouldRepin, + recordCacheUsage, + getCacheState, // legacy getPinned, setPinned, diff --git a/src/routing/telemetry.js b/src/routing/telemetry.js index f7958bd..f9f3f7f 100644 --- a/src/routing/telemetry.js +++ b/src/routing/telemetry.js @@ -165,6 +165,11 @@ function init() { ["candidates", "TEXT"], ["pinned", "INTEGER DEFAULT 0"], ["switch_reason", "TEXT"], + // Phase 6 (cache-aware routing) — per-decision cache economics + // receipt: {decision, reason, warmPrefixTokens, breakEvenTurns, + // projectedSwitchCostUsd, projectedStaySavingsUsd, + // expectedRemainingTurns} as JSON. + ["cache_decision", "TEXT"], ]; for (const [col, type] of additiveCols) { if (!existingCols.has(col)) { @@ -226,7 +231,8 @@ function record(data) { latency_ms, status_code, error_type, cost_usd, tool_calls_made, retry_count, circuit_breaker_state, quality_score, tokens_per_second, cost_efficiency, request_text, response_text, - base_tier, escalation_source, propensity, candidates, pinned, switch_reason + base_tier, escalation_source, propensity, candidates, pinned, switch_reason, + cache_decision ) VALUES ( @request_id, @session_id, @timestamp, @complexity_score, @tier, @agentic_type, @tool_count, @input_tokens, @message_count, @request_type, @@ -234,7 +240,8 @@ function record(data) { @latency_ms, @status_code, @error_type, @cost_usd, @tool_calls_made, @retry_count, @circuit_breaker_state, @quality_score, @tokens_per_second, @cost_efficiency, @request_text, @response_text, - @base_tier, @escalation_source, @propensity, @candidates, @pinned, @switch_reason + @base_tier, @escalation_source, @propensity, @candidates, @pinned, @switch_reason, + @cache_decision )` ); if (!insert) return; @@ -280,6 +287,11 @@ function record(data) { candidates: candidatesJson, pinned: data.pinned ? 1 : 0, switch_reason: data.switch_reason ?? null, + cache_decision: data.cache_decision == null + ? null + : (typeof data.cache_decision === "string" + ? data.cache_decision + : JSON.stringify(data.cache_decision)), }); } catch (err) { logger.debug({ err: err.message }, "Telemetry record failed"); @@ -791,6 +803,129 @@ function _resetForTests() { _testDbDisabled = false; } +// Cache-aware routing (Phase 6) — "cache dollars saved / burned by routing". +// +// Every gated switch/hold decision carries a receipt. Aggregating them +// answers the public question "does model routing break prefix caching?" +// with a dollar figure: +// - a HOLD avoided a switch whose one-time cache write exceeded its +// projected per-turn savings over the expected horizon +// (saved = switchOnce − savings × remainingTurns) +// - a SWITCH cleared break-even; its projected net gain is +// (savings × remainingTurns − switchOnce) +/** + * @param {Object} [opts] + * @param {number} [opts.since] - default: last 7 days. + * @returns {{decisions:number, holds:number, switches:number, + * dollarsSavedByHolds:number, dollarsSavedBySwitches:number, + * totalDollarsSaved:number, byReason:Object}|null} + */ +function getCacheEconomics(opts = {}) { + if (!init()) return null; + const since = opts.since ?? Date.now() - 7 * 24 * 60 * 60 * 1000; + try { + const rows = db + .prepare( + `SELECT cache_decision FROM routing_telemetry + WHERE timestamp > ? AND cache_decision IS NOT NULL + LIMIT 20000` + ) + .all(since); + + const out = { + decisions: 0, + holds: 0, + switches: 0, + dollarsSavedByHolds: 0, + dollarsSavedBySwitches: 0, + totalDollarsSaved: 0, + byReason: {}, + }; + + for (const row of rows) { + let d; + try { + d = JSON.parse(row.cache_decision); + } catch { + continue; + } + if (!d || typeof d !== "object") continue; + out.decisions++; + out.byReason[d.reason ?? "unknown"] = (out.byReason[d.reason ?? "unknown"] ?? 0) + 1; + + const savings = typeof d.projectedStaySavingsUsd === "number" ? d.projectedStaySavingsUsd : null; + const once = typeof d.projectedSwitchCostUsd === "number" ? d.projectedSwitchCostUsd : null; + const horizon = typeof d.expectedRemainingTurns === "number" ? d.expectedRemainingTurns : null; + + if (d.decision === "hold") { + out.holds++; + if (savings != null && once != null && horizon != null) { + out.dollarsSavedByHolds += Math.max(0, once - savings * horizon); + } + } else if (d.decision === "switch") { + out.switches++; + if (savings != null && once != null && horizon != null) { + out.dollarsSavedBySwitches += Math.max(0, savings * horizon - once); + } + } + } + + out.totalDollarsSaved = out.dollarsSavedByHolds + out.dollarsSavedBySwitches; + return out; + } catch (err) { + logger.debug({ err: err.message }, "Telemetry getCacheEconomics failed"); + return null; + } +} + +// Cache-aware routing (Phase 3) — expected remaining turns. +// +// Conditional median: given a session has already reached `currentTurns` +// messages, how many MORE messages do comparable sessions run? Uses the max +// message_count each recent session reached. Conditional (len > current) +// rather than a global median because long sessions systematically outlive +// the average — exactly the sessions whose warm prefixes are worth pricing. +const _remainingTurnsCache = new Map(); +const REMAINING_TURNS_CACHE_TTL_MS = 60 * 1000; +const REMAINING_TURNS_MIN_SESSIONS = 20; + +/** + * @param {number} currentTurns - message_count the session has reached now. + * @returns {number|null} median remaining messages, or null when telemetry + * is too sparse (< 20 comparable sessions in the last 7 days). + */ +function getExpectedRemainingTurns(currentTurns = 0) { + if (!init()) return null; + const bucket = Math.max(0, Math.floor((Number(currentTurns) || 0) / 5) * 5); + const cached = _remainingTurnsCache.get(bucket); + if (cached && Date.now() - cached.ts < REMAINING_TURNS_CACHE_TTL_MS) { + return cached.value; + } + try { + const since = Date.now() - 7 * 24 * 60 * 60 * 1000; + const rows = db + .prepare( + `SELECT MAX(message_count) AS len + FROM routing_telemetry + WHERE timestamp > ? AND session_id IS NOT NULL AND message_count IS NOT NULL + GROUP BY session_id + HAVING len > ? + LIMIT 5000` + ) + .all(since, bucket); + let value = null; + if (rows.length >= REMAINING_TURNS_MIN_SESSIONS) { + const remaining = rows.map((r) => r.len - bucket).sort((a, b) => a - b); + value = remaining[Math.floor(remaining.length / 2)]; + } + _remainingTurnsCache.set(bucket, { value, ts: Date.now() }); + return value; + } catch (err) { + logger.debug({ err: err.message }, "Telemetry getExpectedRemainingTurns failed"); + return null; + } +} + module.exports = { record, query, @@ -799,6 +934,8 @@ module.exports = { getRoutingAccuracy, getEscalationStats, getQualityByTierAndType, + getExpectedRemainingTurns, + getCacheEconomics, recordSavings, getSavingsSummary, cleanup, diff --git a/src/server.js b/src/server.js index 5bf769c..5be6377 100644 --- a/src/server.js +++ b/src/server.js @@ -1,5 +1,4 @@ const express = require("express"); -const compression = require("compression"); const config = require("./config"); const loggingMiddleware = require("./api/middleware/logging"); const router = require("./api/router"); @@ -19,7 +18,7 @@ const metrics = require("./metrics"); const logger = require("./logger"); const { initialiseMcp } = require("./mcp"); const { initConfigWatcher, getConfigWatcher } = require("./config/watcher"); -const { initializeHeadroom, shutdownHeadroom, getHeadroomManager } = require("./headroom"); +const { initializeHeadroom, shutdownHeadroom } = require("./headroom"); const { getWorkerPool, isWorkerPoolReady } = require("./workers/pool"); const { waitForOllama } = require("./clients/ollama-startup"); @@ -213,6 +212,27 @@ async function start() { server.once("error", reject); }); + // TencentDB-Agent-Memory sidecar — non-blocking (first start pulls two + // Docker images). Launched after listen so the proxy is usable immediately; + // the memory stack's LLM calls route back through Lynkr, which is why + // Lynkr must already be accepting requests when the containers come up. + if (config.tencentdbMemory?.enabled) { + (async () => { + try { + const tencentdbLauncher = require("./memory/tencentdb-launcher"); + const result = await tencentdbLauncher.ensureRunning(); + if (result.started) { + logger.info(result.endpoints, "[tdai] TencentDB Agent Memory stack ready"); + console.log(`TencentDB Agent Memory ready — Panel UI: ${result.endpoints.panel}`); + } else { + logger.debug({ reason: result.reason }, "[tdai] Sidecar not started"); + } + } catch (err) { + logger.warn({ err }, "[tdai] TencentDB memory sidecar failed to start — continuing without it"); + } + })(); + } + // Classifier bootstrap check — non-blocking, log-only. // Detects ollama + confirms the classifier model is pulled. Never auto- // installs (that's `lynkr init`'s job); warns and lets scoring fall back diff --git a/src/tools/code-graph.js b/src/tools/code-graph.js index 0b319e5..a0708c5 100644 --- a/src/tools/code-graph.js +++ b/src/tools/code-graph.js @@ -313,9 +313,10 @@ async function getBlastRadius(filePaths, options = {}) { if (cached) return cached; // Query neighbors for each file to estimate blast radius + const depth = String(config.codeGraph?.maxDepth ?? 2); const result = await execGraph( "query", - ["get_neighbors", "--files", ...filePaths, "--depth", "2", "--json"], + ["get_neighbors", "--files", ...filePaths, "--depth", depth, "--json"], ws ); if (!result) return null; @@ -364,11 +365,11 @@ async function getBlastRadius(filePaths, options = {}) { * Uses Graphify's BFS-based query to find related nodes. * * @param {string[]} filePaths — seed file paths - * @param {number} [maxFiles=20] — maximum files to return + * @param {number} [maxFiles] — maximum files to return (default from config.codeGraph.maxFiles) * @param {CodeGraphOptions} [options] * @returns {Promise} */ -async function getRelevantContext(filePaths, maxFiles = 20, options = {}) { +async function getRelevantContext(filePaths, maxFiles = config.codeGraph?.maxFiles ?? 5, options = {}) { if (!Array.isArray(filePaths) || filePaths.length === 0) return null; const ws = resolveWorkspace({ ...options, filePaths }); @@ -419,8 +420,9 @@ async function getComplexitySignals(filePaths, options = {}) { if (cached) return cached; // Run parallel queries: neighbors (blast radius) + god_nodes + graph_stats + const signalDepth = String(config.codeGraph?.maxDepth ?? 2); const [neighborsResult, godNodesResult, statsResult] = await Promise.all([ - execGraph("query", ["get_neighbors", "--files", ...filePaths, "--depth", "2", "--json"], ws), + execGraph("query", ["get_neighbors", "--files", ...filePaths, "--depth", signalDepth, "--json"], ws), execGraph("query", ["god_nodes", "--json"], ws), execGraph("query", ["graph_stats", "--json"], ws), ]); diff --git a/test/cache-state.test.js b/test/cache-state.test.js new file mode 100644 index 0000000..29fe4f0 --- /dev/null +++ b/test/cache-state.test.js @@ -0,0 +1,189 @@ +const assert = require("assert"); +const { describe, it, beforeEach, after } = require("node:test"); +const fs = require("fs"); +const os = require("os"); +const path = require("path"); + +// Redirect the shared telemetry SQLite (which affinity-store piggybacks on) +// at a temp file so tests never write to .lynkr/telemetry.db. +const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), "lynkr-cache-state-")); +const telemetry = require("../src/routing/telemetry"); +telemetry._setDbPathForTests(path.join(tmpDir, "telemetry.db")); + +const affinity = require("../src/routing/session-affinity"); +const store = require("../src/routing/affinity-store"); +const { resolveCacheEconomics, getProviderCacheDefaults } = require("../src/routing/cache-economics"); + +after(() => { + try { + fs.rmSync(tmpDir, { recursive: true, force: true }); + } catch { /* best-effort */ } +}); + +describe("cache-economics: provider fallback table", () => { + it("Anthropic-hosted providers are explicit with 1.25x writes and 5-min TTL", () => { + for (const p of ["azure-anthropic", "bedrock", "databricks"]) { + const d = getProviderCacheDefaults(p); + assert.strictEqual(d.mechanism, "explicit", p); + assert.strictEqual(d.writeMult, 1.25, p); + assert.strictEqual(d.readMult, 0.1, p); + assert.strictEqual(d.ttlMs, 5 * 60 * 1000, p); + } + }); + + it("OpenAI-style providers cache automatically with free writes", () => { + for (const p of ["openai", "azure-openai"]) { + const d = getProviderCacheDefaults(p); + assert.strictEqual(d.mechanism, "automatic", p); + assert.strictEqual(d.writeMult, 0, p); + } + }); + + it("local providers have zero-dollar cache economics", () => { + for (const p of ["ollama", "llamacpp", "lmstudio"]) { + const d = getProviderCacheDefaults(p); + assert.strictEqual(d.mechanism, "local", p); + const econ = resolveCacheEconomics(p, "some-local-model"); + assert.strictEqual(econ.cacheReadPerM, 0, p); + assert.strictEqual(econ.cacheWritePerM, 0, p); + } + }); + + it("unknown provider gets the generic automatic default", () => { + const d = getProviderCacheDefaults("no-such-provider"); + assert.strictEqual(d.mechanism, "automatic"); + assert.strictEqual(d.writeMult, 0); + }); + + it("multipliers apply to the model's input price for unknown-cache models", () => { + // A name no registry source knows resolves to DEFAULT_COST (input=1.0) + // with unknown:true — multipliers then produce read=0.1, write=1.25. + const econ = resolveCacheEconomics("azure-anthropic", "zz-nonexistent-model-for-tests"); + assert.strictEqual(econ.mechanism, "explicit"); + assert.ok(Math.abs(econ.cacheReadPerM - econ.inputPerM * 0.1) < 1e-9); + assert.ok(Math.abs(econ.cacheWritePerM - econ.inputPerM * 1.25) < 1e-9); + assert.strictEqual(econ.unknownPricing, true); + }); +}); + +describe("affinity-store: cache_state persistence", () => { + beforeEach(() => { + affinity._clearAll(); + }); + + it("roundtrips cache state on an existing pin row", () => { + affinity.setPin("s1", { provider: "databricks", model: "m1", tier: "MEDIUM" }); + store.saveCacheState("s1", { + warmPrefixTokens: 1234, + provider: "databricks", + model: "m1", + lastRequestAt: Date.now(), + ttlMs: 300000, + }); + const state = store.loadCacheState("s1"); + assert.strictEqual(state.warmPrefixTokens, 1234); + assert.strictEqual(state.provider, "databricks"); + // Pin fields untouched + const pin = affinity.getPin("s1"); + assert.strictEqual(pin.tier, "MEDIUM"); + }); + + it("creates a minimal row when the session has no pin yet", () => { + store.saveCacheState("s2", { + warmPrefixTokens: 50, + provider: "openai", + model: "gpt-x", + lastRequestAt: Date.now(), + ttlMs: 600000, + }); + assert.strictEqual(store.loadCacheState("s2").warmPrefixTokens, 50); + }); + + it("pin upsert does not clobber cache_state", () => { + affinity.setPin("s3", { provider: "databricks", model: "m1" }); + store.saveCacheState("s3", { + warmPrefixTokens: 999, + provider: "databricks", + model: "m1", + lastRequestAt: Date.now(), + ttlMs: 300000, + }); + // Re-pin (e.g. compaction refresh) — cache state must survive. + affinity.setPin("s3", { provider: "databricks", model: "m1", tier: "COMPLEX" }); + assert.strictEqual(store.loadCacheState("s3").warmPrefixTokens, 999); + }); + + it("returns null for missing or unknown sessions", () => { + assert.strictEqual(store.loadCacheState("nope"), null); + assert.strictEqual(store.loadCacheState(null), null); + }); +}); + +describe("session-affinity: recordCacheUsage / getCacheState", () => { + beforeEach(() => { + affinity._clearAll(); + }); + + it("records warm prefix as read + creation tokens", () => { + affinity.recordCacheUsage("c1", { + provider: "azure-anthropic", + model: "claude-x", + cacheReadTokens: 90000, + cacheCreationTokens: 10000, + }); + const state = affinity.getCacheState("c1"); + assert.strictEqual(state.warmPrefixTokens, 100000); + assert.strictEqual(state.provider, "azure-anthropic"); + assert.strictEqual(state.ttlMs, 5 * 60 * 1000); + assert.strictEqual(state.cold, false); + }); + + it("stays absent when the provider reports no cache signal", () => { + affinity.recordCacheUsage("c2", { + provider: "ollama", + model: "llama3", + cacheReadTokens: 0, + cacheCreationTokens: 0, + }); + assert.strictEqual(affinity.getCacheState("c2"), null); + }); + + it("a zero-signal response does not wipe existing state", () => { + affinity.recordCacheUsage("c3", { + provider: "databricks", model: "m1", cacheReadTokens: 500, cacheCreationTokens: 0, + }); + affinity.recordCacheUsage("c3", { + provider: "databricks", model: "m1", cacheReadTokens: 0, cacheCreationTokens: 0, + }); + assert.strictEqual(affinity.getCacheState("c3").warmPrefixTokens, 500); + }); + + it("overwrites unconditionally on a model switch (caches are model-scoped)", () => { + affinity.recordCacheUsage("c4", { + provider: "databricks", model: "opus", cacheReadTokens: 100000, cacheCreationTokens: 0, + }); + affinity.recordCacheUsage("c4", { + provider: "databricks", model: "haiku", cacheReadTokens: 0, cacheCreationTokens: 2000, + }); + const state = affinity.getCacheState("c4"); + assert.strictEqual(state.model, "haiku"); + assert.strictEqual(state.warmPrefixTokens, 2000); + }); + + it("marks state cold once TTL has elapsed without a refresh", () => { + store.saveCacheState("c5", { + warmPrefixTokens: 1000, + provider: "databricks", + model: "m1", + lastRequestAt: Date.now() - 10 * 60 * 1000, // 10 min ago + ttlMs: 5 * 60 * 1000, + }); + assert.strictEqual(affinity.getCacheState("c5").cold, true); + }); + + it("handles missing sessionId/provider gracefully", () => { + affinity.recordCacheUsage(null, { provider: "databricks", cacheReadTokens: 5 }); + affinity.recordCacheUsage("c6", { cacheReadTokens: 5 }); + assert.strictEqual(affinity.getCacheState("c6"), null); + }); +}); diff --git a/test/cache-switch-cost.test.js b/test/cache-switch-cost.test.js new file mode 100644 index 0000000..b7e2720 --- /dev/null +++ b/test/cache-switch-cost.test.js @@ -0,0 +1,252 @@ +const assert = require("assert"); +const { describe, it, after } = require("node:test"); +const fs = require("fs"); +const os = require("os"); +const path = require("path"); + +// Isolate the shared telemetry SQLite before anything touches it. +const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), "lynkr-switch-cost-")); +const telemetry = require("../src/routing/telemetry"); +telemetry._setDbPathForTests(path.join(tmpDir, "telemetry.db")); + +const { evaluateSwitch } = require("../src/routing/cache-switch-cost"); + +after(() => { + try { + fs.rmSync(tmpDir, { recursive: true, force: true }); + } catch { /* best-effort */ } +}); + +// Deterministic economics resolver — Anthropic-style Aug 2026 prices +// (opus $5/$25, sonnet $3/$15, haiku $1/$5; explicit cache 0.1x read, +// 1.25x write, 5-min TTL). +const ECON = { + opus: { inputPerM: 5, outputPerM: 25, cacheReadPerM: 0.5, cacheWritePerM: 6.25, ttlMs: 300000, mechanism: "explicit" }, + sonnet: { inputPerM: 3, outputPerM: 15, cacheReadPerM: 0.3, cacheWritePerM: 3.75, ttlMs: 300000, mechanism: "explicit" }, + haiku: { inputPerM: 1, outputPerM: 5, cacheReadPerM: 0.1, cacheWritePerM: 1.25, ttlMs: 300000, mechanism: "explicit" }, + local: { inputPerM: 0, outputPerM: 0, cacheReadPerM: 0, cacheWritePerM: 0, ttlMs: 300000, mechanism: "local" }, +}; +const deps = { resolveCacheEconomics: (_provider, model) => ECON[model] }; + +function warmState(tokens, model = "opus") { + return { + warmPrefixTokens: tokens, + provider: "databricks", + model, + lastRequestAt: Date.now(), + ttlMs: 300000, + cold: false, + }; +} + +describe("cache-switch-cost: gate short-circuits", () => { + it("allows when there is no cache state to protect", () => { + const r = evaluateSwitch({ + cacheState: null, + current: { provider: "databricks", model: "opus" }, + target: { provider: "databricks", model: "haiku" }, + deps, + }); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "no_cache_state"); + }); + + it("allows when the prefix is already cold (Phase 2 TTL clock)", () => { + const state = { ...warmState(200000), cold: true }; + const r = evaluateSwitch({ + cacheState: state, + current: { provider: "databricks", model: "opus" }, + target: { provider: "databricks", model: "haiku" }, + deps, + }); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "cache_cold"); + }); + + it("treats state recorded for a different model as stale", () => { + const r = evaluateSwitch({ + cacheState: warmState(200000, "sonnet"), + current: { provider: "databricks", model: "opus" }, + target: { provider: "databricks", model: "haiku" }, + deps, + }); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "stale_cache_state"); + }); + + it("no-ops on same provider+model", () => { + const r = evaluateSwitch({ + cacheState: warmState(200000), + current: { provider: "databricks", model: "opus" }, + target: { provider: "databricks", model: "opus" }, + deps, + }); + assert.strictEqual(r.reason, "same_model"); + }); +}); + +describe("cache-switch-cost: break-even math (plan's worked example)", () => { + // 100k warm prefix on Opus, ~2k new input + 800 output per turn. + const cacheState = warmState(100000); + const current = { provider: "databricks", model: "opus" }; + + it("Opus→Haiku breaks even in ~2 turns and clears the default horizon", () => { + const r = evaluateSwitch({ + cacheState, + current, + target: { provider: "databricks", model: "haiku" }, + newTokensPerTurn: 2000, + outputTokensPerTurn: 800, + deps, + }); + // stay=(100k*0.5+2k*5+0.8k*25)/1M=$0.08; once=102k*1.25/1M=$0.1275; + // switch=(100k*0.1+2k*1+0.8k*5)/1M=$0.016; be=0.1275/0.064≈2.0 + assert.ok(r.breakEvenTurns > 1.5 && r.breakEvenTurns < 2.5, `be=${r.breakEvenTurns}`); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "break_even_cleared"); + }); + + it("Opus→Sonnet needs ~12 turns and is blocked at the conservative default", () => { + const r = evaluateSwitch({ + cacheState, + current, + target: { provider: "databricks", model: "sonnet" }, + newTokensPerTurn: 2000, + outputTokensPerTurn: 800, + deps, + }); + // once=102k*3.75/1M=$0.3825; savings=$0.08-$0.048=$0.032; be≈11.95 + assert.ok(r.breakEvenTurns > 10 && r.breakEvenTurns < 14, `be=${r.breakEvenTurns}`); + assert.strictEqual(r.switchAllowed, false); + assert.strictEqual(r.reason, "break_even_blocked"); + }); + + it("Opus→Sonnet clears once the session is expected to run long enough", () => { + const r = evaluateSwitch({ + cacheState, + current, + target: { provider: "databricks", model: "sonnet" }, + newTokensPerTurn: 2000, + outputTokensPerTurn: 800, + expectedRemainingTurns: 30, + deps, + }); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "break_even_cleared"); + }); + + it("switching to a pricier model is never profitable (escalations bypass the gate)", () => { + const r = evaluateSwitch({ + cacheState: warmState(100000, "haiku"), + current: { provider: "databricks", model: "haiku" }, + target: { provider: "databricks", model: "opus" }, + deps, + }); + assert.strictEqual(r.switchAllowed, false); + assert.strictEqual(r.reason, "never_profitable"); + assert.strictEqual(r.breakEvenTurns, Infinity); + }); + + it("exposes the dollar receipt fields for Phase 6", () => { + const r = evaluateSwitch({ + cacheState, + current, + target: { provider: "databricks", model: "haiku" }, + newTokensPerTurn: 2000, + outputTokensPerTurn: 800, + deps, + }); + assert.ok(Math.abs(r.stayPerTurnUsd - 0.08) < 1e-9); + assert.ok(Math.abs(r.switchOnceUsd - 0.1275) < 1e-9); + assert.ok(Math.abs(r.switchPerTurnUsd - 0.016) < 1e-9); + assert.ok(Math.abs(r.projectedStaySavingsUsd - 0.064) < 1e-9); + }); +}); + +describe("cache-switch-cost: local models (latency, not dollars)", () => { + it("holds a large warm local prefix", () => { + const r = evaluateSwitch({ + cacheState: { ...warmState(50000, "local"), provider: "ollama" }, + current: { provider: "ollama", model: "local" }, + target: { provider: "lmstudio", model: "local" }, + deps: { resolveCacheEconomics: () => ECON.local }, + }); + assert.strictEqual(r.switchAllowed, false); + assert.strictEqual(r.reason, "local_prefill_hold"); + }); + + it("lets a small local prefix go — prefill is cheap", () => { + const r = evaluateSwitch({ + cacheState: { ...warmState(4000, "local"), provider: "ollama" }, + current: { provider: "ollama", model: "local" }, + target: { provider: "lmstudio", model: "local" }, + deps: { resolveCacheEconomics: () => ECON.local }, + }); + assert.strictEqual(r.switchAllowed, true); + assert.strictEqual(r.reason, "local_prefix_small"); + }); +}); + +describe("telemetry: cache_decision receipt + getCacheEconomics (Phase 6)", () => { + it("persists cache_decision via record() and aggregates dollars saved", async () => { + telemetry.record({ + request_id: "cd-1", + provider: "databricks", + cache_decision: { + decision: "hold", + reason: "break_even_blocked", + warmPrefixTokens: 100000, + breakEvenTurns: 11.95, + projectedSwitchCostUsd: 0.3825, + projectedStaySavingsUsd: 0.032, + expectedRemainingTurns: 10, + }, + }); + telemetry.record({ + request_id: "cd-2", + provider: "databricks", + cache_decision: { + decision: "switch", + reason: "break_even_cleared", + warmPrefixTokens: 100000, + breakEvenTurns: 1.99, + projectedSwitchCostUsd: 0.1275, + projectedStaySavingsUsd: 0.064, + expectedRemainingTurns: 10, + }, + }); + // record() writes on setImmediate — let the queue drain. + await new Promise((resolve) => setImmediate(() => setImmediate(resolve))); + + const econ = telemetry.getCacheEconomics(); + assert.strictEqual(econ.decisions, 2); + assert.strictEqual(econ.holds, 1); + assert.strictEqual(econ.switches, 1); + // hold: 0.3825 - 0.032*10 = 0.0625; switch: 0.064*10 - 0.1275 = 0.5125 + assert.ok(Math.abs(econ.dollarsSavedByHolds - 0.0625) < 1e-9); + assert.ok(Math.abs(econ.dollarsSavedBySwitches - 0.5125) < 1e-9); + assert.ok(Math.abs(econ.totalDollarsSaved - 0.575) < 1e-9); + assert.strictEqual(econ.byReason.break_even_blocked, 1); + assert.strictEqual(econ.byReason.break_even_cleared, 1); + }); +}); + +describe("telemetry: getExpectedRemainingTurns", () => { + it("returns null on sparse data, conditional median with enough sessions", () => { + assert.strictEqual(telemetry.getExpectedRemainingTurns(0), null); + + const db = telemetry.getDb(); + const insert = db.prepare( + `INSERT INTO routing_telemetry (request_id, session_id, timestamp, provider, message_count) + VALUES (?, ?, ?, 'databricks', ?)` + ); + // 25 sessions whose conversations reach 11..35 messages. + for (let i = 0; i < 25; i++) { + insert.run(`r${i}`, `sess-${i}`, Date.now(), 11 + i); + } + + const remaining = telemetry.getExpectedRemainingTurns(10); + // Lengths 11..35 → remaining-past-10 = 1..25, median = 13. + assert.strictEqual(remaining, 13); + }); +}); diff --git a/test/context-compression.test.js b/test/context-compression.test.js new file mode 100644 index 0000000..2b163f7 --- /dev/null +++ b/test/context-compression.test.js @@ -0,0 +1,167 @@ +/** + * Tests for context compression integration with complexity scoring + */ + +const { compressMessages, shouldCompress } = require('../src/routing/context-compressor'); +const { analyzeComplexity } = require('../src/routing/complexity-analyzer'); + +describe('Context Compression', () => { + describe('shouldCompress', () => { + it('should compress long conversations (>10 messages)', () => { + const payload = { + messages: Array(15).fill({ role: 'user', content: 'test' }), + }; + expect(shouldCompress(payload)).toBe(true); + }); + + it('should compress conversations with many tool results (>3)', () => { + const payload = { + messages: [ + { role: 'user', content: 'test' }, + { role: 'assistant', content: [{ type: 'tool_result', content: 'output1' }] }, + { role: 'assistant', content: [{ type: 'tool_result', content: 'output2' }] }, + { role: 'assistant', content: [{ type: 'tool_result', content: 'output3' }] }, + { role: 'assistant', content: [{ type: 'tool_result', content: 'output4' }] }, + ], + }; + expect(shouldCompress(payload)).toBe(true); + }); + + it('should not compress short conversations', () => { + const payload = { + messages: [ + { role: 'user', content: 'hello' }, + { role: 'assistant', content: 'hi there' }, + ], + }; + expect(shouldCompress(payload)).toBe(false); + }); + }); + + describe('compressMessages', () => { + it('should offload tool_result content', () => { + const messages = [ + { + role: 'assistant', + content: [ + { type: 'tool_result', tool_use_id: '1', content: 'A'.repeat(1000) }, + ], + }, + ]; + + const { compressed, stats } = compressMessages(messages); + + expect(compressed[0].content[0].content).toBe('[offloaded]'); + expect(stats.toolResultsOffloaded).toBe(1); + expect(stats.reduction).toBeGreaterThan(50); // Should save >50% + }); + + it('should keep sliding window of recent messages', () => { + const messages = Array(20).fill(null).map((_, i) => ({ + role: 'user', + content: `message ${i}`, + })); + + const { compressed } = compressMessages(messages); + + // Should keep only last 5 messages (window size = 5) + expect(compressed.length).toBe(5); + expect(compressed[compressed.length - 1].content).toBe('message 19'); + }); + + it('should always keep system message', () => { + const messages = [ + { role: 'system', content: 'You are a helpful assistant' }, + ...Array(20).fill(null).map((_, i) => ({ + role: 'user', + content: `message ${i}`, + })), + ]; + + const { compressed } = compressMessages(messages); + + expect(compressed[0].role).toBe('system'); + expect(compressed[0].content).toBe('You are a helpful assistant'); + }); + }); + + describe('Integration with analyzeComplexity', () => { + it('should use compressed context for scoring when enabled', async () => { + const payload = { + messages: [ + { role: 'user', content: 'test' }, + { + role: 'assistant', + content: [ + { type: 'tool_result', tool_use_id: '1', content: 'A'.repeat(5000) }, + { type: 'tool_result', tool_use_id: '2', content: 'B'.repeat(5000) }, + { type: 'tool_result', tool_use_id: '3', content: 'C'.repeat(5000) }, + { type: 'tool_result', tool_use_id: '4', content: 'D'.repeat(5000) }, + ], + }, + { role: 'user', content: 'What did you find?' }, + ], + }; + + const result = await analyzeComplexity(payload, { compression: true, weighted: true }); + + expect(result.compression).toBeDefined(); + expect(result.compression.reduction).toBeGreaterThan(50); + expect(result.compression.toolResultsOffloaded).toBe(4); + }); + + it('should skip compression when disabled', async () => { + const payload = { + messages: Array(15).fill(null).map((_, i) => ({ + role: 'user', + content: `message ${i}`, + })), + }; + + const result = await analyzeComplexity(payload, { compression: false }); + + expect(result.compression).toBeNull(); + }); + + it('should route to cheaper tier with compression vs without', async () => { + const payload = { + messages: [ + { role: 'system', content: 'You are a helpful assistant' }, + ...Array(15).fill(null).map((_, i) => ({ + role: 'user', + content: `Simple question ${i}`, + })), + { + role: 'assistant', + content: [ + { type: 'tool_result', tool_use_id: '1', content: 'A'.repeat(10000) }, + ], + }, + { role: 'user', content: 'hello' }, // Simple request + ], + }; + + const withCompression = await analyzeComplexity(payload, { compression: true, weighted: true }); + const withoutCompression = await analyzeComplexity(payload, { compression: false, weighted: true }); + + // With compression, token count should be lower + expect(withCompression.meta.tokens).toBeLessThan( + withoutCompression.meta.tokens + ); + + // Lower token count may lead to cheaper routing + console.log('With compression:', { + score: withCompression.score, + tokens: withCompression.meta.tokens, + recommendation: withCompression.recommendation, + reduction: withCompression.compression?.reduction, + }); + + console.log('Without compression:', { + score: withoutCompression.score, + tokens: withoutCompression.meta.tokens, + recommendation: withoutCompression.recommendation, + }); + }); + }); +}); diff --git a/test/difficulty-classifier.test.js b/test/difficulty-classifier.test.js index df1a161..d8ef635 100644 --- a/test/difficulty-classifier.test.js +++ b/test/difficulty-classifier.test.js @@ -13,6 +13,8 @@ const { _parseResult, _cacheKey, _buildPrompt, + _buildTierContext, + _setTierContextForTests, _clearCacheForTests, _getCacheStats, VALID_TIERS, @@ -77,9 +79,11 @@ describe("difficulty-classifier — cache key stability", () => { describe("difficulty-classifier — skip conditions", () => { beforeEach(() => _clearCacheForTests()); - it("returns null for text shorter than 15 chars", async () => { - const r = await classifyDifficulty("hi"); - assert.strictEqual(r, null); + it("returns null only for empty/whitespace text (min-length gate removed)", async () => { + // Short prompts are now classified (with conversation context when + // available) — only genuinely empty input skips. + assert.strictEqual(await classifyDifficulty(""), null); + assert.strictEqual(await classifyDifficulty(" "), null); }); it("returns null when caller signals a force pattern matched", async () => { @@ -186,4 +190,59 @@ describe("intent-score — _reconcile band cap (Phase A)", () => { assert.strictEqual(r.reconciled, false); assert.strictEqual(r.score, 35); }); -}); \ No newline at end of file +}); +describe("difficulty-classifier — tier deployment context", () => { + const TIERS = { + SIMPLE: "ollama:tiny-model-zz", + MEDIUM: "ollama:mid-model-zz", + COMPLEX: "azure-openai:big-model-zz", + REASONING: "moonshot:frontier-model-zz", + }; + + it("builds one line per configured tier with provider locality", () => { + const ctx = _buildTierContext(TIERS); + assert.ok(ctx.block.includes("- SIMPLE routes to tiny-model-zz (local")); + assert.ok(ctx.block.includes("- MEDIUM routes to mid-model-zz (local")); + assert.ok(ctx.block.includes("- COMPLEX routes to big-model-zz (cloud")); + assert.ok(ctx.block.includes("- REASONING routes to frontier-model-zz (cloud")); + // Rubric primacy must be stated in the block itself. + assert.ok(ctx.block.includes("tie-breaker")); + assert.strictEqual(typeof ctx.fingerprint, "string"); + assert.strictEqual(ctx.fingerprint.length, 16); + }); + + it("returns null when no tiers are configured", () => { + assert.strictEqual(_buildTierContext({}), null); + assert.strictEqual(_buildTierContext(null), null); + assert.strictEqual(_buildTierContext({ SIMPLE: "not-a-spec" }), null); + }); + + it("fingerprint changes when the deployment changes", () => { + const a = _buildTierContext(TIERS); + const b = _buildTierContext({ ...TIERS, COMPLEX: "ollama:other-model-zz" }); + assert.notStrictEqual(a.fingerprint, b.fingerprint); + }); + + it("_buildPrompt embeds the block between the rubric and the prompt", () => { + _setTierContextForTests(TIERS); + try { + const p = _buildPrompt("refactor the entire ingestion pipeline", null); + const depIdx = p.indexOf("Deployment (what each tier currently routes to"); + assert.ok(depIdx > 0, "deployment block missing"); + assert.ok(depIdx > p.indexOf("Reply format"), "block must come after the rubric"); + assert.ok(depIdx < p.indexOf("User prompt:"), "block must come before the prompt"); + } finally { + _setTierContextForTests(undefined); // reset memo for other tests + } + }); + + it("_buildPrompt omits the block when no deployment is known", () => { + _setTierContextForTests(null); + try { + const p = _buildPrompt("refactor the entire ingestion pipeline", null); + assert.ok(!p.includes("Deployment (")); + } finally { + _setTierContextForTests(undefined); + } + }); +}); diff --git a/test/dispatch-registry.test.js b/test/dispatch-registry.test.js index 847fe73..d488128 100644 --- a/test/dispatch-registry.test.js +++ b/test/dispatch-registry.test.js @@ -91,3 +91,38 @@ test('buildDecision caller fields override defaults, extras pass through', () => assert.strictEqual(d.model, 'ornith'); assert.deepStrictEqual(d._queryEmbedding, [0.1, 0.2]); }); + +test('buildDecision derives analysis.requestType (WS2.3 telemetry repair)', () => { + const { buildDecision } = require('../src/routing'); + + // The telemetry record sites read analysis.requestType; before this fix + // nothing set it and every routing_telemetry row recorded request_type + // NULL, starving the evidence-based deescalator. Derivation must match + // the deescalator's: breakdown.taskType.reason ?? taskType. + const fromBreakdown = buildDecision({ + provider: 'ollama', + method: 'tier_config', + analysis: { score: 63, breakdown: { taskType: { reason: 'code_generation', score: 40 } } }, + }); + assert.strictEqual(fromBreakdown.analysis.requestType, 'code_generation'); + + const fromTaskType = buildDecision({ + provider: 'ollama', + method: 'tier_config', + analysis: { score: 20, taskType: 'conversational' }, + }); + assert.strictEqual(fromTaskType.analysis.requestType, 'conversational'); + + // Pre-set requestType is preserved, not overwritten. + const preset = buildDecision({ + provider: 'ollama', + method: 'tier_config', + analysis: { requestType: 'already_set', breakdown: { taskType: { reason: 'other' } } }, + }); + assert.strictEqual(preset.analysis.requestType, 'already_set'); + + // No analysis / no task-type signal → no crash, no fabricated value. + assert.strictEqual(buildDecision({ provider: 'ollama', method: 'x' }).analysis, null); + const bare = buildDecision({ provider: 'ollama', method: 'x', analysis: { score: 5 } }); + assert.ok(!('requestType' in bare.analysis) || bare.analysis.requestType == null); +}); diff --git a/test/memory/distiller-freeze.test.js b/test/memory/distiller-freeze.test.js new file mode 100644 index 0000000..0f76fa8 --- /dev/null +++ b/test/memory/distiller-freeze.test.js @@ -0,0 +1,151 @@ +const assert = require("assert"); +const { describe, it, beforeEach, afterEach } = require("node:test"); +const fs = require("fs"); +const path = require("path"); + +const MODULES = [ + "../../src/config", + "../../src/db", + "../../src/memory/store", + "../../src/memory/extractor", + "../../src/memory/wiki", + "../../src/memory/skills-cache", + "../../src/memory/distiller", +]; + +function clearModules() { + for (const mod of MODULES) { + try { + delete require.cache[require.resolve(mod)]; + } catch { /* not loaded */ } + } +} + +function turn(userText, assistantText) { + return [ + { role: "user", content: userText }, + { role: "assistant", content: assistantText }, + ]; +} + +function conversation(turns) { + const messages = []; + for (let i = 0; i < turns; i++) { + messages.push(...turn(`Question number ${i}: how do I do task ${i}?`, `Answer ${i}: here is how.`)); + } + return messages; +} + +describe("Distiller freeze (Phase 5 — cache-aware routing)", () => { + let distiller; + let store; + let testDbPath; + + beforeEach(() => { + const timestamp = Date.now(); + const random = Math.floor(Math.random() * 1000000); + testDbPath = path.join(__dirname, `../../data/test-distiller-freeze-${timestamp}-${random}.db`); + process.env.SESSION_DB_PATH = testDbPath; + + clearModules(); + require("../../src/db"); + store = require("../../src/memory/store"); + distiller = require("../../src/memory/distiller"); + }); + + afterEach(() => { + try { + const db = require("../../src/db"); + if (typeof db.close === "function") db.close(); + } catch { /* already closed */ } + clearModules(); + for (const suffix of ["", "-wal", "-shm"]) { + try { + fs.unlinkSync(`${testDbPath}${suffix}`); + } catch { /* missing */ } + } + delete process.env.SESSION_DB_PATH; + }); + + it("serves the frozen block verbatim on the next turn", () => { + const sessionId = "freeze-1"; + const first = distiller.distillMessages(conversation(12), { sessionId }); + assert.strictEqual(first.applied, true); + assert.strictEqual(first.stats.fromFrozenCache, undefined); + const frozenContent = first.messages[0].content; + + // One more turn arrives — the block must be byte-identical. + const next = distiller.distillMessages(conversation(13), { sessionId }); + assert.strictEqual(next.applied, true); + assert.strictEqual(next.stats.fromFrozenCache, true); + assert.strictEqual(next.messages[0].content, frozenContent); + + // The newest turn is still present verbatim after the frozen block. + const lastMsg = next.messages[next.messages.length - 1]; + assert.ok(String(lastMsg.content).includes("Answer 12")); + }); + + it("stays byte-identical even when persona memories change between requests", () => { + const sessionId = "freeze-persona"; + const first = distiller.distillMessages(conversation(12), { sessionId }); + const frozenContent = first.messages[0].content; + + // A new preference memory lands mid-window. Without the freeze this + // would rewrite the distilled block (and bust the provider cache). + store.createMemory({ + content: "User prefers TypeScript with strict mode", + type: "preference", + category: "user", + importance: 0.9, + sessionId: null, + }); + + const next = distiller.distillMessages(conversation(13), { sessionId }); + assert.strictEqual(next.stats.fromFrozenCache, true); + assert.strictEqual(next.messages[0].content, frozenContent); + }); + + it("re-distills after K more user turns (default 5)", () => { + const sessionId = "freeze-refresh"; + const first = distiller.distillMessages(conversation(12), { sessionId }); + const frozenContent = first.messages[0].content; + + // 12 → 17 turns: refresh window elapsed, block must be rebuilt. + const refreshed = distiller.distillMessages(conversation(17), { sessionId }); + assert.strictEqual(refreshed.applied, true); + assert.strictEqual(refreshed.stats.fromFrozenCache, undefined); + assert.notStrictEqual(refreshed.messages[0].content, frozenContent); + + // And the new block freezes in turn. + const after = distiller.distillMessages(conversation(18), { sessionId }); + assert.strictEqual(after.stats.fromFrozenCache, true); + assert.strictEqual(after.messages[0].content, refreshed.messages[0].content); + }); + + it("invalidates the frozen block when the covered history is rewritten", () => { + const sessionId = "freeze-rewrite"; + distiller.distillMessages(conversation(12), { sessionId }); + + // Client-side rewrite (compaction/edit): first message changes. + const rewritten = conversation(13); + rewritten[0] = { role: "user", content: "TOTALLY DIFFERENT OPENER" }; + const result = distiller.distillMessages(rewritten, { sessionId }); + assert.strictEqual(result.applied, true); + assert.strictEqual(result.stats.fromFrozenCache, undefined); + }); + + it("does not freeze without a sessionId", () => { + const first = distiller.distillMessages(conversation(12), {}); + assert.strictEqual(first.applied, true); + const second = distiller.distillMessages(conversation(13), {}); + assert.strictEqual(second.stats.fromFrozenCache, undefined); + }); + + it("_clearFrozen drops cached blocks (test isolation)", () => { + const sessionId = "freeze-clear"; + distiller.distillMessages(conversation(12), { sessionId }); + distiller._clearFrozen(); + const result = distiller.distillMessages(conversation(13), { sessionId }); + assert.strictEqual(result.stats.fromFrozenCache, undefined); + }); +}); diff --git a/test/memory/distiller.test.js b/test/memory/distiller.test.js new file mode 100644 index 0000000..fb2d9ee --- /dev/null +++ b/test/memory/distiller.test.js @@ -0,0 +1,249 @@ +const assert = require("assert"); +const { describe, it, beforeEach, afterEach } = require("node:test"); +const fs = require("fs"); +const path = require("path"); + +const MODULES = [ + "../../src/config", + "../../src/db", + "../../src/memory/store", + "../../src/memory/extractor", + "../../src/memory/wiki", + "../../src/memory/skills-cache", + "../../src/memory/distiller", +]; + +function clearModules() { + for (const mod of MODULES) { + try { + delete require.cache[require.resolve(mod)]; + } catch { /* not loaded */ } + } +} + +/** Build a user+assistant exchange (one turn). */ +function turn(userText, assistantText) { + return [ + { role: "user", content: userText }, + { role: "assistant", content: assistantText }, + ]; +} + +/** Build a conversation with N user turns. */ +function conversation(turns) { + const messages = []; + for (let i = 0; i < turns; i++) { + messages.push(...turn(`Question number ${i}: how do I do task ${i}?`, `Answer ${i}: here is how.`)); + } + return messages; +} + +describe("Conversation Distiller", () => { + let distiller; + let store; + let testDbPath; + + beforeEach(() => { + const timestamp = Date.now(); + const random = Math.floor(Math.random() * 1000000); + testDbPath = path.join(__dirname, `../../data/test-distiller-${timestamp}-${random}.db`); + process.env.SESSION_DB_PATH = testDbPath; + + clearModules(); + require("../../src/db"); + store = require("../../src/memory/store"); + distiller = require("../../src/memory/distiller"); + }); + + afterEach(() => { + try { + const db = require("../../src/db"); + if (db && typeof db.close === "function") db.close(); + } catch { /* already closed */ } + + clearModules(); + + try { + for (const file of [testDbPath, `${testDbPath}-wal`, `${testDbPath}-shm`, `${testDbPath}-journal`]) { + if (fs.existsSync(file)) fs.unlinkSync(file); + } + } catch { /* ignore cleanup errors */ } + }); + + describe("needsDistillation()", () => { + it("returns false below the turn threshold", () => { + assert.strictEqual(distiller.needsDistillation(conversation(9)), false); + }); + + it("returns true at the turn threshold (10 turns)", () => { + assert.strictEqual(distiller.needsDistillation(conversation(10)), true); + }); + + it("returns false for empty input", () => { + assert.strictEqual(distiller.needsDistillation([]), false); + assert.strictEqual(distiller.needsDistillation(null), false); + }); + + it("fires the size-based rescue trigger before the turn threshold", () => { + // 5 turns (below the 10-turn threshold) but with a huge pasted block + // — the scenario that overflows a 4k-context local model + const bigBlock = "config line: value = setting\n".repeat(500); // ~14.5k chars ≈ 3.6k tokens + const messages = [ + ...turn("Review this config:\n" + bigBlock, "Looks mostly fine."), + ...turn("What about the timeout?", "Timeout is 65s."), + ...turn("And the gzip settings?", "Enabled globally."), + ...turn("Should I change keepalive?", "No, 65 is standard."), + ...turn("What about worker processes?", "Set to auto."), + ]; + assert.strictEqual(distiller.needsDistillation(messages), true); + + const result = distiller.distillMessages(messages); + assert.strictEqual(result.applied, true); + // Last 3 turns verbatim + distilled block + assert.strictEqual(result.messages.length, 7); + assert.ok(result.messages[0].content.startsWith("[Distilled context")); + }); + + it("does not fire the size trigger when turns fit in the keep window", () => { + const bigBlock = "config line: value = setting\n".repeat(500); + // Only 3 turns — everything is in the keep window, nothing to distill + const messages = [ + ...turn("Review this:\n" + bigBlock, "OK."), + ...turn("Question two?", "Answer two."), + ...turn("Question three?", "Answer three."), + ]; + assert.strictEqual(distiller.needsDistillation(messages), false); + }); + + it("does not fire the size trigger on small conversations", () => { + assert.strictEqual(distiller.needsDistillation(conversation(5)), false); + }); + + it("does not count tool_result-only user messages as turns", () => { + const messages = []; + for (let i = 0; i < 6; i++) { + messages.push({ role: "user", content: `Question ${i}` }); + messages.push({ + role: "assistant", + content: [{ type: "tool_use", id: `t${i}`, name: "Bash", input: {} }], + }); + messages.push({ + role: "user", + content: [{ type: "tool_result", tool_use_id: `t${i}`, content: "output" }], + }); + messages.push({ role: "assistant", content: `Answer ${i}` }); + } + // 6 real turns but 12 user-role messages — must not distill + assert.strictEqual(distiller.needsDistillation(messages), false); + }); + }); + + describe("distillMessages()", () => { + it("returns unchanged messages when below threshold", () => { + const messages = conversation(5); + const result = distiller.distillMessages(messages); + assert.strictEqual(result.applied, false); + assert.strictEqual(result.messages, messages); + }); + + it("keeps the last 3 user turns verbatim and prepends one distilled block", () => { + const messages = conversation(12); + const result = distiller.distillMessages(messages); + + assert.strictEqual(result.applied, true); + // 3 turns × 2 messages + 1 distilled block + assert.strictEqual(result.messages.length, 7); + assert.ok(result.messages[0].content.startsWith("[Distilled context")); + // Recent turns preserved exactly + assert.strictEqual(result.messages[1].content, "Question number 9: how do I do task 9?"); + assert.strictEqual(result.messages[6].content, "Answer 11: here is how."); + }); + + it("summarizes dropped requests in the distilled block", () => { + const result = distiller.distillMessages(conversation(12)); + assert.ok(result.messages[0].content.includes("Requests")); + assert.ok(result.messages[0].content.includes("Question number")); + }); + + it("never splits a tool_use / tool_result pair", () => { + const messages = []; + for (let i = 0; i < 12; i++) { + messages.push({ role: "user", content: `Question ${i}` }); + messages.push({ + role: "assistant", + content: [ + { type: "text", text: `Working on ${i}` }, + { type: "tool_use", id: `t${i}`, name: "Read", input: { file: "a.js" } }, + ], + }); + messages.push({ + role: "user", + content: [{ type: "tool_result", tool_use_id: `t${i}`, content: `result ${i}` }], + }); + messages.push({ role: "assistant", content: `Done with ${i}` }); + } + + const result = distiller.distillMessages(messages); + assert.strictEqual(result.applied, true); + + // Every tool_result in the kept window must have its tool_use present + const kept = result.messages; + const toolUseIds = new Set(); + for (const msg of kept) { + if (!Array.isArray(msg.content)) continue; + for (const block of msg.content) { + if (block.type === "tool_use") toolUseIds.add(block.id); + if (block.type === "tool_result") { + assert.ok( + toolUseIds.has(block.tool_use_id), + `tool_result ${block.tool_use_id} orphaned from its tool_use` + ); + } + } + } + }); + + it("reports meaningful savings on long conversations", () => { + const result = distiller.distillMessages(conversation(20)); + assert.strictEqual(result.applied, true); + assert.ok(parseFloat(result.stats.savingsPct) > 50, `expected >50% savings, got ${result.stats.savingsPct}%`); + }); + + it("includes stored user preferences as persona (L3)", () => { + store.createMemory({ + content: "User prefers TypeScript with strict mode", + type: "preference", + category: "user", + importance: 0.9, + sessionId: null, + }); + + const result = distiller.distillMessages(conversation(12), { sessionId: null }); + assert.ok(result.messages[0].content.includes("User profile:")); + assert.ok(result.messages[0].content.includes("TypeScript")); + }); + + it("counts tool usage in the scenario (L2)", () => { + const messages = []; + for (let i = 0; i < 12; i++) { + messages.push({ role: "user", content: `Question ${i}` }); + messages.push({ + role: "assistant", + content: [ + { type: "text", text: `Answer ${i}` }, + { type: "tool_use", id: `t${i}`, name: "Bash", input: {} }, + ], + }); + messages.push({ + role: "user", + content: [{ type: "tool_result", tool_use_id: `t${i}`, content: "ok" }], + }); + messages.push({ role: "assistant", content: `Done ${i}` }); + } + + const result = distiller.distillMessages(messages); + assert.ok(result.messages[0].content.includes("Tools used:")); + assert.ok(result.messages[0].content.includes("Bash")); + }); + }); +}); diff --git a/test/memory/skills-cache.test.js b/test/memory/skills-cache.test.js new file mode 100644 index 0000000..aba2d42 --- /dev/null +++ b/test/memory/skills-cache.test.js @@ -0,0 +1,168 @@ +const assert = require("assert"); +const { describe, it, beforeEach, afterEach } = require("node:test"); +const fs = require("fs"); +const path = require("path"); + +const MODULES = [ + "../../src/config", + "../../src/db", + "../../src/memory/store", + "../../src/memory/skills-cache", +]; + +function clearModules() { + for (const mod of MODULES) { + try { + delete require.cache[require.resolve(mod)]; + } catch { /* not loaded */ } + } +} + +/** Grep-like output — same shape regardless of the matched values. */ +function grepOutput(query) { + const lines = []; + for (let i = 0; i < 40; i++) { + lines.push(`src/file${i}.js:${i * 10}: const ${query}_${i} = require("./dep${i}");`); + } + return lines.join("\n"); +} + +describe("Skills Cache", () => { + let skills; + let store; + let testDbPath; + + beforeEach(() => { + const timestamp = Date.now(); + const random = Math.floor(Math.random() * 1000000); + testDbPath = path.join(__dirname, `../../data/test-skills-${timestamp}-${random}.db`); + process.env.SESSION_DB_PATH = testDbPath; + + clearModules(); + require("../../src/db"); + store = require("../../src/memory/store"); + skills = require("../../src/memory/skills-cache"); + skills.resetCache(); + }); + + afterEach(() => { + try { + const db = require("../../src/db"); + if (db && typeof db.close === "function") db.close(); + } catch { /* already closed */ } + + clearModules(); + + try { + for (const file of [testDbPath, `${testDbPath}-wal`, `${testDbPath}-shm`, `${testDbPath}-journal`]) { + if (fs.existsSync(file)) fs.unlinkSync(file); + } + } catch { /* ignore cleanup errors */ } + }); + + describe("shapeSignature()", () => { + it("is stable for identical text", () => { + const text = grepOutput("auth"); + assert.strictEqual(skills.shapeSignature(text), skills.shapeSignature(text)); + }); + + it("matches structurally identical outputs with different values", () => { + assert.strictEqual( + skills.shapeSignature(grepOutput("auth")), + skills.shapeSignature(grepOutput("database")) + ); + }); + + it("differs for structurally different outputs", () => { + const json = JSON.stringify({ results: [1, 2, 3], status: "ok" }, null, 2); + assert.notStrictEqual(skills.shapeSignature(grepOutput("x")), skills.shapeSignature(json)); + }); + + it("handles empty input", () => { + assert.strictEqual(skills.shapeSignature(""), "empty"); + assert.strictEqual(skills.shapeSignature(null), "empty"); + }); + }); + + describe("record() / lookup()", () => { + it("returns null for unknown signatures", () => { + assert.strictEqual(skills.lookup("nonexistent"), null); + }); + + it("records outcomes and tracks a running average", () => { + const sig = skills.shapeSignature(grepOutput("x")); + skills.record(sig, "distill", 0.8); + skills.record(sig, "distill", 0.6); + + const skill = skills.lookup(sig); + assert.ok(skill); + assert.strictEqual(skill.hits, 2); + assert.ok(Math.abs(skill.avgSavings - 0.7) < 0.001); + }); + + it("persists validated skills (savings >= 0.6) to the store", async () => { + const sig = skills.shapeSignature(grepOutput("y")); + skills.record(sig, "distill", 0.85); + + // Persistence happens via setImmediate — let it flush + await new Promise(resolve => setImmediate(() => setImmediate(resolve))); + + const persisted = store.getMemoriesByType("skill", 10); + assert.strictEqual(persisted.length, 1); + const meta = typeof persisted[0].metadata === "string" + ? JSON.parse(persisted[0].metadata) + : persisted[0].metadata; + assert.strictEqual(meta.signature, sig); + assert.strictEqual(meta.method, "distill"); + }); + + it("does not persist skills below the minimum savings bar", async () => { + const sig = skills.shapeSignature(grepOutput("z")); + skills.record(sig, "distill", 0.3); + + await new Promise(resolve => setImmediate(() => setImmediate(resolve))); + + assert.strictEqual(store.getMemoriesByType("skill", 10).length, 0); + // Still tracked in memory for hint purposes + assert.ok(skills.lookup(sig)); + }); + }); + + describe("getCompressionHints()", () => { + it("returns neutral hints for unknown shapes", () => { + const hints = skills.getCompressionHints(grepOutput("new")); + assert.strictEqual(hints.skipDedup, false); + assert.strictEqual(hints.maxLengthFactor, 1); + assert.ok(hints.signature); + }); + + it("requires at least 2 observations before applying hints", () => { + const text = grepOutput("once"); + skills.record(skills.shapeSignature(text), "passthrough", 0.0); + + const hints = skills.getCompressionHints(text); + assert.strictEqual(hints.skipDedup, false); + }); + + it("skips dedup for shapes that repeatedly failed to compress", () => { + const text = grepOutput("futile"); + const sig = skills.shapeSignature(text); + skills.record(sig, "passthrough", 0.0); + skills.record(sig, "passthrough", 0.0); + + const hints = skills.getCompressionHints(text); + assert.strictEqual(hints.skipDedup, true); + }); + + it("tightens the budget for proven-compressible shapes", () => { + const text = grepOutput("compressible"); + const sig = skills.shapeSignature(text); + skills.record(sig, "distill", 0.8); + skills.record(sig, "distill", 0.75); + + const hints = skills.getCompressionHints(text); + assert.strictEqual(hints.maxLengthFactor, 0.8); + assert.strictEqual(hints.skipDedup, false); + }); + }); +}); diff --git a/test/memory/tencentdb-launcher.test.js b/test/memory/tencentdb-launcher.test.js new file mode 100644 index 0000000..6421e7b --- /dev/null +++ b/test/memory/tencentdb-launcher.test.js @@ -0,0 +1,135 @@ +const assert = require("assert"); +const { describe, it, beforeEach, afterEach } = require("node:test"); + +const MODULES = [ + "../../src/config", + "../../src/memory/tencentdb-launcher", +]; + +function clearModules() { + for (const mod of MODULES) { + try { + delete require.cache[require.resolve(mod)]; + } catch { /* not loaded */ } + } +} + +describe("TencentDB Memory Launcher", () => { + let savedEnv; + + beforeEach(() => { + savedEnv = { + TENCENTDB_MEMORY_ENABLED: process.env.TENCENTDB_MEMORY_ENABLED, + TENCENTDB_MEMORY_DOCKER_ENABLED: process.env.TENCENTDB_MEMORY_DOCKER_ENABLED, + TENCENTDB_MEMORY_LLM_BASE_URL: process.env.TENCENTDB_MEMORY_LLM_BASE_URL, + TENCENTDB_MEMORY_LLM_MODEL: process.env.TENCENTDB_MEMORY_LLM_MODEL, + }; + clearModules(); + }); + + afterEach(() => { + for (const [key, value] of Object.entries(savedEnv)) { + if (value === undefined) delete process.env[key]; + else process.env[key] = value; + } + clearModules(); + }); + + describe("config defaults", () => { + it("is disabled by default", () => { + delete process.env.TENCENTDB_MEMORY_ENABLED; + const config = require("../../src/config"); + assert.strictEqual(config.tencentdbMemory.enabled, false); + }); + + it("enables via TENCENTDB_MEMORY_ENABLED=true", () => { + process.env.TENCENTDB_MEMORY_ENABLED = "true"; + const config = require("../../src/config"); + assert.strictEqual(config.tencentdbMemory.enabled, true); + assert.strictEqual(config.tencentdbMemory.docker.enabled, true); + }); + + it("defaults the memory LLM to Lynkr's own endpoint", () => { + delete process.env.TENCENTDB_MEMORY_LLM_BASE_URL; + const config = require("../../src/config"); + const { llm } = config.tencentdbMemory; + assert.ok(llm.baseUrl.includes("host.docker.internal")); + assert.ok(llm.baseUrl.endsWith("/v1")); + assert.strictEqual(llm.model, "auto"); + assert.strictEqual(llm.protocol, "openai"); + }); + + it("uses the upstream container names so external scripts are detected", () => { + const config = require("../../src/config"); + assert.strictEqual(config.tencentdbMemory.docker.core.containerName, "tdai-memory-core"); + assert.strictEqual(config.tencentdbMemory.docker.hub.containerName, "tdai-memory-hub"); + assert.strictEqual(config.tencentdbMemory.docker.network, "tdai-memory-stack"); + }); + }); + + describe("ensureRunning() gating", () => { + it("skips when disabled", async () => { + delete process.env.TENCENTDB_MEMORY_ENABLED; + const launcher = require("../../src/memory/tencentdb-launcher"); + const result = await launcher.ensureRunning(); + assert.strictEqual(result.started, false); + assert.strictEqual(result.reason, "disabled"); + }); + + it("skips when docker management is disabled", async () => { + process.env.TENCENTDB_MEMORY_ENABLED = "true"; + process.env.TENCENTDB_MEMORY_DOCKER_ENABLED = "false"; + const launcher = require("../../src/memory/tencentdb-launcher"); + const result = await launcher.ensureRunning(); + assert.strictEqual(result.started, false); + assert.strictEqual(result.reason, "docker_disabled"); + }); + }); + + describe("buildCoreConfigYaml()", () => { + it("embeds the configured LLM settings", () => { + process.env.TENCENTDB_MEMORY_LLM_BASE_URL = "https://api.example.com/v1"; + process.env.TENCENTDB_MEMORY_LLM_MODEL = "test-model"; + const launcher = require("../../src/memory/tencentdb-launcher"); + const yaml = launcher.buildCoreConfigYaml(); + + assert.ok(yaml.includes('baseUrl: "https://api.example.com/v1"')); + assert.ok(yaml.includes('model: "test-model"')); + assert.ok(yaml.includes("deployMode: standalone")); + assert.ok(yaml.includes("storeBackend: sqlite")); + }); + + it("defaults promptMode to code", () => { + const launcher = require("../../src/memory/tencentdb-launcher"); + assert.ok(launcher.buildCoreConfigYaml().includes("promptMode: code")); + }); + }); + + describe("generateUserKey()", () => { + it("produces the sk-mem-<32 alphanumeric> format", () => { + const launcher = require("../../src/memory/tencentdb-launcher"); + const key = launcher.generateUserKey(); + assert.match(key, /^sk-mem-[A-Za-z0-9]{32}$/); + }); + + it("produces unique keys", () => { + const launcher = require("../../src/memory/tencentdb-launcher"); + assert.notStrictEqual(launcher.generateUserKey(), launcher.generateUserKey()); + }); + }); + + describe("maskKey()", () => { + it("masks the middle of a key", () => { + const launcher = require("../../src/memory/tencentdb-launcher"); + const masked = launcher.maskKey("sk-mem-abcdefghijklmnopqrstuvwxyz123456"); + assert.strictEqual(masked, "sk-mem-abcd****3456"); + assert.ok(!masked.includes("efghijkl")); + }); + + it("fully masks short or missing keys", () => { + const launcher = require("../../src/memory/tencentdb-launcher"); + assert.strictEqual(launcher.maskKey("short"), "****"); + assert.strictEqual(launcher.maskKey(null), "****"); + }); + }); +}); diff --git a/test/memory/wiki.test.js b/test/memory/wiki.test.js new file mode 100644 index 0000000..2026c4a --- /dev/null +++ b/test/memory/wiki.test.js @@ -0,0 +1,137 @@ +const assert = require("assert"); +const { describe, it, beforeEach, afterEach } = require("node:test"); +const fs = require("fs"); +const path = require("path"); + +const MODULES = [ + "../../src/config", + "../../src/db", + "../../src/memory/store", + "../../src/memory/wiki", +]; + +function clearModules() { + for (const mod of MODULES) { + try { + delete require.cache[require.resolve(mod)]; + } catch { /* not loaded */ } + } +} + +/** Generate a large distinctive text block (> 500 tokens ≈ 2000 chars). */ +function largeBlock(seed) { + const lines = []; + for (let i = 0; i < 60; i++) { + lines.push(`${seed} configuration line ${i}: value_${seed}_${i} = setting-${i}`); + } + return lines.join("\n"); +} + +describe("Wiki Registry", () => { + let wiki; + let testDbPath; + + beforeEach(() => { + const timestamp = Date.now(); + const random = Math.floor(Math.random() * 1000000); + testDbPath = path.join(__dirname, `../../data/test-wiki-${timestamp}-${random}.db`); + process.env.SESSION_DB_PATH = testDbPath; + + clearModules(); + require("../../src/db"); + wiki = require("../../src/memory/wiki"); + wiki.resetCache(); + }); + + afterEach(() => { + try { + const db = require("../../src/db"); + if (db && typeof db.close === "function") db.close(); + } catch { /* already closed */ } + + clearModules(); + + try { + for (const file of [testDbPath, `${testDbPath}-wal`, `${testDbPath}-shm`, `${testDbPath}-journal`]) { + if (fs.existsSync(file)) fs.unlinkSync(file); + } + } catch { /* ignore cleanup errors */ } + }); + + describe("registerOrDereference()", () => { + it("ignores blocks below the size threshold", () => { + assert.strictEqual(wiki.registerOrDereference("short text"), null); + assert.strictEqual(wiki.registerOrDereference(""), null); + assert.strictEqual(wiki.registerOrDereference(null), null); + }); + + it("registers a new large block and returns null the first time", () => { + assert.strictEqual(wiki.registerOrDereference(largeBlock("alpha")), null); + }); + + it("returns a compact reference for a near-identical repeat", () => { + const block = largeBlock("beta"); + wiki.registerOrDereference(block); + + // Same block with a couple of changed lines — still >= 85% similar + const repeat = block.replace("line 0:", "line 0 (edited):"); + const result = wiki.registerOrDereference(repeat); + + assert.ok(result, "expected a wiki reference for repeated content"); + assert.ok(result.ref.startsWith("[wiki:")); + assert.ok(result.saved > repeat.length * 0.9, "reference should be far smaller than original"); + }); + + it("does not match dissimilar blocks", () => { + wiki.registerOrDereference(largeBlock("gamma")); + assert.strictEqual(wiki.registerOrDereference(largeBlock("delta")), null); + }); + + it("persists entries across cache resets (cross-session dedup)", () => { + const block = largeBlock("epsilon"); + wiki.registerOrDereference(block); + + wiki.resetCache(); // simulate a new session reloading from SQLite + + const result = wiki.registerOrDereference(block); + assert.ok(result, "expected persisted entry to be found after reload"); + }); + }); + + describe("dereferenceMessages()", () => { + it("replaces repeated large blocks inside old history", () => { + const block = largeBlock("zeta"); + wiki.registerOrDereference(block); + + const messages = [ + { role: "user", content: "small message" }, + { role: "user", content: block }, + { + role: "user", + content: [{ type: "tool_result", tool_use_id: "t1", content: block }], + }, + ]; + + const { messages: processed, stats } = wiki.dereferenceMessages(messages); + + assert.strictEqual(stats.dereferenced, 2); + assert.ok(stats.charsSaved > 0); + assert.strictEqual(processed[0].content, "small message"); + assert.ok(processed[1].content.startsWith("[wiki:")); + assert.ok(processed[2].content[0].content.startsWith("[wiki:")); + }); + + it("registers unseen large blocks for future dedup", () => { + const messages = [{ role: "user", content: largeBlock("eta") }]; + const { stats } = wiki.dereferenceMessages(messages); + assert.strictEqual(stats.registered, 1); + assert.strictEqual(stats.dereferenced, 0); + }); + + it("handles empty input", () => { + const { messages, stats } = wiki.dereferenceMessages([]); + assert.deepStrictEqual(messages, []); + assert.strictEqual(stats.dereferenced, 0); + }); + }); +}); diff --git a/test/prompt-cache-injection.test.js b/test/prompt-cache-injection.test.js index b9b6b64..fa50068 100644 --- a/test/prompt-cache-injection.test.js +++ b/test/prompt-cache-injection.test.js @@ -65,7 +65,7 @@ describe('injectAnthropicCacheBreakpoints', () => { assert.deepEqual(body.system[1].cache_control, { type: 'ephemeral' }); }); - it('marks last 3 messages', () => { + it('marks only the newest message on a young conversation (stable hierarchy)', () => { const body = { messages: [ { role: 'user', content: 'msg1' }, @@ -76,20 +76,21 @@ describe('injectAnthropicCacheBreakpoints', () => { ], }; const count = injectAnthropicCacheBreakpoints(body); - assert.equal(count, 3); // last 3 messages (no system = 3 breakpoints max) + // 3 user turns < K(5): no frozen boundary yet — rolling marker only. + assert.equal(count, 1); - // First 2 messages: no cache_control - assert.equal(body.messages[0].content, 'msg1'); // unchanged string + // Earlier messages untouched (they'd churn the cache if marked+moved). + assert.equal(body.messages[0].content, 'msg1'); assert.equal(body.messages[1].content, 'msg2'); + assert.equal(body.messages[2].content, 'msg3'); + assert.equal(body.messages[3].content, 'msg4'); - // Last 3 messages: converted to array with cache_control - assert.ok(Array.isArray(body.messages[2].content)); - assert.ok(Array.isArray(body.messages[3].content)); + // Newest message carries the rolling marker. assert.ok(Array.isArray(body.messages[4].content)); assert.deepEqual(body.messages[4].content[0].cache_control, { type: 'ephemeral' }); }); - it('marks system + last 3 messages = 4 total breakpoints', () => { + it('marks system + rolling = 2 on a young conversation', () => { const body = { system: 'System prompt', messages: [ @@ -101,7 +102,69 @@ describe('injectAnthropicCacheBreakpoints', () => { ], }; const count = injectAnthropicCacheBreakpoints(body); - assert.equal(count, 4); // 1 system + 3 messages = 4 (max) + assert.equal(count, 2); // system + rolling newest + }); + + it('marks the tools block as its own breakpoint', () => { + const body = { + system: 'System', + tools: [ + { name: 'read', input_schema: { type: 'object' } }, + { name: 'write', input_schema: { type: 'object' } }, + ], + messages: [{ role: 'user', content: 'hi' }], + }; + const count = injectAnthropicCacheBreakpoints(body); + assert.equal(count, 3); // tools + system + rolling + assert.equal(body.tools[0].cache_control, undefined); + assert.deepEqual(body.tools[1].cache_control, { type: 'ephemeral' }); + }); + + function conversationWithUserTurns(n) { + const messages = []; + for (let i = 1; i <= n; i++) { + messages.push({ role: 'user', content: `question ${i}` }); + messages.push({ role: 'assistant', content: `answer ${i}` }); + } + return messages; + } + + it('places a frozen boundary at the K-turn bucket once the conversation is old enough', () => { + const body = { messages: conversationWithUserTurns(12) }; // K default 5 → bucket 10 + const count = injectAnthropicCacheBreakpoints(body); + assert.equal(count, 2); // boundary + rolling + + // 10th user message is messages[18] (user turns at even indices). + assert.ok(Array.isArray(body.messages[18].content)); + assert.deepEqual(body.messages[18].content[0].cache_control, { type: 'ephemeral' }); + }); + + it('keeps the boundary byte-stable across turns inside a bucket', () => { + // Turn 12 and turn 14 of the same conversation must mark the SAME + // message — that stability is what compounds provider cache hits. + const at12 = { messages: conversationWithUserTurns(12) }; + const at14 = { messages: conversationWithUserTurns(14) }; + injectAnthropicCacheBreakpoints(at12); + injectAnthropicCacheBreakpoints(at14); + + const marked = (body) => body.messages + .map((m, i) => (Array.isArray(m.content) && m.content.some(b => b.cache_control) ? i : -1)) + .filter(i => i >= 0); + + const m12 = marked(at12); + const m14 = marked(at14); + // Boundary (first marked index) identical; rolling marker differs. + assert.equal(m12[0], 18); + assert.equal(m14[0], 18); + }); + + it('advances the boundary only when the bucket rolls over', () => { + const at15 = { messages: conversationWithUserTurns(15) }; + injectAnthropicCacheBreakpoints(at15); + // bucket = 15 → boundary at the 15th user message... which is also + // covered by the rolling marker region; boundary lands at index 28. + assert.ok(Array.isArray(at15.messages[28].content)); + assert.deepEqual(at15.messages[28].content[0].cache_control, { type: 'ephemeral' }); }); it('respects max 4 breakpoints', () => {