From cf4084e3d830905d482f30b297980849a1f48964 Mon Sep 17 00:00:00 2001 From: Balogun Feranmi Date: Thu, 1 Oct 2026 12:42:07 +0100 Subject: [PATCH 1/3] feat(kg): support local/OpenAI-compatible LLM endpoints (#598) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When KG_LLM_BASE_URL is set, the KG enrichment/skill LLM path talks to that endpoint (Ollama, LM Studio, ...) using KG_LLM_MODEL. KG_LLM_API_KEY is optional and only sent as Authorization when present; response_format is never sent on this path — the JSON instruction plus shared parsing apply, and unusable replies are logged (model + endpoint) and skipped as an empty result instead of crashing the flow. Hosted OpenAI / OpenRouter / Anthropic behaviour and Anthropic-only batch mode are unchanged. --- .env.example | 7 + docs/knowledge-graph.md | 10 ++ .../src/knowledge-graph/kg-llm.service.ts | 5 +- .../src/knowledge-graph/llm-client.spec.ts | 137 ++++++++++++++++++ .../backend/src/knowledge-graph/llm-client.ts | 75 ++++++++-- 5 files changed, 219 insertions(+), 15 deletions(-) create mode 100644 packages/backend/src/knowledge-graph/llm-client.spec.ts diff --git a/.env.example b/.env.example index f9473d49..865530ad 100644 --- a/.env.example +++ b/.env.example @@ -130,6 +130,13 @@ MCP_RATE_LIMIT_PER_MINUTE=60 # KG_LLM_MODEL=gpt-4o-mini # OPENAI_API_KEY= # or OPENROUTER_API_KEY / ANTHROPIC_API_KEY # +# Local / OpenAI-compatible endpoint (Ollama, LM Studio, ...). When set, the KG +# LLM path talks to this base URL instead of OpenAI/OpenRouter/Anthropic, still +# using KG_LLM_MODEL. No key needed — leave KG_LLM_API_KEY empty; it is only +# sent as an Authorization header when set. Validated against local qwen2.5. +# KG_LLM_BASE_URL=http://localhost:11434/v1 +# KG_LLM_API_KEY= +# # Scheduled AI extension (cloud cron): periodically extends the graph + skills # from captured user intents. Off unless ALL of: this flag is true, the cron # runs (POST /api/cron/kg-discovery with CRON_SECRET), and the workspace turned diff --git a/docs/knowledge-graph.md b/docs/knowledge-graph.md index f4843d7a..427258c6 100644 --- a/docs/knowledge-graph.md +++ b/docs/knowledge-graph.md @@ -179,6 +179,16 @@ KG_LLM_PROVIDER=openai # openai | openrouter | anthropic KG_LLM_MODEL=gpt-4o-mini # anthropic default: claude-haiku-4-5 OPENAI_API_KEY= # or OPENROUTER_API_KEY / ANTHROPIC_API_KEY +# Local / OpenAI-compatible endpoint (Ollama, LM Studio, ...). When set, the +# KG LLM path talks here instead of the hosted providers, still using +# KG_LLM_MODEL. No key needed: KG_LLM_API_KEY is optional and only sent as an +# Authorization header when set. No response_format is sent on this path, so +# the model answers from the JSON instruction; unparsable replies are logged +# (model + endpoint) and that pass is skipped. Validated against a local +# qwen2.5 (Qwen2.5-0.5B-Instruct via an OpenAI-compatible front). +KG_LLM_BASE_URL=http://localhost:11434/v1 +KG_LLM_API_KEY= + # Scheduled extension (cloud cron) — all must align: this flag, the cron call, # and the per-workspace "Scheduled AI extension" switch KG_LLM_CRON_ENABLED=false diff --git a/packages/backend/src/knowledge-graph/kg-llm.service.ts b/packages/backend/src/knowledge-graph/kg-llm.service.ts index 047ea461..d770657a 100644 --- a/packages/backend/src/knowledge-graph/kg-llm.service.ts +++ b/packages/backend/src/knowledge-graph/kg-llm.service.ts @@ -27,7 +27,8 @@ Rules: /** * Optional LLM-assisted KG enrichment. Opt-in twice: a global env flag - * (KG_LLM_ENABLED + an API key) AND a per-workspace switch (kg_llm_enabled, + * (KG_LLM_ENABLED + a resolvable LLM config — an API key, or KG_LLM_BASE_URL + * for local models which need no key) AND a per-workspace switch (kg_llm_enabled, * default off, because it costs money). PII-safe: only entity + field NAMES are * sent to the model — never values. Results are stored as suggested LLM edges * for a human to confirm. Cached by a content hash so unchanged graphs don't @@ -42,7 +43,7 @@ export class KgLlmService { private readonly kgStatic: KgStaticService, ) {} - /** Globally available (env flag + a configured API key). */ + /** Globally available (env flag + a resolvable LLM config). */ globallyAvailable(): boolean { return process.env.KG_LLM_ENABLED === 'true' && !!resolveLlmConfig(); } diff --git a/packages/backend/src/knowledge-graph/llm-client.spec.ts b/packages/backend/src/knowledge-graph/llm-client.spec.ts new file mode 100644 index 00000000..7e28033b --- /dev/null +++ b/packages/backend/src/knowledge-graph/llm-client.spec.ts @@ -0,0 +1,137 @@ +import { chatJson, resolveLlmConfig } from './llm-client'; + +const ENV_KEYS = [ + 'KG_LLM_BASE_URL', + 'KG_LLM_API_KEY', + 'KG_LLM_MODEL', + 'KG_LLM_PROVIDER', + 'OPENAI_API_KEY', + 'OPENROUTER_API_KEY', + 'ANTHROPIC_API_KEY', +]; + +function setEnv(vars: Record) { + for (const k of ENV_KEYS) delete process.env[k]; + for (const [k, v] of Object.entries(vars)) { + if (v !== undefined) process.env[k] = v; + } +} + +function okJson(body: any) { + return { + ok: true, + status: 200, + text: async () => JSON.stringify(body), + json: async () => body, + }; +} + +describe('resolveLlmConfig', () => { + afterEach(() => setEnv({})); + + it('returns null when nothing is configured', () => { + setEnv({}); + expect(resolveLlmConfig()).toBeNull(); + }); + + it('keeps the existing OpenAI behaviour without a base URL', () => { + setEnv({ OPENAI_API_KEY: 'sk-x', KG_LLM_MODEL: 'gpt-4o-mini' }); + expect(resolveLlmConfig()).toEqual({ provider: 'openai', model: 'gpt-4o-mini', apiKey: 'sk-x' }); + }); + + it('routes to the custom endpoint when KG_LLM_BASE_URL is set', () => { + setEnv({ KG_LLM_BASE_URL: 'http://localhost:11434/v1/', KG_LLM_MODEL: 'qwen2.5' }); + expect(resolveLlmConfig()).toEqual({ + provider: 'custom', + model: 'qwen2.5', + apiKey: '', + baseUrl: 'http://localhost:11434/v1', + }); + }); + + it('accepts an empty local key and keeps KG_LLM_MODEL', () => { + setEnv({ KG_LLM_BASE_URL: 'http://localhost:1234/v1', KG_LLM_MODEL: 'llama3.1', KG_LLM_API_KEY: '' }); + const cfg = resolveLlmConfig()!; + expect(cfg.provider).toBe('custom'); + expect(cfg.model).toBe('llama3.1'); + expect(cfg.apiKey).toBe(''); + }); +}); + +describe('chatJson', () => { + const realFetch = global.fetch; + + beforeEach(() => { + global.fetch = jest.fn(); + }); + afterEach(() => { + global.fetch = realFetch; + setEnv({}); + jest.restoreAllMocks(); + }); + + it('custom endpoint: no Authorization header with an empty key, no response_format', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: '{"relationships":[]}' } }], usage: {} }), + ); + const res = await chatJson( + { provider: 'custom', model: 'qwen2.5', apiKey: '', baseUrl: 'http://localhost:11434/v1' }, + 'sys', + 'user', + ); + expect(res.json).toEqual({ relationships: [] }); + expect(global.fetch).toHaveBeenCalledTimes(1); + const [url, init] = (global.fetch as jest.Mock).mock.calls[0]; + expect(url).toBe('http://localhost:11434/v1/chat/completions'); + expect(init.headers.Authorization).toBeUndefined(); + const body = JSON.parse(init.body); + expect(body.response_format).toBeUndefined(); + expect(body.messages[0].content).toContain('Respond with a single JSON object'); + }); + + it('custom endpoint: sends Authorization when a key is configured', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: '{}' } }], usage: {} }), + ); + await chatJson( + { provider: 'custom', model: 'm', apiKey: 'secret', baseUrl: 'http://host/v1' }, + 'sys', + 'user', + ); + const [, init] = (global.fetch as jest.Mock).mock.calls[0]; + expect(init.headers.Authorization).toBe('Bearer secret'); + expect(JSON.parse(init.body).response_format).toBeUndefined(); + }); + + it('custom endpoint: unusable JSON is skipped, not thrown', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: 'Sure, here are some thoughts...' } }], usage: {} }), + ); + const res = await chatJson( + { provider: 'custom', model: 'qwen2.5', apiKey: '', baseUrl: 'http://localhost:11434/v1' }, + 'sys', + 'user', + ); + expect(res).toEqual({ json: {} }); + }); + + it('openai path still sends response_format and Authorization', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: '{}' } }], usage: {} }), + ); + await chatJson({ provider: 'openai', model: 'gpt-4o-mini', apiKey: 'sk-x' }, 'sys', 'user'); + const [url, init] = (global.fetch as jest.Mock).mock.calls[0]; + expect(url).toBe('https://api.openai.com/v1/chat/completions'); + expect(init.headers.Authorization).toBe('Bearer sk-x'); + expect(JSON.parse(init.body).response_format).toEqual({ type: 'json_object' }); + }); + + it('hosted paths still throw on unusable JSON', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: 'not json' } }], usage: {} }), + ); + await expect( + chatJson({ provider: 'openai', model: 'gpt-4o-mini', apiKey: 'sk-x' }, 'sys', 'user'), + ).rejects.toThrow(); + }); +}); diff --git a/packages/backend/src/knowledge-graph/llm-client.ts b/packages/backend/src/knowledge-graph/llm-client.ts index 68e8411c..5e543c8d 100644 --- a/packages/backend/src/knowledge-graph/llm-client.ts +++ b/packages/backend/src/knowledge-graph/llm-client.ts @@ -1,17 +1,37 @@ /** * Minimal provider-agnostic LLM client for KG enrichment. - * Supports OpenAI, OpenRouter (OpenAI-compatible) and Anthropic. No SDK — just - * fetch — so it adds no dependencies. JSON-only responses. + * Supports OpenAI, OpenRouter (OpenAI-compatible), Anthropic, and a custom + * OpenAI-compatible base URL for local models (Ollama, LM Studio, ...). + * No SDK — just fetch — so it adds no dependencies. JSON-only responses. */ +import { Logger } from '@nestjs/common'; + +const logger = new Logger('KgLlmClient'); export interface LlmConfig { - provider: 'openai' | 'openrouter' | 'anthropic'; + provider: 'openai' | 'openrouter' | 'anthropic' | 'custom'; model: string; + /** Empty on the custom path: Ollama/LM Studio need no key. */ apiKey: string; + /** Set on the custom path from KG_LLM_BASE_URL (trailing slashes trimmed). */ + baseUrl?: string; } -/** Resolve provider/model/key from env, or null when no key is configured. */ +/** + * Resolve provider/model/key from env, or null when nothing is configured. + * KG_LLM_BASE_URL takes precedence and selects the custom OpenAI-compatible + * path: no API key is required, KG_LLM_API_KEY is optional. + */ export function resolveLlmConfig(): LlmConfig | null { + const customBase = (process.env.KG_LLM_BASE_URL || '').trim().replace(/\/+$/, ''); + if (customBase) { + return { + provider: 'custom', + model: process.env.KG_LLM_MODEL || 'qwen2.5', + apiKey: process.env.KG_LLM_API_KEY || '', + baseUrl: customBase, + }; + } const provider = (process.env.KG_LLM_PROVIDER || 'openai').toLowerCase() as LlmConfig['provider']; const apiKey = provider === 'openrouter' @@ -40,7 +60,10 @@ export interface LlmResult { */ export const KG_LLM_MAX_OUTPUT_TOKENS = 4000; -/** Call the model and parse its reply as JSON. Throws on transport/parse error. */ +/** Call the model and parse its reply as JSON. Throws on transport/parse error, + * except on the custom path where unusable JSON is logged (model + endpoint) + * and answered as an empty result so the enrichment/skill flow is skipped + * rather than crashed. */ export async function chatJson( cfg: LlmConfig, system: string, @@ -85,17 +108,35 @@ export async function chatJson( const base = cfg.provider === 'openrouter' ? 'https://openrouter.ai/api/v1' - : 'https://api.openai.com/v1'; + : cfg.provider === 'custom' + ? (cfg.baseUrl ?? '').replace(/\/+$/, '') + : 'https://api.openai.com/v1'; + // Local models often lack JSON mode: never send response_format there and + // rely on the prompt plus the shared parsing instead. The Authorization + // header stays unconditional on the hosted paths and is only omitted on + // the custom path when no key is configured (Ollama/LM Studio need none). + const headers: Record = + cfg.provider === 'custom' && !cfg.apiKey + ? { 'Content-Type': 'application/json' } + : { Authorization: `Bearer ${cfg.apiKey}`, 'Content-Type': 'application/json' }; const r = await fetch(`${base}/chat/completions`, { method: 'POST', - headers: { Authorization: `Bearer ${cfg.apiKey}`, 'Content-Type': 'application/json' }, + headers, body: JSON.stringify({ model: cfg.model, temperature: 0, max_tokens: maxTokens, - response_format: { type: 'json_object' }, + ...(cfg.provider === 'custom' + ? {} + : { response_format: { type: 'json_object' } }), messages: [ - { role: 'system', content: system }, + { + role: 'system', + content: + cfg.provider === 'custom' + ? system + '\nRespond with a single JSON object and nothing else.' + : system, + }, { role: 'user', content: user }, ], }), @@ -103,10 +144,18 @@ export async function chatJson( if (!r.ok) throw new Error(`LLM ${r.status}: ${(await r.text()).slice(0, 200)}`); const j = await r.json(); const text = j.choices?.[0]?.message?.content ?? '{}'; - return { - json: JSON.parse(stripFences(text)), - usage: { inputTokens: j.usage?.prompt_tokens, outputTokens: j.usage?.completion_tokens }, - }; + try { + return { + json: JSON.parse(stripFences(text)), + usage: { inputTokens: j.usage?.prompt_tokens, outputTokens: j.usage?.completion_tokens }, + }; + } catch { + if (cfg.provider !== 'custom') throw new Error(`LLM returned unusable JSON: ${String(text).slice(0, 200)}`); + logger.warn( + `KG LLM custom endpoint returned unusable JSON (model=${cfg.model} endpoint=${base}); skipping this pass.`, + ); + return { json: {} }; + } } export interface BatchRequest { From d8138af5a5b7972a698ce6522c7f7e243bb107bf Mon Sep 17 00:00:00 2001 From: Balogun Feranmi Date: Fri, 2 Oct 2026 15:41:46 +0100 Subject: [PATCH 2/3] fix(kg): explicit skip on unusable local replies; origin-only logging; docker/docs wiring (#598 review) Custom path now resolves { json: null, skipped: true } instead of an empty answer, and enrich()/generateForConnectors()/generateForServer() return early: no kg_llm_hash write, no pending-suggestion replacement. Endpoint logging uses new URL(base).origin only. Docs block comments out the localhost URL and covers Docker networking; docker-compose.yml passes the five KG_LLM_* vars through with empty defaults. --- docker-compose.yml | 11 ++ docs/knowledge-graph.md | 16 ++- .../knowledge-graph/kg-llm.service.spec.ts | 72 ++++++++++++ .../src/knowledge-graph/kg-llm.service.ts | 6 +- .../knowledge-graph/kg-skill.service.spec.ts | 104 ++++++++++++++++++ .../src/knowledge-graph/kg-skill.service.ts | 12 +- .../src/knowledge-graph/llm-client.spec.ts | 4 +- .../backend/src/knowledge-graph/llm-client.ts | 24 +++- 8 files changed, 235 insertions(+), 14 deletions(-) create mode 100644 packages/backend/src/knowledge-graph/kg-llm.service.spec.ts diff --git a/docker-compose.yml b/docker-compose.yml index 87d95abf..703532ad 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -81,6 +81,17 @@ services: # address. Needed for the bundled MOTIS (`motis`) and any other API you # run on the same Docker network. - SSRF_ALLOWED_HOSTS=${SSRF_ALLOWED_HOSTS:-} + # Knowledge Graph LLM (optional): hosted providers need their API key, + # a local OpenAI-compatible endpoint needs KG_LLM_BASE_URL and no key + # (KG_LLM_API_KEY is only sent when set). From inside this container + # `localhost` is the app itself — point at the host or a sibling + # service instead, e.g. http://host.docker.internal:11434/v1 + # (Linux: extra_hosts: ["host.docker.internal:host-gateway"]). + - KG_LLM_ENABLED=${KG_LLM_ENABLED:-} + - KG_LLM_PROVIDER=${KG_LLM_PROVIDER:-} + - KG_LLM_MODEL=${KG_LLM_MODEL:-} + - KG_LLM_BASE_URL=${KG_LLM_BASE_URL:-} + - KG_LLM_API_KEY=${KG_LLM_API_KEY:-} depends_on: postgres: diff --git a/docs/knowledge-graph.md b/docs/knowledge-graph.md index 427258c6..a3fffcbb 100644 --- a/docs/knowledge-graph.md +++ b/docs/knowledge-graph.md @@ -184,10 +184,20 @@ OPENAI_API_KEY= # or OPENROUTER_API_KEY / ANTHROPIC_API_KEY # KG_LLM_MODEL. No key needed: KG_LLM_API_KEY is optional and only sent as an # Authorization header when set. No response_format is sent on this path, so # the model answers from the JSON instruction; unparsable replies are logged -# (model + endpoint) and that pass is skipped. Validated against a local +# (model + endpoint origin) and that pass is skipped. Validated against a local # qwen2.5 (Qwen2.5-0.5B-Instruct via an OpenAI-compatible front). -KG_LLM_BASE_URL=http://localhost:11434/v1 -KG_LLM_API_KEY= +# KG_LLM_BASE_URL=http://localhost:11434/v1 +# KG_LLM_API_KEY= +``` + +Docker note: `localhost` inside the app container means the AnythingMCP +container itself, not your host. To reach an Ollama running on the host, use +`http://host.docker.internal:11434/v1` — on Linux that name needs an explicit +mapping (`extra_hosts: ["host.docker.internal:host-gateway"]`). When Ollama +runs as a service in the same Compose project/network, use its service name +instead (e.g. `http://ollama:11434/v1`). The five `KG_LLM_*` variables are +passed through to the app container in `docker-compose.yml` with empty +defaults, so setting them in `.env` is enough. # Scheduled extension (cloud cron) — all must align: this flag, the cron call, # and the per-workspace "Scheduled AI extension" switch diff --git a/packages/backend/src/knowledge-graph/kg-llm.service.spec.ts b/packages/backend/src/knowledge-graph/kg-llm.service.spec.ts new file mode 100644 index 00000000..c60f9c1f --- /dev/null +++ b/packages/backend/src/knowledge-graph/kg-llm.service.spec.ts @@ -0,0 +1,72 @@ +import { KgLlmService } from './kg-llm.service'; + +const ORG = 'org-A'; + +function setKgEnv() { + process.env.KG_LLM_ENABLED = 'true'; + process.env.KG_LLM_BASE_URL = 'http://localhost:11434/v1'; + process.env.KG_LLM_MODEL = 'qwen2.5'; + delete process.env.KG_LLM_API_KEY; +} + +function clearKgEnv() { + delete process.env.KG_LLM_ENABLED; + delete process.env.KG_LLM_BASE_URL; + delete process.env.KG_LLM_MODEL; +} + +function okJson(body: any) { + return { ok: true, status: 200, text: async () => JSON.stringify(body), json: async () => body }; +} + +function make(prisma: any) { + const kgStatic = { isEnabled: async () => true, getFlag: async () => true }; + return new KgLlmService(prisma, kgStatic as any); +} + +function nodes() { + return [0, 1].map((i) => ({ + id: `n${i}`, + entity: `ent${i}`, + fields: [{ name: 'email' }], + outputFields: [], + connector: { name: 'crm' }, + })); +} + +/** Unusable custom reply must not store the hash (test 1 for #816 review). */ +describe('KgLlmService.enrich custom skip', () => { + const realFetch = global.fetch; + + beforeEach(() => { + setKgEnv(); + global.fetch = jest.fn(); + }); + afterEach(() => { + global.fetch = realFetch; + clearKgEnv(); + jest.restoreAllMocks(); + }); + + it('does not update kg_llm_hash on an unusable reply and stays retryable', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: 'Sure, here are some thoughts...' } }] }), + ); + const prisma = { + kgNode: { findMany: jest.fn().mockResolvedValue(nodes()) }, + orgSettings: { findUnique: jest.fn().mockResolvedValue(null), upsert: jest.fn() }, + }; + const res = await make(prisma).enrich(ORG); + expect(res).toEqual({ suggested: 0, skipped: true, model: 'qwen2.5' }); + expect(prisma.orgSettings.upsert).not.toHaveBeenCalled(); + + // Retry with a usable reply stores the hash normally. + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: '{"relationships":[]}' } }] }), + ); + const retry = await make(prisma).enrich(ORG); + expect(retry.suggested).toBe(0); + expect(retry.skipped).toBeUndefined(); + expect(prisma.orgSettings.upsert).toHaveBeenCalledTimes(1); + }); +}); diff --git a/packages/backend/src/knowledge-graph/kg-llm.service.ts b/packages/backend/src/knowledge-graph/kg-llm.service.ts index d770657a..5a136130 100644 --- a/packages/backend/src/knowledge-graph/kg-llm.service.ts +++ b/packages/backend/src/knowledge-graph/kg-llm.service.ts @@ -67,7 +67,11 @@ export class KgLlmService { if (!built) return { suggested: 0, model: cfg.model }; if ('skipped' in built) return { suggested: 0, skipped: true, model: cfg.model }; - const { json, usage } = await chatJson(cfg, built.system, built.user); + const { json, skipped, usage } = await chatJson(cfg, built.system, built.user); + // Custom endpoint with an unusable reply: return early before + // applyEnrichResult so kg_llm_hash is NOT stored and the next run retries + // normally instead of seeing an unchanged graph and reporting `skipped`. + if (skipped) return { suggested: 0, skipped: true, model: cfg.model }; const suggested = await this.applyEnrichResult(organizationId, json, built); this.logger.log( `KG LLM enrich ${organizationId}: ${suggested} suggested (${cfg.model}, in=${usage?.inputTokens ?? '?'} out=${usage?.outputTokens ?? '?'})`, diff --git a/packages/backend/src/knowledge-graph/kg-skill.service.spec.ts b/packages/backend/src/knowledge-graph/kg-skill.service.spec.ts index aab2bdae..02b2bdd1 100644 --- a/packages/backend/src/knowledge-graph/kg-skill.service.spec.ts +++ b/packages/backend/src/knowledge-graph/kg-skill.service.spec.ts @@ -1,6 +1,10 @@ import { ConflictException, NotFoundException } from '@nestjs/common'; import { KgSkillService } from './kg-skill.service'; +function okJson(body: any) { + return { ok: true, status: 200, text: async () => JSON.stringify(body), json: async () => body }; +} + /** Tenant isolation + defaults for manual skill creation. Prisma/LLM mocked. */ describe('KgSkillService.create', () => { const ORG = 'org-A'; @@ -58,3 +62,103 @@ describe('KgSkillService.create', () => { ); }); }); + +/** Unusable custom replies must leave pending suggestions untouched (test 2 for #816 review). */ +describe('KgSkillService custom skip', () => { + const ORG = 'org-A'; + const realFetch = global.fetch; + + function setKgEnv() { + process.env.KG_LLM_BASE_URL = 'http://localhost:11434/v1'; + process.env.KG_LLM_MODEL = 'qwen2.5'; + delete process.env.KG_LLM_API_KEY; + } + function clearKgEnv() { + delete process.env.KG_LLM_BASE_URL; + delete process.env.KG_LLM_MODEL; + } + + beforeEach(() => { + setKgEnv(); + global.fetch = jest.fn(); + }); + afterEach(() => { + global.fetch = realFetch; + clearKgEnv(); + jest.restoreAllMocks(); + }); + + function generatePrisma() { + return { + toolInvocation: { + findMany: jest.fn().mockResolvedValue([ + { + intent: 'quote net price', + status: 'SUCCESS', + tool: { name: 'get_price', connector: { name: 'Billing' } }, + }, + ]), + }, + connector: { findMany: jest.fn().mockResolvedValue([{ id: 'c1', name: 'Billing' }]) }, + orgSettings: { findUnique: jest.fn().mockResolvedValue(null) }, + kgSkillSuggestion: { deleteMany: jest.fn(), create: jest.fn() }, + }; + } + + it('generate() leaves pending suggestions untouched on an unusable reply', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: 'Sure, here are some thoughts...' } }] }), + ); + const prisma = generatePrisma(); + const svc = new KgSkillService(prisma as any, { isEnabled: async () => true } as any); + const res = await svc.generate(ORG); + expect(res).toEqual({ created: 0, skipped: true, model: 'qwen2.5' }); + expect(prisma.kgSkillSuggestion.deleteMany).not.toHaveBeenCalled(); + expect(prisma.kgSkillSuggestion.create).not.toHaveBeenCalled(); + }); + + it('generate() still replaces pending suggestions on a usable reply', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ + choices: [ + { + message: { + content: JSON.stringify({ + skills: [{ connector: 'Billing', title: 'T', instruction: 'I', confidence: 0.5 }], + }), + }, + }, + ], + }), + ); + const prisma = generatePrisma(); + const svc = new KgSkillService(prisma as any, { isEnabled: async () => true } as any); + const res = await svc.generate(ORG); + expect(res.created).toBe(1); + expect(res.skipped).toBeUndefined(); + expect(prisma.kgSkillSuggestion.deleteMany).toHaveBeenCalledTimes(1); + }); + + it('consolidate() leaves applied skills untouched on an unusable reply', async () => { + (global.fetch as jest.Mock).mockResolvedValue( + okJson({ choices: [{ message: { content: 'Sure, here are some thoughts...' } }] }), + ); + const applied = [0, 1].map((i) => ({ + id: `s${i}`, + title: `rule ${i}`, + whenToUse: 'w', + instruction: 'do it', + connector: { name: 'Billing' }, + })); + const prisma = { + kgSkillSuggestion: { findMany: jest.fn().mockResolvedValue(applied), deleteMany: jest.fn() }, + connector: { findMany: jest.fn() }, + }; + const svc = new KgSkillService(prisma as any, { isEnabled: async () => true } as any); + const res = await svc.consolidate(ORG); + expect(res).toEqual( + expect.objectContaining({ before: 2, after: 2, model: 'qwen2.5' }), + ); + expect(prisma.kgSkillSuggestion.deleteMany).not.toHaveBeenCalled(); + }); +}); diff --git a/packages/backend/src/knowledge-graph/kg-skill.service.ts b/packages/backend/src/knowledge-graph/kg-skill.service.ts index 3dad23fb..6acc867f 100644 --- a/packages/backend/src/knowledge-graph/kg-skill.service.ts +++ b/packages/backend/src/knowledge-graph/kg-skill.service.ts @@ -63,7 +63,7 @@ export class KgSkillService { async generate( organizationId: string, opts?: { mcpServerId?: string }, - ): Promise<{ created: number; model?: string; usage?: any }> { + ): Promise<{ created: number; skipped?: boolean; model?: string; usage?: any }> { if (!(await this.llm.isEnabled(organizationId))) { throw new ConflictException('AI features are disabled for this workspace.'); } @@ -85,7 +85,10 @@ export class KgSkillService { const cfg = resolveLlmConfig()!; const built = await this.buildConnectorRequest(organizationId); if (!built) return { created: 0, model: cfg.model }; - const { json, usage } = await chatJson(cfg, built.system, built.user); + const { json, skipped, usage } = await chatJson(cfg, built.system, built.user); + // Custom endpoint with an unusable reply: return early before + // applyConnectorResult so pending suggestions are left untouched. + if (skipped) return { created: 0, skipped: true, model: cfg.model }; const created = await this.applyConnectorResult(organizationId, json); this.logger.log(`KG skills (connectors) ${organizationId}: ${created}`); return { created, model: cfg.model, usage }; @@ -182,11 +185,14 @@ export class KgSkillService { ok: i.status === 'SUCCESS', })); - const { json, usage } = await chatJson( + const { json, skipped, usage } = await chatJson( cfg, SERVER_PROMPT, JSON.stringify({ server: server.name, connectors: connectorsContext, calls }), ); + // Custom endpoint with an unusable reply: return early before the + // deleteMany below so pending suggestions are left untouched. + if (skipped) return { created: 0, skipped: true, model: cfg.model }; const skills: any[] = Array.isArray(json?.skills) ? json.skills : []; await this.prisma.kgSkillSuggestion.deleteMany({ diff --git a/packages/backend/src/knowledge-graph/llm-client.spec.ts b/packages/backend/src/knowledge-graph/llm-client.spec.ts index 7e28033b..fbcd08ef 100644 --- a/packages/backend/src/knowledge-graph/llm-client.spec.ts +++ b/packages/backend/src/knowledge-graph/llm-client.spec.ts @@ -103,7 +103,7 @@ describe('chatJson', () => { expect(JSON.parse(init.body).response_format).toBeUndefined(); }); - it('custom endpoint: unusable JSON is skipped, not thrown', async () => { + it('custom endpoint: unusable JSON resolves an explicit skip, not an empty answer', async () => { (global.fetch as jest.Mock).mockResolvedValue( okJson({ choices: [{ message: { content: 'Sure, here are some thoughts...' } }], usage: {} }), ); @@ -112,7 +112,7 @@ describe('chatJson', () => { 'sys', 'user', ); - expect(res).toEqual({ json: {} }); + expect(res).toEqual({ json: null, skipped: true }); }); it('openai path still sends response_format and Authorization', async () => { diff --git a/packages/backend/src/knowledge-graph/llm-client.ts b/packages/backend/src/knowledge-graph/llm-client.ts index 5e543c8d..e0235d2f 100644 --- a/packages/backend/src/knowledge-graph/llm-client.ts +++ b/packages/backend/src/knowledge-graph/llm-client.ts @@ -49,6 +49,9 @@ export function resolveLlmConfig(): LlmConfig | null { export interface LlmResult { json: any; + /** Custom path only: the reply was not usable JSON, so there is nothing to + * apply. Callers must return early without persisting anything. */ + skipped?: boolean; usage?: { inputTokens?: number; outputTokens?: number }; } @@ -61,9 +64,9 @@ export interface LlmResult { export const KG_LLM_MAX_OUTPUT_TOKENS = 4000; /** Call the model and parse its reply as JSON. Throws on transport/parse error, - * except on the custom path where unusable JSON is logged (model + endpoint) - * and answered as an empty result so the enrichment/skill flow is skipped - * rather than crashed. */ + * except on the custom path where an unusable reply resolves + * `{ json: null, skipped: true }` so callers can return early without + * persisting anything (no hash update, no suggestion replacement). */ export async function chatJson( cfg: LlmConfig, system: string, @@ -152,9 +155,20 @@ export async function chatJson( } catch { if (cfg.provider !== 'custom') throw new Error(`LLM returned unusable JSON: ${String(text).slice(0, 200)}`); logger.warn( - `KG LLM custom endpoint returned unusable JSON (model=${cfg.model} endpoint=${base}); skipping this pass.`, + `KG LLM custom endpoint returned unusable JSON (model=${cfg.model} endpoint=${endpointOrigin(base)}); skipping this pass.`, ); - return { json: {} }; + return { json: null, skipped: true }; + } +} + +/** Origin only (scheme + host + port) so credentials, paths and query data + * in a custom base URL can never leak into the logs. Falls back to a fixed + * marker when the base is not a parseable URL. */ +function endpointOrigin(base: string): string { + try { + return new URL(base).origin; + } catch { + return '(unparseable endpoint)'; } } From 66945a6e08c3a282e38f3bf918d607637aaa186a Mon Sep 17 00:00:00 2001 From: Matteo Morelli Date: Fri, 2 Oct 2026 17:45:53 +0200 Subject: [PATCH 3/3] docs(kg): keep the env block in one code fence; pass provider keys in compose --- docker-compose.yml | 3 +++ docs/knowledge-graph.md | 19 +++++++++---------- 2 files changed, 12 insertions(+), 10 deletions(-) diff --git a/docker-compose.yml b/docker-compose.yml index 703532ad..453a80b7 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -92,6 +92,9 @@ services: - KG_LLM_MODEL=${KG_LLM_MODEL:-} - KG_LLM_BASE_URL=${KG_LLM_BASE_URL:-} - KG_LLM_API_KEY=${KG_LLM_API_KEY:-} + - OPENAI_API_KEY=${OPENAI_API_KEY:-} + - OPENROUTER_API_KEY=${OPENROUTER_API_KEY:-} + - ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY:-} depends_on: postgres: diff --git a/docs/knowledge-graph.md b/docs/knowledge-graph.md index a3fffcbb..95936558 100644 --- a/docs/knowledge-graph.md +++ b/docs/knowledge-graph.md @@ -188,16 +188,6 @@ OPENAI_API_KEY= # or OPENROUTER_API_KEY / ANTHROPIC_API_KEY # qwen2.5 (Qwen2.5-0.5B-Instruct via an OpenAI-compatible front). # KG_LLM_BASE_URL=http://localhost:11434/v1 # KG_LLM_API_KEY= -``` - -Docker note: `localhost` inside the app container means the AnythingMCP -container itself, not your host. To reach an Ollama running on the host, use -`http://host.docker.internal:11434/v1` — on Linux that name needs an explicit -mapping (`extra_hosts: ["host.docker.internal:host-gateway"]`). When Ollama -runs as a service in the same Compose project/network, use its service name -instead (e.g. `http://ollama:11434/v1`). The five `KG_LLM_*` variables are -passed through to the app container in `docker-compose.yml` with empty -defaults, so setting them in `.env` is enough. # Scheduled extension (cloud cron) — all must align: this flag, the cron call, # and the per-workspace "Scheduled AI extension" switch @@ -210,6 +200,15 @@ KG_LLM_BATCH=false # Anthropic Message Batches (~50% cheaper) KG_LLM_REDACT_INTENTS=true ``` +Docker note: `localhost` inside the app container means the AnythingMCP +container itself, not your host. To reach an Ollama running on the host, use +`http://host.docker.internal:11434/v1` — on Linux that name needs an explicit +mapping (`extra_hosts: ["host.docker.internal:host-gateway"]`). When Ollama +runs as a service in the same Compose project/network, use its service name +instead (e.g. `http://ollama:11434/v1`). The `KG_LLM_*` variables and the +provider keys are passed through to the app container in `docker-compose.yml` +with empty defaults, so setting them in `.env` is enough. + The graph itself (static + observational layers, manual editing, the `kg_how_to_obtain` tool) works with **no LLM key** — the AI flags only add the optional enrichment and skill-generation passes on top.