diff --git a/.env.example b/.env.example index ddfebb1..f889faa 100644 --- a/.env.example +++ b/.env.example @@ -16,11 +16,31 @@ ALLOW_SERVER_DEEPGRAM_FALLBACK=false SERPER_API_KEY= ALLOW_SERVER_SERPER_FALLBACK=false -# Voice broker mode: -# - deepgram: Deepgram Voice Agent path. -# - custom: browser -> local broker -> OpenRouter foreground/background models. +# Voice mode is now chosen at runtime in Settings (no rebuild needed): +# - deepgram-duplex (default): the two-model "interaction" mimic. A fast +# foreground model keeps the conversation moving while a background model does +# the heavy lifting. Runs inside this app's own dev/start server — no separate +# host — and needs only a Deepgram key plus an LLM key (OpenAI or OpenRouter; +# the broker auto-detects the provider from the key shape). +# - deepgram-agent: the single Deepgram Voice Agent path. +# - openai-realtime: test/comparison-only speech-to-speech over WebRTC. +# VITE_VOICE_BROKER_MODE below is only the legacy initial default for the +# runtime setting ("deepgram" -> deepgram-agent, anything else -> duplex). VITE_VOICE_BROKER_MODE=deepgram +# Real OpenAI key. Used by the read-aloud gpt-4o-mini-tts route and, when the +# fallback flag is enabled, to mint OpenAI Realtime client secrets server-side +# for the test voice mode. Realtime is BYOK-first: the browser can send its own +# OpenAI key, so this server key is optional. +OPENAI_API_KEY= +ALLOW_SERVER_OPENAI_FALLBACK=false +# OpenAI Realtime model for the test/comparison voice mode (browser WebRTC). +VITE_OPENAI_REALTIME_MODEL=gpt-realtime + +# Set to false on deployed hosts to ignore the client-supplied x-miso-tts-api-url +# header and use only MISO_TTS_API_URL below (blocks localhost port probing). +MISO_TTS_ALLOW_HEADER_URL=true + # Remote voice/signaling server. Leave VITE_VOICE_WS_URL empty to use the same # host the app is served from. When the app host cannot serve WebSockets # (e.g. Vercel), point this at the dedicated voice server, e.g. diff --git a/.eslintrc.json b/.eslintrc.json deleted file mode 100644 index 7e2b4f1..0000000 --- a/.eslintrc.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "parser": "@typescript-eslint/parser", - "plugins": ["@typescript-eslint", "react", "react-hooks"], - "extends": [ - "eslint:recommended", - "plugin:@typescript-eslint/recommended", - "plugin:react/recommended", - "plugin:react-hooks/recommended" - ], - "settings": { - "react": { - "version": "detect" - } - }, - "env": { - "browser": true, - "es2021": true - } -} diff --git a/BUG_REPORT.md b/BUG_REPORT.md index fd4b278..0f0b68a 100644 --- a/BUG_REPORT.md +++ b/BUG_REPORT.md @@ -1,68 +1,97 @@ -# Bug Review Report +# Audit Report -Review of the Tutor-System codebase (server, client memory engines, views, -and the voice/bilateral architecture). Typecheck passes; 277/281 tests pass -(the 3 failures are environmental — `pymupdf` is not installed in the review -container — not code defects). +Full audit of the Tutor-System codebase (server, client memory engines, views, +PDF pipeline, Mermaid rendering, UI, and the voice architecture) plus the +product changes requested alongside it. Verification gate after the work: +`format:check`, `lint` (tsc), `test` (280 node pass / 1 skipped, 591 DOM pass), +and `build` all green, with `pymupdf`/`pymupdf4llm` installed. -## Fixed in this branch +## Security — fixed -### 1. Chat SSE stream was buffered by the compression middleware (high impact) -`server.ts` applies `app.use(compression())` globally, and `/api/chat` streams -Server-Sent Events without flushing. `compression()` gzip-buffers a -`text/event-stream` body and releases it only when the response ends, so the -token-by-token streaming in `ChatPanel` never runs — the user sees a spinner, -then the whole answer at once. Verified in isolation: five events arrived in a -single trailing chunk. +1. **Cross-site WebSocket hijack on `/api/voice-agent` (High).** The Deepgram + Voice Agent upgrade accepted any origin, unlike the custom broker. It now + uses the same `isAuthorizedLocalVoiceBrokerRequest` (same-origin / debug + token) gate, closing the hijack + Deepgram-credit-drain vector. +2. **Mermaid XSS via `securityLevel: "loose"` (High).** Chat diagrams are + influenced by untrusted PDF/OCR/web content and were rendered with HTML + sanitization disabled + `innerHTML`. Switched to `securityLevel: "strict"` + (matching RevisionView), which DOMPurifies the SVG and blocks in-diagram + `click`/`javascript:` directives. +3. **Unbounded per-user SQLite handles (High, DoS + FD/memory leak).** Each + client-supplied `userId` opened a WAL handle that was cached forever. + `LearnerStore` now evicts idle handles with an LRU cap + (`LEARNINGAI_MAX_OPEN_DBS`, default 64). +4. **Document-read identity mismatch (High, best-effort).** `/documents/:id/file` + and `/text` trusted `req.query.userId`. They now reject a request whose + header identity and query identity explicitly disagree, while still allowing + the header-less react-pdf fetch. (True tenant isolation still requires real + auth, which the local-first model defers.) +5. **Oversized-body DoS (Med).** `express.urlencoded` limit lowered from 100mb + to 1mb. +6. **MisoTTS localhost probing (Low).** `MISO_TTS_ALLOW_HEADER_URL=false` lets + deployments ignore the client-supplied Miso URL header. -**Fix:** mark the SSE response `Cache-Control: no-cache, no-transform` -(which `compression` respects and skips), set `X-Accel-Buffering: no`, and call -`res.flush()` after each event write. After the fix, events stream ~200ms apart. +## Memory leaks — fixed -### 2. Full-duplex voice broker was unreachable on a deployed host (high impact) -The `/api/voice-broker` WebSocket upgrade was gated by -`isAuthorizedLocalVoiceBrokerRequest`, which only passed for loopback or a debug -token. On a deployed persistent Node host the broker returned `403`, so the -foreground/background duplex voice design could never run in production. +7. **Hot-mic on unmount.** A live voice session (mic `MediaStream`, + `AudioContext`, `ScriptProcessorNode`, WebSocket) was never released if + `ChatPanel` unmounted mid-session. Added an unmount cleanup. +8. **Web-search cache never evicted.** Expired entries are now dropped and the + map is size-capped. +9. See security #3 (SQLite handle eviction). -**Fix:** also accept the upgrade when the request `Origin` matches the app host -(same-origin, with `x-forwarded-host` support behind a proxy). This blocks -cross-site WebSocket hijacking while letting the deployed app's own page connect. -Keys remain BYOK per connection; server fallback keys stay gated behind the -existing `ALLOW_SERVER_*` env flags. Verified with real handshakes under -`NODE_ENV=production`: same-origin → `101`, cross-site → `403`, no-origin → `403`. +## Correctness — fixed (previously open findings #3–#5) -## Open findings (not yet fixed) +10. **Blank reply when the tool budget is exhausted mid-tool-call (#3).** The + `/api/chat` agent loop could exit with tool results appended but never + synthesized. Added a final no-tools synthesis turn; telemetry now reports an + honest model-turn count (fixes the #5 off-by-one). +11. **TTS deadline too tight (#4).** `VOICE_BROKER_TTS_DEADLINE_MS` default + raised from 180ms to 1500ms. +12. **Misleading TTS model (#5).** `/api/tts` now calls the `gpt-4o-mini-tts` + model it reports instead of silently substituting `tts-1`. +13. **Silent $0 cost (#5).** `openRouterCost` matches pricing across provider + prefixes, so non-`openai/` models no longer read as $0. -### 3. Tool loop can terminate with no synthesized answer (medium) -In the `/api/chat` agent loop (`while (iterations < MAX_ITERATIONS)`), if the -model emits tool calls on the final allowed iteration, the tool results are -appended to the message list but the loop exits before another model turn turns -them into text. `finalContent` for that turn is empty, so the user can get a -blank/truncated reply when the tool-iteration budget is hit. Consider one extra -synthesis turn after the loop when the last turn was a tool call. +## Product changes shipped alongside the audit -### 4. Voice broker TTS deadline is unrealistically tight (medium) -`VOICE_BROKER_TTS_DEADLINE_MS` defaults to `180`ms. MisoTTS/Deepgram audio -rarely returns that fast, so synthesis is almost always aborted before it -arrives. A value around `1500`ms is more realistic. +- **PDF page-aware context.** Extraction switched to page-indexed markdown + (`pymupdf4llm` `page_chunks`); the current reader page's text is now injected + as a labeled `CURRENT PAGE N of M` block (plus adjacent pages) into the tutor + prompt. Previously the model was never told which page the learner was on and + only ever saw the document's opening characters. +- **Mermaid redesign.** Diagrams render only once the source parses cleanly (no + more raw parser errors flashing mid-stream), are static and fully visible + (removed the perpetual dimming + auto-panning "focus tour"), with click-to- + focus zoom, higher contrast, and reduced-motion support. +- **AdminView dark theme.** Converted the light cream/serif diagnostics surface + to the dark Cosmic Obsidian theme for cross-view consistency (RevisionView's + intentional paper look is preserved). +- **Voice: friction-free duplex + Realtime test mode.** Voice mode is a runtime + Setting; the two-model Deepgram duplex is the default and runs in the app's own + server with just a Deepgram key + an OpenAI or OpenRouter key (auto-detected); + a default-off OpenAI Realtime (WebRTC) mode was added for full-duplex + benchmarking with no persistent server. -### 5. Minor -- `iterations + 1` is reported in telemetry after the loop already incremented, - so iteration counts are off by one (`server.ts`). -- `openRouterCost` only falls back on an `openai/` prefix mismatch, so a - non-OpenAI model with a pricing-key mismatch silently reads as `$0`. +## Deferred / not changed (with rationale) -## Deployment note +- **StatusBadge light palette.** Its wrapper is unused in-app (only its + `currentColor` icons are used, which adapt to dark), and its classes are + pinned by tests — re-theming would break tests for no app benefit. +- **`.eslintrc.json` removed.** It referenced parsers/plugins that were never + installed; `npm run lint` runs `tsc`, so the config only implied a lint that + never ran. +- **True multi-tenant auth.** Local profiles remain "not real auth" by design; + document-read hardening is best-effort within that model. +- **`new Function` JS runner / arbitrary model-authored code.** Retained behind + the explicit "Run" gate; the passive Mermaid XSS path is now closed. -Vercel serverless cannot host the persistent WebSocket the voice broker needs -(`server/vercel-handler.ts` never calls `attachWebSockets`). Voice mode requires -a long-running Node host. With finding #2 fixed, setting -`VITE_VOICE_BROKER_MODE=custom` at build time enables the full-duplex broker -same-origin on such a host. +## Deployment notes -## Responsiveness - -Measured in a real mobile browser (Chromium, 320–390px) across Study, Analytics, -Revision, and Admin: no horizontal overflow, cards stack, nav collapses to -icons, composer and voice blob fit. The app is already substantially responsive. +- Vercel serverless still cannot host the persistent voice-broker WebSocket, so + `deepgram-duplex` / `deepgram-agent` need a long-running Node host (or the + separate voice server via `VITE_VOICE_WS_URL`). The `openai-realtime` test + mode is the one voice path that works on serverless (browser↔OpenAI WebRTC + + the tiny `/api/realtime/token` endpoint). +- A proxy terminating the voice WebSocket must strip inbound `X-Forwarded-Host` + so the same-origin broker check cannot be spoofed. diff --git a/README.md b/README.md index 003e2a7..1c7f41f 100644 --- a/README.md +++ b/README.md @@ -202,16 +202,37 @@ to cloud Postgres/object storage later. ## Voice Modes -- `deepgram` mode uses the Deepgram Voice Agent path. -- `custom` mode opens a browser WebSocket to the local broker. The broker sends - foreground teaching to an OpenRouter-compatible `VOICE_FOREGROUND_MODEL` and - delegates web/code/PDF/tool work to `VOICE_BACKGROUND_MODEL`. +Voice mode is a **runtime setting** in Settings (no rebuild required): + +- **`deepgram-duplex` (default)** — the two-model "interaction" mimic of + Thinking Machines' interaction models. A fast foreground model + (`VOICE_FOREGROUND_MODEL`) keeps the live conversation moving while an async + background model (`VOICE_BACKGROUND_MODEL`) does web/code/PDF/tool heavy + lifting, whose result is stitched back in as a spoken aside. It runs inside + the same `npm run dev` / `npm start` Node process that serves the app — **no + separate server to spin up** — and needs only a **Deepgram key** (STT + Aura + TTS) plus an **LLM key**. The LLM key can be OpenAI **or** OpenRouter; the + broker auto-detects the provider from the key shape (`sk-or-…` → OpenRouter, + `sk-…` → OpenAI direct), so "just a Deepgram key + a ChatGPT key" works. +- **`deepgram-agent`** — the single Deepgram Voice Agent path. +- **`openai-realtime` (test / comparison only)** — connects the browser + straight to OpenAI's Realtime API over **WebRTC**, for a true full-duplex, + uninterrupted benchmark against the cheaper mimic. It needs **no persistent + server** (only a tiny `/api/realtime/token` endpoint that also runs on Vercel + serverless) and is BYOK-first (the browser's OpenAI key mints a short-lived + ephemeral secret; the standard key never reaches the WebRTC exchange). This is + intentionally **not** the default and is billed at premium realtime rates. + - Background answers are cleaned before insertion so raw markdown such as `**Apple**` is not read aloud. - MisoTTS is optional and experimental. The local broker only accepts loopback - Miso URLs such as `http://127.0.0.1:8080`. + Miso URLs such as `http://127.0.0.1:8080`; set `MISO_TTS_ALLOW_HEADER_URL=false` + on deployed hosts to ignore the client-supplied Miso URL header. - No route should claim a universal sub-200 ms guarantee. Report latency as measured p50, p95, failure rate, route, provider, region, and hardware. +- Deployments that terminate the voice WebSocket behind a proxy must strip any + inbound `X-Forwarded-Host` header so the same-origin broker check can't be + spoofed. ## Getting Started diff --git a/scripts/classify_and_extract.py b/scripts/classify_and_extract.py index 5b273f1..76203e2 100644 --- a/scripts/classify_and_extract.py +++ b/scripts/classify_and_extract.py @@ -69,6 +69,7 @@ def main(): "total_pages": total_pages, "pages_with_text": pages_with_text, "content": "", + "pages": [], "images": [], "vision_page_limit": MAX_VISION_PAGES, } @@ -77,8 +78,23 @@ def main(): try: with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): import pymupdf4llm - md_text = pymupdf4llm.to_markdown(doc) - result["content"] = md_text + # Single-pass, page-indexed extraction. page_chunks=True returns + # one dict per page so we can hand the model the exact page the + # learner is looking at, and it is no slower than the plain + # whole-document call. + page_chunks = pymupdf4llm.to_markdown(doc, page_chunks=True) + pages = [] + content_parts = [] + for index, chunk in enumerate(page_chunks): + page_text = (chunk.get("text") if isinstance(chunk, dict) else str(chunk)) or "" + meta = chunk.get("metadata", {}) if isinstance(chunk, dict) else {} + page_num = meta.get("page", index) if isinstance(meta, dict) else index + pages.append({"page_num": int(page_num), "text": page_text}) + # Keep an explicit page marker in the joined markdown so the + # whole-document excerpt retains page boundaries. + content_parts.append(f"\n{page_text}") + result["pages"] = pages + result["content"] = "\n\n".join(content_parts) except Exception as e: doc.close() result["error"] = f"Failed to extract markdown: {str(e)}" diff --git a/server.ts b/server.ts index 5ccc872..7ddfa11 100644 --- a/server.ts +++ b/server.ts @@ -73,15 +73,41 @@ const VOICE_BACKGROUND_MODEL = process.env.VOICE_BACKGROUND_MODEL || process.env.GPT55_MODEL || "openai/gpt-5.5"; + +// The two-model duplex broker can run its foreground/background LLM calls on a +// raw OpenAI ("ChatGPT") key or on OpenRouter, chosen automatically from the +// key shape: OpenRouter keys are "sk-or-...", OpenAI keys are "sk-..." only. +// This lets the duplex work with just a Deepgram key + an OpenAI key, no +// OpenRouter account required. OpenAI wants bare model ids, so the OpenRouter +// "openai/" prefix is stripped, and OpenRouter-only hosted tools are dropped. +const resolveBrokerLlmProvider = (apiKey: string, model: string) => { + const key = String(apiKey || ""); + const isOpenAiDirect = /^sk-/.test(key) && !/^sk-or-/.test(key); + if (isOpenAiDirect) { + return { + baseURL: "https://api.openai.com/v1", + model: model.replace(/^openai\//, ""), + supportsOpenRouterTools: false, + }; + } + return { + baseURL: "https://openrouter.ai/api/v1", + model, + supportsOpenRouterTools: true, + }; +}; const VOICE_BROKER_FAST_ACK = /^(1|true|yes|on)$/i.test( String(process.env.VOICE_BROKER_FAST_ACK || "").trim(), ); const VOICE_BROKER_FAST_ACK_TEXT = process.env.VOICE_BROKER_FAST_ACK_TEXT || "Okay."; const VOICE_BROKER_FAST_ACK_FILE = process.env.VOICE_BROKER_FAST_ACK_FILE || ""; +// Deepgram Aura / MisoTTS synthesis rarely returns in under ~180ms, so the old +// default aborted almost every utterance before audio arrived. 1500ms is a +// realistic upper bound that still fails fast on a genuinely stuck provider. const VOICE_BROKER_TTS_DEADLINE_MS = Math.max( 50, - Number(process.env.VOICE_BROKER_TTS_DEADLINE_MS || 180), + Number(process.env.VOICE_BROKER_TTS_DEADLINE_MS || 1500), ); const VOICE_BROKER_STT_MODEL = process.env.VOICE_BROKER_STT_MODEL || "nova-3"; const VOICE_BROKER_TTS_MODEL = @@ -178,7 +204,14 @@ const MISO_TTS_HEALTH_TIMEOUT_MS = 800; const VOICE_WS_BUFFER_HIGH_WATER_BYTES = 1_000_000; const VOICE_AGENT_MESSAGE_BUFFER_LIMIT = 80; +// The client can point TTS at a local Miso server via this header, which is +// convenient for local dev but lets a caller probe arbitrary loopback ports on +// a deployed host. Deployments can set MISO_TTS_ALLOW_HEADER_URL=false to ignore +// the header and use only the server-configured MISO_TTS_API_URL. Defaults to +// on to preserve the existing local workflow. +const ALLOW_MISO_HEADER_URL = process.env.MISO_TTS_ALLOW_HEADER_URL !== "false"; const readMisoTtsApiUrlOverride = (headers: IncomingHttpHeaders) => { + if (!ALLOW_MISO_HEADER_URL) return ""; const raw = headers["x-miso-tts-api-url"]; if (Array.isArray(raw)) return raw[0] || ""; return typeof raw === "string" ? raw : ""; @@ -626,10 +659,19 @@ const openRouterCost = ( inputTokens: number, outputTokens: number, ) => { + // Resolve pricing across provider-prefix variants so a model such as + // "anthropic/claude-..." or a bare "gpt-4o-mini" doesn't silently read as $0 + // when the pricing map keys it under a different prefix. + const bareModel = model.includes("/") + ? model.slice(model.indexOf("/") + 1) + : model; const modelPricing = pricing[model] || - pricing[model.replace(/^openai\//, "")] || - pricing[`openai/${model}`]; + pricing[bareModel] || + pricing[`openai/${bareModel}`] || + Object.entries(pricing).find( + ([key]) => key === model || key.endsWith(`/${bareModel}`), + )?.[1]; if (!modelPricing) return 0; return roundCost( inputTokens * modelPricing.prompt + outputTokens * modelPricing.completion, @@ -877,7 +919,10 @@ export async function createTutorServerApp( app.use(compression()); app.use(express.json({ limit: "10mb" })); - app.use(express.urlencoded({ limit: "100mb", extended: true })); + // Cap urlencoded bodies far below the previous 100mb: no route needs a large + // form body, and an oversized `extended` (qs) parse is a cheap memory-DoS + // amplifier. Large binary uploads go through multer, not this parser. + app.use(express.urlencoded({ limit: "1mb", extended: true })); const uploadDir = process.env.TUTOR_UPLOAD_DIR || @@ -1100,6 +1145,81 @@ export async function createTutorServerApp( }); }); + // Mint a short-lived OpenAI Realtime client secret for the browser's WebRTC + // handshake (the test/comparison voice mode). This is the ONLY server piece + // that path needs — a single POST, so it also runs on Vercel serverless. The + // standard key never reaches the browser: it stays here, and only the + // ephemeral secret is returned. BYOK (the browser's own key via Authorization) + // is preferred; the server key is used only when ALLOW_SERVER_OPENAI_FALLBACK + // is enabled. + app.post("/api/realtime/token", async (req, res) => { + const byokKey = sanitizeApiKey( + String(req.headers["authorization"] || "").replace(/^Bearer\s+/i, ""), + ); + const allowServerOpenAi = /^(1|true|yes|on)$/i.test( + String(process.env.ALLOW_SERVER_OPENAI_FALLBACK || "").trim(), + ); + const apiKey = + byokKey || + (allowServerOpenAi ? sanitizeApiKey(process.env.OPENAI_API_KEY) : ""); + if (!apiKey) { + return res.status(401).json({ + error: + "An OpenAI API key is required for Realtime voice (test mode). Add it in Settings.", + code: "OPENAI_KEY_MISSING", + }); + } + const model = String(req.body?.model || "gpt-realtime").slice(0, 100); + const voice = String(req.body?.voice || "marin").slice(0, 40); + try { + const response = await fetch( + "https://api.openai.com/v1/realtime/client_secrets", + { + method: "POST", + headers: { + Authorization: `Bearer ${apiKey}`, + "Content-Type": "application/json", + }, + body: JSON.stringify({ + session: { + type: "realtime", + model, + audio: { output: { voice } }, + }, + }), + }, + ); + const bodyText = await response.text(); + if (!response.ok) { + return res.status(response.status).json({ + error: "Failed to mint an OpenAI Realtime client secret.", + detail: bodyText.slice(0, 500), + }); + } + let parsed: any = {}; + try { + parsed = JSON.parse(bodyText); + } catch { + parsed = {}; + } + const value = parsed?.value || parsed?.client_secret?.value || ""; + if (!value) { + return res + .status(502) + .json({ error: "OpenAI did not return a client secret." }); + } + return res.json({ + value, + expiresAt: parsed?.expires_at || parsed?.expires_after || null, + model, + }); + } catch (error: any) { + return res.status(502).json({ + error: error?.message || "Realtime token mint failed.", + }); + } + }); + app.get("/api/learner/profile", (req, res) => { const userId = learnerUserIdFromHeaders(req.headers); const profile = learnerStore.ensureProfile( @@ -1171,10 +1291,36 @@ export async function createTutorServerApp( res.json({ ok: true, ...result }); }); - app.get("/api/learner/documents/:documentId/file", (req, res) => { - const userId = normalizeLearnerUserId( - req.query.userId || learnerUserIdFromHeaders(req.headers), + // Local profiles are not real auth (see README): both the request header and + // the ?userId= query are client-supplied, so this cannot enforce true tenant + // isolation. What it *can* do is reject a request whose header identity and + // query identity explicitly disagree (a confused-deputy / URL-smuggling + // vector) while still allowing the header-less react-pdf `` + // fetch, which can only carry identity via the query string. Prefer the + // header when present; fall back to the query otherwise. + const resolveDocumentUserId = (req: express.Request): string | null => { + const headerRaw = + req.headers["x-learningai-user-id"] || + req.headers["X-LearningAI-User-Id"] || + req.headers["x-user-id"]; + const hasHeader = Boolean( + Array.isArray(headerRaw) ? headerRaw[0] : headerRaw, ); + const headerUser = learnerUserIdFromHeaders(req.headers); + const queryUser = req.query.userId + ? normalizeLearnerUserId(req.query.userId) + : ""; + if (hasHeader && queryUser && headerUser !== queryUser) { + return null; + } + return hasHeader ? headerUser : queryUser || headerUser; + }; + + app.get("/api/learner/documents/:documentId/file", (req, res) => { + const userId = resolveDocumentUserId(req); + if (!userId) { + return res.status(403).json({ error: "User identity mismatch." }); + } const document = learnerStore.getDocument(userId, req.params.documentId); if (!document || !fs.existsSync(document.filePath)) { return res.status(404).json({ error: "Document file not found." }); @@ -1184,9 +1330,10 @@ export async function createTutorServerApp( }); app.get("/api/learner/documents/:documentId/text", (req, res) => { - const userId = normalizeLearnerUserId( - req.query.userId || learnerUserIdFromHeaders(req.headers), - ); + const userId = resolveDocumentUserId(req); + if (!userId) { + return res.status(403).json({ error: "User identity mismatch." }); + } const document = learnerStore.getDocument(userId, req.params.documentId); if (!document) { return res.status(404).json({ error: "Document text not found." }); @@ -1387,6 +1534,17 @@ export async function createTutorServerApp( let extractedText = result.content || ""; + // Page-indexed text so the tutor can be told exactly what is on the + // page the learner is viewing. Seed from the pymupdf4llm page chunks; + // vision-OCR pages below are merged in by page number. + const pageTextMap = new Map(); + for (const page of Array.isArray(result.pages) ? result.pages : []) { + const pageNum = Number(page?.page_num); + if (Number.isFinite(pageNum)) { + pageTextMap.set(pageNum, String(page?.text || "")); + } + } + // If Scanned or Mixed, perform Vision Parsing on page images. if ( result.classification === "Scanned" || @@ -1426,7 +1584,13 @@ export async function createTutorServerApp( const pageText = response.choices[0]?.message?.content?.trim(); if (pageText) { - extractedText += `\n\n## OCR / Vision Page ${Number(img.page_num ?? 0) + 1}\n\n${pageText}`; + const pageNum = Number(img.page_num ?? 0); + extractedText += `\n\n## OCR / Vision Page ${pageNum + 1}\n\n${pageText}`; + const existing = pageTextMap.get(pageNum); + pageTextMap.set( + pageNum, + existing ? `${existing}\n\n${pageText}` : pageText, + ); } } } catch (visionError) { @@ -1435,6 +1599,10 @@ export async function createTutorServerApp( } } + const pages = Array.from(pageTextMap.entries()) + .sort((a, b) => a[0] - b[0]) + .map(([page_num, text]) => ({ page_num, text })); + let serverDocument = null; if (documentId && bookId) { try { @@ -1447,6 +1615,7 @@ export async function createTutorServerApp( size: req.file?.size || 0, sourcePath: filePath, extractedText, + pages, classification: result.classification, extractionMode: result.extraction_mode, totalPages: result.total_pages, @@ -2150,7 +2319,10 @@ Use a Markdown table when comparing 2+ things. Keep every section scannable; whi apiKey: openaiKey, }); const mp3 = await openai.audio.speech.create({ - model: "tts-1", + // Call the model actually reported below in X-Usage-Model rather + // than silently substituting tts-1, so usage/cost telemetry is + // truthful. gpt-4o-mini-tts is a current OpenAI speech model. + model: "gpt-4o-mini-tts", voice: "alloy", input: billedText, }); @@ -2585,7 +2757,32 @@ Use a Markdown table when comparing 2+ things. Keep every section scannable; whi serperApiKey: bodySerperKey, language, requestId: clientRequestId, + activeDocumentId: bodyActiveDocumentId, + currentPage: bodyCurrentPage, + currentPageTotal: bodyCurrentPageTotal, } = req.body; + // Resolve the extracted text of the page the learner is currently viewing + // so it can be injected as an explicit "CURRENT PAGE" block below. + const currentPageContext = (() => { + const documentId = + typeof bodyActiveDocumentId === "string" ? bodyActiveDocumentId : ""; + const pageNumber = Number(bodyCurrentPage); + if (!documentId || !Number.isFinite(pageNumber) || pageNumber < 1) { + return null; + } + try { + const userId = learnerUserIdFromHeaders(req.headers); + const page = learnerStore.readDocumentPageText( + userId, + documentId, + pageNumber, + ); + if (!page || !page.pageText.trim()) return null; + return page; + } catch { + return null; + } + })(); requestId = normalizeClientRequestId(clientRequestId) || requestId; const runtimeSettings = normalizeBrainRuntimeSettings(rawRuntimeSettings); const runtimeSettingsSnapshot = compactRuntimeSettings(runtimeSettings); @@ -2876,6 +3073,25 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: systemInstruction += `\n\n${memoryContext}`; } + // Tell the tutor exactly what is on the page the learner is looking at. + // This is the extracted text of the current reader page (plus immediate + // neighbours for continuity), so questions like "what's on this page?" are + // answered from the real page rather than the document's opening excerpt. + if (currentPageContext) { + const pageBody = currentPageContext.pageText + .replace(/\s+\n/g, "\n") + .trim() + .slice(0, 4000); + const neighborBody = currentPageContext.neighborText + .trim() + .slice(0, 2000); + systemInstruction += `\n\nCURRENT PAGE ${currentPageContext.page} of ${currentPageContext.totalPages} (the page the learner is viewing right now):\n${pageBody}`; + if (neighborBody) { + systemInstruction += `\n\nADJACENT PAGES (for continuity only):\n${neighborBody}`; + } + systemInstruction += `\n\nWhen the learner refers to "this page", "here", or "the current page", answer from the CURRENT PAGE text above.`; + } + if (currentPageImage && sourceMaterialRequest) { systemInstruction += `\n\nCURRENT PAGE IMAGE IS ATTACHED THROUGH THE look_at_current_page TOOL. For this source-material request, call look_at_current_page before answering and answer from the page image plus selected/library context. Do not use web_search unless the user explicitly asks for web search.`; } @@ -2924,6 +3140,15 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: let evaluatedAnswers: any[] = []; let iterations = 0; + // Count actual model completions (each streamed turn), including the + // final answer turn that `break`s out before iterations++ and the + // post-loop synthesis turn. This is the honest number for telemetry; + // `iterations` alone only counts tool-executing turns. + let modelTurns = 0; + // True once a turn has emitted tool calls and false again once a turn + // answers with text. If the loop exits with this still true, the budget + // was exhausted mid-tool-call and we owe the user a synthesis turn. + let lastTurnHadToolCalls = false; const MAX_ITERATIONS = runtimeSettings.toolIterationLimit; let finalContent = ""; @@ -3086,11 +3311,15 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: } } + modelTurns++; + if (!isToolCall) { + lastTurnHadToolCalls = false; break; // Done! } // We have tool calls + lastTurnHadToolCalls = true; sendEvent("status", { phase: "tool_execution" }); const validToolCalls = currentToolCalls.filter(Boolean); recordSystemActivity({ @@ -3582,6 +3811,46 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: } } + // If the loop exhausted its tool-iteration budget while the last model + // turn was still emitting tool calls, the tool results were appended to + // the message list but never turned into a spoken answer — the user + // would otherwise get a blank/truncated reply. Run one final synthesis + // turn with no tools so the accumulated tool results become text. + if (lastTurnHadToolCalls) { + try { + sendEvent("status", { phase: "synthesizing" }); + const synthesisStream: any = await openai.chat.completions.create({ + model: usedModelForUsage, + messages: formattedMessages as any, + stream: true, + stream_options: { include_usage: true } as any, + } as any); + modelTurns++; + for await (const chunk of synthesisStream) { + const usage = (chunk as any).usage; + if (usage) { + inputTokens = usage.prompt_tokens ?? inputTokens; + outputTokens = usage.completion_tokens ?? outputTokens; + usageEstimated = false; + } + const delta = chunk.choices?.[0]?.delta; + if (delta?.content) { + finalContent += delta.content; + sendEvent("chunk", { content: delta.content }); + } + } + } catch (synthesisError: any) { + recordSystemActivity({ + kind: "model", + status: "failed", + title: "Final synthesis turn failed", + detail: String(synthesisError?.message || synthesisError), + requestId, + phase: "tool_followup", + }); + } + } + if (inputTokens === 0 && outputTokens === 0) { inputTokens = estimateTokensFromText(formattedMessages); outputTokens = estimateTokensFromText(finalContent); @@ -3613,7 +3882,7 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: graphUpdates: graphUpdates.length, flashcards: flashcardsUpdates.length, evaluatedAnswers: evaluatedAnswers.length, - iterations: iterations + 1, + iterations: modelTurns, runtimeSettings: runtimeSettingsSnapshot, }); recordSystemActivity({ @@ -3636,7 +3905,7 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: flashcards: flashcardsUpdates.length, evaluatedAnswers: evaluatedAnswers.length, webSources: webSources.length, - iterations: iterations + 1, + iterations: modelTurns, runtimeSettings: runtimeSettingsSnapshot, }, }); @@ -3726,6 +3995,16 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: wss.emit("connection", ws, request); }); } else if (pathname === "/api/voice-agent") { + // Gate the Deepgram Voice Agent upgrade with the same same-origin / + // debug-token check the custom broker uses. Without it, any external + // page could open this socket (cross-site WebSocket hijacking) and, if + // ALLOW_SERVER_DEEPGRAM_FALLBACK is enabled, drain the deployment's + // Deepgram credits or exhaust connections. + if (!isAuthorizedLocalVoiceBrokerRequest(request)) { + socket.write("HTTP/1.1 403 Forbidden\r\nConnection: close\r\n\r\n"); + socket.destroy(); + return; + } wss.handleUpgrade(request, socket, head, (ws) => { wss.emit("connection", ws, request); }); @@ -4674,12 +4953,16 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: return "I have the local learning book, previous context, and document memory loaded. I can answer the live part now, and any web, PDF, code, or pricing work is staged for the GPT-5.5 background layer once the provider key is connected."; } const startedAt = Date.now(); + const fgProvider = resolveBrokerLlmProvider( + brokerOpenRouterApiKey, + foregroundModel, + ); const openai = new OpenAI({ - baseURL: "https://openrouter.ai/api/v1", + baseURL: fgProvider.baseURL, apiKey: brokerOpenRouterApiKey, }); const response = await openai.chat.completions.create({ - model: foregroundModel, + model: fgProvider.model, temperature: 0.35, max_tokens: 180, messages: [ @@ -4754,12 +5037,16 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: ); } + const bgProvider = resolveBrokerLlmProvider( + brokerBackgroundApiKey, + backgroundModel, + ); const openai = new OpenAI({ - baseURL: "https://openrouter.ai/api/v1", + baseURL: bgProvider.baseURL, apiKey: brokerBackgroundApiKey, }); const modelResponse = await openai.chat.completions.create({ - model: backgroundModel, + model: bgProvider.model, temperature: 0.18, max_tokens: 420, messages: [ @@ -4781,11 +5068,12 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: .join("\n\n"), }, ], - tools: [ - { - type: "openrouter:web_search", - }, - ], + // OpenRouter's hosted web tool only exists on OpenRouter; on a + // direct OpenAI key the background model still reasons over the + // delegated request, just without hosted web search. + tools: bgProvider.supportsOpenRouterTools + ? [{ type: "openrouter:web_search" }] + : undefined, } as any); const responseMessage = (modelResponse.choices[0]?.message as Record) || diff --git a/server/learner-store.ts b/server/learner-store.ts index a84bc00..9f18412 100644 --- a/server/learner-store.ts +++ b/server/learner-store.ts @@ -50,6 +50,7 @@ type StoreDocumentInput = { size: number; sourcePath: string; extractedText: string; + pages?: { page_num: number; text: string }[]; classification?: string; extractionMode?: string; totalPages?: number; @@ -141,10 +142,40 @@ export class LearnerStore { return userDir; } + // Each distinct client-supplied userId opens (and, previously, never closed) + // its own WAL SQLite handle. An unauthenticated caller looping distinct ids + // could therefore exhaust file descriptors / process memory. Bound the number + // of concurrently open handles with a simple LRU: re-inserting a key moves it + // to the most-recently-used end of the Map, and the oldest is closed once the + // cap is exceeded. Handles reopen lazily, so eviction is transparent. + private static readonly MAX_OPEN_DATABASES = Math.max( + 4, + Number(process.env.LEARNINGAI_MAX_OPEN_DBS || 64), + ); + + private evictIdleDatabases() { + while (this.databases.size > LearnerStore.MAX_OPEN_DATABASES) { + const oldestKey = this.databases.keys().next().value; + if (oldestKey === undefined) break; + const oldestDb = this.databases.get(oldestKey); + this.databases.delete(oldestKey); + try { + oldestDb?.close(); + } catch { + // The handle is being discarded regardless; ignore close races. + } + } + } + private dbFor(userIdInput: unknown) { const userId = normalizeLearnerUserId(userIdInput); const cached = this.databases.get(userId); - if (cached) return cached; + if (cached) { + // Mark as most-recently-used. + this.databases.delete(userId); + this.databases.set(userId, cached); + return cached; + } const dbPath = path.join(this.getUserDir(userId), "brain.sqlite"); const Database = loadSqliteDatabase(); const db = new Database(dbPath); @@ -203,6 +234,7 @@ export class LearnerStore { ON background_tasks (user_id, request_id, updated_at); `); this.databases.set(userId, db); + this.evictIdleDatabases(); return db; } @@ -236,10 +268,23 @@ export class LearnerStore { "extracted-text", `${documentSegment}.txt`, ); + const pagesRelativePath = path.join( + "extracted-text", + `${documentSegment}.pages.json`, + ); const pdfPath = path.join(userDir, pdfRelativePath); const textPath = path.join(userDir, textRelativePath); + const pagesPath = path.join(userDir, pagesRelativePath); fs.copyFileSync(input.sourcePath, pdfPath); fs.writeFileSync(textPath, input.extractedText || "", "utf8"); + // Persist the page-indexed text next to the flat extracted text so the + // tutor can be handed the exact page the learner is viewing. + if (Array.isArray(input.pages) && input.pages.length > 0) { + fs.writeFileSync(pagesPath, JSON.stringify(input.pages), "utf8"); + } else if (fs.existsSync(pagesPath)) { + // A re-ingest with no page map should not leave a stale one behind. + fs.rmSync(pagesPath, { force: true }); + } const now = Date.now(); const db = this.dbFor(userId); db.prepare( @@ -329,6 +374,65 @@ export class LearnerStore { return fs.readFileSync(document.textPath, "utf8"); } + // Return the extracted text of a single page (1-based, matching the reader's + // page number) plus the immediate neighbours for local context. Falls back + // to an empty result when no page map was stored for the document. + readDocumentPageText( + userIdInput: unknown, + documentIdInput: unknown, + pageNumber: number, + options: { neighborRadius?: number } = {}, + ): { + page: number; + totalPages: number; + pageText: string; + neighborText: string; + } | null { + const userId = normalizeLearnerUserId(userIdInput); + const documentId = String(documentIdInput || "").trim(); + if (!documentId) return null; + const documentSegment = sanitizePathSegment(documentId, "document"); + const pagesPath = path.join( + this.getUserDir(userId), + "extracted-text", + `${documentSegment}.pages.json`, + ); + if (!fs.existsSync(pagesPath)) return null; + let pages: { page_num: number; text: string }[]; + try { + pages = JSON.parse(fs.readFileSync(pagesPath, "utf8")); + } catch { + return null; + } + if (!Array.isArray(pages) || pages.length === 0) return null; + // Reader page numbers are 1-based; stored page_num is 0-based. + const targetIndex = Math.max( + 0, + Math.min(pages.length - 1, Math.round(pageNumber) - 1), + ); + const radius = Math.max(0, options.neighborRadius ?? 1); + const byPageNum = new Map(); + for (const entry of pages) { + byPageNum.set(Number(entry.page_num), String(entry.text || "")); + } + const target = pages[targetIndex]; + const targetPageNum = Number(target.page_num); + const neighborParts: string[] = []; + for (let offset = -radius; offset <= radius; offset += 1) { + if (offset === 0) continue; + const neighbor = byPageNum.get(targetPageNum + offset); + if (neighbor && neighbor.trim()) { + neighborParts.push(`[page ${targetPageNum + offset + 1}] ${neighbor}`); + } + } + return { + page: targetPageNum + 1, + totalPages: pages.length, + pageText: String(target.text || ""), + neighborText: neighborParts.join("\n\n"), + }; + } + copyMigrationRecords(records: MigrationRecordInput[]) { let copied = 0; const now = Date.now(); diff --git a/server/web-search.ts b/server/web-search.ts index 9a6214b..02eeae4 100644 --- a/server/web-search.ts +++ b/server/web-search.ts @@ -34,11 +34,28 @@ const SERPER_ENDPOINTS: Record = { const CACHE_TTL_MS = 10 * 60 * 1000; const REQUEST_TIMEOUT_MS = 8000; const MAX_ATTEMPTS = 2; +// Cap the number of cached queries so a long-lived host doesn't grow this Map +// once per unique (mode, key, query, maxResults) combination forever. +const MAX_CACHE_ENTRIES = 500; const cache = new Map< string, { expiresAt: number; results: NormalizedWebSource[] } >(); +// Drop expired entries, then enforce the size cap by evicting oldest-first +// (Map preserves insertion order). Called on every write. +const pruneCache = () => { + const now = Date.now(); + for (const [key, entry] of cache) { + if (entry.expiresAt <= now) cache.delete(key); + } + while (cache.size > MAX_CACHE_ENTRIES) { + const oldest = cache.keys().next().value; + if (oldest === undefined) break; + cache.delete(oldest); + } +}; + const abortError = () => new DOMException("The operation was aborted.", "AbortError"); @@ -235,6 +252,7 @@ export async function searchSerper( const cacheKey = `${mode}:${apiKeyFingerprint}:${query.toLowerCase()}:${maxResults}`; const cached = cache.get(cacheKey); if (cached && cached.expiresAt > Date.now()) return cached.results; + if (cached) cache.delete(cacheKey); if (options.signal?.aborted) throw abortError(); @@ -263,6 +281,7 @@ export async function searchSerper( const payload = await response.json(); const results = normalizeRows(payload, mode, maxResults); cache.set(cacheKey, { expiresAt: Date.now() + CACHE_TTL_MS, results }); + pruneCache(); return results; } catch (error) { if (options.signal?.aborted) throw error; diff --git a/src/components/ChatPanel.tsx b/src/components/ChatPanel.tsx index 901e121..8525107 100644 --- a/src/components/ChatPanel.tsx +++ b/src/components/ChatPanel.tsx @@ -89,8 +89,14 @@ import { buildBrainContextPacket } from "../memory/brain.context"; import { buildVoiceFunctionCallResponse, parseVoiceFunctionArguments, + VOICE_AGENT_TOOL_DEFINITIONS, type VoiceAgentFunctionCall, } from "../lib/voiceAgentTools"; +import { + startRealtimeSession, + type RealtimeSessionHandle, + type RealtimeToolDefinition, +} from "../lib/realtimeVoice"; import { chatTitleFromMessageSet, flattenChatMessagesForPrompt, @@ -293,48 +299,51 @@ const loadMermaid = () => { const mermaid = module.default; mermaid.initialize({ startOnLoad: false, + // strict sanitizes model-generated diagram source (DOMPurify) and + // blocks in-diagram click/javascript directives — the diagram text is + // influenced by untrusted PDF/web content, so "loose" was an XSS path. + securityLevel: "strict", theme: "base", - securityLevel: "loose", fontFamily: - "'Geist Sans', Inter, 'Hiragino Sans', 'Yu Gothic UI', 'Noto Sans JP', system-ui, sans-serif", + "'Geist', 'Geist Sans', Inter, 'Hiragino Sans', 'Yu Gothic UI', 'Noto Sans JP', system-ui, sans-serif", themeVariables: { background: "transparent", fontSize: "14px", - // Node surfaces: one calm zinc family instead of mermaid's default - // olive/pink mix, with the app's orange reserved for focus states. - primaryColor: "#1f1f23", - primaryTextColor: "#f4f4f5", - primaryBorderColor: "#4b4b52", - secondaryColor: "#2a2a30", - secondaryTextColor: "#e4e4e7", - secondaryBorderColor: "#4b4b52", - tertiaryColor: "#232327", - tertiaryTextColor: "#e4e4e7", - tertiaryBorderColor: "#3f3f46", - mainBkg: "#1f1f23", - nodeBorder: "#4b4b52", - nodeTextColor: "#f4f4f5", - textColor: "#d4d4d8", - titleColor: "#f4f4f5", - lineColor: "#8a8a93", - arrowheadColor: "#8a8a93", - edgeLabelBackground: "#0f0f11", - clusterBkg: "rgba(42,42,48,0.45)", - clusterBorder: "#3f3f46", + // Calmer zinc family, lifted for legible contrast on the near-black + // card. Orange stays reserved for the click-to-focus highlight. + primaryColor: "#26262c", + primaryTextColor: "#fafafa", + primaryBorderColor: "#6b6b76", + secondaryColor: "#2f2f36", + secondaryTextColor: "#f4f4f5", + secondaryBorderColor: "#6b6b76", + tertiaryColor: "#28282e", + tertiaryTextColor: "#f4f4f5", + tertiaryBorderColor: "#57575f", + mainBkg: "#26262c", + nodeBorder: "#6b6b76", + nodeTextColor: "#fafafa", + textColor: "#e7e7ea", + titleColor: "#fafafa", + lineColor: "#a8a8b3", + arrowheadColor: "#a8a8b3", + edgeLabelBackground: "#18181b", + clusterBkg: "rgba(42,42,48,0.55)", + clusterBorder: "#57575f", noteBkgColor: "#292524", noteTextColor: "#fcd34d", - noteBorderColor: "#57534e", - actorBkg: "#1f1f23", - actorTextColor: "#f4f4f5", - actorBorder: "#4b4b52", - labelBoxBkgColor: "#1f1f23", - labelTextColor: "#f4f4f5", + noteBorderColor: "#78716c", + actorBkg: "#26262c", + actorTextColor: "#fafafa", + actorBorder: "#6b6b76", + labelBoxBkgColor: "#26262c", + labelTextColor: "#fafafa", }, flowchart: { curve: "basis", - padding: 14, - nodeSpacing: 48, - rankSpacing: 58, + padding: 16, + nodeSpacing: 50, + rankSpacing: 60, htmlLabels: true, useMaxWidth: true, }, @@ -409,316 +418,265 @@ const Mermaid = ({ const chartRef = useRef(null); const originalViewBoxRef = useRef(null); const viewBoxAnimationRef = useRef(null); - const [tourNodes, setTourNodes] = useState([]); - const [activeNodeIndex, setActiveNodeIndex] = useState(0); + const focusNodesRef = useRef([]); + const [status, setStatus] = useState<"loading" | "ready">("loading"); + const [focusIndex, setFocusIndex] = useState(null); const isStage = variant === "stage"; + const prefersReducedMotion = () => + typeof window !== "undefined" && + typeof window.matchMedia === "function" && + window.matchMedia("(prefers-reduced-motion: reduce)").matches; + useEffect(() => { let cancelled = false; - if (!chartRef.current) return; - chartRef.current.textContent = ""; - originalViewBoxRef.current = null; + const container = chartRef.current; + if (!container) return; + if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } - setTourNodes([]); - setActiveNodeIndex(0); - - loadMermaid() - .then((mermaid) => - mermaid.render( - `mermaid-${Math.random().toString(36).substring(7)}`, - chart, - ), - ) - .then((res) => { - if (cancelled || !chartRef.current) return; - chartRef.current.innerHTML = res.svg; - const svg = chartRef.current.querySelector("svg"); - if (!svg) return; - svg.setAttribute("role", "img"); - svg.setAttribute("aria-label", "Mermaid diagram with focus tour"); - svg.setAttribute("preserveAspectRatio", "xMidYMid meet"); - svg.removeAttribute("width"); - svg.removeAttribute("height"); - svg.style.display = "block"; - svg.style.width = "100%"; - svg.style.maxWidth = "100%"; - svg.style.maxHeight = isStage ? "72vh" : "70dvh"; - svg.style.minWidth = "0"; - svg.style.height = "auto"; - svg.style.margin = "0 auto"; - if (!svg.getAttribute("viewBox")) { + focusNodesRef.current = []; + setFocusIndex(null); + + // Debounce so streaming tokens don't render on every keystroke, and only + // render once the source PARSES cleanly. This is what stops half-finished + // ```mermaid fences from flashing raw parser errors: while the block is + // still streaming (or genuinely invalid) we simply leave the last good + // diagram / skeleton in place instead of dumping an error string. + const timer = window.setTimeout(() => { + loadMermaid() + .then(async (mermaid) => { + if (cancelled) return; + let parseable = false; try { - const bounds = (svg as unknown as SVGGraphicsElement).getBBox(); - svg.setAttribute( - "viewBox", - `${bounds.x} ${bounds.y} ${Math.max(bounds.width, 1)} ${Math.max(bounds.height, 1)}`, + parseable = Boolean( + await mermaid.parse(chart, { suppressErrors: true }), ); - } catch {} - } - originalViewBoxRef.current = parseMermaidViewBox(svg); - const nodes = collectMermaidTourNodes(svg); - nodes.forEach((node, index) => { - node.setAttribute("data-mermaid-tour-node", "true"); - node.setAttribute("data-mermaid-tour-index", String(index)); - node.setAttribute("tabindex", "0"); - node.addEventListener("click", () => setActiveNodeIndex(index)); - }); - setTourNodes( - nodes.map((node, index) => { + } catch { + parseable = false; + } + if (!parseable) return; + const res = await mermaid.render( + `mermaid-${Math.random().toString(36).slice(2)}`, + chart, + ); + if (cancelled || !chartRef.current) return; + chartRef.current.innerHTML = res.svg; + const svg = chartRef.current.querySelector("svg"); + if (!svg) return; + svg.setAttribute("role", "img"); + svg.setAttribute("aria-label", "Diagram"); + svg.setAttribute("preserveAspectRatio", "xMidYMid meet"); + svg.removeAttribute("width"); + svg.removeAttribute("height"); + svg.style.display = "block"; + svg.style.width = "100%"; + svg.style.maxWidth = "100%"; + svg.style.maxHeight = isStage ? "72vh" : "70dvh"; + svg.style.minWidth = "0"; + svg.style.height = "auto"; + svg.style.margin = "0 auto"; + if (!svg.getAttribute("viewBox")) { + try { + const bounds = (svg as unknown as SVGGraphicsElement).getBBox(); + svg.setAttribute( + "viewBox", + `${bounds.x} ${bounds.y} ${Math.max(bounds.width, 1)} ${Math.max(bounds.height, 1)}`, + ); + } catch {} + } + originalViewBoxRef.current = parseMermaidViewBox(svg); + // Optional click-to-focus: clicking a node zooms to it. Nodes stay at + // full opacity by default — no perpetual dimming, no auto-panning. + const nodes = collectMermaidTourNodes(svg); + focusNodesRef.current = nodes; + nodes.forEach((node, index) => { + node.setAttribute("data-mermaid-node", "true"); + node.setAttribute("tabindex", "0"); + node.setAttribute("role", "button"); const label = cleanMermaidTourLabel(node.textContent); - return label || `Step ${index + 1}`; - }), - ); - }) - .catch((error) => { - console.warn("Mermaid error", error); - if (!cancelled && chartRef.current) { - chartRef.current.textContent = - error instanceof Error ? error.message : String(error); - setTourNodes([]); - } - }); + if (label) node.setAttribute("aria-label", `Focus ${label}`); + node.style.cursor = "zoom-in"; + const toggleFocus = () => + setFocusIndex((current) => (current === index ? null : index)); + node.addEventListener("click", toggleFocus); + node.addEventListener("keydown", (event) => { + const key = (event as KeyboardEvent).key; + if (key === "Enter" || key === " ") { + event.preventDefault(); + toggleFocus(); + } + }); + }); + setStatus("ready"); + }) + .catch((error) => { + console.warn("Mermaid render error", error); + }); + }, 120); return () => { cancelled = true; + window.clearTimeout(timer); if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } }; - }, [chart]); - - useEffect(() => { - if (tourNodes.length <= 1) return; - const reduceMotion = window.matchMedia( - "(prefers-reduced-motion: reduce)", - ).matches; - if (reduceMotion) return; - const timer = window.setInterval( - () => { - setActiveNodeIndex((current) => (current + 1) % tourNodes.length); - }, - isStage ? 5600 : 4400, - ); - return () => window.clearInterval(timer); - }, [isStage, tourNodes.length]); + }, [chart, isStage]); + // Animate the viewBox to the focused node, or back to the full diagram when + // focus is cleared. Focus is user-initiated (click/enter) only. useEffect(() => { const container = chartRef.current; const svg = container?.querySelector("svg"); - if (!container || !svg || tourNodes.length <= 0) return; - const nodes = Array.from( - svg.querySelectorAll("[data-mermaid-tour-node='true']"), - ); - nodes.forEach((node) => node.removeAttribute("data-mermaid-active")); - const activeNode = nodes[activeNodeIndex]; - if (!activeNode) return; - activeNode.setAttribute("data-mermaid-active", "true"); const original = originalViewBoxRef.current; - if (!original) return; + if (!svg || !original) return; if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } - // Focus by animating the SVG viewBox rather than a CSS transform. The - // viewBox is in diagram coordinates, so the pan/zoom lands exactly on the - // node (no clamped-pixel drift, no mid-transition measurement races) and - // the diagram can never escape its container. - const current = parseMermaidViewBox(svg) || original; - const svgRect = svg.getBoundingClientRect(); - const nodeRect = activeNode.getBoundingClientRect(); - if (svgRect.width < 1 || svgRect.height < 1) return; - // Invert the xMidYMid-meet mapping to express the node in viewBox coords. - const renderScale = Math.min( - svgRect.width / current[2], - svgRect.height / current[3], - ); - const contentLeft = - svgRect.left + (svgRect.width - current[2] * renderScale) / 2; - const contentTop = - svgRect.top + (svgRect.height - current[3] * renderScale) / 2; - const nodeX = current[0] + (nodeRect.left - contentLeft) / renderScale; - const nodeY = current[1] + (nodeRect.top - contentTop) / renderScale; - const nodeW = nodeRect.width / renderScale; - const nodeH = nodeRect.height / renderScale; - - const [origX, origY, origW, origH] = original; - const aspect = origW / origH; - // Zoom so the node fills roughly a third of the frame, but never zoom in - // past 2x or out past the full diagram. - let targetW = Math.min( - origW, - Math.max(nodeW * (isStage ? 2.6 : 3.1), origW * 0.5), - ); - let targetH = targetW / aspect; - if (targetH < nodeH * 1.7) { - targetH = Math.min(origH, nodeH * 1.7); - targetW = targetH * aspect; + const nodes = focusNodesRef.current; + nodes.forEach((node) => node.removeAttribute("data-mermaid-active")); + + let target: MermaidViewBox = original; + if (focusIndex !== null && nodes[focusIndex]) { + const activeNode = nodes[focusIndex]; + activeNode.setAttribute("data-mermaid-active", "true"); + // Focus by animating the SVG viewBox (diagram coordinates) so the pan/zoom + // lands exactly on the node and the diagram can never escape its frame. + const current = parseMermaidViewBox(svg) || original; + const svgRect = svg.getBoundingClientRect(); + const nodeRect = activeNode.getBoundingClientRect(); + if (svgRect.width >= 1 && svgRect.height >= 1) { + const renderScale = Math.min( + svgRect.width / current[2], + svgRect.height / current[3], + ); + const contentLeft = + svgRect.left + (svgRect.width - current[2] * renderScale) / 2; + const contentTop = + svgRect.top + (svgRect.height - current[3] * renderScale) / 2; + const nodeX = current[0] + (nodeRect.left - contentLeft) / renderScale; + const nodeY = current[1] + (nodeRect.top - contentTop) / renderScale; + const nodeW = nodeRect.width / renderScale; + const nodeH = nodeRect.height / renderScale; + const [origX, origY, origW, origH] = original; + const aspect = origW / origH; + let targetW = Math.min( + origW, + Math.max(nodeW * (isStage ? 2.6 : 3.1), origW * 0.45), + ); + let targetH = targetW / aspect; + if (targetH < nodeH * 1.7) { + targetH = Math.min(origH, nodeH * 1.7); + targetW = targetH * aspect; + } + let targetX = nodeX + nodeW / 2 - targetW / 2; + let targetY = nodeY + nodeH / 2 - targetH / 2; + targetX = Math.max(origX, Math.min(origX + origW - targetW, targetX)); + targetY = Math.max(origY, Math.min(origY + origH - targetH, targetY)); + target = [targetX, targetY, targetW, targetH]; + } } - let targetX = nodeX + nodeW / 2 - targetW / 2; - let targetY = nodeY + nodeH / 2 - targetH / 2; - targetX = Math.max(origX, Math.min(origX + origW - targetW, targetX)); - targetY = Math.max(origY, Math.min(origY + origH - targetH, targetY)); - const target: MermaidViewBox = [targetX, targetY, targetW, targetH]; const applyViewBox = (box: MermaidViewBox) => svg.setAttribute("viewBox", box.map((v) => v.toFixed(2)).join(" ")); - - if (window.matchMedia("(prefers-reduced-motion: reduce)").matches) { + const from = parseMermaidViewBox(svg) || original; + if (prefersReducedMotion()) { applyViewBox(target); return; } - - const durationMs = isStage ? 1050 : 800; + const durationMs = isStage ? 700 : 560; const startedAt = performance.now(); - const from = current; const easeInOutCubic = (t: number) => t < 0.5 ? 4 * t * t * t : 1 - Math.pow(-2 * t + 2, 3) / 2; const tick = (now: number) => { const progress = Math.min(1, (now - startedAt) / durationMs); const eased = easeInOutCubic(progress); applyViewBox( - from.map((value, i) => value + (target[i] - value) * eased) as MermaidViewBox, + from.map( + (value, i) => value + (target[i] - value) * eased, + ) as MermaidViewBox, ); viewBoxAnimationRef.current = progress < 1 ? requestAnimationFrame(tick) : null; }; viewBoxAnimationRef.current = requestAnimationFrame(tick); - }, [activeNodeIndex, isStage, tourNodes.length]); - - const activeTourLabel = tourNodes[activeNodeIndex]; + }, [focusIndex, isStage]); return (
- {tourNodes.length > 1 && !isStage && ( -
- -
- - {activeTourLabel || `Step ${activeNodeIndex + 1}`} - - - {tourNodes.map((node, index) => ( - - ))} - + {status === "loading" && ( +
+
+ + Rendering diagram…
-
)} + {status === "ready" && focusIndex !== null && !isStage && ( + + )}
); }; @@ -3904,6 +3862,8 @@ export function ChatPanel({ (state) => state.betaProofTrafficApproval, ); const activeDocumentId = useStore((state) => state.activeDocumentId); + const pdfPage = useStore((state) => state.pdfPage); + const pdfTotalPages = useStore((state) => state.pdfTotalPages); const ttsVoice = useStore((state) => state.ttsVoice); const misoTtsApiUrl = useStore((state) => state.misoTtsApiUrl); const setActiveView = useStore((state) => state.setActiveView); @@ -3959,9 +3919,15 @@ export function ChatPanel({ const [thinkingStep, setThinkingStep] = useState(0); const [serverOpenRouterReady, setServerOpenRouterReady] = useState(false); const [serverDeepgramReady, setServerDeepgramReady] = useState(false); - const voiceBrokerMode = - import.meta.env.VITE_VOICE_BROKER_MODE === "custom" ? "custom" : "deepgram"; - const usesCustomVoiceBroker = voiceBrokerMode === "custom"; + // Voice mode is a runtime setting now (no rebuild to switch paths). + const voiceMode = useStore((state) => state.voiceMode); + const usesCustomVoiceBroker = voiceMode === "deepgram-duplex"; + const usesOpenAiRealtime = voiceMode === "openai-realtime"; + const voiceBrokerMode = usesOpenAiRealtime + ? "openai-realtime" + : usesCustomVoiceBroker + ? "custom" + : "deepgram"; const voiceBrokerTtsModel = import.meta.env.VITE_VOICE_BROKER_TTS_MODEL || "aura-2-thalia-en"; const usesBrowserVoiceTts = @@ -4064,6 +4030,7 @@ export function ChatPanel({ string | null >(null); const wsRef = useRef(null); + const realtimeHandleRef = useRef(null); const mediaStreamRef = useRef(null); const audioContextRef = useRef(null); const processorRef = useRef(null); @@ -5122,6 +5089,17 @@ export function ChatPanel({ cancelAnimationFrame(outputRafRef.current); outputRafRef.current = null; } + // OpenAI Realtime (test mode) runs its own WebRTC peer connection; null the + // ref before stopping so the module's "closed" callback doesn't re-enter. + if (realtimeHandleRef.current) { + const realtime = realtimeHandleRef.current; + realtimeHandleRef.current = null; + try { + realtime.stop(); + } catch { + /* ignore */ + } + } if (wsRef.current) { wsRef.current.close(); wsRef.current = null; @@ -5235,6 +5213,19 @@ export function ChatPanel({ setVoiceState("idle"); }; + // Ensure a live voice session is fully torn down if ChatPanel unmounts while + // it is active — otherwise the mic MediaStream, AudioContext, ScriptProcessor + // node and voice WebSocket leak, and the microphone stays hot. stopVoice is + // not memoized, so route the unmount call through a ref to always invoke the + // latest version without re-registering this effect on every render. + const stopVoiceRef = useRef(stopVoice); + stopVoiceRef.current = stopVoice; + useEffect(() => { + return () => { + stopVoiceRef.current?.(); + }; + }, []); + const sendVoiceText = (text: string) => { const trimmed = text.trim(); const ws = wsRef.current; @@ -5896,7 +5887,160 @@ export function ChatPanel({ ], ); + // OpenAI Realtime (WebRTC) — the test/comparison-only voice path. Runs + // entirely browser<->OpenAI, so it needs no persistent server (and works on + // Vercel). Isolated from the default Deepgram duplex flow above. + const startRealtimeVoice = async () => { + if (activeBetaProofTrafficLocked) { + alertProofTrafficApprovalNeeded(); + return; + } + try { + endingRef.current = false; + voiceTurnsRef.current = []; + const sessionId = `voice-${Date.now()}`; + voiceSessionIdRef.current = sessionId; + voiceProofAttemptIdRef.current = activeBetaProofAttemptId || null; + voiceStartedAtRef.current = Date.now(); + voiceSessionCountedRef.current = false; + voiceSessionErrorRef.current = null; + recordVoiceAgentEvent({ + type: "session_started", + status: "started", + sessionId, + summary: `OpenAI Realtime (test) session starting for ${activeLearningBookTitle}.`, + metadata: { + language, + bookId: canonicalActiveBookId, + documentId: activeDocumentId, + voiceBrokerMode, + proofAttemptId: getVoiceProofAttemptId(), + }, + }); + recordVoiceModelRun("started", sessionId, { + phase: "session_started", + language, + }); + setMessages((prev) => [ + ...prev, + { + id: sessionId, + requestId: sessionId, + role: "assistant", + content: "", + isVoice: true, + voiceSession: { turns: [], startedAt: Date.now(), durationSeconds: 0 }, + }, + ]); + setVoiceState("listening"); + + const voiceContextPayload = await buildVoiceStudyContext().catch( + (): null => null, + ); + voiceStudyContextRef.current = voiceContextPayload; + const studyContext = voiceContextPayload?.studyContext || ""; + + const instructions = [ + "You are Tutor, a warm and concise spoken tutor in a live full-duplex voice conversation. Keep replies short and natural for speech. Never read markdown, code fences, or bracketed citations aloud.", + "Always reply in the language the learner is speaking.", + studyContext + ? `Learner context packet (local book, memory, selected text, and current document):\n${studyContext.slice(0, 8000)}` + : "No additional local context is attached.", + ] + .filter(Boolean) + .join("\n\n"); + + const tools: RealtimeToolDefinition[] = VOICE_AGENT_TOOL_DEFINITIONS.filter( + (tool) => + tool.name === "look_at_study_context" || tool.name === "web_search", + ).map((tool) => ({ + name: tool.name, + description: tool.description, + parameters: tool.parameters, + })); + + const onToolCall = async ( + name: string, + args: Record, + ): Promise => { + if (name === "look_at_study_context") { + return { + context: studyContext || "No additional local context is attached.", + }; + } + if (name === "web_search") { + const query = String((args as { query?: unknown }).query || "").trim(); + if (!query) return { error: "web_search requires a query." }; + try { + const response = await fetch("/api/web-search", { + method: "POST", + headers: { + "Content-Type": "application/json", + ...(serperApiKey ? { "X-Serper-API-Key": serperApiKey } : {}), + }, + body: JSON.stringify({ + query, + mode: "search", + maxResults: 5, + serperApiKey: serperApiKey || undefined, + }), + }); + const data = await response.json(); + return { sources: data?.sources || data?.results || [] }; + } catch (searchError) { + return { + error: + searchError instanceof Error + ? searchError.message + : "web_search failed.", + }; + } + } + return { note: `${name} is not available in the Realtime test mode.` }; + }; + + const handle = await startRealtimeSession({ + openAiKey: apiKey.trim(), + model: import.meta.env.VITE_OPENAI_REALTIME_MODEL || "gpt-realtime", + voice: "marin", + instructions, + tools, + onToolCall, + onUserTranscript: (text) => { + appendVoiceTurn("user", text); + }, + onAssistantTranscript: (text, done) => { + setVoiceCaption({ role: "assistant", text }); + if (done) appendVoiceTurn("assistant", text); + }, + onStateChange: (state) => { + if (state === "live") setVoiceState("listening"); + if (state === "closed" && realtimeHandleRef.current) { + realtimeHandleRef.current = null; + if (!endingRef.current) stopVoice(); + } + }, + onError: (message) => { + voiceSessionErrorRef.current = message; + console.warn("[Realtime] error:", message); + }, + }); + realtimeHandleRef.current = handle; + } catch (error) { + voiceSessionErrorRef.current = + error instanceof Error ? error.message : String(error); + alert( + `OpenAI Realtime voice (test mode) could not start: ${voiceSessionErrorRef.current}. Add an OpenAI API key in Settings, or enable a server key with ALLOW_SERVER_OPENAI_FALLBACK.`, + ); + stopVoice(); + } + }; + const startVoice = async () => { + if (usesOpenAiRealtime) { + await startRealtimeVoice(); + return; + } if (!usesCustomVoiceBroker && !hasDeepgramRuntimeKey) { alert( "Please configure your Deepgram API Key in settings or expose the local server fallback before using Voice features.", @@ -7165,6 +7309,12 @@ export function ChatPanel({ activeProject: activeLearningBook?.title || activeProject, activeBookId: canonicalActiveBookId, activeDocumentId, + // The page the learner is currently viewing, so the tutor can be told + // exactly what is on screen rather than always the document's opening. + currentPage: activeDocumentId ? pdfPage : undefined, + currentPageTotal: activeDocumentId + ? pdfTotalPages || undefined + : undefined, documentContexts: orderedBookDocuments.map((document) => ({ id: document.id, title: document.title, diff --git a/src/components/SettingsModal.tsx b/src/components/SettingsModal.tsx index dbd46bd..09be193 100644 --- a/src/components/SettingsModal.tsx +++ b/src/components/SettingsModal.tsx @@ -21,7 +21,7 @@ import { TimerReset, UserCog, } from "lucide-react"; -import { useStore } from "../store"; +import { useStore, type VoiceMode } from "../store"; import { useTranslation } from "../lib/translations"; import { useMotionPreference } from "../hooks/useMotionPreference"; import { @@ -488,6 +488,8 @@ export function SettingsButton() { setTtsVoice, misoTtsApiUrl, setMisoTtsApiUrl, + voiceMode, + setVoiceMode, aiModel, setAiModel, animationsEnabled, @@ -1052,6 +1054,38 @@ export function SettingsButton() {

+
+ + +

+ {voiceMode === "deepgram-duplex" + ? "Default. A fast foreground model keeps the conversation moving while a background model does the heavy lifting. Needs a Deepgram key plus an OpenAI or OpenRouter key; runs in this app's own server — no separate host." + : voiceMode === "deepgram-agent" + ? "A single Deepgram Voice Agent handles speech and reasoning end to end." + : "Test / comparison only — connects your browser straight to OpenAI's Realtime API over WebRTC for true full-duplex speech. Uses your OpenAI key and is billed at premium realtime rates."} +

+
+