From 4ffebe012ad4af3129f179a9775af0d381774489 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 23 Jul 2026 11:42:20 +0000 Subject: [PATCH 1/4] Audit fixes: security, memory leaks, correctness, and page-aware PDF context Security: - Gate /api/voice-agent WS upgrade with the same same-origin check as the custom broker (closes cross-site WebSocket hijack / Deepgram credit drain). - Reject document reads whose header identity and ?userId= query disagree. - Lower express.urlencoded limit from 100mb to 1mb (memory-DoS amplifier). - Add MISO_TTS_ALLOW_HEADER_URL switch to ignore client-supplied Miso URL. Memory leaks: - Tear down the voice session (mic/AudioContext/WebSocket) on ChatPanel unmount. - Bound open per-user SQLite handles with an LRU (was unbounded/never closed). - Evict expired/oversized web-search cache entries. Correctness: - Add a final synthesis turn when the tool-iteration budget is exhausted mid-tool-call, so the user no longer gets a blank reply; report honest model-turn counts. - Raise voice broker TTS deadline default 180ms -> 1500ms. - Call the reported gpt-4o-mini-tts model instead of silently substituting tts-1. - Broaden OpenRouter pricing key matching so non-openai models aren't $0. PDF page-aware context: - Extract page-indexed markdown via pymupdf4llm page_chunks (single pass). - Persist a per-page text map and add readDocumentPageText(). - Send the current reader page to /api/chat and inject a labeled "CURRENT PAGE N of M" block plus adjacent pages into the tutor prompt. Remove dead .eslintrc.json (no eslint installed; lint script runs tsc). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01Vqigtqdnq6RkLrGp4VNBBd --- .eslintrc.json | 19 --- scripts/classify_and_extract.py | 20 ++- server.ts | 209 +++++++++++++++++++++++++++++--- server/learner-store.ts | 106 +++++++++++++++- server/web-search.ts | 19 +++ src/components/ChatPanel.tsx | 21 ++++ 6 files changed, 358 insertions(+), 36 deletions(-) delete mode 100644 .eslintrc.json diff --git a/.eslintrc.json b/.eslintrc.json deleted file mode 100644 index 7e2b4f1..0000000 --- a/.eslintrc.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "parser": "@typescript-eslint/parser", - "plugins": ["@typescript-eslint", "react", "react-hooks"], - "extends": [ - "eslint:recommended", - "plugin:@typescript-eslint/recommended", - "plugin:react/recommended", - "plugin:react-hooks/recommended" - ], - "settings": { - "react": { - "version": "detect" - } - }, - "env": { - "browser": true, - "es2021": true - } -} diff --git a/scripts/classify_and_extract.py b/scripts/classify_and_extract.py index 5b273f1..76203e2 100644 --- a/scripts/classify_and_extract.py +++ b/scripts/classify_and_extract.py @@ -69,6 +69,7 @@ def main(): "total_pages": total_pages, "pages_with_text": pages_with_text, "content": "", + "pages": [], "images": [], "vision_page_limit": MAX_VISION_PAGES, } @@ -77,8 +78,23 @@ def main(): try: with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): import pymupdf4llm - md_text = pymupdf4llm.to_markdown(doc) - result["content"] = md_text + # Single-pass, page-indexed extraction. page_chunks=True returns + # one dict per page so we can hand the model the exact page the + # learner is looking at, and it is no slower than the plain + # whole-document call. + page_chunks = pymupdf4llm.to_markdown(doc, page_chunks=True) + pages = [] + content_parts = [] + for index, chunk in enumerate(page_chunks): + page_text = (chunk.get("text") if isinstance(chunk, dict) else str(chunk)) or "" + meta = chunk.get("metadata", {}) if isinstance(chunk, dict) else {} + page_num = meta.get("page", index) if isinstance(meta, dict) else index + pages.append({"page_num": int(page_num), "text": page_text}) + # Keep an explicit page marker in the joined markdown so the + # whole-document excerpt retains page boundaries. + content_parts.append(f"\n{page_text}") + result["pages"] = pages + result["content"] = "\n\n".join(content_parts) except Exception as e: doc.close() result["error"] = f"Failed to extract markdown: {str(e)}" diff --git a/server.ts b/server.ts index 5ccc872..40c5db8 100644 --- a/server.ts +++ b/server.ts @@ -79,9 +79,12 @@ const VOICE_BROKER_FAST_ACK = /^(1|true|yes|on)$/i.test( const VOICE_BROKER_FAST_ACK_TEXT = process.env.VOICE_BROKER_FAST_ACK_TEXT || "Okay."; const VOICE_BROKER_FAST_ACK_FILE = process.env.VOICE_BROKER_FAST_ACK_FILE || ""; +// Deepgram Aura / MisoTTS synthesis rarely returns in under ~180ms, so the old +// default aborted almost every utterance before audio arrived. 1500ms is a +// realistic upper bound that still fails fast on a genuinely stuck provider. const VOICE_BROKER_TTS_DEADLINE_MS = Math.max( 50, - Number(process.env.VOICE_BROKER_TTS_DEADLINE_MS || 180), + Number(process.env.VOICE_BROKER_TTS_DEADLINE_MS || 1500), ); const VOICE_BROKER_STT_MODEL = process.env.VOICE_BROKER_STT_MODEL || "nova-3"; const VOICE_BROKER_TTS_MODEL = @@ -178,7 +181,14 @@ const MISO_TTS_HEALTH_TIMEOUT_MS = 800; const VOICE_WS_BUFFER_HIGH_WATER_BYTES = 1_000_000; const VOICE_AGENT_MESSAGE_BUFFER_LIMIT = 80; +// The client can point TTS at a local Miso server via this header, which is +// convenient for local dev but lets a caller probe arbitrary loopback ports on +// a deployed host. Deployments can set MISO_TTS_ALLOW_HEADER_URL=false to ignore +// the header and use only the server-configured MISO_TTS_API_URL. Defaults to +// on to preserve the existing local workflow. +const ALLOW_MISO_HEADER_URL = process.env.MISO_TTS_ALLOW_HEADER_URL !== "false"; const readMisoTtsApiUrlOverride = (headers: IncomingHttpHeaders) => { + if (!ALLOW_MISO_HEADER_URL) return ""; const raw = headers["x-miso-tts-api-url"]; if (Array.isArray(raw)) return raw[0] || ""; return typeof raw === "string" ? raw : ""; @@ -626,10 +636,19 @@ const openRouterCost = ( inputTokens: number, outputTokens: number, ) => { + // Resolve pricing across provider-prefix variants so a model such as + // "anthropic/claude-..." or a bare "gpt-4o-mini" doesn't silently read as $0 + // when the pricing map keys it under a different prefix. + const bareModel = model.includes("/") + ? model.slice(model.indexOf("/") + 1) + : model; const modelPricing = pricing[model] || - pricing[model.replace(/^openai\//, "")] || - pricing[`openai/${model}`]; + pricing[bareModel] || + pricing[`openai/${bareModel}`] || + Object.entries(pricing).find( + ([key]) => key === model || key.endsWith(`/${bareModel}`), + )?.[1]; if (!modelPricing) return 0; return roundCost( inputTokens * modelPricing.prompt + outputTokens * modelPricing.completion, @@ -877,7 +896,10 @@ export async function createTutorServerApp( app.use(compression()); app.use(express.json({ limit: "10mb" })); - app.use(express.urlencoded({ limit: "100mb", extended: true })); + // Cap urlencoded bodies far below the previous 100mb: no route needs a large + // form body, and an oversized `extended` (qs) parse is a cheap memory-DoS + // amplifier. Large binary uploads go through multer, not this parser. + app.use(express.urlencoded({ limit: "1mb", extended: true })); const uploadDir = process.env.TUTOR_UPLOAD_DIR || @@ -1171,10 +1193,36 @@ export async function createTutorServerApp( res.json({ ok: true, ...result }); }); - app.get("/api/learner/documents/:documentId/file", (req, res) => { - const userId = normalizeLearnerUserId( - req.query.userId || learnerUserIdFromHeaders(req.headers), + // Local profiles are not real auth (see README): both the request header and + // the ?userId= query are client-supplied, so this cannot enforce true tenant + // isolation. What it *can* do is reject a request whose header identity and + // query identity explicitly disagree (a confused-deputy / URL-smuggling + // vector) while still allowing the header-less react-pdf `` + // fetch, which can only carry identity via the query string. Prefer the + // header when present; fall back to the query otherwise. + const resolveDocumentUserId = (req: express.Request): string | null => { + const headerRaw = + req.headers["x-learningai-user-id"] || + req.headers["X-LearningAI-User-Id"] || + req.headers["x-user-id"]; + const hasHeader = Boolean( + Array.isArray(headerRaw) ? headerRaw[0] : headerRaw, ); + const headerUser = learnerUserIdFromHeaders(req.headers); + const queryUser = req.query.userId + ? normalizeLearnerUserId(req.query.userId) + : ""; + if (hasHeader && queryUser && headerUser !== queryUser) { + return null; + } + return hasHeader ? headerUser : queryUser || headerUser; + }; + + app.get("/api/learner/documents/:documentId/file", (req, res) => { + const userId = resolveDocumentUserId(req); + if (!userId) { + return res.status(403).json({ error: "User identity mismatch." }); + } const document = learnerStore.getDocument(userId, req.params.documentId); if (!document || !fs.existsSync(document.filePath)) { return res.status(404).json({ error: "Document file not found." }); @@ -1184,9 +1232,10 @@ export async function createTutorServerApp( }); app.get("/api/learner/documents/:documentId/text", (req, res) => { - const userId = normalizeLearnerUserId( - req.query.userId || learnerUserIdFromHeaders(req.headers), - ); + const userId = resolveDocumentUserId(req); + if (!userId) { + return res.status(403).json({ error: "User identity mismatch." }); + } const document = learnerStore.getDocument(userId, req.params.documentId); if (!document) { return res.status(404).json({ error: "Document text not found." }); @@ -1387,6 +1436,17 @@ export async function createTutorServerApp( let extractedText = result.content || ""; + // Page-indexed text so the tutor can be told exactly what is on the + // page the learner is viewing. Seed from the pymupdf4llm page chunks; + // vision-OCR pages below are merged in by page number. + const pageTextMap = new Map(); + for (const page of Array.isArray(result.pages) ? result.pages : []) { + const pageNum = Number(page?.page_num); + if (Number.isFinite(pageNum)) { + pageTextMap.set(pageNum, String(page?.text || "")); + } + } + // If Scanned or Mixed, perform Vision Parsing on page images. if ( result.classification === "Scanned" || @@ -1426,7 +1486,13 @@ export async function createTutorServerApp( const pageText = response.choices[0]?.message?.content?.trim(); if (pageText) { - extractedText += `\n\n## OCR / Vision Page ${Number(img.page_num ?? 0) + 1}\n\n${pageText}`; + const pageNum = Number(img.page_num ?? 0); + extractedText += `\n\n## OCR / Vision Page ${pageNum + 1}\n\n${pageText}`; + const existing = pageTextMap.get(pageNum); + pageTextMap.set( + pageNum, + existing ? `${existing}\n\n${pageText}` : pageText, + ); } } } catch (visionError) { @@ -1435,6 +1501,10 @@ export async function createTutorServerApp( } } + const pages = Array.from(pageTextMap.entries()) + .sort((a, b) => a[0] - b[0]) + .map(([page_num, text]) => ({ page_num, text })); + let serverDocument = null; if (documentId && bookId) { try { @@ -1447,6 +1517,7 @@ export async function createTutorServerApp( size: req.file?.size || 0, sourcePath: filePath, extractedText, + pages, classification: result.classification, extractionMode: result.extraction_mode, totalPages: result.total_pages, @@ -2150,7 +2221,10 @@ Use a Markdown table when comparing 2+ things. Keep every section scannable; whi apiKey: openaiKey, }); const mp3 = await openai.audio.speech.create({ - model: "tts-1", + // Call the model actually reported below in X-Usage-Model rather + // than silently substituting tts-1, so usage/cost telemetry is + // truthful. gpt-4o-mini-tts is a current OpenAI speech model. + model: "gpt-4o-mini-tts", voice: "alloy", input: billedText, }); @@ -2585,7 +2659,32 @@ Use a Markdown table when comparing 2+ things. Keep every section scannable; whi serperApiKey: bodySerperKey, language, requestId: clientRequestId, + activeDocumentId: bodyActiveDocumentId, + currentPage: bodyCurrentPage, + currentPageTotal: bodyCurrentPageTotal, } = req.body; + // Resolve the extracted text of the page the learner is currently viewing + // so it can be injected as an explicit "CURRENT PAGE" block below. + const currentPageContext = (() => { + const documentId = + typeof bodyActiveDocumentId === "string" ? bodyActiveDocumentId : ""; + const pageNumber = Number(bodyCurrentPage); + if (!documentId || !Number.isFinite(pageNumber) || pageNumber < 1) { + return null; + } + try { + const userId = learnerUserIdFromHeaders(req.headers); + const page = learnerStore.readDocumentPageText( + userId, + documentId, + pageNumber, + ); + if (!page || !page.pageText.trim()) return null; + return page; + } catch { + return null; + } + })(); requestId = normalizeClientRequestId(clientRequestId) || requestId; const runtimeSettings = normalizeBrainRuntimeSettings(rawRuntimeSettings); const runtimeSettingsSnapshot = compactRuntimeSettings(runtimeSettings); @@ -2876,6 +2975,25 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: systemInstruction += `\n\n${memoryContext}`; } + // Tell the tutor exactly what is on the page the learner is looking at. + // This is the extracted text of the current reader page (plus immediate + // neighbours for continuity), so questions like "what's on this page?" are + // answered from the real page rather than the document's opening excerpt. + if (currentPageContext) { + const pageBody = currentPageContext.pageText + .replace(/\s+\n/g, "\n") + .trim() + .slice(0, 4000); + const neighborBody = currentPageContext.neighborText + .trim() + .slice(0, 2000); + systemInstruction += `\n\nCURRENT PAGE ${currentPageContext.page} of ${currentPageContext.totalPages} (the page the learner is viewing right now):\n${pageBody}`; + if (neighborBody) { + systemInstruction += `\n\nADJACENT PAGES (for continuity only):\n${neighborBody}`; + } + systemInstruction += `\n\nWhen the learner refers to "this page", "here", or "the current page", answer from the CURRENT PAGE text above.`; + } + if (currentPageImage && sourceMaterialRequest) { systemInstruction += `\n\nCURRENT PAGE IMAGE IS ATTACHED THROUGH THE look_at_current_page TOOL. For this source-material request, call look_at_current_page before answering and answer from the page image plus selected/library context. Do not use web_search unless the user explicitly asks for web search.`; } @@ -2924,6 +3042,15 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: let evaluatedAnswers: any[] = []; let iterations = 0; + // Count actual model completions (each streamed turn), including the + // final answer turn that `break`s out before iterations++ and the + // post-loop synthesis turn. This is the honest number for telemetry; + // `iterations` alone only counts tool-executing turns. + let modelTurns = 0; + // True once a turn has emitted tool calls and false again once a turn + // answers with text. If the loop exits with this still true, the budget + // was exhausted mid-tool-call and we owe the user a synthesis turn. + let lastTurnHadToolCalls = false; const MAX_ITERATIONS = runtimeSettings.toolIterationLimit; let finalContent = ""; @@ -3086,11 +3213,15 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: } } + modelTurns++; + if (!isToolCall) { + lastTurnHadToolCalls = false; break; // Done! } // We have tool calls + lastTurnHadToolCalls = true; sendEvent("status", { phase: "tool_execution" }); const validToolCalls = currentToolCalls.filter(Boolean); recordSystemActivity({ @@ -3582,6 +3713,46 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: } } + // If the loop exhausted its tool-iteration budget while the last model + // turn was still emitting tool calls, the tool results were appended to + // the message list but never turned into a spoken answer — the user + // would otherwise get a blank/truncated reply. Run one final synthesis + // turn with no tools so the accumulated tool results become text. + if (lastTurnHadToolCalls) { + try { + sendEvent("status", { phase: "synthesizing" }); + const synthesisStream: any = await openai.chat.completions.create({ + model: usedModelForUsage, + messages: formattedMessages as any, + stream: true, + stream_options: { include_usage: true } as any, + } as any); + modelTurns++; + for await (const chunk of synthesisStream) { + const usage = (chunk as any).usage; + if (usage) { + inputTokens = usage.prompt_tokens ?? inputTokens; + outputTokens = usage.completion_tokens ?? outputTokens; + usageEstimated = false; + } + const delta = chunk.choices?.[0]?.delta; + if (delta?.content) { + finalContent += delta.content; + sendEvent("chunk", { content: delta.content }); + } + } + } catch (synthesisError: any) { + recordSystemActivity({ + kind: "model", + status: "failed", + title: "Final synthesis turn failed", + detail: String(synthesisError?.message || synthesisError), + requestId, + phase: "tool_followup", + }); + } + } + if (inputTokens === 0 && outputTokens === 0) { inputTokens = estimateTokensFromText(formattedMessages); outputTokens = estimateTokensFromText(finalContent); @@ -3613,7 +3784,7 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: graphUpdates: graphUpdates.length, flashcards: flashcardsUpdates.length, evaluatedAnswers: evaluatedAnswers.length, - iterations: iterations + 1, + iterations: modelTurns, runtimeSettings: runtimeSettingsSnapshot, }); recordSystemActivity({ @@ -3636,7 +3807,7 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: flashcards: flashcardsUpdates.length, evaluatedAnswers: evaluatedAnswers.length, webSources: webSources.length, - iterations: iterations + 1, + iterations: modelTurns, runtimeSettings: runtimeSettingsSnapshot, }, }); @@ -3726,6 +3897,16 @@ IMPORTANT TOOL USAGE INSTRUCTIONS: wss.emit("connection", ws, request); }); } else if (pathname === "/api/voice-agent") { + // Gate the Deepgram Voice Agent upgrade with the same same-origin / + // debug-token check the custom broker uses. Without it, any external + // page could open this socket (cross-site WebSocket hijacking) and, if + // ALLOW_SERVER_DEEPGRAM_FALLBACK is enabled, drain the deployment's + // Deepgram credits or exhaust connections. + if (!isAuthorizedLocalVoiceBrokerRequest(request)) { + socket.write("HTTP/1.1 403 Forbidden\r\nConnection: close\r\n\r\n"); + socket.destroy(); + return; + } wss.handleUpgrade(request, socket, head, (ws) => { wss.emit("connection", ws, request); }); diff --git a/server/learner-store.ts b/server/learner-store.ts index a84bc00..9f18412 100644 --- a/server/learner-store.ts +++ b/server/learner-store.ts @@ -50,6 +50,7 @@ type StoreDocumentInput = { size: number; sourcePath: string; extractedText: string; + pages?: { page_num: number; text: string }[]; classification?: string; extractionMode?: string; totalPages?: number; @@ -141,10 +142,40 @@ export class LearnerStore { return userDir; } + // Each distinct client-supplied userId opens (and, previously, never closed) + // its own WAL SQLite handle. An unauthenticated caller looping distinct ids + // could therefore exhaust file descriptors / process memory. Bound the number + // of concurrently open handles with a simple LRU: re-inserting a key moves it + // to the most-recently-used end of the Map, and the oldest is closed once the + // cap is exceeded. Handles reopen lazily, so eviction is transparent. + private static readonly MAX_OPEN_DATABASES = Math.max( + 4, + Number(process.env.LEARNINGAI_MAX_OPEN_DBS || 64), + ); + + private evictIdleDatabases() { + while (this.databases.size > LearnerStore.MAX_OPEN_DATABASES) { + const oldestKey = this.databases.keys().next().value; + if (oldestKey === undefined) break; + const oldestDb = this.databases.get(oldestKey); + this.databases.delete(oldestKey); + try { + oldestDb?.close(); + } catch { + // The handle is being discarded regardless; ignore close races. + } + } + } + private dbFor(userIdInput: unknown) { const userId = normalizeLearnerUserId(userIdInput); const cached = this.databases.get(userId); - if (cached) return cached; + if (cached) { + // Mark as most-recently-used. + this.databases.delete(userId); + this.databases.set(userId, cached); + return cached; + } const dbPath = path.join(this.getUserDir(userId), "brain.sqlite"); const Database = loadSqliteDatabase(); const db = new Database(dbPath); @@ -203,6 +234,7 @@ export class LearnerStore { ON background_tasks (user_id, request_id, updated_at); `); this.databases.set(userId, db); + this.evictIdleDatabases(); return db; } @@ -236,10 +268,23 @@ export class LearnerStore { "extracted-text", `${documentSegment}.txt`, ); + const pagesRelativePath = path.join( + "extracted-text", + `${documentSegment}.pages.json`, + ); const pdfPath = path.join(userDir, pdfRelativePath); const textPath = path.join(userDir, textRelativePath); + const pagesPath = path.join(userDir, pagesRelativePath); fs.copyFileSync(input.sourcePath, pdfPath); fs.writeFileSync(textPath, input.extractedText || "", "utf8"); + // Persist the page-indexed text next to the flat extracted text so the + // tutor can be handed the exact page the learner is viewing. + if (Array.isArray(input.pages) && input.pages.length > 0) { + fs.writeFileSync(pagesPath, JSON.stringify(input.pages), "utf8"); + } else if (fs.existsSync(pagesPath)) { + // A re-ingest with no page map should not leave a stale one behind. + fs.rmSync(pagesPath, { force: true }); + } const now = Date.now(); const db = this.dbFor(userId); db.prepare( @@ -329,6 +374,65 @@ export class LearnerStore { return fs.readFileSync(document.textPath, "utf8"); } + // Return the extracted text of a single page (1-based, matching the reader's + // page number) plus the immediate neighbours for local context. Falls back + // to an empty result when no page map was stored for the document. + readDocumentPageText( + userIdInput: unknown, + documentIdInput: unknown, + pageNumber: number, + options: { neighborRadius?: number } = {}, + ): { + page: number; + totalPages: number; + pageText: string; + neighborText: string; + } | null { + const userId = normalizeLearnerUserId(userIdInput); + const documentId = String(documentIdInput || "").trim(); + if (!documentId) return null; + const documentSegment = sanitizePathSegment(documentId, "document"); + const pagesPath = path.join( + this.getUserDir(userId), + "extracted-text", + `${documentSegment}.pages.json`, + ); + if (!fs.existsSync(pagesPath)) return null; + let pages: { page_num: number; text: string }[]; + try { + pages = JSON.parse(fs.readFileSync(pagesPath, "utf8")); + } catch { + return null; + } + if (!Array.isArray(pages) || pages.length === 0) return null; + // Reader page numbers are 1-based; stored page_num is 0-based. + const targetIndex = Math.max( + 0, + Math.min(pages.length - 1, Math.round(pageNumber) - 1), + ); + const radius = Math.max(0, options.neighborRadius ?? 1); + const byPageNum = new Map(); + for (const entry of pages) { + byPageNum.set(Number(entry.page_num), String(entry.text || "")); + } + const target = pages[targetIndex]; + const targetPageNum = Number(target.page_num); + const neighborParts: string[] = []; + for (let offset = -radius; offset <= radius; offset += 1) { + if (offset === 0) continue; + const neighbor = byPageNum.get(targetPageNum + offset); + if (neighbor && neighbor.trim()) { + neighborParts.push(`[page ${targetPageNum + offset + 1}] ${neighbor}`); + } + } + return { + page: targetPageNum + 1, + totalPages: pages.length, + pageText: String(target.text || ""), + neighborText: neighborParts.join("\n\n"), + }; + } + copyMigrationRecords(records: MigrationRecordInput[]) { let copied = 0; const now = Date.now(); diff --git a/server/web-search.ts b/server/web-search.ts index 9a6214b..02eeae4 100644 --- a/server/web-search.ts +++ b/server/web-search.ts @@ -34,11 +34,28 @@ const SERPER_ENDPOINTS: Record = { const CACHE_TTL_MS = 10 * 60 * 1000; const REQUEST_TIMEOUT_MS = 8000; const MAX_ATTEMPTS = 2; +// Cap the number of cached queries so a long-lived host doesn't grow this Map +// once per unique (mode, key, query, maxResults) combination forever. +const MAX_CACHE_ENTRIES = 500; const cache = new Map< string, { expiresAt: number; results: NormalizedWebSource[] } >(); +// Drop expired entries, then enforce the size cap by evicting oldest-first +// (Map preserves insertion order). Called on every write. +const pruneCache = () => { + const now = Date.now(); + for (const [key, entry] of cache) { + if (entry.expiresAt <= now) cache.delete(key); + } + while (cache.size > MAX_CACHE_ENTRIES) { + const oldest = cache.keys().next().value; + if (oldest === undefined) break; + cache.delete(oldest); + } +}; + const abortError = () => new DOMException("The operation was aborted.", "AbortError"); @@ -235,6 +252,7 @@ export async function searchSerper( const cacheKey = `${mode}:${apiKeyFingerprint}:${query.toLowerCase()}:${maxResults}`; const cached = cache.get(cacheKey); if (cached && cached.expiresAt > Date.now()) return cached.results; + if (cached) cache.delete(cacheKey); if (options.signal?.aborted) throw abortError(); @@ -263,6 +281,7 @@ export async function searchSerper( const payload = await response.json(); const results = normalizeRows(payload, mode, maxResults); cache.set(cacheKey, { expiresAt: Date.now() + CACHE_TTL_MS, results }); + pruneCache(); return results; } catch (error) { if (options.signal?.aborted) throw error; diff --git a/src/components/ChatPanel.tsx b/src/components/ChatPanel.tsx index 901e121..4df5a9a 100644 --- a/src/components/ChatPanel.tsx +++ b/src/components/ChatPanel.tsx @@ -3904,6 +3904,8 @@ export function ChatPanel({ (state) => state.betaProofTrafficApproval, ); const activeDocumentId = useStore((state) => state.activeDocumentId); + const pdfPage = useStore((state) => state.pdfPage); + const pdfTotalPages = useStore((state) => state.pdfTotalPages); const ttsVoice = useStore((state) => state.ttsVoice); const misoTtsApiUrl = useStore((state) => state.misoTtsApiUrl); const setActiveView = useStore((state) => state.setActiveView); @@ -5235,6 +5237,19 @@ export function ChatPanel({ setVoiceState("idle"); }; + // Ensure a live voice session is fully torn down if ChatPanel unmounts while + // it is active — otherwise the mic MediaStream, AudioContext, ScriptProcessor + // node and voice WebSocket leak, and the microphone stays hot. stopVoice is + // not memoized, so route the unmount call through a ref to always invoke the + // latest version without re-registering this effect on every render. + const stopVoiceRef = useRef(stopVoice); + stopVoiceRef.current = stopVoice; + useEffect(() => { + return () => { + stopVoiceRef.current?.(); + }; + }, []); + const sendVoiceText = (text: string) => { const trimmed = text.trim(); const ws = wsRef.current; @@ -7165,6 +7180,12 @@ export function ChatPanel({ activeProject: activeLearningBook?.title || activeProject, activeBookId: canonicalActiveBookId, activeDocumentId, + // The page the learner is currently viewing, so the tutor can be told + // exactly what is on screen rather than always the document's opening. + currentPage: activeDocumentId ? pdfPage : undefined, + currentPageTotal: activeDocumentId + ? pdfTotalPages || undefined + : undefined, documentContexts: orderedBookDocuments.map((document) => ({ id: document.id, title: document.title, From 18de729520f331ac2e7f8809e89f8d4ef7020170 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 23 Jul 2026 11:52:56 +0000 Subject: [PATCH 2/4] Redesign Mermaid rendering and convert AdminView to the dark theme Mermaid (chat flowcharts): - Render only once the diagram source parses cleanly, so half-streamed ```mermaid fences no longer flash raw parser errors; show a quiet skeleton. - Remove the perpetual node dimming + auto-panning "focus tour" that greyed out and constantly moved the diagram; render it static and fully visible. - Make zoom optional (click / Enter to focus a node, "Reset view" to restore), honoring prefers-reduced-motion. - Switch securityLevel from "loose" to "strict" (diagram source is influenced by untrusted PDF/web content) and lift node/edge/text contrast; align font. AdminView: - Convert the light cream/serif diagnostics surface to the dark Cosmic Obsidian theme (surfaces, text, borders) to match Chat/Study/Analytics, while leaving RevisionView's intentional paper look untouched. StatusBadge is intentionally left as-is: its wrapper is unused in-app (only its currentColor icons are used, which adapt to dark) and its palette is pinned by tests. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01Vqigtqdnq6RkLrGp4VNBBd --- src/components/ChatPanel.tsx | 476 +++++----- src/views/AdminView.tsx | 1588 +++++++++++++++++----------------- 2 files changed, 1008 insertions(+), 1056 deletions(-) diff --git a/src/components/ChatPanel.tsx b/src/components/ChatPanel.tsx index 4df5a9a..5f6b81b 100644 --- a/src/components/ChatPanel.tsx +++ b/src/components/ChatPanel.tsx @@ -293,48 +293,51 @@ const loadMermaid = () => { const mermaid = module.default; mermaid.initialize({ startOnLoad: false, + // strict sanitizes model-generated diagram source (DOMPurify) and + // blocks in-diagram click/javascript directives — the diagram text is + // influenced by untrusted PDF/web content, so "loose" was an XSS path. + securityLevel: "strict", theme: "base", - securityLevel: "loose", fontFamily: - "'Geist Sans', Inter, 'Hiragino Sans', 'Yu Gothic UI', 'Noto Sans JP', system-ui, sans-serif", + "'Geist', 'Geist Sans', Inter, 'Hiragino Sans', 'Yu Gothic UI', 'Noto Sans JP', system-ui, sans-serif", themeVariables: { background: "transparent", fontSize: "14px", - // Node surfaces: one calm zinc family instead of mermaid's default - // olive/pink mix, with the app's orange reserved for focus states. - primaryColor: "#1f1f23", - primaryTextColor: "#f4f4f5", - primaryBorderColor: "#4b4b52", - secondaryColor: "#2a2a30", - secondaryTextColor: "#e4e4e7", - secondaryBorderColor: "#4b4b52", - tertiaryColor: "#232327", - tertiaryTextColor: "#e4e4e7", - tertiaryBorderColor: "#3f3f46", - mainBkg: "#1f1f23", - nodeBorder: "#4b4b52", - nodeTextColor: "#f4f4f5", - textColor: "#d4d4d8", - titleColor: "#f4f4f5", - lineColor: "#8a8a93", - arrowheadColor: "#8a8a93", - edgeLabelBackground: "#0f0f11", - clusterBkg: "rgba(42,42,48,0.45)", - clusterBorder: "#3f3f46", + // Calmer zinc family, lifted for legible contrast on the near-black + // card. Orange stays reserved for the click-to-focus highlight. + primaryColor: "#26262c", + primaryTextColor: "#fafafa", + primaryBorderColor: "#6b6b76", + secondaryColor: "#2f2f36", + secondaryTextColor: "#f4f4f5", + secondaryBorderColor: "#6b6b76", + tertiaryColor: "#28282e", + tertiaryTextColor: "#f4f4f5", + tertiaryBorderColor: "#57575f", + mainBkg: "#26262c", + nodeBorder: "#6b6b76", + nodeTextColor: "#fafafa", + textColor: "#e7e7ea", + titleColor: "#fafafa", + lineColor: "#a8a8b3", + arrowheadColor: "#a8a8b3", + edgeLabelBackground: "#18181b", + clusterBkg: "rgba(42,42,48,0.55)", + clusterBorder: "#57575f", noteBkgColor: "#292524", noteTextColor: "#fcd34d", - noteBorderColor: "#57534e", - actorBkg: "#1f1f23", - actorTextColor: "#f4f4f5", - actorBorder: "#4b4b52", - labelBoxBkgColor: "#1f1f23", - labelTextColor: "#f4f4f5", + noteBorderColor: "#78716c", + actorBkg: "#26262c", + actorTextColor: "#fafafa", + actorBorder: "#6b6b76", + labelBoxBkgColor: "#26262c", + labelTextColor: "#fafafa", }, flowchart: { curve: "basis", - padding: 14, - nodeSpacing: 48, - rankSpacing: 58, + padding: 16, + nodeSpacing: 50, + rankSpacing: 60, htmlLabels: true, useMaxWidth: true, }, @@ -409,316 +412,265 @@ const Mermaid = ({ const chartRef = useRef(null); const originalViewBoxRef = useRef(null); const viewBoxAnimationRef = useRef(null); - const [tourNodes, setTourNodes] = useState([]); - const [activeNodeIndex, setActiveNodeIndex] = useState(0); + const focusNodesRef = useRef([]); + const [status, setStatus] = useState<"loading" | "ready">("loading"); + const [focusIndex, setFocusIndex] = useState(null); const isStage = variant === "stage"; + const prefersReducedMotion = () => + typeof window !== "undefined" && + typeof window.matchMedia === "function" && + window.matchMedia("(prefers-reduced-motion: reduce)").matches; + useEffect(() => { let cancelled = false; - if (!chartRef.current) return; - chartRef.current.textContent = ""; - originalViewBoxRef.current = null; + const container = chartRef.current; + if (!container) return; + if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } - setTourNodes([]); - setActiveNodeIndex(0); - - loadMermaid() - .then((mermaid) => - mermaid.render( - `mermaid-${Math.random().toString(36).substring(7)}`, - chart, - ), - ) - .then((res) => { - if (cancelled || !chartRef.current) return; - chartRef.current.innerHTML = res.svg; - const svg = chartRef.current.querySelector("svg"); - if (!svg) return; - svg.setAttribute("role", "img"); - svg.setAttribute("aria-label", "Mermaid diagram with focus tour"); - svg.setAttribute("preserveAspectRatio", "xMidYMid meet"); - svg.removeAttribute("width"); - svg.removeAttribute("height"); - svg.style.display = "block"; - svg.style.width = "100%"; - svg.style.maxWidth = "100%"; - svg.style.maxHeight = isStage ? "72vh" : "70dvh"; - svg.style.minWidth = "0"; - svg.style.height = "auto"; - svg.style.margin = "0 auto"; - if (!svg.getAttribute("viewBox")) { + focusNodesRef.current = []; + setFocusIndex(null); + + // Debounce so streaming tokens don't render on every keystroke, and only + // render once the source PARSES cleanly. This is what stops half-finished + // ```mermaid fences from flashing raw parser errors: while the block is + // still streaming (or genuinely invalid) we simply leave the last good + // diagram / skeleton in place instead of dumping an error string. + const timer = window.setTimeout(() => { + loadMermaid() + .then(async (mermaid) => { + if (cancelled) return; + let parseable = false; try { - const bounds = (svg as unknown as SVGGraphicsElement).getBBox(); - svg.setAttribute( - "viewBox", - `${bounds.x} ${bounds.y} ${Math.max(bounds.width, 1)} ${Math.max(bounds.height, 1)}`, + parseable = Boolean( + await mermaid.parse(chart, { suppressErrors: true }), ); - } catch {} - } - originalViewBoxRef.current = parseMermaidViewBox(svg); - const nodes = collectMermaidTourNodes(svg); - nodes.forEach((node, index) => { - node.setAttribute("data-mermaid-tour-node", "true"); - node.setAttribute("data-mermaid-tour-index", String(index)); - node.setAttribute("tabindex", "0"); - node.addEventListener("click", () => setActiveNodeIndex(index)); - }); - setTourNodes( - nodes.map((node, index) => { + } catch { + parseable = false; + } + if (!parseable) return; + const res = await mermaid.render( + `mermaid-${Math.random().toString(36).slice(2)}`, + chart, + ); + if (cancelled || !chartRef.current) return; + chartRef.current.innerHTML = res.svg; + const svg = chartRef.current.querySelector("svg"); + if (!svg) return; + svg.setAttribute("role", "img"); + svg.setAttribute("aria-label", "Diagram"); + svg.setAttribute("preserveAspectRatio", "xMidYMid meet"); + svg.removeAttribute("width"); + svg.removeAttribute("height"); + svg.style.display = "block"; + svg.style.width = "100%"; + svg.style.maxWidth = "100%"; + svg.style.maxHeight = isStage ? "72vh" : "70dvh"; + svg.style.minWidth = "0"; + svg.style.height = "auto"; + svg.style.margin = "0 auto"; + if (!svg.getAttribute("viewBox")) { + try { + const bounds = (svg as unknown as SVGGraphicsElement).getBBox(); + svg.setAttribute( + "viewBox", + `${bounds.x} ${bounds.y} ${Math.max(bounds.width, 1)} ${Math.max(bounds.height, 1)}`, + ); + } catch {} + } + originalViewBoxRef.current = parseMermaidViewBox(svg); + // Optional click-to-focus: clicking a node zooms to it. Nodes stay at + // full opacity by default — no perpetual dimming, no auto-panning. + const nodes = collectMermaidTourNodes(svg); + focusNodesRef.current = nodes; + nodes.forEach((node, index) => { + node.setAttribute("data-mermaid-node", "true"); + node.setAttribute("tabindex", "0"); + node.setAttribute("role", "button"); const label = cleanMermaidTourLabel(node.textContent); - return label || `Step ${index + 1}`; - }), - ); - }) - .catch((error) => { - console.warn("Mermaid error", error); - if (!cancelled && chartRef.current) { - chartRef.current.textContent = - error instanceof Error ? error.message : String(error); - setTourNodes([]); - } - }); + if (label) node.setAttribute("aria-label", `Focus ${label}`); + node.style.cursor = "zoom-in"; + const toggleFocus = () => + setFocusIndex((current) => (current === index ? null : index)); + node.addEventListener("click", toggleFocus); + node.addEventListener("keydown", (event) => { + const key = (event as KeyboardEvent).key; + if (key === "Enter" || key === " ") { + event.preventDefault(); + toggleFocus(); + } + }); + }); + setStatus("ready"); + }) + .catch((error) => { + console.warn("Mermaid render error", error); + }); + }, 120); return () => { cancelled = true; + window.clearTimeout(timer); if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } }; - }, [chart]); - - useEffect(() => { - if (tourNodes.length <= 1) return; - const reduceMotion = window.matchMedia( - "(prefers-reduced-motion: reduce)", - ).matches; - if (reduceMotion) return; - const timer = window.setInterval( - () => { - setActiveNodeIndex((current) => (current + 1) % tourNodes.length); - }, - isStage ? 5600 : 4400, - ); - return () => window.clearInterval(timer); - }, [isStage, tourNodes.length]); + }, [chart, isStage]); + // Animate the viewBox to the focused node, or back to the full diagram when + // focus is cleared. Focus is user-initiated (click/enter) only. useEffect(() => { const container = chartRef.current; const svg = container?.querySelector("svg"); - if (!container || !svg || tourNodes.length <= 0) return; - const nodes = Array.from( - svg.querySelectorAll("[data-mermaid-tour-node='true']"), - ); - nodes.forEach((node) => node.removeAttribute("data-mermaid-active")); - const activeNode = nodes[activeNodeIndex]; - if (!activeNode) return; - activeNode.setAttribute("data-mermaid-active", "true"); const original = originalViewBoxRef.current; - if (!original) return; + if (!svg || !original) return; if (viewBoxAnimationRef.current !== null) { cancelAnimationFrame(viewBoxAnimationRef.current); viewBoxAnimationRef.current = null; } - // Focus by animating the SVG viewBox rather than a CSS transform. The - // viewBox is in diagram coordinates, so the pan/zoom lands exactly on the - // node (no clamped-pixel drift, no mid-transition measurement races) and - // the diagram can never escape its container. - const current = parseMermaidViewBox(svg) || original; - const svgRect = svg.getBoundingClientRect(); - const nodeRect = activeNode.getBoundingClientRect(); - if (svgRect.width < 1 || svgRect.height < 1) return; - // Invert the xMidYMid-meet mapping to express the node in viewBox coords. - const renderScale = Math.min( - svgRect.width / current[2], - svgRect.height / current[3], - ); - const contentLeft = - svgRect.left + (svgRect.width - current[2] * renderScale) / 2; - const contentTop = - svgRect.top + (svgRect.height - current[3] * renderScale) / 2; - const nodeX = current[0] + (nodeRect.left - contentLeft) / renderScale; - const nodeY = current[1] + (nodeRect.top - contentTop) / renderScale; - const nodeW = nodeRect.width / renderScale; - const nodeH = nodeRect.height / renderScale; - - const [origX, origY, origW, origH] = original; - const aspect = origW / origH; - // Zoom so the node fills roughly a third of the frame, but never zoom in - // past 2x or out past the full diagram. - let targetW = Math.min( - origW, - Math.max(nodeW * (isStage ? 2.6 : 3.1), origW * 0.5), - ); - let targetH = targetW / aspect; - if (targetH < nodeH * 1.7) { - targetH = Math.min(origH, nodeH * 1.7); - targetW = targetH * aspect; + const nodes = focusNodesRef.current; + nodes.forEach((node) => node.removeAttribute("data-mermaid-active")); + + let target: MermaidViewBox = original; + if (focusIndex !== null && nodes[focusIndex]) { + const activeNode = nodes[focusIndex]; + activeNode.setAttribute("data-mermaid-active", "true"); + // Focus by animating the SVG viewBox (diagram coordinates) so the pan/zoom + // lands exactly on the node and the diagram can never escape its frame. + const current = parseMermaidViewBox(svg) || original; + const svgRect = svg.getBoundingClientRect(); + const nodeRect = activeNode.getBoundingClientRect(); + if (svgRect.width >= 1 && svgRect.height >= 1) { + const renderScale = Math.min( + svgRect.width / current[2], + svgRect.height / current[3], + ); + const contentLeft = + svgRect.left + (svgRect.width - current[2] * renderScale) / 2; + const contentTop = + svgRect.top + (svgRect.height - current[3] * renderScale) / 2; + const nodeX = current[0] + (nodeRect.left - contentLeft) / renderScale; + const nodeY = current[1] + (nodeRect.top - contentTop) / renderScale; + const nodeW = nodeRect.width / renderScale; + const nodeH = nodeRect.height / renderScale; + const [origX, origY, origW, origH] = original; + const aspect = origW / origH; + let targetW = Math.min( + origW, + Math.max(nodeW * (isStage ? 2.6 : 3.1), origW * 0.45), + ); + let targetH = targetW / aspect; + if (targetH < nodeH * 1.7) { + targetH = Math.min(origH, nodeH * 1.7); + targetW = targetH * aspect; + } + let targetX = nodeX + nodeW / 2 - targetW / 2; + let targetY = nodeY + nodeH / 2 - targetH / 2; + targetX = Math.max(origX, Math.min(origX + origW - targetW, targetX)); + targetY = Math.max(origY, Math.min(origY + origH - targetH, targetY)); + target = [targetX, targetY, targetW, targetH]; + } } - let targetX = nodeX + nodeW / 2 - targetW / 2; - let targetY = nodeY + nodeH / 2 - targetH / 2; - targetX = Math.max(origX, Math.min(origX + origW - targetW, targetX)); - targetY = Math.max(origY, Math.min(origY + origH - targetH, targetY)); - const target: MermaidViewBox = [targetX, targetY, targetW, targetH]; const applyViewBox = (box: MermaidViewBox) => svg.setAttribute("viewBox", box.map((v) => v.toFixed(2)).join(" ")); - - if (window.matchMedia("(prefers-reduced-motion: reduce)").matches) { + const from = parseMermaidViewBox(svg) || original; + if (prefersReducedMotion()) { applyViewBox(target); return; } - - const durationMs = isStage ? 1050 : 800; + const durationMs = isStage ? 700 : 560; const startedAt = performance.now(); - const from = current; const easeInOutCubic = (t: number) => t < 0.5 ? 4 * t * t * t : 1 - Math.pow(-2 * t + 2, 3) / 2; const tick = (now: number) => { const progress = Math.min(1, (now - startedAt) / durationMs); const eased = easeInOutCubic(progress); applyViewBox( - from.map((value, i) => value + (target[i] - value) * eased) as MermaidViewBox, + from.map( + (value, i) => value + (target[i] - value) * eased, + ) as MermaidViewBox, ); viewBoxAnimationRef.current = progress < 1 ? requestAnimationFrame(tick) : null; }; viewBoxAnimationRef.current = requestAnimationFrame(tick); - }, [activeNodeIndex, isStage, tourNodes.length]); - - const activeTourLabel = tourNodes[activeNodeIndex]; + }, [focusIndex, isStage]); return (
- {tourNodes.length > 1 && !isStage && ( -
- -
- - {activeTourLabel || `Step ${activeNodeIndex + 1}`} - - - {tourNodes.map((node, index) => ( - - ))} - + {status === "loading" && ( +
+
+ + Rendering diagram…
-
)} + {status === "ready" && focusIndex !== null && !isStage && ( + + )}
); }; diff --git a/src/views/AdminView.tsx b/src/views/AdminView.tsx index dd5a48a..26bdd1f 100644 --- a/src/views/AdminView.tsx +++ b/src/views/AdminView.tsx @@ -297,7 +297,7 @@ const statusTone = (status: string) => { if (status === "deferred") return "border-violet-200 bg-violet-50 text-violet-700"; if (status === "dismissed") - return "border-zinc-300 bg-zinc-100 text-zinc-600"; + return "border-white/15 bg-white/[0.06] text-zinc-400"; return "border-blue-200 bg-blue-50 text-blue-700"; }; @@ -1716,7 +1716,7 @@ export function AdminView() { ? "border-green-200 bg-green-50 text-green-600" : serverConsoleStatus === "connecting" ? "border-blue-200 bg-blue-50 text-blue-600" - : "border-zinc-200 bg-zinc-50 text-zinc-500"; + : "border-white/10 bg-white/[0.04] text-zinc-400"; const activityLabel = activityStatus === "ready" ? "Live" @@ -1732,7 +1732,7 @@ export function AdminView() { ? "border-blue-200 bg-blue-50 text-blue-600" : activityStatus === "error" ? "border-red-200 bg-red-50 text-red-600" - : "border-zinc-200 bg-zinc-50 text-zinc-500"; + : "border-white/10 bg-white/[0.04] text-zinc-400"; const updateRuntimeSetting = ( key: K, value: BrainRuntimeSettings[K], @@ -2025,7 +2025,7 @@ export function AdminView() { ]); return ( -
+
{/* Subtle Paper Texture Overlay */}
{/* Sidebar Navigation */} -
+
@@ -2052,35 +2052,35 @@ export function AdminView() {