From a69e81c3ff299b5267baff40ebed37d56fd6b127 Mon Sep 17 00:00:00 2001 From: Jose Montes de Oca Date: Tue, 22 Sep 2026 14:27:25 -0400 Subject: [PATCH 1/4] fix: make every advertised Markdown alternate resolve Every docs page advertises a text/markdown alternate at .mdx, and six of the seven returned 404. The proxy built its rewrite patterns by concatenating docsRoute ("/") with the optional `{/*path}` group, which supplies its own slash, so the patterns only ever matched the root and double-slash paths. Strip the trailing slash from the prefix, the way upstream's Fumadocs CLI does for any base URL. Fixing the pattern also turns the negotiated rewrite into a catch-all, so the mapping now lives behind one exported helper that only negotiates on paths that look like docs pages. The root page advertises /index.mdx instead of /.mdx, and both aliases resolve. A new test walks every page and ties the advertised alternate, the proxy rewrite, and the /llms.mdx handler URL together, and the standalone smoke run now fetches the alternates over HTTP. --- __tests__/canonical.test.ts | 34 ++++++++------ __tests__/markdown-alternates.test.ts | 38 +++++++++++++++ __tests__/proxy.test.ts | 61 +++++++++++++++++++++++- lib/canonical.ts | 8 ++-- lib/shared.ts | 11 +++-- proxy.ts | 67 ++++++++++++++++++++++----- scripts/smoke-standalone.ts | 12 +++++ 7 files changed, 199 insertions(+), 32 deletions(-) create mode 100644 __tests__/markdown-alternates.test.ts diff --git a/__tests__/canonical.test.ts b/__tests__/canonical.test.ts index 1f9e925..b01fad8 100644 --- a/__tests__/canonical.test.ts +++ b/__tests__/canonical.test.ts @@ -37,34 +37,42 @@ describe("robotsContent", () => { }); describe("buildPageMetadata", () => { - it("emits absolute canonical URL for a nested docs path", () => { - const md = buildPageMetadata("/docs/get-started/install"); - expect(md.alternates?.canonical).toBe( - "https://docs.prose.md/docs/get-started/install", - ); + it("emits absolute canonical URL for a root-mounted docs path", () => { + const md = buildPageMetadata("/setup"); + expect(md.alternates?.canonical).toBe("https://docs.prose.md/setup"); }); - it("emits .mdx markdown alternate so Fumadocs's proxy can rewrite it", () => { - const md = buildPageMetadata("/docs/get-started/install"); - expect(md.alternates?.types?.["text/markdown"]).toBe( - "/docs/get-started/install.mdx", - ); + it("emits absolute canonical URL for the root page", () => { + const md = buildPageMetadata("/"); + expect(md.alternates?.canonical).toBe("https://docs.prose.md/"); + }); + + it("emits .mdx as the markdown alternate so the proxy can rewrite it", () => { + const md = buildPageMetadata("/setup"); + expect(md.alternates?.types?.["text/markdown"]).toBe("/setup.mdx"); + }); + + it("emits /index.mdx as the root page's markdown alternate", () => { + // Appending .mdx to "/" would advertise "/.mdx", which reads as a dotfile + // and only resolved by accident of the old rewrite pattern. + const md = buildPageMetadata("/"); + expect(md.alternates?.types?.["text/markdown"]).toBe("/index.mdx"); }); it("always emits /llms.txt as the text/plain alternate", () => { - const md = buildPageMetadata("/docs/anywhere"); + const md = buildPageMetadata("/anywhere"); expect(md.alternates?.types?.["text/plain"]).toBe("/llms.txt"); }); it("sets robots index/follow false in preview mode", () => { vi.stubEnv("DOCS_PREVIEW_MODE", "true"); - const md = buildPageMetadata("/docs/foo"); + const md = buildPageMetadata("/foo"); expect(md.robots).toEqual({ index: false, follow: false }); }); it("sets robots index/follow true when preview mode is off", () => { vi.stubEnv("DOCS_PREVIEW_MODE", "false"); - const md = buildPageMetadata("/docs/foo"); + const md = buildPageMetadata("/foo"); expect(md.robots).toEqual({ index: true, follow: true }); }); }); diff --git a/__tests__/markdown-alternates.test.ts b/__tests__/markdown-alternates.test.ts new file mode 100644 index 0000000..872912e --- /dev/null +++ b/__tests__/markdown-alternates.test.ts @@ -0,0 +1,38 @@ +// @vitest-environment node +// The proxy runs on the server; keep this in the same environment as +// proxy.test.ts so the helper sees exactly what production sees. +import { describe, expect, it } from "vitest"; +import { buildPageMetadata } from "../lib/canonical"; +import { getPageMarkdownUrl, source } from "../lib/source"; +import { resolveMarkdownRewrite } from "../proxy"; + +// Three layers have to agree on one string: buildPageMetadata advertises +// the alternate, resolveMarkdownRewrite maps it, and the /llms.mdx handler +// prerenders exactly the paths getPageMarkdownUrl produces. A page whose +// slug breaks the shape the proxy expects, or a change to any one layer, +// shows up here without a build. +describe("markdown alternates", () => { + it("advertise a URL the proxy rewrites to the page's prerendered Markdown", () => { + const pages = source.getPages(); + expect(pages.length).toBeGreaterThan(0); + + const mismatches: string[] = []; + for (const page of pages) { + const alternate = buildPageMetadata(page.url).alternates?.types?.[ + "text/markdown" + ]; + if (typeof alternate !== "string") { + mismatches.push(`${page.url}: no text/markdown alternate`); + continue; + } + const rewritten = resolveMarkdownRewrite(alternate, false); + const expected = getPageMarkdownUrl(page).url; + if (rewritten !== expected) { + mismatches.push( + `${page.url}: ${alternate} -> ${String(rewritten)}, expected ${expected}`, + ); + } + } + expect(mismatches).toEqual([]); + }); +}); diff --git a/__tests__/proxy.test.ts b/__tests__/proxy.test.ts index 4122185..cc6e8e4 100644 --- a/__tests__/proxy.test.ts +++ b/__tests__/proxy.test.ts @@ -3,7 +3,11 @@ // which the legacy-host redirect depends on. import { describe, expect, it } from 'vitest'; import { NextRequest } from 'next/server'; -import proxy, { hasEncodedPathname, resolveDocsHostRedirect } from '../proxy'; +import proxy, { + hasEncodedPathname, + resolveDocsHostRedirect, + resolveMarkdownRewrite, +} from '../proxy'; describe('resolveDocsHostRedirect', () => { it('redirects the legacy docs host to docs.prose.md', () => { @@ -41,7 +45,62 @@ describe('hasEncodedPathname', () => { }); }); +describe('resolveMarkdownRewrite', () => { + it('maps an advertised .mdx alternate to its prerendered Markdown', () => { + expect(resolveMarkdownRewrite('/setup.mdx', false)).toBe( + '/llms.mdx/setup/content.md', + ); + }); + + it('maps the advertised root alternate /index.mdx to the root Markdown', () => { + expect(resolveMarkdownRewrite('/index.mdx', false)).toBe( + '/llms.mdx/content.md', + ); + }); + + it('keeps the legacy /.mdx root alternate working', () => { + expect(resolveMarkdownRewrite('/.mdx', false)).toBe( + '/llms.mdx/content.md', + ); + }); + + it('no longer matches the old double-slash form', () => { + // Next 308s these before the proxy runs, so a match here would only + // ever have been reachable by accident. + expect(resolveMarkdownRewrite('//setup.mdx', false)).toBeNull(); + }); + + it('leaves a page URL alone without a Markdown Accept', () => { + expect(resolveMarkdownRewrite('/setup', false)).toBeNull(); + }); + + it('leaves /llms.txt alone without a Markdown Accept', () => { + expect(resolveMarkdownRewrite('/llms.txt', false)).toBeNull(); + }); +}); + describe('proxy', () => { + it('rewrites an advertised .mdx alternate to the Markdown route', () => { + const res = proxy(new NextRequest('https://docs.prose.md/setup.mdx')); + expect(res.headers.get('x-middleware-rewrite')).toBe( + 'https://docs.prose.md/llms.mdx/setup/content.md', + ); + expect(res.headers.get('x-middleware-next')).toBeNull(); + }); + + it('rewrites the root alternate /index.mdx to the root Markdown', () => { + const res = proxy(new NextRequest('https://docs.prose.md/index.mdx')); + expect(res.headers.get('x-middleware-rewrite')).toBe( + 'https://docs.prose.md/llms.mdx/content.md', + ); + }); + + it('rejects an encoded .mdx path before rewriting it', () => { + const res = proxy(new NextRequest('https://docs.prose.md/setup%2Emdx')); + expect(res.status).toBe(404); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + }); + it('returns 404 for a percent-encoded robots path', () => { const res = proxy(new NextRequest('https://docs.prose.md/robots%2Etxt')); expect(res.status).toBe(404); diff --git a/lib/canonical.ts b/lib/canonical.ts index 37ab1fe..aa1490d 100644 --- a/lib/canonical.ts +++ b/lib/canonical.ts @@ -28,8 +28,10 @@ export function robotsContent(): string | null { * plus the agent corpus at /llms.txt), and robots index/follow directives * gated by preview mode. * - * The markdown alternate uses the `.mdx` URL suffix because Fumadocs's - * proxy.ts rewrites `${docsRoute}{/*path}.mdx` to the markdown route. + * The markdown alternate uses the `.mdx` URL suffix because + * resolveMarkdownRewrite in proxy.ts maps `.mdx` to the + * prerendered /llms.mdx//content.md file. The root page has no slug, + * so it advertises /index.mdx, which the proxy maps explicitly. */ export function buildPageMetadata(path: string): Metadata { const preview = isPreviewMode(); @@ -37,7 +39,7 @@ export function buildPageMetadata(path: string): Metadata { alternates: { canonical: canonicalUrl(path), types: { - "text/markdown": `${path}.mdx`, + "text/markdown": path === "/" ? "/index.mdx" : `${path}.mdx`, "text/plain": "/llms.txt", }, }, diff --git a/lib/shared.ts b/lib/shared.ts index 2bc7bc9..046c75b 100644 --- a/lib/shared.ts +++ b/lib/shared.ts @@ -1,7 +1,12 @@ export const appName = "OpenProse"; -// Docs are mounted at the root: docs.openprose.ai/ rather than -// docs.openprose.ai/docs/. The (docs) route group at app/(docs)/ -// applies the DocsLayout to everything without adding a URL segment. +// Docs are mounted at the root: docs.prose.md/ rather than +// docs.prose.md/docs/. The catch-all page at app/[[...slug]]/ serves +// every page without adding a URL segment. +// +// docsRoute feeds both the Fumadocs loader (which normalizes a trailing +// slash away when building page.url) and the proxy's rewrite patterns +// (which strip it before concatenating). Keep it either "/" or a prefix +// without a trailing slash, such as "/docs". export const docsRoute = "/"; export const docsImageRoute = "/og"; export const docsContentRoute = "/llms.mdx"; diff --git a/proxy.ts b/proxy.ts index d0ab112..0bf1b9f 100644 --- a/proxy.ts +++ b/proxy.ts @@ -2,12 +2,18 @@ import { NextRequest, NextResponse } from 'next/server'; import { isMarkdownPreferred, rewritePath } from 'fumadocs-core/negotiation'; import { docsContentRoute, docsRoute } from '@/lib/shared'; +// The optional group `{/*path}` supplies its own leading slash, so the prefix +// in front of it must not end in one. With docsRoute = "/" the raw +// concatenation `/{/*path}` only ever matched "/" and double-slash paths, +// which Next collapses with a 308 before the proxy runs. +const docsPrefix = docsRoute.replace(/\/$/, ''); + const { rewrite: rewriteDocs } = rewritePath( - `${docsRoute}{/*path}`, + `${docsPrefix}{/*path}`, `${docsContentRoute}{/*path}/content.md`, ); const { rewrite: rewriteSuffix } = rewritePath( - `${docsRoute}{/*path}.mdx`, + `${docsPrefix}{/*path}.mdx`, `${docsContentRoute}{/*path}/content.md`, ); @@ -37,6 +43,47 @@ export function hasEncodedPathname(pathname: string): boolean { } } +// With the docs mounted at the root, the negotiated pattern matches every +// path, and isMarkdownPreferred is true for `Accept: text/plain` as well as +// `text/markdown`. Docs slugs are plain kebab-case words; every other route +// on the site (/robots.txt, /llms.txt, /llms.mdx/**, /og/**, /.well-known/**) +// carries a dot. A wrong guess here rewrites into a path the build never +// generated and 404s under `fallback: false`, never a wrong body. +function looksLikeDocsPage(pathname: string): boolean { + return !pathname.includes('.') && !pathname.startsWith('/_next'); +} + +// Returns the /llms.mdx path that serves a request as Markdown, or null. +// Three ways in: the root aliases (/index.mdx is advertised, /.mdx is what +// crawlers saw before), the `.mdx` alternate every other page +// advertises, and Accept negotiation on a page's HTML URL. +export function resolveMarkdownRewrite( + pathname: string, + prefersMarkdown: boolean, +): string | null { + // path-to-regexp's wildcard accepts an empty first segment, so `//setup.mdx` + // would compile to /llms.mdx//setup/content.md. Next collapses repeated + // slashes with a 308 before the proxy runs, so this is unreachable over + // HTTP; refuse it here so the helper never emits a path with an empty + // segment, which no prerendered file has. + if (pathname.includes('//')) return null; + + // The suffix pattern would send /index.mdx to /llms.mdx/index/content.md, + // which the build never generates; the root's Markdown has no slug segment. + if (pathname === '/index.mdx' || pathname === '/.mdx') { + return `${docsContentRoute}/content.md`; + } + + const suffixed = rewriteSuffix(pathname); + if (suffixed) return suffixed; + + if (prefersMarkdown && looksLikeDocsPage(pathname)) { + return rewriteDocs(pathname) || null; + } + + return null; +} + export default function proxy(request: NextRequest) { if (hasEncodedPathname(request.nextUrl.pathname)) { return new NextResponse(null, { status: 404 }); @@ -51,17 +98,13 @@ export default function proxy(request: NextRequest) { return NextResponse.redirect(hostRedirect, 301); } - const result = rewriteSuffix(request.nextUrl.pathname); - if (result) { - return NextResponse.rewrite(new URL(result, request.nextUrl)); - } - - if (isMarkdownPreferred(request)) { - const result = rewriteDocs(request.nextUrl.pathname); + const target = resolveMarkdownRewrite( + request.nextUrl.pathname, + isMarkdownPreferred(request), + ); - if (result) { - return NextResponse.rewrite(new URL(result, request.nextUrl)); - } + if (target) { + return NextResponse.rewrite(new URL(target, request.nextUrl)); } return NextResponse.next(); diff --git a/scripts/smoke-standalone.ts b/scripts/smoke-standalone.ts index 7838253..75f7e53 100644 --- a/scripts/smoke-standalone.ts +++ b/scripts/smoke-standalone.ts @@ -410,6 +410,18 @@ async function runChecks( await expectStatus(baseUrl, "/sitemap.xml", 200); await expectStatus(baseUrl, "/setup", 200, { contentType: "text/html" }); await expectStatus(baseUrl, "/llms.mdx/setup/content.md", 200); + // Every page advertises .mdx as its Markdown alternate; the root + // advertises /index.mdx and the legacy /.mdx still resolves. + await expectStatus(baseUrl, "/setup.mdx", 200, { + contentType: "text/markdown", + }); + await expectStatus(baseUrl, "/index.mdx", 200, { + contentType: "text/markdown", + }); + await expectStatus(baseUrl, "/.mdx", 200, { contentType: "text/markdown" }); + // A rewrite can only land on a prerendered file or a 404 (fallback: false). + // The tree diff at the end proves this one wrote nothing to disk. + await expectStatus(baseUrl, `/${UNKNOWN_PAGE}.mdx`, 404); // Markdown negotiation must still rewrite the docs root. await expectStatus(baseUrl, "/", 200, { contentType: "text/markdown", From ddc658e9c60b0fca5d8cc6c3e36dc037f910e952 Mon Sep 17 00:00:00 2001 From: Jose Montes de Oca Date: Tue, 22 Sep 2026 14:41:52 -0400 Subject: [PATCH 2/4] fix: negotiate Markdown on every page and tell caches about it A page's HTML URL now answers Markdown when the client asks for it with Accept: text/markdown or text/plain, and the rewrite carries Vary: Accept so a cache never hands a browser the Markdown body. The .mdx alternates are not negotiated and do not get the header. The tests enumerate every non-page route on the site (llms.txt, robots, sitemap, the /llms.mdx and /og trees, /_next, /.well-known) and pin that a Markdown-preferring client still gets that route rather than a rewrite into a path the build never generated. A second cross-layer loop ties the negotiated URL of every page to its /llms.mdx handler URL. The standalone smoke server now binds to 0.0.0.0 like the Dockerfile. With a loopback HOSTNAME, Next normalises request.nextUrl to localhost, mismatches its own origin, treats every rewrite as external, and proxies the request back to itself, which dropped the Vary header and doubled the work. Production never takes that path, so the smoke run must not. --- __tests__/markdown-alternates.test.ts | 20 +++++ __tests__/proxy.test.ts | 113 ++++++++++++++++++++++++++ proxy.ts | 17 ++-- scripts/smoke-standalone.ts | 41 +++++++++- 4 files changed, 185 insertions(+), 6 deletions(-) diff --git a/__tests__/markdown-alternates.test.ts b/__tests__/markdown-alternates.test.ts index 872912e..8226af8 100644 --- a/__tests__/markdown-alternates.test.ts +++ b/__tests__/markdown-alternates.test.ts @@ -35,4 +35,24 @@ describe("markdown alternates", () => { } expect(mismatches).toEqual([]); }); + + it("negotiate every page's HTML URL to the same prerendered Markdown", () => { + // The agent journey: llms.txt lists HTML URLs, and a client asks for + // them with Accept: text/markdown. The proxy's page-shape gate has to + // accept every real page URL, or that page silently answers HTML. + const pages = source.getPages(); + expect(pages.length).toBeGreaterThan(0); + + const mismatches: string[] = []; + for (const page of pages) { + const rewritten = resolveMarkdownRewrite(page.url, true); + const expected = getPageMarkdownUrl(page).url; + if (rewritten !== expected) { + mismatches.push( + `${page.url}: -> ${String(rewritten)}, expected ${expected}`, + ); + } + } + expect(mismatches).toEqual([]); + }); }); diff --git a/__tests__/proxy.test.ts b/__tests__/proxy.test.ts index cc6e8e4..bda4314 100644 --- a/__tests__/proxy.test.ts +++ b/__tests__/proxy.test.ts @@ -77,6 +77,40 @@ describe('resolveMarkdownRewrite', () => { it('leaves /llms.txt alone without a Markdown Accept', () => { expect(resolveMarkdownRewrite('/llms.txt', false)).toBeNull(); }); + + describe('with a Markdown Accept', () => { + it("negotiates a page's HTML URL to its prerendered Markdown", () => { + expect(resolveMarkdownRewrite('/setup', true)).toBe( + '/llms.mdx/setup/content.md', + ); + }); + + it('negotiates the root to the root Markdown', () => { + expect(resolveMarkdownRewrite('/', true)).toBe('/llms.mdx/content.md'); + }); + + it('no longer matches the old double-slash form', () => { + expect(resolveMarkdownRewrite('//setup', true)).toBeNull(); + }); + + // The gate: with the docs at the root the negotiated pattern matches + // every path, and isMarkdownPreferred is true for text/plain too. Each + // row is a request a Markdown-preferring client makes today that a + // naive catch-all would have rewritten into a 404. + it.each([ + '/llms.txt', + '/llms-full.txt', + '/robots.txt', + '/sitemap.xml', + '/llms.mdx/setup/content.md', + '/llms.mdx/content.md', + '/og/setup/image.png', + '/_next/static/chunk.js', + '/.well-known/agent-skills/index.json', + ])('leaves the non-page route %s alone', (pathname) => { + expect(resolveMarkdownRewrite(pathname, true)).toBeNull(); + }); + }); }); describe('proxy', () => { @@ -95,6 +129,85 @@ describe('proxy', () => { ); }); + it('does not mark the .mdx alternate as negotiated', () => { + // The alternate answers Markdown to every client, so caches need no + // Accept key for it. + const res = proxy(new NextRequest('https://docs.prose.md/setup.mdx')); + expect(res.headers.get('vary')).toBeNull(); + }); + + it("negotiates a page's HTML URL when the client prefers Markdown", () => { + const res = proxy( + new NextRequest('https://docs.prose.md/setup', { + headers: { accept: 'text/markdown' }, + }), + ); + expect(res.headers.get('x-middleware-rewrite')).toBe( + 'https://docs.prose.md/llms.mdx/setup/content.md', + ); + expect(res.headers.get('vary')).toBe('Accept'); + expect(res.headers.get('x-middleware-next')).toBeNull(); + }); + + it("negotiates a page's HTML URL when the client prefers plain text", () => { + const res = proxy( + new NextRequest('https://docs.prose.md/setup', { + headers: { accept: 'text/plain' }, + }), + ); + expect(res.headers.get('x-middleware-rewrite')).toBe( + 'https://docs.prose.md/llms.mdx/setup/content.md', + ); + expect(res.headers.get('vary')).toBe('Accept'); + }); + + it('negotiates the root when the client prefers Markdown', () => { + const res = proxy( + new NextRequest('https://docs.prose.md/', { + headers: { accept: 'text/markdown' }, + }), + ); + expect(res.headers.get('x-middleware-rewrite')).toBe( + 'https://docs.prose.md/llms.mdx/content.md', + ); + expect(res.headers.get('vary')).toBe('Accept'); + }); + + it('serves HTML to a browser Accept header', () => { + const res = proxy( + new NextRequest('https://docs.prose.md/setup', { + headers: { + accept: + 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', + }, + }), + ); + expect(res.headers.get('x-middleware-next')).toBe('1'); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + }); + + it('leaves /llms.txt alone when the client prefers plain text', () => { + const res = proxy( + new NextRequest('https://docs.prose.md/llms.txt', { + headers: { accept: 'text/plain' }, + }), + ); + expect(res.headers.get('x-middleware-next')).toBe('1'); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + }); + + it('leaves a /llms.mdx file alone when the client prefers Markdown', () => { + // This is the URL the negotiation lands on; re-negotiating it would + // rewrite into a path the build never generated. + const res = proxy( + new NextRequest('https://docs.prose.md/llms.mdx/setup/content.md', { + headers: { accept: 'text/markdown' }, + }), + ); + expect(res.headers.get('x-middleware-next')).toBe('1'); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + }); + it('rejects an encoded .mdx path before rewriting it', () => { const res = proxy(new NextRequest('https://docs.prose.md/setup%2Emdx')); expect(res.status).toBe(404); diff --git a/proxy.ts b/proxy.ts index 0bf1b9f..0844d67 100644 --- a/proxy.ts +++ b/proxy.ts @@ -98,13 +98,20 @@ export default function proxy(request: NextRequest) { return NextResponse.redirect(hostRedirect, 301); } - const target = resolveMarkdownRewrite( - request.nextUrl.pathname, - isMarkdownPreferred(request), - ); + const { pathname } = request.nextUrl; + const prefersMarkdown = isMarkdownPreferred(request); + const target = resolveMarkdownRewrite(pathname, prefersMarkdown); if (target) { - return NextResponse.rewrite(new URL(target, request.nextUrl)); + // A page's HTML URL answers Markdown only because of the Accept header, + // so caches must key on it. The `.mdx` alternates answer Markdown to + // every client and are not negotiated. (Every non-negotiated target + // comes from an `.mdx` path: the root aliases and the suffix pattern.) + const negotiated = prefersMarkdown && !pathname.endsWith('.mdx'); + return NextResponse.rewrite( + new URL(target, request.nextUrl), + negotiated ? { headers: { Vary: 'Accept' } } : undefined, + ); } return NextResponse.next(); diff --git a/scripts/smoke-standalone.ts b/scripts/smoke-standalone.ts index 75f7e53..7db31d4 100644 --- a/scripts/smoke-standalone.ts +++ b/scripts/smoke-standalone.ts @@ -99,6 +99,8 @@ function parseCli(): { mode: Mode; port: number } { interface HttpResult { status: number; contentType: string; + /** The Vary header as sent, or "" when absent. */ + vary: string; body: string; } @@ -138,6 +140,7 @@ async function get( return { status: res.status, contentType: res.headers.get("content-type") ?? "", + vary: res.headers.get("vary") ?? "", body: await res.text(), }; } @@ -246,12 +249,20 @@ function startServer(port: number): Server { // Mirror the run stage's environment. DOCS_PREVIEW_MODE is dropped on // purpose: the image never carries it, and the robots body must come from // the build, not from whatever the shell running this script has set. + // + // HOSTNAME must be the Dockerfile's 0.0.0.0, not a loopback address. Next + // rewrites a loopback hostname to "localhost" in request.nextUrl, so with + // HOSTNAME=127.0.0.1 every proxy rewrite carries an origin the router does + // not recognise as its own. It then treats the rewrite as external and + // proxies the request back to itself over HTTP, which doubles the work and + // drops the headers the proxy set (Vary: Accept). Production never takes + // that path, so the smoke run must not either. const env: NodeJS.ProcessEnv = { ...process.env, NODE_ENV: "production", NEXT_TELEMETRY_DISABLED: "1", PORT: String(port), - HOSTNAME: "127.0.0.1", + HOSTNAME: "0.0.0.0", }; delete env.DOCS_PREVIEW_MODE; @@ -427,6 +438,34 @@ async function runChecks( contentType: "text/markdown", headers: { Accept: "text/markdown" }, }); + // Negotiation works on every page, not just the root, and tells caches + // so: the same URL answers HTML or Markdown depending on Accept. Next + // appends its own Vary line (rsc, next-router-state-tree, ...) after the + // proxy's, and fetch() joins repeated headers with ", ", so look for + // "accept" as one member of the list rather than the whole value. + const negotiated = await expectStatus(baseUrl, "/setup", 200, { + contentType: "text/markdown", + headers: { Accept: "text/markdown" }, + }); + if (negotiated) { + check( + "GET /setup (Accept: text/markdown) sets Vary: Accept", + /(^|,)\s*accept\s*(,|$)/i.test(negotiated.vary), + `vary: ${JSON.stringify(negotiated.vary)}`, + ); + } + // The gate: a Markdown-preferring client on a non-page route gets that + // route, not a rewrite into a /llms.mdx path the build never generated. + // text/plain is the natural Accept for llms.txt, and it also satisfies + // isMarkdownPreferred; /llms.mdx/… is where negotiation itself lands. + await expectStatus(baseUrl, "/llms.txt", 200, { + contentType: "text/plain", + headers: { Accept: "text/plain" }, + }); + await expectStatus(baseUrl, "/llms.mdx/setup/content.md", 200, { + contentType: "text/markdown", + headers: { Accept: "text/markdown" }, + }); // 3. Unknown paths 404 without writing anything to disk. This one reaches // the catch-all, so it is `dynamicParams = false` that keeps it off disk. From 3e43e96b202ffd376b4caf6da7208a44bf530941 Mon Sep 17 00:00:00 2001 From: Jose Montes de Oca Date: Tue, 22 Sep 2026 14:45:43 -0400 Subject: [PATCH 3/4] fix: make the page-move redirects permanent The /start and /openprose pages moved to the root for good, but their redirects answered 307. A temporary redirect tells a crawler the old URL may come back, so search engines keep listing the old URLs as separate pages and pass none of their signals to the new ones. The entry for /start/what-is-openprose was permanent from the day it was added and only became temporary when the harness routes were introduced alongside it. That change justified temporary redirects for the harness routes alone, which may host docs again, so those four stay 307. The standalone smoke run can now observe a redirect at all: it records the Location header and asserts a 308 page move, a 307 harness route, and that an old /openprose/.mdx alternate lands on a working one. --- next.config.mjs | 12 ++++++--- scripts/smoke-standalone.ts | 53 +++++++++++++++++++++++++++++++++++++ 2 files changed, 61 insertions(+), 4 deletions(-) diff --git a/next.config.mjs b/next.config.mjs index 6901f49..f86a2cf 100644 --- a/next.config.mjs +++ b/next.config.mjs @@ -24,23 +24,27 @@ const config = { return [ // The early docs lived under /start/*, then under /openprose/*. The // site now covers one topic, so the pages live at the root; keep every - // old link alive. + // old link alive. These moves are final, so the redirects are + // permanent (308): browsers cache them and search engines drop the + // old URL and pass its signals to the new one. A temporary redirect + // would keep the old URLs listed as separate pages. The harness group + // below stays temporary; see its comment. { source: "/start/what-is-openprose", destination: "/", - permanent: false, + permanent: true, }, // The bare entry is required: `/:path*` cannot produce `/` when the // wildcard matches zero segments. { source: "/openprose", destination: "/", - permanent: false, + permanent: true, }, { source: "/openprose/:path*", destination: "/:path*", - permanent: false, + permanent: true, }, ...Object.entries(harnessRoutes).map(([prefix, destination]) => ({ source: `/${prefix}/:path*`, diff --git a/scripts/smoke-standalone.ts b/scripts/smoke-standalone.ts index 7db31d4..6800654 100644 --- a/scripts/smoke-standalone.ts +++ b/scripts/smoke-standalone.ts @@ -101,6 +101,8 @@ interface HttpResult { contentType: string; /** The Vary header as sent, or "" when absent. */ vary: string; + /** The Location header as sent, or null when absent. */ + location: string | null; body: string; } @@ -141,6 +143,7 @@ async function get( status: res.status, contentType: res.headers.get("content-type") ?? "", vary: res.headers.get("vary") ?? "", + location: res.headers.get("location"), body: await res.text(), }; } @@ -172,6 +175,35 @@ async function expectStatus( return res; } +/** + * Fetches a path and asserts a redirect status and where it points. `get` + * never follows redirects, so the Location header is what the client saw. + * Both sides are resolved against the server's origin because Next may emit + * a relative or an absolute Location depending on the destination. + */ +async function expectRedirect( + baseUrl: string, + path: string, + status: 307 | 308, + target: string, +): Promise { + const label = `GET ${path} -> ${status} ${target}`; + let res: HttpResult; + try { + res = await get(baseUrl, path); + } catch (error) { + fail(label, `request failed: ${String(error)}`); + return; + } + const got = res.location ? new URL(res.location, baseUrl).href : null; + const want = new URL(target, baseUrl).href; + check( + label, + res.status === status && got === want, + `status ${res.status}, location ${JSON.stringify(res.location)}`, + ); +} + function checkRobotsBody(label: string, mode: Mode, res: HttpResult): void { if (mode === "public") { check( @@ -467,6 +499,27 @@ async function runChecks( headers: { Accept: "text/markdown" }, }); + // 2b. Redirects from next.config.mjs. Page moves are permanent (308), so + // search engines retire the old URL; the harness routes stay temporary + // (307) because they may host docs again. Redirects answer before the + // proxy runs, so none of these can touch the cache tree. + await expectRedirect(baseUrl, "/openprose/contracts", 308, "/contracts"); + await expectRedirect(baseUrl, "/start/what-is-openprose", 308, "/"); + await expectRedirect( + baseUrl, + "/cli", + 307, + "https://github.com/openprose/prose/tree/main/packages/reactor-cli", + ); + // An old page's advertised alternate heals too: the redirect lands on a + // .mdx URL the proxy now serves as Markdown. + await expectRedirect( + baseUrl, + "/openprose/contracts.mdx", + 308, + "/contracts.mdx", + ); + // 3. Unknown paths 404 without writing anything to disk. This one reaches // the catch-all, so it is `dynamicParams = false` that keeps it off disk. await expectStatus(baseUrl, `/${UNKNOWN_PAGE}`, 404); From cac5e8cbbd20dd092cae3753d7e3867e71d2a131 Mon Sep 17 00:00:00 2001 From: Jose Montes de Oca Date: Tue, 29 Sep 2026 21:16:49 -0400 Subject: [PATCH 4/4] fix: keep Markdown negotiation away from the search API The docs-page gate treated any extensionless path outside /_next as a page, and the search API has no extension. Its clients commonly send `Accept: application/json, text/plain, */*`, and text/plain alone counts as a Markdown preference, so /api/search and its OpenAPI document were rewritten into /llms.mdx paths the build never generates and answered 404. The gate now excludes /api, and the unit tests and smoke run both send that Accept header to prove the API still answers JSON. The gate is exported as isNegotiablePage so the proxy makes one decision about which rewrites are content-negotiated and therefore carry Vary: Accept. The HTML side of a page cannot carry it: Next overwrites Vary on App Router page responses before rendering, and neither a proxy header nor a headers() rule in next.config survives that. The proxy comment records the constraint so a future cache in front of the site is configured to key page URLs on Accept itself. --- __tests__/proxy.test.ts | 50 +++++++++++++++++++++++++++++++++++++ proxy.ts | 27 +++++++++++++++----- scripts/smoke-standalone.ts | 14 ++++++++++- 3 files changed, 84 insertions(+), 7 deletions(-) diff --git a/__tests__/proxy.test.ts b/__tests__/proxy.test.ts index bda4314..9444cbe 100644 --- a/__tests__/proxy.test.ts +++ b/__tests__/proxy.test.ts @@ -5,6 +5,7 @@ import { describe, expect, it } from 'vitest'; import { NextRequest } from 'next/server'; import proxy, { hasEncodedPathname, + isNegotiablePage, resolveDocsHostRedirect, resolveMarkdownRewrite, } from '../proxy'; @@ -107,12 +108,37 @@ describe('resolveMarkdownRewrite', () => { '/og/setup/image.png', '/_next/static/chunk.js', '/.well-known/agent-skills/index.json', + '/api/search', + '/api/search/openapi', ])('leaves the non-page route %s alone', (pathname) => { expect(resolveMarkdownRewrite(pathname, true)).toBeNull(); }); }); }); +describe('isNegotiablePage', () => { + it.each(['/', '/setup', '/contracts', '/harness-agnostic'])( + 'treats the page URL %s as negotiable', + (pathname) => { + expect(isNegotiablePage(pathname)).toBe(true); + }, + ); + + it.each([ + '/setup.mdx', + '/index.mdx', + '/llms.txt', + '/robots.txt', + '/llms.mdx/setup/content.md', + '/_next/static/chunk.js', + '/api', + '/api/search', + '/api/search/openapi', + ])('does not treat %s as negotiable', (pathname) => { + expect(isNegotiablePage(pathname)).toBe(false); + }); +}); + describe('proxy', () => { it('rewrites an advertised .mdx alternate to the Markdown route', () => { const res = proxy(new NextRequest('https://docs.prose.md/setup.mdx')); @@ -186,6 +212,30 @@ describe('proxy', () => { expect(res.headers.get('x-middleware-rewrite')).toBeNull(); }); + it('leaves the search API alone for a JSON client that also accepts text', () => { + // `application/json, text/plain, */*` is a common HTTP client default, + // and text/plain alone satisfies isMarkdownPreferred. + const res = proxy( + new NextRequest('https://docs.prose.md/api/search?q=setup', { + headers: { accept: 'application/json, text/plain, */*' }, + }), + ); + expect(res.headers.get('x-middleware-next')).toBe('1'); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + expect(res.headers.get('vary')).toBeNull(); + }); + + it('leaves the search OpenAPI document alone for the same client', () => { + const res = proxy( + new NextRequest('https://docs.prose.md/api/search/openapi', { + headers: { accept: 'application/json, text/plain, */*' }, + }), + ); + expect(res.headers.get('x-middleware-next')).toBe('1'); + expect(res.headers.get('x-middleware-rewrite')).toBeNull(); + expect(res.headers.get('vary')).toBeNull(); + }); + it('leaves /llms.txt alone when the client prefers plain text', () => { const res = proxy( new NextRequest('https://docs.prose.md/llms.txt', { diff --git a/proxy.ts b/proxy.ts index 0844d67..a1aad85 100644 --- a/proxy.ts +++ b/proxy.ts @@ -47,10 +47,17 @@ export function hasEncodedPathname(pathname: string): boolean { // path, and isMarkdownPreferred is true for `Accept: text/plain` as well as // `text/markdown`. Docs slugs are plain kebab-case words; every other route // on the site (/robots.txt, /llms.txt, /llms.mdx/**, /og/**, /.well-known/**) -// carries a dot. A wrong guess here rewrites into a path the build never -// generated and 404s under `fallback: false`, never a wrong body. -function looksLikeDocsPage(pathname: string): boolean { - return !pathname.includes('.') && !pathname.startsWith('/_next'); +// carries a dot, except the search API, whose clients commonly send +// `Accept: application/json, text/plain, */*` and must never be rewritten. +// A wrong guess here rewrites into a path the build never generated and +// 404s under `fallback: false`, never a wrong body. +export function isNegotiablePage(pathname: string): boolean { + return ( + !pathname.includes('.') && + !pathname.startsWith('/_next') && + !pathname.startsWith('/api/') && + pathname !== '/api' + ); } // Returns the /llms.mdx path that serves a request as Markdown, or null. @@ -77,7 +84,7 @@ export function resolveMarkdownRewrite( const suffixed = rewriteSuffix(pathname); if (suffixed) return suffixed; - if (prefersMarkdown && looksLikeDocsPage(pathname)) { + if (prefersMarkdown && isNegotiablePage(pathname)) { return rewriteDocs(pathname) || null; } @@ -107,7 +114,15 @@ export default function proxy(request: NextRequest) { // so caches must key on it. The `.mdx` alternates answer Markdown to // every client and are not negotiated. (Every non-negotiated target // comes from an `.mdx` path: the root aliases and the suffix pattern.) - const negotiated = prefersMarkdown && !pathname.endsWith('.mdx'); + // + // Only the Markdown side can carry this. Route handler responses keep a + // Vary set here (Next appends its own list after it), but App Router + // page responses do not: Next overwrites Vary with its RSC list before + // rendering, and neither this proxy nor a headers() rule in next.config + // survives that. So the HTML side of a page ships without `Accept` in + // Vary, and a shared cache in front of the site must key on Accept for + // page URLs itself. There is no such cache today. + const negotiated = prefersMarkdown && isNegotiablePage(pathname); return NextResponse.rewrite( new URL(target, request.nextUrl), negotiated ? { headers: { Vary: 'Accept' } } : undefined, diff --git a/scripts/smoke-standalone.ts b/scripts/smoke-standalone.ts index 6800654..5852cd3 100644 --- a/scripts/smoke-standalone.ts +++ b/scripts/smoke-standalone.ts @@ -474,7 +474,8 @@ async function runChecks( // so: the same URL answers HTML or Markdown depending on Accept. Next // appends its own Vary line (rsc, next-router-state-tree, ...) after the // proxy's, and fetch() joins repeated headers with ", ", so look for - // "accept" as one member of the list rather than the whole value. + // "accept" as one member of the list rather than the whole value. (The + // HTML side cannot carry it: Next overwrites Vary on page responses.) const negotiated = await expectStatus(baseUrl, "/setup", 200, { contentType: "text/markdown", headers: { Accept: "text/markdown" }, @@ -498,6 +499,17 @@ async function runChecks( contentType: "text/markdown", headers: { Accept: "text/markdown" }, }); + // The search API has no extension, and its clients commonly send + // `application/json, text/plain, */*`, which satisfies isMarkdownPreferred + // on text/plain alone. It must answer JSON, not a rewrite into /llms.mdx. + await expectStatus(baseUrl, "/api/search?q=setup", 200, { + contentType: "application/json", + headers: { Accept: "application/json, text/plain, */*" }, + }); + await expectStatus(baseUrl, "/api/search/openapi", 200, { + contentType: "application/json", + headers: { Accept: "application/json, text/plain, */*" }, + }); // 2b. Redirects from next.config.mjs. Page moves are permanent (308), so // search engines retire the old URL; the harness routes stay temporary