Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 21 additions & 13 deletions __tests__/canonical.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -37,34 +37,42 @@ describe("robotsContent", () => {
});

describe("buildPageMetadata", () => {
it("emits absolute canonical URL for a nested docs path", () => {
const md = buildPageMetadata("/docs/get-started/install");
expect(md.alternates?.canonical).toBe(
"https://docs.prose.md/docs/get-started/install",
);
it("emits absolute canonical URL for a root-mounted docs path", () => {
const md = buildPageMetadata("/setup");
expect(md.alternates?.canonical).toBe("https://docs.prose.md/setup");
});

it("emits .mdx markdown alternate so Fumadocs's proxy can rewrite it", () => {
const md = buildPageMetadata("/docs/get-started/install");
expect(md.alternates?.types?.["text/markdown"]).toBe(
"/docs/get-started/install.mdx",
);
it("emits absolute canonical URL for the root page", () => {
const md = buildPageMetadata("/");
expect(md.alternates?.canonical).toBe("https://docs.prose.md/");
});

it("emits <slug>.mdx as the markdown alternate so the proxy can rewrite it", () => {
const md = buildPageMetadata("/setup");
expect(md.alternates?.types?.["text/markdown"]).toBe("/setup.mdx");
});

it("emits /index.mdx as the root page's markdown alternate", () => {
// Appending .mdx to "/" would advertise "/.mdx", which reads as a dotfile
// and only resolved by accident of the old rewrite pattern.
const md = buildPageMetadata("/");
expect(md.alternates?.types?.["text/markdown"]).toBe("/index.mdx");
});

it("always emits /llms.txt as the text/plain alternate", () => {
const md = buildPageMetadata("/docs/anywhere");
const md = buildPageMetadata("/anywhere");
expect(md.alternates?.types?.["text/plain"]).toBe("/llms.txt");
});

it("sets robots index/follow false in preview mode", () => {
vi.stubEnv("DOCS_PREVIEW_MODE", "true");
const md = buildPageMetadata("/docs/foo");
const md = buildPageMetadata("/foo");
expect(md.robots).toEqual({ index: false, follow: false });
});

it("sets robots index/follow true when preview mode is off", () => {
vi.stubEnv("DOCS_PREVIEW_MODE", "false");
const md = buildPageMetadata("/docs/foo");
const md = buildPageMetadata("/foo");
expect(md.robots).toEqual({ index: true, follow: true });
});
});
58 changes: 58 additions & 0 deletions __tests__/markdown-alternates.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
// @vitest-environment node
// The proxy runs on the server; keep this in the same environment as
// proxy.test.ts so the helper sees exactly what production sees.
import { describe, expect, it } from "vitest";
import { buildPageMetadata } from "../lib/canonical";
import { getPageMarkdownUrl, source } from "../lib/source";
import { resolveMarkdownRewrite } from "../proxy";

// Three layers have to agree on one string: buildPageMetadata advertises
// the alternate, resolveMarkdownRewrite maps it, and the /llms.mdx handler
// prerenders exactly the paths getPageMarkdownUrl produces. A page whose
// slug breaks the shape the proxy expects, or a change to any one layer,
// shows up here without a build.
describe("markdown alternates", () => {
it("advertise a URL the proxy rewrites to the page's prerendered Markdown", () => {
const pages = source.getPages();
expect(pages.length).toBeGreaterThan(0);

const mismatches: string[] = [];
for (const page of pages) {
const alternate = buildPageMetadata(page.url).alternates?.types?.[
"text/markdown"
];
if (typeof alternate !== "string") {
mismatches.push(`${page.url}: no text/markdown alternate`);
continue;
}
const rewritten = resolveMarkdownRewrite(alternate, false);
const expected = getPageMarkdownUrl(page).url;
if (rewritten !== expected) {
mismatches.push(
`${page.url}: ${alternate} -> ${String(rewritten)}, expected ${expected}`,
);
}
}
expect(mismatches).toEqual([]);
});

it("negotiate every page's HTML URL to the same prerendered Markdown", () => {
// The agent journey: llms.txt lists HTML URLs, and a client asks for
// them with Accept: text/markdown. The proxy's page-shape gate has to
// accept every real page URL, or that page silently answers HTML.
const pages = source.getPages();
expect(pages.length).toBeGreaterThan(0);

const mismatches: string[] = [];
for (const page of pages) {
const rewritten = resolveMarkdownRewrite(page.url, true);
const expected = getPageMarkdownUrl(page).url;
if (rewritten !== expected) {
mismatches.push(
`${page.url}: -> ${String(rewritten)}, expected ${expected}`,
);
}
}
expect(mismatches).toEqual([]);
});
});
224 changes: 223 additions & 1 deletion __tests__/proxy.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,12 @@
// which the legacy-host redirect depends on.
import { describe, expect, it } from 'vitest';
import { NextRequest } from 'next/server';
import proxy, { hasEncodedPathname, resolveDocsHostRedirect } from '../proxy';
import proxy, {
hasEncodedPathname,
isNegotiablePage,
resolveDocsHostRedirect,
resolveMarkdownRewrite,
} from '../proxy';

describe('resolveDocsHostRedirect', () => {
it('redirects the legacy docs host to docs.prose.md', () => {
Expand Down Expand Up @@ -41,7 +46,224 @@ describe('hasEncodedPathname', () => {
});
});

describe('resolveMarkdownRewrite', () => {
it('maps an advertised <slug>.mdx alternate to its prerendered Markdown', () => {
expect(resolveMarkdownRewrite('/setup.mdx', false)).toBe(
'/llms.mdx/setup/content.md',
);
});

it('maps the advertised root alternate /index.mdx to the root Markdown', () => {
expect(resolveMarkdownRewrite('/index.mdx', false)).toBe(
'/llms.mdx/content.md',
);
});

it('keeps the legacy /.mdx root alternate working', () => {
expect(resolveMarkdownRewrite('/.mdx', false)).toBe(
'/llms.mdx/content.md',
);
});

it('no longer matches the old double-slash form', () => {
// Next 308s these before the proxy runs, so a match here would only
// ever have been reachable by accident.
expect(resolveMarkdownRewrite('//setup.mdx', false)).toBeNull();
});

it('leaves a page URL alone without a Markdown Accept', () => {
expect(resolveMarkdownRewrite('/setup', false)).toBeNull();
});

it('leaves /llms.txt alone without a Markdown Accept', () => {
expect(resolveMarkdownRewrite('/llms.txt', false)).toBeNull();
});

describe('with a Markdown Accept', () => {
it("negotiates a page's HTML URL to its prerendered Markdown", () => {
expect(resolveMarkdownRewrite('/setup', true)).toBe(
'/llms.mdx/setup/content.md',
);
});

it('negotiates the root to the root Markdown', () => {
expect(resolveMarkdownRewrite('/', true)).toBe('/llms.mdx/content.md');
});

it('no longer matches the old double-slash form', () => {
expect(resolveMarkdownRewrite('//setup', true)).toBeNull();
});

// The gate: with the docs at the root the negotiated pattern matches
// every path, and isMarkdownPreferred is true for text/plain too. Each
// row is a request a Markdown-preferring client makes today that a
// naive catch-all would have rewritten into a 404.
it.each([
'/llms.txt',
'/llms-full.txt',
'/robots.txt',
'/sitemap.xml',
'/llms.mdx/setup/content.md',
'/llms.mdx/content.md',
'/og/setup/image.png',
'/_next/static/chunk.js',
'/.well-known/agent-skills/index.json',
'/api/search',
'/api/search/openapi',
])('leaves the non-page route %s alone', (pathname) => {
expect(resolveMarkdownRewrite(pathname, true)).toBeNull();
});
});
});

describe('isNegotiablePage', () => {
it.each(['/', '/setup', '/contracts', '/harness-agnostic'])(
'treats the page URL %s as negotiable',
(pathname) => {
expect(isNegotiablePage(pathname)).toBe(true);
},
);

it.each([
'/setup.mdx',
'/index.mdx',
'/llms.txt',
'/robots.txt',
'/llms.mdx/setup/content.md',
'/_next/static/chunk.js',
'/api',
'/api/search',
'/api/search/openapi',
])('does not treat %s as negotiable', (pathname) => {
expect(isNegotiablePage(pathname)).toBe(false);
});
});

describe('proxy', () => {
it('rewrites an advertised .mdx alternate to the Markdown route', () => {
const res = proxy(new NextRequest('https://docs.prose.md/setup.mdx'));
expect(res.headers.get('x-middleware-rewrite')).toBe(
'https://docs.prose.md/llms.mdx/setup/content.md',
);
expect(res.headers.get('x-middleware-next')).toBeNull();
});

it('rewrites the root alternate /index.mdx to the root Markdown', () => {
const res = proxy(new NextRequest('https://docs.prose.md/index.mdx'));
expect(res.headers.get('x-middleware-rewrite')).toBe(
'https://docs.prose.md/llms.mdx/content.md',
);
});

it('does not mark the .mdx alternate as negotiated', () => {
// The alternate answers Markdown to every client, so caches need no
// Accept key for it.
const res = proxy(new NextRequest('https://docs.prose.md/setup.mdx'));
expect(res.headers.get('vary')).toBeNull();
});

it("negotiates a page's HTML URL when the client prefers Markdown", () => {
const res = proxy(
new NextRequest('https://docs.prose.md/setup', {
headers: { accept: 'text/markdown' },
}),
);
expect(res.headers.get('x-middleware-rewrite')).toBe(
'https://docs.prose.md/llms.mdx/setup/content.md',
);
expect(res.headers.get('vary')).toBe('Accept');
expect(res.headers.get('x-middleware-next')).toBeNull();
});

it("negotiates a page's HTML URL when the client prefers plain text", () => {
const res = proxy(
new NextRequest('https://docs.prose.md/setup', {
headers: { accept: 'text/plain' },
}),
);
expect(res.headers.get('x-middleware-rewrite')).toBe(
'https://docs.prose.md/llms.mdx/setup/content.md',
);
expect(res.headers.get('vary')).toBe('Accept');
});

it('negotiates the root when the client prefers Markdown', () => {
const res = proxy(
new NextRequest('https://docs.prose.md/', {
headers: { accept: 'text/markdown' },
}),
);
expect(res.headers.get('x-middleware-rewrite')).toBe(
'https://docs.prose.md/llms.mdx/content.md',
);
expect(res.headers.get('vary')).toBe('Accept');
});

it('serves HTML to a browser Accept header', () => {
const res = proxy(
new NextRequest('https://docs.prose.md/setup', {
headers: {
accept:
'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
},
}),
);
expect(res.headers.get('x-middleware-next')).toBe('1');
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
});

it('leaves the search API alone for a JSON client that also accepts text', () => {
// `application/json, text/plain, */*` is a common HTTP client default,
// and text/plain alone satisfies isMarkdownPreferred.
const res = proxy(
new NextRequest('https://docs.prose.md/api/search?q=setup', {
headers: { accept: 'application/json, text/plain, */*' },
}),
);
expect(res.headers.get('x-middleware-next')).toBe('1');
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
expect(res.headers.get('vary')).toBeNull();
});

it('leaves the search OpenAPI document alone for the same client', () => {
const res = proxy(
new NextRequest('https://docs.prose.md/api/search/openapi', {
headers: { accept: 'application/json, text/plain, */*' },
}),
);
expect(res.headers.get('x-middleware-next')).toBe('1');
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
expect(res.headers.get('vary')).toBeNull();
});

it('leaves /llms.txt alone when the client prefers plain text', () => {
const res = proxy(
new NextRequest('https://docs.prose.md/llms.txt', {
headers: { accept: 'text/plain' },
}),
);
expect(res.headers.get('x-middleware-next')).toBe('1');
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
});

it('leaves a /llms.mdx file alone when the client prefers Markdown', () => {
// This is the URL the negotiation lands on; re-negotiating it would
// rewrite into a path the build never generated.
const res = proxy(
new NextRequest('https://docs.prose.md/llms.mdx/setup/content.md', {
headers: { accept: 'text/markdown' },
}),
);
expect(res.headers.get('x-middleware-next')).toBe('1');
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
});

it('rejects an encoded .mdx path before rewriting it', () => {
const res = proxy(new NextRequest('https://docs.prose.md/setup%2Emdx'));
expect(res.status).toBe(404);
expect(res.headers.get('x-middleware-rewrite')).toBeNull();
});

it('returns 404 for a percent-encoded robots path', () => {
const res = proxy(new NextRequest('https://docs.prose.md/robots%2Etxt'));
expect(res.status).toBe(404);
Expand Down
8 changes: 5 additions & 3 deletions lib/canonical.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,16 +28,18 @@ export function robotsContent(): string | null {
* plus the agent corpus at /llms.txt), and robots index/follow directives
* gated by preview mode.
*
* The markdown alternate uses the `.mdx` URL suffix because Fumadocs's
* proxy.ts rewrites `${docsRoute}{/*path}.mdx` to the markdown route.
* The markdown alternate uses the `.mdx` URL suffix because
* resolveMarkdownRewrite in proxy.ts maps `<page.url>.mdx` to the
* prerendered /llms.mdx/<slug>/content.md file. The root page has no slug,
* so it advertises /index.mdx, which the proxy maps explicitly.
*/
export function buildPageMetadata(path: string): Metadata {
const preview = isPreviewMode();
return {
alternates: {
canonical: canonicalUrl(path),
types: {
"text/markdown": `${path}.mdx`,
"text/markdown": path === "/" ? "/index.mdx" : `${path}.mdx`,
"text/plain": "/llms.txt",
},
},
Expand Down
11 changes: 8 additions & 3 deletions lib/shared.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,12 @@
export const appName = "OpenProse";
// Docs are mounted at the root: docs.openprose.ai/<slug> rather than
// docs.openprose.ai/docs/<slug>. The (docs) route group at app/(docs)/
// applies the DocsLayout to everything without adding a URL segment.
// Docs are mounted at the root: docs.prose.md/<slug> rather than
// docs.prose.md/docs/<slug>. The catch-all page at app/[[...slug]]/ serves
// every page without adding a URL segment.
//
// docsRoute feeds both the Fumadocs loader (which normalizes a trailing
// slash away when building page.url) and the proxy's rewrite patterns
// (which strip it before concatenating). Keep it either "/" or a prefix
// without a trailing slash, such as "/docs".
export const docsRoute = "/";
export const docsImageRoute = "/og";
export const docsContentRoute = "/llms.mdx";
Expand Down
Loading
Loading