diff --git a/lib/sp/blueskyFacets.ts b/lib/sp/blueskyFacets.ts new file mode 100644 index 00000000..d0e16f03 --- /dev/null +++ b/lib/sp/blueskyFacets.ts @@ -0,0 +1,162 @@ +// Bluesky rich-text facets. +// +// Bluesky does not parse anything out of post text. A URL posted as plain +// text stays plain text, and a hashtag is just a word starting with '#'. +// Anything that should be clickable has to be described by a facet giving its +// byte range and what it points at. There is no auto-parse flag to set. +// +// The part that bites: those offsets are counted in UTF-8 *bytes*, while +// JavaScript string indices are UTF-16 code units. Any emoji, accented +// character or CJK text before a link shifts the two apart, and the facet +// then highlights the wrong span — usually mid-word, sometimes past the end +// of the string. Every offset here is converted through Buffer.byteLength. + +export type BlueskyFacet = { + index: { byteStart: number; byteEnd: number }; + features: Array< + | { $type: "app.bsky.richtext.facet#link"; uri: string } + | { $type: "app.bsky.richtext.facet#tag"; tag: string } + >; +}; + +/** Bluesky counts 300 graphemes, and separately caps the record at 3000 bytes. */ +const MAX_GRAPHEMES = 300; +const MAX_BYTES = 3000; + +const URL_RE = /https?:\/\/[^\s<>"']+/g; + +// A '#' that starts a word, followed by tag characters. The lookbehind stops +// it firing inside a URL fragment or an id like `foo#bar`. +const TAG_RE = /(?"']+)/g; + +/** Trailing characters that are almost always sentence punctuation, not URL. */ +const TRAILING_PUNCT = /[.,;:!?)\]}'"]+$/; + +function byteLen(s: string): number { + return Buffer.byteLength(s, "utf8"); +} + +/** + * Byte offset of a UTF-16 index, which is what the facet index actually means. + */ +function byteOffsetOf(text: string, charIndex: number): number { + return byteLen(text.slice(0, charIndex)); +} + +/** + * Build the facets for a post. + * + * Returns them sorted by start offset and non-overlapping, which is what the + * API expects; an unsorted or overlapping set is rejected or renders wrong. + */ +export function buildFacets(text: string): BlueskyFacet[] { + const facets: BlueskyFacet[] = []; + const taken: Array<[number, number]> = []; + + const overlaps = (start: number, end: number) => + taken.some(([s, e]) => start < e && end > s); + + for (const m of text.matchAll(URL_RE)) { + const raw = m[0]; + // "Read https://example.com." should link the URL, not the full stop. + // Closing brackets only come off when unbalanced, so a URL that + // legitimately ends in ')' — Wikipedia does this — survives. + let url = raw.replace(TRAILING_PUNCT, ""); + if (url.includes("(") && !url.includes(")") && raw.startsWith(url + ")")) { + url = url + ")"; + } + if (!url) continue; + + const start = m.index ?? 0; + const end = start + url.length; + if (overlaps(start, end)) continue; + taken.push([start, end]); + + facets.push({ + index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) }, + features: [{ $type: "app.bsky.richtext.facet#link", uri: url }], + }); + } + + for (const m of text.matchAll(TAG_RE)) { + const tag = m[1].replace(TRAILING_PUNCT, ""); + // Bluesky rejects an empty tag, caps them at 64 characters, and a + // purely numeric one is nearly always "#1" in prose rather than a tag. + if (!tag || tag.length > 64 || /^\d+$/.test(tag)) continue; + + const start = m.index ?? 0; + const end = start + 1 + tag.length; + if (overlaps(start, end)) continue; + taken.push([start, end]); + + facets.push({ + index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) }, + features: [{ $type: "app.bsky.richtext.facet#tag", tag }], + }); + } + + return facets.sort((a, b) => a.index.byteStart - b.index.byteStart); +} + +/** + * Length as Bluesky counts it: graphemes, not UTF-16 code units. + * + * `"🚀".length` is 2, and a post of 200 emoji is 400 by that measure but 200 + * by Bluesky's — so a naive length check rejects posts the API would accept. + */ +export function graphemeLength(text: string): number { + if (typeof Intl !== "undefined" && "Segmenter" in Intl) { + return [...new Intl.Segmenter(undefined, { granularity: "grapheme" }).segment(text)].length; + } + return [...text].length; +} + +/** + * Trim a post to Bluesky's limits without corrupting it. + * + * `text.slice(0, 300)` is wrong twice over: it counts UTF-16 code units, so an + * emoji spends two of the 300, and it can cut between the halves of a + * surrogate pair, producing a lone surrogate that is not valid UTF-8. This + * cuts on grapheme boundaries where the runtime can identify them, then + * enforces the byte ceiling separately. + */ +export function truncateForBluesky(text: string): string { + let out = text; + + const segmenter = + typeof Intl !== "undefined" && "Segmenter" in Intl + ? new Intl.Segmenter(undefined, { granularity: "grapheme" }) + : null; + + if (segmenter) { + const graphemes = [...segmenter.segment(out)].map((s) => s.segment); + if (graphemes.length > MAX_GRAPHEMES) out = graphemes.slice(0, MAX_GRAPHEMES).join(""); + } else { + // No Segmenter: fall back to code points, which at least never splits a + // surrogate pair the way slice() does. + const points = [...out]; + if (points.length > MAX_GRAPHEMES) out = points.slice(0, MAX_GRAPHEMES).join(""); + } + + // Byte ceiling. Drop whole code points so the result stays valid UTF-8. + while (byteLen(out) > MAX_BYTES) { + const points = [...out]; + points.pop(); + out = points.join(""); + } + + return out; +} + +/** The full post record, facets included. */ +export function buildPostRecord(text: string, createdAt: string) { + const trimmed = truncateForBluesky(text); + const facets = buildFacets(trimmed); + return { + $type: "app.bsky.feed.post" as const, + text: trimmed, + createdAt, + // Omitted entirely when empty: an empty array is legal but noise. + ...(facets.length ? { facets } : {}), + }; +} diff --git a/lib/sp/platforms/bluesky.ts b/lib/sp/platforms/bluesky.ts index 5a9b2c9b..3e465526 100644 --- a/lib/sp/platforms/bluesky.ts +++ b/lib/sp/platforms/bluesky.ts @@ -14,6 +14,8 @@ // Posting: com.atproto.repo.createRecord with the app.bsky.feed.post // collection. Returns the post's at:// URI + cid. +import { buildPostRecord } from "../blueskyFacets"; + const DEFAULT_PDS = "https://bsky.social"; export type BlueskySession = { @@ -78,9 +80,8 @@ export async function refreshBlueskySession(input: { return { accessJwt: json.accessJwt, refreshJwt: json.refreshJwt }; } -// Post a single text post. Bluesky's 300-char limit is enforced by -// the API — we trim client-side and document the limit to the user. -const MAX_POST_CHARS = 300; +// Post a single text post. Trimming to Bluesky's limits and working out the +// rich-text facets both live in ../blueskyFacets.ts. export type BlueskyPostResult = { uri: string; // at://did:plc:.../app.bsky.feed.post/... @@ -96,12 +97,9 @@ export async function createBlueskyPost(input: { pdsUrl?: string; }): Promise { const pds = (input.pdsUrl ?? DEFAULT_PDS).replace(/\/$/, ""); - const text = input.text.slice(0, MAX_POST_CHARS); - const record = { - $type: "app.bsky.feed.post", - text, - createdAt: new Date().toISOString(), - }; + // Facets are not optional decoration: without them a posted URL is inert + // text and a hashtag is just a word. Bluesky parses nothing on its own. + const record = buildPostRecord(input.text, new Date().toISOString()); const res = await fetch(`${pds}/xrpc/com.atproto.repo.createRecord`, { method: "POST", headers: { @@ -133,4 +131,6 @@ export async function createBlueskyPost(input: { }; } -export const BLUESKY_MAX_CHARS = MAX_POST_CHARS; +// Bluesky's limit is 300 graphemes, not 300 JS string units. Callers use +// this for UI counters; truncateForBluesky does the enforcing. +export const BLUESKY_MAX_CHARS = 300; diff --git a/lib/sp/post.ts b/lib/sp/post.ts index dcab8f8b..63defd47 100644 --- a/lib/sp/post.ts +++ b/lib/sp/post.ts @@ -7,6 +7,7 @@ // Extracted so the v1 API can call it without depending on next/cache. import type { SupabaseClient } from "@supabase/supabase-js"; +import { graphemeLength } from "./blueskyFacets"; import { encryptSecret, decryptSecret } from "@/lib/sp/vault"; import { enqueueBrowserPost } from "@/lib/lx/workerClient"; import { resolveSubreddit } from "@/lib/sp/redditSubreddit"; @@ -150,10 +151,13 @@ export async function postViaAccount(args: { let title: string | null = null; let subreddit: string | null = null; if (account.platform === "bluesky") { - if (text.length > BLUESKY_MAX_CHARS) { + // Graphemes, not text.length: Bluesky counts the way a reader does, so + // an emoji is one character to it and two to JavaScript. + const length = graphemeLength(text); + if (length > BLUESKY_MAX_CHARS) { return { ok: false, - error: `Bluesky posts max ${BLUESKY_MAX_CHARS} chars (got ${text.length}).`, + error: `Bluesky posts max ${BLUESKY_MAX_CHARS} characters (got ${length}).`, }; } } else if (account.platform === "reddit") { diff --git a/tests/bluesky-facets.test.ts b/tests/bluesky-facets.test.ts new file mode 100644 index 00000000..966788d8 --- /dev/null +++ b/tests/bluesky-facets.test.ts @@ -0,0 +1,153 @@ +import { describe, it, expect } from "vitest"; +import { + buildFacets, + buildPostRecord, + truncateForBluesky, +} from "@/lib/sp/blueskyFacets"; + +/** What the facet actually highlights, resolved through UTF-8 bytes. */ +function sliceByBytes(text: string, start: number, end: number): string { + return Buffer.from(text, "utf8").subarray(start, end).toString("utf8"); +} + +describe("buildFacets — links", () => { + it("links a bare URL", () => { + const text = "Check out https://pairux.com/@moshcoding"; + const [f] = buildFacets(text); + expect(f.features[0]).toEqual({ + $type: "app.bsky.richtext.facet#link", + uri: "https://pairux.com/@moshcoding", + }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe( + "https://pairux.com/@moshcoding", + ); + }); + + it("leaves sentence punctuation outside the link", () => { + const text = "See https://example.com."; + const [f] = buildFacets(text); + expect(f.features[0]).toMatchObject({ uri: "https://example.com" }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com"); + }); + + it("finds several URLs", () => { + expect(buildFacets("a https://one.test b https://two.test")).toHaveLength(2); + }); +}); + +describe("buildFacets — hashtags", () => { + it("tags a hashtag without the #", () => { + const text = "shipping #moshcoding today"; + const [f] = buildFacets(text); + expect(f.features[0]).toEqual({ $type: "app.bsky.richtext.facet#tag", tag: "moshcoding" }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#moshcoding"); + }); + + it("does not treat a URL fragment as a tag", () => { + const facets = buildFacets("https://example.com/docs#install"); + expect(facets).toHaveLength(1); + expect(facets[0].features[0].$type).toBe("app.bsky.richtext.facet#link"); + }); + + it("ignores a purely numeric tag", () => { + expect(buildFacets("ranked #1 today")).toHaveLength(0); + }); + + it("drops trailing punctuation from a tag", () => { + const [f] = buildFacets("about #opensource, mostly"); + expect(f.features[0]).toMatchObject({ tag: "opensource" }); + }); +}); + +describe("byte offsets, not string indices", () => { + // The bug this guards: JS indices are UTF-16 units, facet offsets are + // UTF-8 bytes. Anything multi-byte earlier in the string shifts them apart. + it("stays correct after an emoji", () => { + const text = "🚀 https://example.com"; + const [f] = buildFacets(text); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com"); + // A naive implementation would report the UTF-16 index of 2 here. + expect(f.index.byteStart).toBe(5); + }); + + it("stays correct after CJK text", () => { + const text = "日本語のテキスト https://example.com #tag"; + const facets = buildFacets(text); + expect(sliceByBytes(text, facets[0].index.byteStart, facets[0].index.byteEnd)).toBe( + "https://example.com", + ); + expect(sliceByBytes(text, facets[1].index.byteStart, facets[1].index.byteEnd)).toBe("#tag"); + }); + + it("stays correct after accented characters", () => { + const text = "café société #naïve"; + const [f] = buildFacets(text); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#naïve"); + }); + + it("returns facets sorted by start offset", () => { + const facets = buildFacets("#one https://a.test #two https://b.test"); + const starts = facets.map((f) => f.index.byteStart); + expect([...starts].sort((a, b) => a - b)).toEqual(starts); + }); + + it("produces no overlapping ranges", () => { + const facets = buildFacets("https://example.com/a#b #real"); + for (let i = 1; i < facets.length; i++) { + expect(facets[i].index.byteStart).toBeGreaterThanOrEqual(facets[i - 1].index.byteEnd); + } + }); +}); + +describe("truncateForBluesky", () => { + it("leaves a short post alone", () => { + expect(truncateForBluesky("hello")).toBe("hello"); + }); + + it("counts emoji as one grapheme, not two units", () => { + const text = "🚀".repeat(300); + expect([...truncateForBluesky(text)].length).toBe(300); + }); + + it("never produces a lone surrogate when the cut lands on an emoji", () => { + // 300 a's then an emoji: the cut falls exactly where slice() would split + // the surrogate pair and emit invalid UTF-8. + const out = truncateForBluesky("a".repeat(300) + "🚀"); + // A high surrogate not followed by a low, or a low not preceded by a high. + const LONE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + const out = truncateForBluesky("x".repeat(299) + "👨‍👩‍👧‍👦"); + expect(Buffer.from(out, "utf8").toString("utf8")).toBe(out); + }); +}); + +describe("buildPostRecord", () => { + it("omits facets entirely when there are none", () => { + const r = buildPostRecord("just text", "2026-07-28T00:00:00Z"); + expect(r).not.toHaveProperty("facets"); + expect(r.$type).toBe("app.bsky.feed.post"); + }); + + it("includes facets when the text has them", () => { + const r = buildPostRecord("see https://example.com #tag", "2026-07-28T00:00:00Z"); + expect(r).toHaveProperty("facets"); + }); + + it("builds facets against the truncated text, so offsets stay in range", () => { + const text = "x".repeat(295) + " https://example.com/very/long/path #tag"; + const r = buildPostRecord(text, "2026-07-28T00:00:00Z"); + const bytes = Buffer.byteLength(r.text, "utf8"); + for (const f of (r as { facets?: { index: { byteEnd: number } }[] }).facets ?? []) { + expect(f.index.byteEnd).toBeLessThanOrEqual(bytes); + } + }); +});