From 05cf9f0f01fb64a5c9e52b9a0f9deacc783bf597 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 02:19:05 +0000 Subject: [PATCH] fix(social): send Bluesky rich-text facets, so links and tags are live MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bluesky parses nothing out of post text. A URL posted as plain text stays plain text and a hashtag is just a word starting with '#'. Anything clickable has to be described by a facet giving its byte range and what it points at, and there is no auto-parse flag to turn on. We were posting neither, so every link and tag we published was inert. The part that makes this easy to get wrong is that facet offsets are counted in UTF-8 bytes while JavaScript string indices are UTF-16 code units. One emoji, accented character or CJK word before a link shifts the two apart, and the facet then highlights the wrong span โ€” mid-word, or past the end of the string. Every offset here goes through Buffer.byteLength, and the tests assert by slicing the UTF-8 buffer at the offsets we emit rather than by trusting the numbers. Two related length bugs came out of the same confusion. Truncation used text.slice(0, 300): that spends two of the 300 on every emoji, and can cut between the halves of a surrogate pair, producing a lone surrogate that is not valid UTF-8. The pre-flight check compared text.length against the limit, so a post of 200 emoji measured 400 and was rejected though the API would have taken it. Both now count graphemes, which is how Bluesky counts and how a reader reads. URLs keep sentence punctuation outside the link, tags drop trailing punctuation, purely numeric tags are ignored as prose ("ranked #1"), and a '#' inside a URL fragment is not mistaken for a tag. Co-Authored-By: Claude Opus 5 (1M context) --- lib/sp/blueskyFacets.ts | 162 +++++++++++++++++++++++++++++++++++ lib/sp/platforms/bluesky.ts | 20 ++--- lib/sp/post.ts | 8 +- tests/bluesky-facets.test.ts | 153 +++++++++++++++++++++++++++++++++ 4 files changed, 331 insertions(+), 12 deletions(-) create mode 100644 lib/sp/blueskyFacets.ts create mode 100644 tests/bluesky-facets.test.ts diff --git a/lib/sp/blueskyFacets.ts b/lib/sp/blueskyFacets.ts new file mode 100644 index 00000000..d0e16f03 --- /dev/null +++ b/lib/sp/blueskyFacets.ts @@ -0,0 +1,162 @@ +// Bluesky rich-text facets. +// +// Bluesky does not parse anything out of post text. A URL posted as plain +// text stays plain text, and a hashtag is just a word starting with '#'. +// Anything that should be clickable has to be described by a facet giving its +// byte range and what it points at. There is no auto-parse flag to set. +// +// The part that bites: those offsets are counted in UTF-8 *bytes*, while +// JavaScript string indices are UTF-16 code units. Any emoji, accented +// character or CJK text before a link shifts the two apart, and the facet +// then highlights the wrong span โ€” usually mid-word, sometimes past the end +// of the string. Every offset here is converted through Buffer.byteLength. + +export type BlueskyFacet = { + index: { byteStart: number; byteEnd: number }; + features: Array< + | { $type: "app.bsky.richtext.facet#link"; uri: string } + | { $type: "app.bsky.richtext.facet#tag"; tag: string } + >; +}; + +/** Bluesky counts 300 graphemes, and separately caps the record at 3000 bytes. */ +const MAX_GRAPHEMES = 300; +const MAX_BYTES = 3000; + +const URL_RE = /https?:\/\/[^\s<>"']+/g; + +// A '#' that starts a word, followed by tag characters. The lookbehind stops +// it firing inside a URL fragment or an id like `foo#bar`. +const TAG_RE = /(?"']+)/g; + +/** Trailing characters that are almost always sentence punctuation, not URL. */ +const TRAILING_PUNCT = /[.,;:!?)\]}'"]+$/; + +function byteLen(s: string): number { + return Buffer.byteLength(s, "utf8"); +} + +/** + * Byte offset of a UTF-16 index, which is what the facet index actually means. + */ +function byteOffsetOf(text: string, charIndex: number): number { + return byteLen(text.slice(0, charIndex)); +} + +/** + * Build the facets for a post. + * + * Returns them sorted by start offset and non-overlapping, which is what the + * API expects; an unsorted or overlapping set is rejected or renders wrong. + */ +export function buildFacets(text: string): BlueskyFacet[] { + const facets: BlueskyFacet[] = []; + const taken: Array<[number, number]> = []; + + const overlaps = (start: number, end: number) => + taken.some(([s, e]) => start < e && end > s); + + for (const m of text.matchAll(URL_RE)) { + const raw = m[0]; + // "Read https://example.com." should link the URL, not the full stop. + // Closing brackets only come off when unbalanced, so a URL that + // legitimately ends in ')' โ€” Wikipedia does this โ€” survives. + let url = raw.replace(TRAILING_PUNCT, ""); + if (url.includes("(") && !url.includes(")") && raw.startsWith(url + ")")) { + url = url + ")"; + } + if (!url) continue; + + const start = m.index ?? 0; + const end = start + url.length; + if (overlaps(start, end)) continue; + taken.push([start, end]); + + facets.push({ + index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) }, + features: [{ $type: "app.bsky.richtext.facet#link", uri: url }], + }); + } + + for (const m of text.matchAll(TAG_RE)) { + const tag = m[1].replace(TRAILING_PUNCT, ""); + // Bluesky rejects an empty tag, caps them at 64 characters, and a + // purely numeric one is nearly always "#1" in prose rather than a tag. + if (!tag || tag.length > 64 || /^\d+$/.test(tag)) continue; + + const start = m.index ?? 0; + const end = start + 1 + tag.length; + if (overlaps(start, end)) continue; + taken.push([start, end]); + + facets.push({ + index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) }, + features: [{ $type: "app.bsky.richtext.facet#tag", tag }], + }); + } + + return facets.sort((a, b) => a.index.byteStart - b.index.byteStart); +} + +/** + * Length as Bluesky counts it: graphemes, not UTF-16 code units. + * + * `"๐Ÿš€".length` is 2, and a post of 200 emoji is 400 by that measure but 200 + * by Bluesky's โ€” so a naive length check rejects posts the API would accept. + */ +export function graphemeLength(text: string): number { + if (typeof Intl !== "undefined" && "Segmenter" in Intl) { + return [...new Intl.Segmenter(undefined, { granularity: "grapheme" }).segment(text)].length; + } + return [...text].length; +} + +/** + * Trim a post to Bluesky's limits without corrupting it. + * + * `text.slice(0, 300)` is wrong twice over: it counts UTF-16 code units, so an + * emoji spends two of the 300, and it can cut between the halves of a + * surrogate pair, producing a lone surrogate that is not valid UTF-8. This + * cuts on grapheme boundaries where the runtime can identify them, then + * enforces the byte ceiling separately. + */ +export function truncateForBluesky(text: string): string { + let out = text; + + const segmenter = + typeof Intl !== "undefined" && "Segmenter" in Intl + ? new Intl.Segmenter(undefined, { granularity: "grapheme" }) + : null; + + if (segmenter) { + const graphemes = [...segmenter.segment(out)].map((s) => s.segment); + if (graphemes.length > MAX_GRAPHEMES) out = graphemes.slice(0, MAX_GRAPHEMES).join(""); + } else { + // No Segmenter: fall back to code points, which at least never splits a + // surrogate pair the way slice() does. + const points = [...out]; + if (points.length > MAX_GRAPHEMES) out = points.slice(0, MAX_GRAPHEMES).join(""); + } + + // Byte ceiling. Drop whole code points so the result stays valid UTF-8. + while (byteLen(out) > MAX_BYTES) { + const points = [...out]; + points.pop(); + out = points.join(""); + } + + return out; +} + +/** The full post record, facets included. */ +export function buildPostRecord(text: string, createdAt: string) { + const trimmed = truncateForBluesky(text); + const facets = buildFacets(trimmed); + return { + $type: "app.bsky.feed.post" as const, + text: trimmed, + createdAt, + // Omitted entirely when empty: an empty array is legal but noise. + ...(facets.length ? { facets } : {}), + }; +} diff --git a/lib/sp/platforms/bluesky.ts b/lib/sp/platforms/bluesky.ts index 5a9b2c9b..3e465526 100644 --- a/lib/sp/platforms/bluesky.ts +++ b/lib/sp/platforms/bluesky.ts @@ -14,6 +14,8 @@ // Posting: com.atproto.repo.createRecord with the app.bsky.feed.post // collection. Returns the post's at:// URI + cid. +import { buildPostRecord } from "../blueskyFacets"; + const DEFAULT_PDS = "https://bsky.social"; export type BlueskySession = { @@ -78,9 +80,8 @@ export async function refreshBlueskySession(input: { return { accessJwt: json.accessJwt, refreshJwt: json.refreshJwt }; } -// Post a single text post. Bluesky's 300-char limit is enforced by -// the API โ€” we trim client-side and document the limit to the user. -const MAX_POST_CHARS = 300; +// Post a single text post. Trimming to Bluesky's limits and working out the +// rich-text facets both live in ../blueskyFacets.ts. export type BlueskyPostResult = { uri: string; // at://did:plc:.../app.bsky.feed.post/... @@ -96,12 +97,9 @@ export async function createBlueskyPost(input: { pdsUrl?: string; }): Promise { const pds = (input.pdsUrl ?? DEFAULT_PDS).replace(/\/$/, ""); - const text = input.text.slice(0, MAX_POST_CHARS); - const record = { - $type: "app.bsky.feed.post", - text, - createdAt: new Date().toISOString(), - }; + // Facets are not optional decoration: without them a posted URL is inert + // text and a hashtag is just a word. Bluesky parses nothing on its own. + const record = buildPostRecord(input.text, new Date().toISOString()); const res = await fetch(`${pds}/xrpc/com.atproto.repo.createRecord`, { method: "POST", headers: { @@ -133,4 +131,6 @@ export async function createBlueskyPost(input: { }; } -export const BLUESKY_MAX_CHARS = MAX_POST_CHARS; +// Bluesky's limit is 300 graphemes, not 300 JS string units. Callers use +// this for UI counters; truncateForBluesky does the enforcing. +export const BLUESKY_MAX_CHARS = 300; diff --git a/lib/sp/post.ts b/lib/sp/post.ts index dcab8f8b..63defd47 100644 --- a/lib/sp/post.ts +++ b/lib/sp/post.ts @@ -7,6 +7,7 @@ // Extracted so the v1 API can call it without depending on next/cache. import type { SupabaseClient } from "@supabase/supabase-js"; +import { graphemeLength } from "./blueskyFacets"; import { encryptSecret, decryptSecret } from "@/lib/sp/vault"; import { enqueueBrowserPost } from "@/lib/lx/workerClient"; import { resolveSubreddit } from "@/lib/sp/redditSubreddit"; @@ -150,10 +151,13 @@ export async function postViaAccount(args: { let title: string | null = null; let subreddit: string | null = null; if (account.platform === "bluesky") { - if (text.length > BLUESKY_MAX_CHARS) { + // Graphemes, not text.length: Bluesky counts the way a reader does, so + // an emoji is one character to it and two to JavaScript. + const length = graphemeLength(text); + if (length > BLUESKY_MAX_CHARS) { return { ok: false, - error: `Bluesky posts max ${BLUESKY_MAX_CHARS} chars (got ${text.length}).`, + error: `Bluesky posts max ${BLUESKY_MAX_CHARS} characters (got ${length}).`, }; } } else if (account.platform === "reddit") { diff --git a/tests/bluesky-facets.test.ts b/tests/bluesky-facets.test.ts new file mode 100644 index 00000000..966788d8 --- /dev/null +++ b/tests/bluesky-facets.test.ts @@ -0,0 +1,153 @@ +import { describe, it, expect } from "vitest"; +import { + buildFacets, + buildPostRecord, + truncateForBluesky, +} from "@/lib/sp/blueskyFacets"; + +/** What the facet actually highlights, resolved through UTF-8 bytes. */ +function sliceByBytes(text: string, start: number, end: number): string { + return Buffer.from(text, "utf8").subarray(start, end).toString("utf8"); +} + +describe("buildFacets โ€” links", () => { + it("links a bare URL", () => { + const text = "Check out https://pairux.com/@moshcoding"; + const [f] = buildFacets(text); + expect(f.features[0]).toEqual({ + $type: "app.bsky.richtext.facet#link", + uri: "https://pairux.com/@moshcoding", + }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe( + "https://pairux.com/@moshcoding", + ); + }); + + it("leaves sentence punctuation outside the link", () => { + const text = "See https://example.com."; + const [f] = buildFacets(text); + expect(f.features[0]).toMatchObject({ uri: "https://example.com" }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com"); + }); + + it("finds several URLs", () => { + expect(buildFacets("a https://one.test b https://two.test")).toHaveLength(2); + }); +}); + +describe("buildFacets โ€” hashtags", () => { + it("tags a hashtag without the #", () => { + const text = "shipping #moshcoding today"; + const [f] = buildFacets(text); + expect(f.features[0]).toEqual({ $type: "app.bsky.richtext.facet#tag", tag: "moshcoding" }); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#moshcoding"); + }); + + it("does not treat a URL fragment as a tag", () => { + const facets = buildFacets("https://example.com/docs#install"); + expect(facets).toHaveLength(1); + expect(facets[0].features[0].$type).toBe("app.bsky.richtext.facet#link"); + }); + + it("ignores a purely numeric tag", () => { + expect(buildFacets("ranked #1 today")).toHaveLength(0); + }); + + it("drops trailing punctuation from a tag", () => { + const [f] = buildFacets("about #opensource, mostly"); + expect(f.features[0]).toMatchObject({ tag: "opensource" }); + }); +}); + +describe("byte offsets, not string indices", () => { + // The bug this guards: JS indices are UTF-16 units, facet offsets are + // UTF-8 bytes. Anything multi-byte earlier in the string shifts them apart. + it("stays correct after an emoji", () => { + const text = "๐Ÿš€ https://example.com"; + const [f] = buildFacets(text); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com"); + // A naive implementation would report the UTF-16 index of 2 here. + expect(f.index.byteStart).toBe(5); + }); + + it("stays correct after CJK text", () => { + const text = "ๆ—ฅๆœฌ่ชžใฎใƒ†ใ‚ญใ‚นใƒˆ https://example.com #tag"; + const facets = buildFacets(text); + expect(sliceByBytes(text, facets[0].index.byteStart, facets[0].index.byteEnd)).toBe( + "https://example.com", + ); + expect(sliceByBytes(text, facets[1].index.byteStart, facets[1].index.byteEnd)).toBe("#tag"); + }); + + it("stays correct after accented characters", () => { + const text = "cafรฉ sociรฉtรฉ #naรฏve"; + const [f] = buildFacets(text); + expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#naรฏve"); + }); + + it("returns facets sorted by start offset", () => { + const facets = buildFacets("#one https://a.test #two https://b.test"); + const starts = facets.map((f) => f.index.byteStart); + expect([...starts].sort((a, b) => a - b)).toEqual(starts); + }); + + it("produces no overlapping ranges", () => { + const facets = buildFacets("https://example.com/a#b #real"); + for (let i = 1; i < facets.length; i++) { + expect(facets[i].index.byteStart).toBeGreaterThanOrEqual(facets[i - 1].index.byteEnd); + } + }); +}); + +describe("truncateForBluesky", () => { + it("leaves a short post alone", () => { + expect(truncateForBluesky("hello")).toBe("hello"); + }); + + it("counts emoji as one grapheme, not two units", () => { + const text = "๐Ÿš€".repeat(300); + expect([...truncateForBluesky(text)].length).toBe(300); + }); + + it("never produces a lone surrogate when the cut lands on an emoji", () => { + // 300 a's then an emoji: the cut falls exactly where slice() would split + // the surrogate pair and emit invalid UTF-8. + const out = truncateForBluesky("a".repeat(300) + "๐Ÿš€"); + // A high surrogate not followed by a low, or a low not preceded by a high. + const LONE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + const out = truncateForBluesky("x".repeat(299) + "๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ"); + expect(Buffer.from(out, "utf8").toString("utf8")).toBe(out); + }); +}); + +describe("buildPostRecord", () => { + it("omits facets entirely when there are none", () => { + const r = buildPostRecord("just text", "2026-07-28T00:00:00Z"); + expect(r).not.toHaveProperty("facets"); + expect(r.$type).toBe("app.bsky.feed.post"); + }); + + it("includes facets when the text has them", () => { + const r = buildPostRecord("see https://example.com #tag", "2026-07-28T00:00:00Z"); + expect(r).toHaveProperty("facets"); + }); + + it("builds facets against the truncated text, so offsets stay in range", () => { + const text = "x".repeat(295) + " https://example.com/very/long/path #tag"; + const r = buildPostRecord(text, "2026-07-28T00:00:00Z"); + const bytes = Buffer.byteLength(r.text, "utf8"); + for (const f of (r as { facets?: { index: { byteEnd: number } }[] }).facets ?? []) { + expect(f.index.byteEnd).toBeLessThanOrEqual(bytes); + } + }); +});