Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
162 changes: 162 additions & 0 deletions lib/sp/blueskyFacets.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,162 @@
// Bluesky rich-text facets.
//
// Bluesky does not parse anything out of post text. A URL posted as plain
// text stays plain text, and a hashtag is just a word starting with '#'.
// Anything that should be clickable has to be described by a facet giving its
// byte range and what it points at. There is no auto-parse flag to set.
//
// The part that bites: those offsets are counted in UTF-8 *bytes*, while
// JavaScript string indices are UTF-16 code units. Any emoji, accented
// character or CJK text before a link shifts the two apart, and the facet
// then highlights the wrong span — usually mid-word, sometimes past the end
// of the string. Every offset here is converted through Buffer.byteLength.

export type BlueskyFacet = {
index: { byteStart: number; byteEnd: number };
features: Array<
| { $type: "app.bsky.richtext.facet#link"; uri: string }
| { $type: "app.bsky.richtext.facet#tag"; tag: string }
>;
};

/** Bluesky counts 300 graphemes, and separately caps the record at 3000 bytes. */
const MAX_GRAPHEMES = 300;
const MAX_BYTES = 3000;

const URL_RE = /https?:\/\/[^\s<>"']+/g;

// A '#' that starts a word, followed by tag characters. The lookbehind stops
// it firing inside a URL fragment or an id like `foo#bar`.
const TAG_RE = /(?<![\w/])#([^\s#.,;:!?()[\]{}<>"']+)/g;

/** Trailing characters that are almost always sentence punctuation, not URL. */
const TRAILING_PUNCT = /[.,;:!?)\]}'"]+$/;

function byteLen(s: string): number {
return Buffer.byteLength(s, "utf8");
}

/**
* Byte offset of a UTF-16 index, which is what the facet index actually means.
*/
function byteOffsetOf(text: string, charIndex: number): number {
return byteLen(text.slice(0, charIndex));
}

/**
* Build the facets for a post.
*
* Returns them sorted by start offset and non-overlapping, which is what the
* API expects; an unsorted or overlapping set is rejected or renders wrong.
*/
export function buildFacets(text: string): BlueskyFacet[] {
const facets: BlueskyFacet[] = [];
const taken: Array<[number, number]> = [];

const overlaps = (start: number, end: number) =>
taken.some(([s, e]) => start < e && end > s);

for (const m of text.matchAll(URL_RE)) {
const raw = m[0];
// "Read https://example.com." should link the URL, not the full stop.
// Closing brackets only come off when unbalanced, so a URL that
// legitimately ends in ')' — Wikipedia does this — survives.
let url = raw.replace(TRAILING_PUNCT, "");
if (url.includes("(") && !url.includes(")") && raw.startsWith(url + ")")) {
url = url + ")";
}
if (!url) continue;

const start = m.index ?? 0;
const end = start + url.length;
if (overlaps(start, end)) continue;
taken.push([start, end]);

facets.push({
index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) },
features: [{ $type: "app.bsky.richtext.facet#link", uri: url }],
});
}

for (const m of text.matchAll(TAG_RE)) {
const tag = m[1].replace(TRAILING_PUNCT, "");
// Bluesky rejects an empty tag, caps them at 64 characters, and a
// purely numeric one is nearly always "#1" in prose rather than a tag.
if (!tag || tag.length > 64 || /^\d+$/.test(tag)) continue;

const start = m.index ?? 0;
const end = start + 1 + tag.length;
if (overlaps(start, end)) continue;
taken.push([start, end]);

facets.push({
index: { byteStart: byteOffsetOf(text, start), byteEnd: byteOffsetOf(text, end) },
features: [{ $type: "app.bsky.richtext.facet#tag", tag }],
});
}

return facets.sort((a, b) => a.index.byteStart - b.index.byteStart);
}

/**
* Length as Bluesky counts it: graphemes, not UTF-16 code units.
*
* `"🚀".length` is 2, and a post of 200 emoji is 400 by that measure but 200
* by Bluesky's — so a naive length check rejects posts the API would accept.
*/
export function graphemeLength(text: string): number {
if (typeof Intl !== "undefined" && "Segmenter" in Intl) {
return [...new Intl.Segmenter(undefined, { granularity: "grapheme" }).segment(text)].length;
}
return [...text].length;
}

/**
* Trim a post to Bluesky's limits without corrupting it.
*
* `text.slice(0, 300)` is wrong twice over: it counts UTF-16 code units, so an
* emoji spends two of the 300, and it can cut between the halves of a
* surrogate pair, producing a lone surrogate that is not valid UTF-8. This
* cuts on grapheme boundaries where the runtime can identify them, then
* enforces the byte ceiling separately.
*/
export function truncateForBluesky(text: string): string {
let out = text;

const segmenter =
typeof Intl !== "undefined" && "Segmenter" in Intl
? new Intl.Segmenter(undefined, { granularity: "grapheme" })
: null;

if (segmenter) {
const graphemes = [...segmenter.segment(out)].map((s) => s.segment);
if (graphemes.length > MAX_GRAPHEMES) out = graphemes.slice(0, MAX_GRAPHEMES).join("");
} else {
// No Segmenter: fall back to code points, which at least never splits a
// surrogate pair the way slice() does.
const points = [...out];
if (points.length > MAX_GRAPHEMES) out = points.slice(0, MAX_GRAPHEMES).join("");
}

// Byte ceiling. Drop whole code points so the result stays valid UTF-8.
while (byteLen(out) > MAX_BYTES) {
const points = [...out];
points.pop();
out = points.join("");
}

return out;
}

/** The full post record, facets included. */
export function buildPostRecord(text: string, createdAt: string) {
const trimmed = truncateForBluesky(text);
const facets = buildFacets(trimmed);
return {
$type: "app.bsky.feed.post" as const,
text: trimmed,
createdAt,
// Omitted entirely when empty: an empty array is legal but noise.
...(facets.length ? { facets } : {}),
};
}
20 changes: 10 additions & 10 deletions lib/sp/platforms/bluesky.ts
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,8 @@
// Posting: com.atproto.repo.createRecord with the app.bsky.feed.post
// collection. Returns the post's at:// URI + cid.

import { buildPostRecord } from "../blueskyFacets";

const DEFAULT_PDS = "https://bsky.social";

export type BlueskySession = {
Expand Down Expand Up @@ -78,9 +80,8 @@ export async function refreshBlueskySession(input: {
return { accessJwt: json.accessJwt, refreshJwt: json.refreshJwt };
}

// Post a single text post. Bluesky's 300-char limit is enforced by
// the API — we trim client-side and document the limit to the user.
const MAX_POST_CHARS = 300;
// Post a single text post. Trimming to Bluesky's limits and working out the
// rich-text facets both live in ../blueskyFacets.ts.

export type BlueskyPostResult = {
uri: string; // at://did:plc:.../app.bsky.feed.post/...
Expand All @@ -96,12 +97,9 @@ export async function createBlueskyPost(input: {
pdsUrl?: string;
}): Promise<BlueskyPostResult> {
const pds = (input.pdsUrl ?? DEFAULT_PDS).replace(/\/$/, "");
const text = input.text.slice(0, MAX_POST_CHARS);
const record = {
$type: "app.bsky.feed.post",
text,
createdAt: new Date().toISOString(),
};
// Facets are not optional decoration: without them a posted URL is inert
// text and a hashtag is just a word. Bluesky parses nothing on its own.
const record = buildPostRecord(input.text, new Date().toISOString());
const res = await fetch(`${pds}/xrpc/com.atproto.repo.createRecord`, {
method: "POST",
headers: {
Expand Down Expand Up @@ -133,4 +131,6 @@ export async function createBlueskyPost(input: {
};
}

export const BLUESKY_MAX_CHARS = MAX_POST_CHARS;
// Bluesky's limit is 300 graphemes, not 300 JS string units. Callers use
// this for UI counters; truncateForBluesky does the enforcing.
export const BLUESKY_MAX_CHARS = 300;
8 changes: 6 additions & 2 deletions lib/sp/post.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
// Extracted so the v1 API can call it without depending on next/cache.

import type { SupabaseClient } from "@supabase/supabase-js";
import { graphemeLength } from "./blueskyFacets";
import { encryptSecret, decryptSecret } from "@/lib/sp/vault";
import { enqueueBrowserPost } from "@/lib/lx/workerClient";
import { resolveSubreddit } from "@/lib/sp/redditSubreddit";
Expand Down Expand Up @@ -150,10 +151,13 @@ export async function postViaAccount(args: {
let title: string | null = null;
let subreddit: string | null = null;
if (account.platform === "bluesky") {
if (text.length > BLUESKY_MAX_CHARS) {
// Graphemes, not text.length: Bluesky counts the way a reader does, so
// an emoji is one character to it and two to JavaScript.
const length = graphemeLength(text);
if (length > BLUESKY_MAX_CHARS) {
return {
ok: false,
error: `Bluesky posts max ${BLUESKY_MAX_CHARS} chars (got ${text.length}).`,
error: `Bluesky posts max ${BLUESKY_MAX_CHARS} characters (got ${length}).`,
};
}
} else if (account.platform === "reddit") {
Expand Down
153 changes: 153 additions & 0 deletions tests/bluesky-facets.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
import { describe, it, expect } from "vitest";
import {
buildFacets,
buildPostRecord,
truncateForBluesky,
} from "@/lib/sp/blueskyFacets";

/** What the facet actually highlights, resolved through UTF-8 bytes. */
function sliceByBytes(text: string, start: number, end: number): string {
return Buffer.from(text, "utf8").subarray(start, end).toString("utf8");
}

describe("buildFacets — links", () => {
it("links a bare URL", () => {
const text = "Check out https://pairux.com/@moshcoding";
const [f] = buildFacets(text);
expect(f.features[0]).toEqual({
$type: "app.bsky.richtext.facet#link",
uri: "https://pairux.com/@moshcoding",
});
expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe(
"https://pairux.com/@moshcoding",
);
});

it("leaves sentence punctuation outside the link", () => {
const text = "See https://example.com.";
const [f] = buildFacets(text);
expect(f.features[0]).toMatchObject({ uri: "https://example.com" });
expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com");
});

it("finds several URLs", () => {
expect(buildFacets("a https://one.test b https://two.test")).toHaveLength(2);
});
});

describe("buildFacets — hashtags", () => {
it("tags a hashtag without the #", () => {
const text = "shipping #moshcoding today";
const [f] = buildFacets(text);
expect(f.features[0]).toEqual({ $type: "app.bsky.richtext.facet#tag", tag: "moshcoding" });
expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#moshcoding");
});

it("does not treat a URL fragment as a tag", () => {
const facets = buildFacets("https://example.com/docs#install");
expect(facets).toHaveLength(1);
expect(facets[0].features[0].$type).toBe("app.bsky.richtext.facet#link");
});

it("ignores a purely numeric tag", () => {
expect(buildFacets("ranked #1 today")).toHaveLength(0);
});

it("drops trailing punctuation from a tag", () => {
const [f] = buildFacets("about #opensource, mostly");
expect(f.features[0]).toMatchObject({ tag: "opensource" });
});
});

describe("byte offsets, not string indices", () => {
// The bug this guards: JS indices are UTF-16 units, facet offsets are
// UTF-8 bytes. Anything multi-byte earlier in the string shifts them apart.
it("stays correct after an emoji", () => {
const text = "🚀 https://example.com";
const [f] = buildFacets(text);
expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("https://example.com");
// A naive implementation would report the UTF-16 index of 2 here.
expect(f.index.byteStart).toBe(5);
});

it("stays correct after CJK text", () => {
const text = "日本語のテキスト https://example.com #tag";
const facets = buildFacets(text);
expect(sliceByBytes(text, facets[0].index.byteStart, facets[0].index.byteEnd)).toBe(
"https://example.com",
);
expect(sliceByBytes(text, facets[1].index.byteStart, facets[1].index.byteEnd)).toBe("#tag");
});

it("stays correct after accented characters", () => {
const text = "café société #naïve";
const [f] = buildFacets(text);
expect(sliceByBytes(text, f.index.byteStart, f.index.byteEnd)).toBe("#naïve");
});

it("returns facets sorted by start offset", () => {
const facets = buildFacets("#one https://a.test #two https://b.test");
const starts = facets.map((f) => f.index.byteStart);
expect([...starts].sort((a, b) => a - b)).toEqual(starts);
});

it("produces no overlapping ranges", () => {
const facets = buildFacets("https://example.com/a#b #real");
for (let i = 1; i < facets.length; i++) {
expect(facets[i].index.byteStart).toBeGreaterThanOrEqual(facets[i - 1].index.byteEnd);
}
});
});

describe("truncateForBluesky", () => {
it("leaves a short post alone", () => {
expect(truncateForBluesky("hello")).toBe("hello");
});

it("counts emoji as one grapheme, not two units", () => {
const text = "🚀".repeat(300);
expect([...truncateForBluesky(text)].length).toBe(300);
});

it("never produces a lone surrogate when the cut lands on an emoji", () => {
// 300 a's then an emoji: the cut falls exactly where slice() would split
// the surrogate pair and emit invalid UTF-8.
const out = truncateForBluesky("a".repeat(300) + "🚀");
// A high surrogate not followed by a low, or a low not preceded by a high.
const LONE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/;
expect(out).not.toMatch(LONE);
// And it round-trips through UTF-8 unchanged, which is the real bar.
expect(Buffer.from(out, "utf8").toString("utf8")).toBe(out);

// Demonstrates the bug being fixed: the naive version does not survive.
const naive = ("a".repeat(300) + "🚀").slice(0, 301);
expect(naive).toMatch(LONE);
});

it("keeps a family emoji whole rather than splitting the sequence", () => {
const out = truncateForBluesky("x".repeat(299) + "👨‍👩‍👧‍👦");
expect(Buffer.from(out, "utf8").toString("utf8")).toBe(out);
});
});

describe("buildPostRecord", () => {
it("omits facets entirely when there are none", () => {
const r = buildPostRecord("just text", "2026-07-28T00:00:00Z");
expect(r).not.toHaveProperty("facets");
expect(r.$type).toBe("app.bsky.feed.post");
});

it("includes facets when the text has them", () => {
const r = buildPostRecord("see https://example.com #tag", "2026-07-28T00:00:00Z");
expect(r).toHaveProperty("facets");
});

it("builds facets against the truncated text, so offsets stay in range", () => {
const text = "x".repeat(295) + " https://example.com/very/long/path #tag";
const r = buildPostRecord(text, "2026-07-28T00:00:00Z");
const bytes = Buffer.byteLength(r.text, "utf8");
for (const f of (r as { facets?: { index: { byteEnd: number } }[] }).facets ?? []) {
expect(f.index.byteEnd).toBeLessThanOrEqual(bytes);
}
});
});
Loading