From 3408480fe8c581befd1b7d25ab3dbd82751259f8 Mon Sep 17 00:00:00 2001
From: Seongho Bae
Date: Mon, 24 Aug 2026 03:36:15 +0900
Subject: [PATCH 1/3] fix: normalize encoded scripts in plain source
---
frontend/src/PostBody.test.tsx | 9 +++++++++
lineageweave/chunking.py | 32 ++++++++++++++++++++++++++------
tests/test_chunking.py | 19 +++++++++++++++++++
3 files changed, 54 insertions(+), 6 deletions(-)
diff --git a/frontend/src/PostBody.test.tsx b/frontend/src/PostBody.test.tsx
index 76837b5b2..3039c0fcb 100644
--- a/frontend/src/PostBody.test.tsx
+++ b/frontend/src/PostBody.test.tsx
@@ -58,6 +58,15 @@ describe("PostBody", () => {
expect(screen.getByText("Embedded image")).toBeInTheDocument();
});
+ it("renders encoded malicious markup as inert text", () => {
+ const { container } = render(
+ ,
+ );
+
+ expect(container.querySelector("script")).toBeNull();
+ expect(screen.getByText("Keep alert(1) literal.")).toBeInTheDocument();
+ });
+
it("renders authoritative LLM structure levels for semantic list units", () => {
render(
bool:
_CARET_EXPONENT = re.compile(
r"(?<=[A-Za-z0-9µμ°ΩÅåÅ)])\^(?:\{([+\-]?\d{1,3}|[nNiI])\}|([+\-]?\d{1,3}|[nNiI]))"
)
+_ENCODED_CARET = re.compile(r"&(?:amp;)*(?:#0*94|#x0*5e);", re.IGNORECASE)
+_ENCODED_SCRIPT_TAG = re.compile(
+ r"&(?:amp;)*(?:lt|#0*60|#x0*3c);\s*/?\s*(?:sup|sub)"
+ r"(?=\s|/|&(?:amp;)*(?:gt|#0*62|#x0*3e);).*?"
+ r"&(?:amp;)*(?:gt|#0*62|#x0*3e);",
+ re.IGNORECASE | re.DOTALL,
+)
def apply_unicode_script(text: str, kind: str) -> str:
@@ -213,13 +220,26 @@ def apply_unicode_script(text: str, kind: str) -> str:
return f"{leading}{prefix}{compact}{trailing}"
-def _replace_html_script(match: re.Match[str], kind: str) -> str:
- inner = match.group(1)
+def _decode_html_entities(text: str) -> str:
for _ in range(3):
- decoded = unescape(inner)
- if decoded == inner:
+ decoded = unescape(text)
+ if decoded == text:
break
- inner = decoded
+ text = decoded
+ return text
+
+
+def _decode_script_entities(text: str) -> str:
+ decoded_tags = _ENCODED_SCRIPT_TAG.sub(
+ lambda match: _decode_html_entities(match.group(0)), text
+ )
+ return _ENCODED_CARET.sub(
+ lambda match: _decode_html_entities(match.group(0)), decoded_tags
+ )
+
+
+def _replace_html_script(match: re.Match[str], kind: str) -> str:
+ inner = _decode_html_entities(match.group(1))
return apply_unicode_script(_INNER_TAG.sub("", inner), kind)
@@ -227,7 +247,7 @@ def normalize_script_text(text: str) -> str:
"""Turn HTML/caret quantity scripts into Unicode without treating comparisons as tags."""
replaced = _CARET_EXPONENT.sub(
lambda match: apply_unicode_script(match.group(1) or match.group(2), "sup"),
- text,
+ _decode_script_entities(text),
)
replaced = _HTML_SUP.sub(lambda match: _replace_html_script(match, "sup"), replaced)
replaced = _HTML_SUB.sub(lambda match: _replace_html_script(match, "sub"), replaced)
diff --git a/tests/test_chunking.py b/tests/test_chunking.py
index ab70048b3..49ecda543 100644
--- a/tests/test_chunking.py
+++ b/tests/test_chunking.py
@@ -284,6 +284,25 @@ def test_chunk_by_source_body_splits_plain_lists_and_markdown_tables() -> None:
]
+def test_chunk_by_source_body_normalizes_entity_encoded_quantity_scripts() -> None:
+ chunks = chunk_by_source_body(
+ "Reserve 12 m^3, x<sup>2</sup>, and H<sub>2</sub>O."
+ )
+
+ assert [chunk.text for chunk in chunks] == ["Reserve 12 m³, x², and H₂O."]
+
+
+def test_chunk_by_source_body_keeps_encoded_non_script_markup_inert() -> None:
+ body = (
+ "Keep <b>bold</b>, <sup-note>2</sup-note>, "
+ "and <script>alert(1)</script> literal."
+ )
+
+ chunks = chunk_by_source_body(body)
+
+ assert [(chunk.unit_type, chunk.text) for chunk in chunks] == [("plain_text", body)]
+
+
def test_chunk_by_source_body_preserves_empty_markdown_table_cells() -> None:
chunks = chunk_by_source_body(
"| Key | Value | State |\n"
From ebe6985f9fab466b5ebabedd538ff5ad93244c5d Mon Sep 17 00:00:00 2001
From: Seongho Bae
Date: Mon, 24 Aug 2026 03:48:42 +0900
Subject: [PATCH 2/3] fix: keep invalid encoded script markup literal
---
frontend/src/PostBody.test.tsx | 35 ++++++++++++++++++++++++----
frontend/src/PostBody.tsx | 16 +++++++++----
frontend/src/postBodyDisplay.test.ts | 16 +++++++++++++
frontend/src/postBodyDisplay.ts | 23 ++++++++++++++----
lineageweave/chunking.py | 23 ++++++++++++------
tests/test_chunking.py | 13 +++++++++++
6 files changed, 106 insertions(+), 20 deletions(-)
diff --git a/frontend/src/PostBody.test.tsx b/frontend/src/PostBody.test.tsx
index 3039c0fcb..1ead554bd 100644
--- a/frontend/src/PostBody.test.tsx
+++ b/frontend/src/PostBody.test.tsx
@@ -58,13 +58,40 @@ describe("PostBody", () => {
expect(screen.getByText("Embedded image")).toBeInTheDocument();
});
- it("renders encoded malicious markup as inert text", () => {
- const { container } = render(
- ,
+ it("renders raw and persisted encoded non-script markup as the same inert text", () => {
+ const encoded =
+ "Keep <b>bold</b>, <sup-note>2</sup-note>, " +
+ "<sub:item>3</sub:item>, and <script>alert(1)</script> literal.";
+ const visible =
+ "Keep bold, 2, 3, and literal.";
+ const { container, rerender } = render(${encoded}
`} />);
+
+ expect(screen.getByText(visible)).toBeInTheDocument();
+ expect(container.querySelector("b")).toBeNull();
+ expect(container.querySelector("script")).toBeNull();
+ expect(container.querySelector("sup-note")).toBeNull();
+
+ rerender(
+ ${encoded}`}
+ structureUnits={[
+ {
+ unit_index: 0,
+ unit_kind_code: "plain_text",
+ unit_text: encoded,
+ indent_level: 0,
+ indent_source_code: "explicit",
+ indent_confidence: 1,
+ indent_evidence: "Synthetic encoded source",
+ },
+ ]}
+ />,
);
+ expect(screen.getByText(visible)).toBeInTheDocument();
+ expect(container.querySelector("b")).toBeNull();
expect(container.querySelector("script")).toBeNull();
- expect(screen.getByText("Keep alert(1) literal.")).toBeInTheDocument();
+ expect(container.querySelector("sup-note")).toBeNull();
});
it("renders authoritative LLM structure levels for semantic list units", () => {
diff --git a/frontend/src/PostBody.tsx b/frontend/src/PostBody.tsx
index 080d375c5..48a708ca0 100644
--- a/frontend/src/PostBody.tsx
+++ b/frontend/src/PostBody.tsx
@@ -1,4 +1,10 @@
-import { splitPostBody, splitScriptRuns, normalizeScriptText, type PostBodySegment } from "./postBodyDisplay";
+import {
+ decodeHtmlEntities,
+ splitPostBody,
+ splitScriptRuns,
+ normalizeScriptText,
+ type PostBodySegment,
+} from "./postBodyDisplay";
import { t } from "./i18n";
import type { PostContentUnit, PostImageContent } from "./api";
import { Fragment, type ReactNode } from "react";
@@ -141,7 +147,7 @@ function isStructuredTableRow(unit: PostContentUnit): boolean {
* indentation for every later unresolved unit.
*/
function normalizedUnitText(value: string): string {
- return normalizeScriptText(value.replace(/\s+/g, " ").trim());
+ return decodeHtmlEntities(normalizeScriptText(value)).replace(/\s+/g, " ").trim();
}
/**
@@ -232,7 +238,9 @@ function renderStructuredUnits(
{rows.map((row, rowIndex) => (
{row.unit_text.split(/\s*\|\s*/).map((cell, cellIndex) => (
- | {renderStyledText(cell)} |
+
+ {renderStyledText(decodeHtmlEntities(cell))}
+ |
))}
))}
@@ -251,7 +259,7 @@ function renderStructuredUnits(
renderSegment(
{
kind: "text",
- text: unit.unit_text,
+ text: decodeHtmlEntities(unit.unit_text),
...(unit.unit_label === "footnote" || sourceText?.role === "footnote"
? { role: "footnote" as const }
: {}),
diff --git a/frontend/src/postBodyDisplay.test.ts b/frontend/src/postBodyDisplay.test.ts
index 20e266122..3a5a5e3af 100644
--- a/frontend/src/postBodyDisplay.test.ts
+++ b/frontend/src/postBodyDisplay.test.ts
@@ -166,6 +166,22 @@ describe("splitPostBody", () => {
).toEqual([{ kind: "text", text: "Reserve 12 m³ and x² units." }]);
});
+ it("keeps invalid encoded script pairs literal", () => {
+ expect(
+ splitPostBody(
+ "Keep x<sup>2 unmatched; x<sup/>2 self-closing; " +
+ "x<sup class="unit">2</sup> attributed; and " +
+ "x<sup>2</sub> mismatched.
",
+ ),
+ ).toEqual([
+ {
+ kind: "text",
+ text:
+ 'Keep x2 unmatched; x2 self-closing; x2 attributed; and x2 mismatched.',
+ },
+ ]);
+ });
+
it("keeps encoded non-script inline markup literal", () => {
expect(splitPostBody("Keep <b>bold</b> literal.
")).toEqual([
{ kind: "text", text: "Keep bold literal." },
diff --git a/frontend/src/postBodyDisplay.ts b/frontend/src/postBodyDisplay.ts
index f037a8f1a..6a4f17540 100644
--- a/frontend/src/postBodyDisplay.ts
+++ b/frontend/src/postBodyDisplay.ts
@@ -39,8 +39,10 @@ export function decodeHtmlEntities(text: string): string {
const decoder = document.createElement("textarea");
let decoded = text;
for (let pass = 0; pass < 3; pass += 1) {
- decoder.innerHTML = decoded;
- const next = decoder.value;
+ const next = decoded.replace(/&(?:#[0-9]+|#x[0-9a-f]+|[a-z][a-z0-9]+);/gi, (entity) => {
+ decoder.innerHTML = entity;
+ return decoder.value;
+ });
if (next === decoded) break;
decoded = next;
}
@@ -156,8 +158,15 @@ const SUB_UNI_TO_ASCII = buildUnicodeToAsciiTable(SUB_ASCII_TO_UNI);
const CARET_EXPONENT =
/(?<=[A-Za-z0-9µμ°ΩÅåÅ)])\^(?:\{([+-]?\d{1,3}|[nNiI])\}|([+-]?\d{1,3}|[nNiI]))/g;
const ENCODED_CARET = /&(?:amp;)*(?:#0*94|#x0*5e);/gi;
-const ENCODED_SCRIPT_TAG =
- /&(?:amp;)*(?:lt|#0*60|#x0*3c);\s*\/?\s*(?:sup|sub)(?=\s|\/|&(?:amp;)*(?:gt|#0*62|#x0*3e);).*?&(?:amp;)*(?:gt|#0*62|#x0*3e);/gis;
+const ENCODED_LT = String.raw`&(?:amp;)*(?:lt|#0*60|#x0*3c);`;
+const ENCODED_GT = String.raw`&(?:amp;)*(?:gt|#0*62|#x0*3e);`;
+const ENCODED_SCRIPT_TOKEN =
+ `${ENCODED_LT}\\s*/?\\s*(?:sup|sub)(?=\\s|/|${ENCODED_GT})`;
+const ENCODED_SCRIPT_PAIR = new RegExp(
+ `${ENCODED_LT}(sup|sub)${ENCODED_GT}` +
+ `((?:(?!${ENCODED_SCRIPT_TOKEN}).)*?)${ENCODED_LT}/\\1${ENCODED_GT}`,
+ "gis",
+);
function applyUnicodeScript(text: string, kind: "super" | "sub"): string {
const table = kind === "super" ? SUPER_ASCII_TO_UNI : SUB_ASCII_TO_UNI;
@@ -185,7 +194,11 @@ function replaceHtmlScripts(text: string): string {
function decodeScriptEntities(text: string): string {
return text
- .replace(ENCODED_SCRIPT_TAG, (tag) => decodeHtmlEntities(tag))
+ .replace(
+ ENCODED_SCRIPT_PAIR,
+ (_pair, kind: string, inner: string) =>
+ `<${kind.toLowerCase()}>${inner}${kind.toLowerCase()}>`,
+ )
.replace(ENCODED_CARET, (caret) => decodeHtmlEntities(caret));
}
diff --git a/lineageweave/chunking.py b/lineageweave/chunking.py
index 70f5eeba5..fc66e2ec3 100644
--- a/lineageweave/chunking.py
+++ b/lineageweave/chunking.py
@@ -195,10 +195,15 @@ def _is_footnote_reference(attrs: list[tuple[str, str | None]]) -> bool:
r"(?<=[A-Za-z0-9µμ°ΩÅåÅ)])\^(?:\{([+\-]?\d{1,3}|[nNiI])\}|([+\-]?\d{1,3}|[nNiI]))"
)
_ENCODED_CARET = re.compile(r"&(?:amp;)*(?:#0*94|#x0*5e);", re.IGNORECASE)
-_ENCODED_SCRIPT_TAG = re.compile(
- r"&(?:amp;)*(?:lt|#0*60|#x0*3c);\s*/?\s*(?:sup|sub)"
- r"(?=\s|/|&(?:amp;)*(?:gt|#0*62|#x0*3e);).*?"
- r"&(?:amp;)*(?:gt|#0*62|#x0*3e);",
+_ENCODED_LT = r"&(?:amp;)*(?:lt|#0*60|#x0*3c);"
+_ENCODED_GT = r"&(?:amp;)*(?:gt|#0*62|#x0*3e);"
+_ENCODED_SCRIPT_TOKEN = (
+ rf"{_ENCODED_LT}\s*/?\s*(?:sup|sub)(?=\s|/|{_ENCODED_GT})"
+)
+_ENCODED_SCRIPT_PAIR = re.compile(
+ rf"{_ENCODED_LT}(?Psup|sub){_ENCODED_GT}"
+ rf"(?P(?:(?!{_ENCODED_SCRIPT_TOKEN}).)*?)"
+ rf"{_ENCODED_LT}/(?P=kind){_ENCODED_GT}",
re.IGNORECASE | re.DOTALL,
)
@@ -230,11 +235,15 @@ def _decode_html_entities(text: str) -> str:
def _decode_script_entities(text: str) -> str:
- decoded_tags = _ENCODED_SCRIPT_TAG.sub(
- lambda match: _decode_html_entities(match.group(0)), text
+ decoded_pairs = _ENCODED_SCRIPT_PAIR.sub(
+ lambda match: (
+ f"<{match.group('kind').lower()}>{match.group('inner')}"
+ f"{match.group('kind').lower()}>"
+ ),
+ text,
)
return _ENCODED_CARET.sub(
- lambda match: _decode_html_entities(match.group(0)), decoded_tags
+ lambda match: _decode_html_entities(match.group(0)), decoded_pairs
)
diff --git a/tests/test_chunking.py b/tests/test_chunking.py
index 49ecda543..21cf75b05 100644
--- a/tests/test_chunking.py
+++ b/tests/test_chunking.py
@@ -292,6 +292,19 @@ def test_chunk_by_source_body_normalizes_entity_encoded_quantity_scripts() -> No
assert [chunk.text for chunk in chunks] == ["Reserve 12 m³, x², and H₂O."]
+def test_chunk_by_source_body_keeps_invalid_encoded_script_pairs_literal() -> None:
+ bodies = (
+ "Keep x<sup>2 unmatched.",
+ "Keep x<sup/>2 self-closing.",
+ "Keep x<sup class="unit">2</sup> attributed.",
+ "Keep x<sup>2</sub> mismatched.",
+ )
+
+ assert [chunk_by_source_body(body)[0].text for body in bodies] == list(bodies)
+ combined = " ".join(bodies)
+ assert chunk_by_source_body(combined)[0].text == combined
+
+
def test_chunk_by_source_body_keeps_encoded_non_script_markup_inert() -> None:
body = (
"Keep <b>bold</b>, <sup-note>2</sup-note>, "
From 0d6e899478ae1923d51819cb370ab50a1fca4a63 Mon Sep 17 00:00:00 2001
From: Seongho Bae
Date: Mon, 24 Aug 2026 04:07:35 +0900
Subject: [PATCH 3/3] fix: normalize persisted script units at display boundary
---
frontend/src/PostBody.test.tsx | 59 ++++++++++++++++++++++++++++++++++
frontend/src/PostBody.tsx | 10 ++++--
2 files changed, 66 insertions(+), 3 deletions(-)
diff --git a/frontend/src/PostBody.test.tsx b/frontend/src/PostBody.test.tsx
index 1ead554bd..3eaba4970 100644
--- a/frontend/src/PostBody.test.tsx
+++ b/frontend/src/PostBody.test.tsx
@@ -94,6 +94,65 @@ describe("PostBody", () => {
expect(container.querySelector("sup-note")).toBeNull();
});
+ it("renders raw and legacy persisted encoded scripts with the same semantics", () => {
+ const encoded =
+ "Volume x<sup>2</sup>, coolant H<sub>2</sub>O, and area m^3.";
+ const { container, rerender } = render(${encoded}`} />);
+
+ expect([...container.querySelectorAll("sup")].map((node) => node.textContent)).toEqual([
+ "2",
+ "3",
+ ]);
+ expect(container.querySelector("sub")?.textContent).toBe("2");
+
+ rerender(
+ ${encoded}`}
+ structureUnits={[
+ {
+ unit_index: 0,
+ unit_kind_code: "plain_text",
+ unit_text: encoded,
+ indent_level: 0,
+ indent_source_code: "explicit",
+ indent_confidence: 1,
+ indent_evidence: "Synthetic legacy persisted unit",
+ },
+ ]}
+ />,
+ );
+
+ expect([...container.querySelectorAll("sup")].map((node) => node.textContent)).toEqual([
+ "2",
+ "3",
+ ]);
+ expect(container.querySelector("sub")?.textContent).toBe("2");
+ });
+
+ it("normalizes legacy encoded scripts in persisted table cells", () => {
+ const { container } = render(
+ ,
+ );
+
+ const superscript = container.querySelector("td sup");
+ expect(superscript?.textContent).toBe("3");
+ expect(superscript?.closest("td")?.textContent).toBe("12 m3");
+ });
+
it("renders authoritative LLM structure levels for semantic list units", () => {
render(
{row.unit_text.split(/\s*\|\s*/).map((cell, cellIndex) => (
- {renderStyledText(decodeHtmlEntities(cell))}
+ {renderStyledText(displayUnitText(cell))}
|
))}
@@ -259,7 +263,7 @@ function renderStructuredUnits(
renderSegment(
{
kind: "text",
- text: decodeHtmlEntities(unit.unit_text),
+ text: displayUnitText(unit.unit_text),
...(unit.unit_label === "footnote" || sourceText?.role === "footnote"
? { role: "footnote" as const }
: {}),