diff --git a/src/infra/outbound/sanitize-text.test.ts b/src/infra/outbound/sanitize-text.test.ts
index 2ab54a31d546..d4e93a64b4ed 100644
--- a/src/infra/outbound/sanitize-text.test.ts
+++ b/src/infra/outbound/sanitize-text.test.ts
@@ -137,6 +137,8 @@ describe("sanitizeForPlainText", () => {
it("strips unknown/remaining tags", () => {
expect(sanitizeForPlainText('text')).toBe("text");
expect(sanitizeForPlainText('link')).toBe("link");
+ expect(sanitizeForPlainText("")).toBe("alert(1)");
+ expect(sanitizeForPlainText("
visible")).toBe("visible");
});
it("strips colon- and dot-qualified tags", () => {
@@ -309,6 +311,67 @@ describe("sanitizeForPlainText", () => {
expect(sanitizeForPlainText("a < b && c > d")).toBe("a < b && c > d");
});
+ it.each([
+ "Guard the retry loop: only retry while attempts0, otherwise give up.",
+ "Set the threshold so that latency4.",
+ "Use timeout<300 and n>0 for the probe.",
+ "a2",
+ "1<2>0",
+ "retry if attempts<3 and wait>5s",
+ "attempts5s",
+ "重试次数5秒",
+ "🙂5s",
+ "Set latencyd; }\n```\n\nand confirm concurrency>4 is safe.",
+ ])("preserves unspaced comparison prose in %s", (input) => {
+ expect(sanitizeForPlainText(input)).toBe(input);
+ });
+
+ it.each([10_000, 40_000])("bounds malformed comparison scanning with %i spaces", (size) => {
+ const input = `x5`;
+ const started = process.hrtime.bigint();
+ const sanitized = sanitizeForPlainText(input);
+ const elapsedMs = Number(process.hrtime.bigint() - started) / 1e6;
+
+ expect(sanitized).toBe("x5");
+ expect(elapsedMs).toBeLessThan(500);
+ });
+
+ it.each([
+ ["checkbox-after-value", 'x^2 • done', "x^2 • done"],
+ ["boolean-after-value", 'todo', "todo"],
+ ["boolean-only", "todo", "todo"],
+ ["boolean-first", 'done', "done"],
+ ["interleaved", 'done', "done"],
+ ["autofocus-after-value", 'ready', "ready"],
+ ["controls-after-value", 'play', "play"],
+ ["autoplay-after-value", 'now', "now"],
+ ["bare-custom", "text", "text"],
+ ["mixed-custom", "text
", "\ntext\n"],
+ ["download", "file", "file"],
+ ["custom-element-boolean", "text", "text"],
+ ["custom-element-bare", "text", "text"],
+ ["custom-element-empty", "", ""],
+ ["qualified-bare", "text", "text"],
+ ["unpaired-dot-qualified-clause", "foo5", "foo5"],
+ ["adjacent-numeric", "foo5", "foo5"],
+ ["paired-clause", "foo5", "foo5"],
+ ["void-numeric", "foo
5", "foo5"],
+ ["multiple-bare-numeric", "foo5", "foo5"],
+ ["unpaired-clause", "foo5", "foo5"],
+ ["uppercase-unpaired-clause", "foo5", "foo5"],
+ ])("strips or converts tags with bare attributes (%s)", (_name, input, expected) => {
+ expect(sanitizeForPlainText(input)).toBe(expected);
+ });
+
+ it.each([
+ ["range ad", "range ad"],
+ ["attempts5s", "attempts5s"],
+ ["x2", "x2"],
+ ])("retains existing stripping of ambiguous markup in %s", (input, expected) => {
+ expect(sanitizeForPlainText(input)).toBe(expected);
+ });
+
// --- mixed content ------------------------------------------------------
it("handles mixed HTML content", () => {
diff --git a/src/infra/outbound/sanitize-text.ts b/src/infra/outbound/sanitize-text.ts
index 3766d1f9ab62..d77ec95f3580 100644
--- a/src/infra/outbound/sanitize-text.ts
+++ b/src/infra/outbound/sanitize-text.ts
@@ -7,8 +7,16 @@ import { stripInternalRuntimeScaffolding } from "./protocol-scaffolding.js";
// Retained for the deprecated plugin-sdk/infra-runtime compatibility barrel.
export { stripInternalRuntimeScaffolding };
-// A tag name ends at whitespace, `/`, or `>`; `` is prose, not markup.
+// Preserve the existing tag grammar; only exclude unspaced comparison prose.
const HTML_TAG_RE = /<\/?[a-z][a-z0-9_.:-]*(?=[\s/>])[^>]*>/gi;
+// Disjoint whitespace/prose branches avoid quadratic backtracking on malformed tags.
+const COMPARISON_PROSE_RE = /^<([a-z][a-z0-9_]*\.?)\s+[^<>=/"'\s][^<>=/"']*>$/i;
+const COMPARISON_LEFT_OPERAND_RE = /[\p{L}\p{N}_\p{S}]$/u;
+const COMPARISON_CLAUSE_RE = /\b(?:and|or)\s|[.!?;:]\s|且/iu;
+// Standard HTML element names are never comparison operands: retain main's
+// stripping even beside numeric text or prose-like bare attributes.
+const HTML_ELEMENT_NAME_RE =
+ /^(?:a|abbr|address|area|article|aside|audio|b|base|bdi|bdo|blockquote|body|br|button|canvas|caption|cite|code|col|colgroup|data|datalist|dd|del|details|dfn|dialog|div|dl|dt|em|embed|fieldset|figcaption|figure|footer|form|h[1-6]|head|header|hgroup|hr|html|i|iframe|img|input|ins|kbd|label|legend|li|link|main|map|mark|menu|meta|meter|nav|noscript|object|ol|optgroup|option|output|p|picture|pre|progress|q|rp|rt|ruby|s|samp|script|search|section|select|selectedcontent|slot|small|source|span|strong|style|sub|summary|sup|table|tbody|td|template|textarea|tfoot|th|thead|time|title|tr|track|u|ul|var|video|wbr)$/i;
const LABELED_ANGLE_LINK_RE =
/<(?:https?:\/\/|mailto:)[^<>\s|]+\|([^<>\r\n|]*[^<>\s|][^<>\r\n|]*)>/gi;
const MAY_CONTAIN_MARKDOWN_CODE_RE = /[`~]|\t| {4}/;
@@ -22,16 +30,42 @@ const CONVERTIBLE_HTML_OPEN_TAG_RE =
const EMPTY_HTML_ELEMENT_RE =
/<((?!(?:br|p|div)(?=[\s>]))[a-z][a-z0-9_.:-]*)(?=[\s>])(?:[^"'<>]|"[^"]*"|'[^']*')*>(?:[^\S\r\n\u2028\u2029]|<(?!\/?(?:br|p|div)(?=[\s/>]))\/?[a-z][a-z0-9_.:-]*(?=[\s/>])(?:[^"'<>]|"[^"]*"|'[^']*')*>)*<\/\1\s*>/gi;
-function removeMatchesUntilStable(text: string, pattern: RegExp): string {
+function removeMatchesUntilStable(
+ text: string,
+ pattern: RegExp,
+ replacement?: (match: string, offset: number, source: string) => string,
+): string {
let previous: string;
let current = text;
do {
previous = current;
- current = current.replace(pattern, "");
+ current = replacement ? current.replace(pattern, replacement) : current.replace(pattern, "");
} while (current !== previous);
return current;
}
+function stripHtmlTagUnlessComparison(
+ tag: string,
+ offset: number,
+ source: string,
+ closingTagNames: ReadonlySet,
+): string {
+ const rightOperand = source.charCodeAt(offset + tag.length);
+ if (
+ !(rightOperand >= 48 && rightOperand <= 57) ||
+ !COMPARISON_LEFT_OPERAND_RE.test(source.slice(Math.max(0, offset - 2), offset))
+ ) {
+ return "";
+ }
+ const comparisonName = COMPARISON_PROSE_RE.exec(tag)?.[1];
+ return comparisonName !== undefined &&
+ !HTML_ELEMENT_NAME_RE.test(comparisonName) &&
+ COMPARISON_CLAUSE_RE.test(tag) &&
+ !closingTagNames.has(comparisonName.toLowerCase())
+ ? tag
+ : "";
+}
+
function convertHtmlOutsideCode(text: string, options: { style?: "markdown" }): string {
const boldMarker = options.style === "markdown" ? "**" : "*";
const strikeMarker = options.style === "markdown" ? "~~" : "~";
@@ -56,7 +90,14 @@ function convertHtmlOutsideCode(text: string, options: { style?: "markdown" }):
.replace(/(.*?)<\/h[1-6]>/gi, `\n${boldMarker}$1${boldMarker}\n`)
.replace(/(.*?)<\/li>/gi, "• $1\n");
- return removeMatchesUntilStable(converted, HTML_TAG_RE).replace(/\n{3,}/g, "\n\n");
+ // A matching closer is positive markup evidence, even when its content is numeric.
+ const closingTagNames = new Set();
+ for (const tag of converted.matchAll(/<\/[a-z][a-z0-9_.:-]*\s*>/gi)) {
+ closingTagNames.add(tag[0].slice(2, -1).trim().toLowerCase());
+ }
+ return removeMatchesUntilStable(converted, HTML_TAG_RE, (tag, offset, source) =>
+ stripHtmlTagUnlessComparison(tag, offset, source, closingTagNames),
+ ).replace(/\n{3,}/g, "\n\n");
}
/**