mirror of
https://github.com/openclaw/openclaw.git
synced 2026-08-08 02:52:15 +00:00
fix(markdown): preserve astral chunk boundaries
This commit is contained in:
@@ -439,6 +439,16 @@ describe("markdownToTelegramHtml", () => {
|
||||
it("keeps an astral char whole when a positive limit starts on its pair", () => {
|
||||
expect(splitTelegramHtmlChunks("A😀B", 1)).toEqual(["A", "😀", "B"]);
|
||||
});
|
||||
|
||||
it("keeps astral chars whole in rendered Markdown chunks", () => {
|
||||
const chunks = markdownToTelegramChunks("A😀B", 1);
|
||||
|
||||
expect(chunks.map((chunk) => chunk.text)).toEqual(["A", "😀", "B"]);
|
||||
for (const chunk of chunks) {
|
||||
expect(containsLoneSurrogate(chunk.html)).toBe(false);
|
||||
expect(containsLoneSurrogate(chunk.text)).toBe(false);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
function containsLoneSurrogate(text: string): boolean {
|
||||
|
||||
@@ -42,6 +42,23 @@ function scanParenAwareBreakpoints(text: string): { lastNewline: number; lastWhi
|
||||
return { lastNewline, lastWhitespace };
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps UTF-16 chunk boundaries from separating a supplementary-plane character.
|
||||
* A one-unit positive limit still needs to emit an entire surrogate pair.
|
||||
*/
|
||||
export function avoidTrailingHighSurrogateBreak(text: string, start: number, end: number): number {
|
||||
if (
|
||||
end >= text.length ||
|
||||
text.charCodeAt(end - 1) < 0xd800 ||
|
||||
text.charCodeAt(end - 1) > 0xdbff ||
|
||||
text.charCodeAt(end) < 0xdc00 ||
|
||||
text.charCodeAt(end) > 0xdfff
|
||||
) {
|
||||
return end;
|
||||
}
|
||||
return end - 1 > start ? end - 1 : end + 1;
|
||||
}
|
||||
|
||||
/**
|
||||
* Splits plain text into size-bounded chunks at readable boundaries.
|
||||
*
|
||||
@@ -66,7 +83,11 @@ export function chunkText(text: string, limit: number): string[] {
|
||||
// Prefer block boundaries, then spaces, then a hard size cut when no
|
||||
// readable breakpoint exists inside this window.
|
||||
const breakOffset = lastNewline > 0 ? lastNewline : lastWhitespace;
|
||||
const end = breakOffset > 0 ? cursor + breakOffset : windowEnd;
|
||||
const end = avoidTrailingHighSurrogateBreak(
|
||||
text,
|
||||
cursor,
|
||||
breakOffset > 0 ? cursor + breakOffset : windowEnd,
|
||||
);
|
||||
chunks.push(text.slice(cursor, end));
|
||||
cursor = end;
|
||||
while (cursor < text.length && /\s/.test(text[cursor] ?? "")) {
|
||||
|
||||
@@ -85,6 +85,28 @@ describe("renderMarkdownIRChunksWithinLimit", () => {
|
||||
expect(chunks.every((chunk) => chunk.rendered.length <= 1)).toBe(true);
|
||||
});
|
||||
|
||||
it("keeps astral characters whole when a positive limit reaches their pair", () => {
|
||||
const chunks = renderMarkdownIRChunksWithinLimit({
|
||||
ir: markdownToIR("A😀B"),
|
||||
limit: 1,
|
||||
renderChunk: (chunk) => chunk.text,
|
||||
measureRendered: (rendered) => rendered.length,
|
||||
});
|
||||
|
||||
expect(chunks.map((chunk) => chunk.source.text)).toEqual(["A", "😀", "B"]);
|
||||
});
|
||||
|
||||
it("keeps astral characters whole when rendered size requires a retry split", () => {
|
||||
const chunks = renderMarkdownIRChunksWithinLimit({
|
||||
ir: markdownToIR("A😀"),
|
||||
limit: 3,
|
||||
renderChunk: (chunk) => (chunk.text === "A😀" ? "too long" : chunk.text),
|
||||
measureRendered: (rendered) => rendered.length,
|
||||
});
|
||||
|
||||
expect(chunks.map((chunk) => chunk.source.text)).toEqual(["A", "😀"]);
|
||||
});
|
||||
|
||||
it("treats Infinity as no size cap and returns a single chunk", () => {
|
||||
const text = "one two three four five six seven eight nine ten";
|
||||
const ir = markdownToIR(text);
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { avoidTrailingHighSurrogateBreak } from "./chunk-text.js";
|
||||
// Markdown Core module implements render aware chunking behavior.
|
||||
import {
|
||||
chunkMarkdownIR,
|
||||
@@ -127,10 +128,11 @@ function findLargestChunkTextLengthWithinRenderedLimit<TRendered>(
|
||||
// Rendered length is not guaranteed to be monotonic after escaping/link or
|
||||
// file-reference rewriting, so test exact candidates from longest to shortest.
|
||||
for (let candidateLength = currentTextLength - 1; candidateLength >= 1; candidateLength -= 1) {
|
||||
const candidate = sliceMarkdownIR(chunk, 0, candidateLength);
|
||||
const safeCandidateLength = avoidTrailingHighSurrogateBreak(chunk.text, 0, candidateLength);
|
||||
const candidate = sliceMarkdownIR(chunk, 0, safeCandidateLength);
|
||||
const rendered = options.renderChunk(candidate);
|
||||
if (options.measureRendered(rendered) <= renderedLimit) {
|
||||
return candidateLength;
|
||||
return safeCandidateLength;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
@@ -215,7 +217,7 @@ function findMarkdownIRPreservedSplitIndex(text: string, start: number, limit: n
|
||||
if (lastAnyWhitespaceBreak > start) {
|
||||
return resolveWhitespaceBreak(lastAnyWhitespaceBreak, lastAnyWhitespaceRunStart);
|
||||
}
|
||||
return maxEnd;
|
||||
return avoidTrailingHighSurrogateBreak(text, start, maxEnd);
|
||||
}
|
||||
|
||||
function splitMarkdownIRPreserveWhitespace(ir: MarkdownIR, limit: number): MarkdownIR[] {
|
||||
|
||||
Reference in New Issue
Block a user