Compare commits

...

2 Commits

Author SHA1 Message Date
diegosouzapw
455aade660 fix(i18n): split oversized docs sections before translating
chunkMarkdown only cut on '## ' headings, so a single long section (README.md
carries a 16 KB one, USER_GUIDE.md a 20 KB one) became one oversized request.
On the slow fallback model that request could not finish inside the backend's
10-minute fetch timeout, and the three biggest docs of every new locale failed
with 'fetch failed' on every retry — Odia burned eight attempts on them.

A section still longer than maxChars is now split again on sub-headings and
paragraph boundaries, never inside a fenced code block; a block that is itself
larger than the limit stays whole rather than being cut mid-paragraph.
2026-09-11 16:08:49 -03:00
diegosouzapw
e89a4da2b0 fix(i18n): keep the ICU literal escape around angle placeholders when translating
English writes '<name>' — the single quotes are ICU's escape, so the span
renders as literal text. Every backend drops them, and the translated message
then parses as an unclosed ICU tag: the nine locales of the first batch each
shipped two such strings and only CI caught them.

translateString and translateBatch now restore the quoting, doubling an
apostrophe inside the span so it does not close the literal early (Estonian
OmniRoute'i, Irish d'eochair). Messages whose English mixes real markup with a
literal span are left untouched, since there is no safe way to tell the two
apart.
2026-09-10 12:39:15 -03:00
4 changed files with 211 additions and 7 deletions

View File

@@ -119,6 +119,32 @@ export const TRANSLATION_SYSTEM = (englishName, native) =>
`Keep punctuation and trailing whitespace identical to the source.`,
].join(" ");
/**
* Restores the ICU literal escape the backends drop around angle placeholders.
*
* English writes `'<name>'`: those single quotes are ICU's escape, so the span
* renders as the literal text `<name>`. Translations come back as a bare
* `<nome>`, which ICU then parses as an (unclosed) tag and the message fails to
* compile — every locale in the first batch shipped two of these.
*
* Only messages whose English side quotes EVERY angle span are touched: when the
* source mixes real markup (`<b>`) with a literal span there is no safe way to
* tell which is which, so the translation is left exactly as it came back. An
* apostrophe inside the span is doubled, otherwise it closes the literal early.
*/
export function preserveIcuLiteralQuotes(englishValue, translated) {
if (typeof englishValue !== "string" || typeof translated !== "string") return translated;
if (!englishValue.includes("'<")) return translated;
// Every "<" in the source must be the start of an escaped span.
for (let i = 0; i < englishValue.length; i++) {
if (englishValue[i] === "<" && englishValue[i - 1] !== "'") return translated;
}
return translated.replace(
/(?<!')<([^<>]*)>(?!')/g,
(_m, inner) => `'<${inner.replace(/'/g, "''")}>'`
);
}
export async function translateString(englishValue, localeEntry, backend) {
const englishName = localeEntry.english ?? localeEntry.name;
const native = localeEntry.native ?? localeEntry.name;
@@ -127,7 +153,7 @@ export async function translateString(englishValue, localeEntry, backend) {
{ role: "user", content: englishValue },
];
const out = await callChat(messages, backend);
return out.trim();
return preserveIcuLiteralQuotes(englishValue, out.trim());
}
// ----- Batch mode ----------------------------------------------------------
@@ -198,8 +224,14 @@ export async function translateBatch(entries, localeEntry, backend) {
{ role: "user", content: JSON.stringify(payload) },
];
const out = await callChat(messages, backend);
return parseBatchResponse(
const parsed = parseBatchResponse(
out,
entries.map((e) => e.id)
);
for (const entry of entries) {
if (typeof parsed[entry.id] === "string") {
parsed[entry.id] = preserveIcuLiteralQuotes(entry.text, parsed[entry.id]);
}
}
return parsed;
}

View File

@@ -380,16 +380,22 @@ const SYSTEM_PROMPT = (englishName, native) =>
`Return ONLY the translated markdown — no preamble, no explanation, no surrounding fences.`,
].join(" ");
// Splits a markdown body into chunks of <= maxChars, breaking on top-level `## ` headings only.
function chunkMarkdown(markdown, maxChars = 6000) {
// Splits a markdown body into chunks of <= maxChars. Top-level `## ` headings
// are the preferred cut; a section that is still longer than maxChars is then
// split again on `### ` headings and paragraph boundaries, never inside a
// fenced code block. Before the second pass a single long section (README.md
// has a 16 KB one, USER_GUIDE.md a 20 KB one) became one oversized request
// that the slow fallback model could not answer inside the backend's
// 10-minute fetch timeout, and the biggest docs failed on every retry.
export function chunkMarkdown(markdown, maxChars = 6000) {
if (markdown.length <= maxChars) return [markdown];
const lines = markdown.split("\n");
const chunks = [];
const sections = [];
let buf = [];
let size = 0;
for (const line of lines) {
if (line.startsWith("## ") && size > maxChars * 0.5) {
chunks.push(buf.join("\n"));
sections.push(buf.join("\n"));
buf = [line];
size = line.length;
} else {
@@ -397,7 +403,65 @@ function chunkMarkdown(markdown, maxChars = 6000) {
size += line.length + 1;
}
}
if (buf.length) chunks.push(buf.join("\n"));
if (buf.length) sections.push(buf.join("\n"));
return sections.flatMap((section) =>
section.length <= maxChars ? [section] : splitOversizedSection(section, maxChars)
);
}
const FENCE_LINE = /^\s*(```|~~~)/;
// Groups a section into blocks — a whole fenced code block, a heading-led run,
// or a paragraph ending at a blank line — and packs them greedily. A block that
// is itself larger than maxChars stays whole: cutting mid-paragraph or inside a
// fence would hand the model a fragment it cannot translate faithfully.
function splitOversizedSection(section, maxChars) {
const blocks = [];
let block = [];
let inFence = false;
for (const line of section.split("\n")) {
const isFence = FENCE_LINE.test(line);
if (inFence) {
block.push(line);
if (isFence) {
inFence = false;
blocks.push(block);
block = [];
}
continue;
}
if (isFence) {
if (block.length) blocks.push(block);
block = [line];
inFence = true;
continue;
}
if (/^##+ /.test(line) && block.length) {
blocks.push(block);
block = [];
}
block.push(line);
if (line.trim() === "") {
blocks.push(block);
block = [];
}
}
if (block.length) blocks.push(block);
const chunks = [];
let current = [];
let size = 0;
for (const lines of blocks) {
const length = lines.join("\n").length + 1;
if (size > 0 && size + length > maxChars) {
chunks.push(current.join("\n"));
current = [];
size = 0;
}
current.push(...lines);
size += length;
}
if (current.length) chunks.push(current.join("\n"));
return chunks;
}

View File

@@ -0,0 +1,65 @@
import test from "node:test";
import assert from "node:assert/strict";
import { chunkMarkdown } from "../../scripts/i18n/run-translation.mjs";
// The docs translator splits a page into chunks so one upstream call stays
// short. It used to break only on `## ` headings, so a single long section
// (README.md carries a 16 KB one, USER_GUIDE.md a 20 KB one) became one
// oversized request that the slow fallback model could not answer inside the
// backend's 10-minute fetch timeout — the three biggest docs of every locale
// then failed with "fetch failed" on every retry.
const para = (label: string, n = 12) =>
Array.from({ length: n }, (_, i) => `${label} sentence ${i + 1} with some filler text.`).join(
" "
);
test("a section longer than maxChars is split on sub-headings and paragraphs", () => {
const body = [
"## Big",
para("a"),
"",
"### Part one",
para("b"),
"",
para("c"),
"",
"### Part two",
para("d"),
].join("\n");
const chunks = chunkMarkdown(body, 700);
assert.ok(chunks.length > 1, "must split an oversized section");
for (const chunk of chunks) assert.ok(chunk.length <= 700, `chunk too big: ${chunk.length}`);
// Nothing lost: re-joining with blank lines reproduces every line of the source.
const lines = (s: string) => s.split("\n").filter((l) => l.trim() !== "");
assert.deepEqual(lines(chunks.join("\n\n")), lines(body));
});
test("never splits inside a fenced code block", () => {
const code = [
"```ts",
...Array.from({ length: 30 }, (_, i) => `const v${i} = ${i};`),
"```",
].join("\n");
const body = ["## Code", para("x"), "", code, "", para("y")].join("\n");
const chunks = chunkMarkdown(body, 500);
const withFence = chunks.filter((c) => c.includes("```"));
for (const chunk of withFence) {
assert.equal(
(chunk.match(/```/g) ?? []).length % 2,
0,
"fence must open and close in the same chunk"
);
}
});
test("keeps the old behaviour for short pages and for normal ## sections", () => {
assert.deepEqual(chunkMarkdown("# Title\n\nshort", 6000), ["# Title\n\nshort"]);
const body = ["## A", para("a", 4), "", "## B", para("b", 4), "", "## C", para("c", 4)].join(
"\n"
);
const chunks = chunkMarkdown(body, 400);
assert.ok(chunks.every((c) => c.length <= 400));
assert.ok(chunks.length > 1);
for (const chunk of chunks.slice(1)) assert.match(chunk, /^## /, "cuts land on ## boundaries");
});

View File

@@ -0,0 +1,43 @@
import test from "node:test";
import assert from "node:assert/strict";
import { preserveIcuLiteralQuotes } from "../../scripts/i18n/lib/translate-backend.mjs";
// The English catalog escapes angle placeholders for ICU: the single quotes in
// '<name>' make the span literal text. Translation backends drop them, and the
// message then parses as an unclosed ICU tag — every locale added in batch 1
// shipped two such strings before CI caught them.
test("re-escapes an angle span the translation left bare", () => {
assert.equal(
preserveIcuLiteralQuotes(
"Regenerate ~/.claude/profiles/'<name>'/settings.json",
"Regenerar ~/.claude/profiles/<nome>/settings.json"
),
"Regenerar ~/.claude/profiles/'<nome>'/settings.json"
);
});
test("doubles an apostrophe inside the span, which would close the literal early", () => {
assert.equal(
preserveIcuLiteralQuotes("'<your OmniRoute API key>'", "<teie OmniRoute'i API-võti>"),
"'<teie OmniRoute''i API-võti>'"
);
});
test("leaves a translation that already carries the quotes untouched", () => {
const already = "'<seu token>'";
assert.equal(preserveIcuLiteralQuotes("'<your token>'", already), already);
});
test("does not touch messages whose English has a real unquoted tag", () => {
// <b> here is markup the translation must keep as markup, not literal text.
const translated = "Clique em <b>Salvar</b> e em '<nome>'";
assert.equal(preserveIcuLiteralQuotes("Click <b>Save</b> and '<name>'", translated), translated);
});
test("leaves messages without angle spans alone", () => {
assert.equal(preserveIcuLiteralQuotes("Settings", "Nastavitve"), "Nastavitve");
assert.equal(
preserveIcuLiteralQuotes("Restricted to {count} endpoints", "Omejeno na {count} točk"),
"Omejeno na {count} točk"
);
});