mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-09-12 01:42:27 +03:00
Compare commits
2 Commits
fix/12568-
...
feat/i18n-
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
455aade660 | ||
|
|
e89a4da2b0 |
@@ -119,6 +119,32 @@ export const TRANSLATION_SYSTEM = (englishName, native) =>
|
||||
`Keep punctuation and trailing whitespace identical to the source.`,
|
||||
].join(" ");
|
||||
|
||||
/**
|
||||
* Restores the ICU literal escape the backends drop around angle placeholders.
|
||||
*
|
||||
* English writes `'<name>'`: those single quotes are ICU's escape, so the span
|
||||
* renders as the literal text `<name>`. Translations come back as a bare
|
||||
* `<nome>`, which ICU then parses as an (unclosed) tag and the message fails to
|
||||
* compile — every locale in the first batch shipped two of these.
|
||||
*
|
||||
* Only messages whose English side quotes EVERY angle span are touched: when the
|
||||
* source mixes real markup (`<b>`) with a literal span there is no safe way to
|
||||
* tell which is which, so the translation is left exactly as it came back. An
|
||||
* apostrophe inside the span is doubled, otherwise it closes the literal early.
|
||||
*/
|
||||
export function preserveIcuLiteralQuotes(englishValue, translated) {
|
||||
if (typeof englishValue !== "string" || typeof translated !== "string") return translated;
|
||||
if (!englishValue.includes("'<")) return translated;
|
||||
// Every "<" in the source must be the start of an escaped span.
|
||||
for (let i = 0; i < englishValue.length; i++) {
|
||||
if (englishValue[i] === "<" && englishValue[i - 1] !== "'") return translated;
|
||||
}
|
||||
return translated.replace(
|
||||
/(?<!')<([^<>]*)>(?!')/g,
|
||||
(_m, inner) => `'<${inner.replace(/'/g, "''")}>'`
|
||||
);
|
||||
}
|
||||
|
||||
export async function translateString(englishValue, localeEntry, backend) {
|
||||
const englishName = localeEntry.english ?? localeEntry.name;
|
||||
const native = localeEntry.native ?? localeEntry.name;
|
||||
@@ -127,7 +153,7 @@ export async function translateString(englishValue, localeEntry, backend) {
|
||||
{ role: "user", content: englishValue },
|
||||
];
|
||||
const out = await callChat(messages, backend);
|
||||
return out.trim();
|
||||
return preserveIcuLiteralQuotes(englishValue, out.trim());
|
||||
}
|
||||
|
||||
// ----- Batch mode ----------------------------------------------------------
|
||||
@@ -198,8 +224,14 @@ export async function translateBatch(entries, localeEntry, backend) {
|
||||
{ role: "user", content: JSON.stringify(payload) },
|
||||
];
|
||||
const out = await callChat(messages, backend);
|
||||
return parseBatchResponse(
|
||||
const parsed = parseBatchResponse(
|
||||
out,
|
||||
entries.map((e) => e.id)
|
||||
);
|
||||
for (const entry of entries) {
|
||||
if (typeof parsed[entry.id] === "string") {
|
||||
parsed[entry.id] = preserveIcuLiteralQuotes(entry.text, parsed[entry.id]);
|
||||
}
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
@@ -380,16 +380,22 @@ const SYSTEM_PROMPT = (englishName, native) =>
|
||||
`Return ONLY the translated markdown — no preamble, no explanation, no surrounding fences.`,
|
||||
].join(" ");
|
||||
|
||||
// Splits a markdown body into chunks of <= maxChars, breaking on top-level `## ` headings only.
|
||||
function chunkMarkdown(markdown, maxChars = 6000) {
|
||||
// Splits a markdown body into chunks of <= maxChars. Top-level `## ` headings
|
||||
// are the preferred cut; a section that is still longer than maxChars is then
|
||||
// split again on `### ` headings and paragraph boundaries, never inside a
|
||||
// fenced code block. Before the second pass a single long section (README.md
|
||||
// has a 16 KB one, USER_GUIDE.md a 20 KB one) became one oversized request
|
||||
// that the slow fallback model could not answer inside the backend's
|
||||
// 10-minute fetch timeout, and the biggest docs failed on every retry.
|
||||
export function chunkMarkdown(markdown, maxChars = 6000) {
|
||||
if (markdown.length <= maxChars) return [markdown];
|
||||
const lines = markdown.split("\n");
|
||||
const chunks = [];
|
||||
const sections = [];
|
||||
let buf = [];
|
||||
let size = 0;
|
||||
for (const line of lines) {
|
||||
if (line.startsWith("## ") && size > maxChars * 0.5) {
|
||||
chunks.push(buf.join("\n"));
|
||||
sections.push(buf.join("\n"));
|
||||
buf = [line];
|
||||
size = line.length;
|
||||
} else {
|
||||
@@ -397,7 +403,65 @@ function chunkMarkdown(markdown, maxChars = 6000) {
|
||||
size += line.length + 1;
|
||||
}
|
||||
}
|
||||
if (buf.length) chunks.push(buf.join("\n"));
|
||||
if (buf.length) sections.push(buf.join("\n"));
|
||||
return sections.flatMap((section) =>
|
||||
section.length <= maxChars ? [section] : splitOversizedSection(section, maxChars)
|
||||
);
|
||||
}
|
||||
|
||||
const FENCE_LINE = /^\s*(```|~~~)/;
|
||||
|
||||
// Groups a section into blocks — a whole fenced code block, a heading-led run,
|
||||
// or a paragraph ending at a blank line — and packs them greedily. A block that
|
||||
// is itself larger than maxChars stays whole: cutting mid-paragraph or inside a
|
||||
// fence would hand the model a fragment it cannot translate faithfully.
|
||||
function splitOversizedSection(section, maxChars) {
|
||||
const blocks = [];
|
||||
let block = [];
|
||||
let inFence = false;
|
||||
for (const line of section.split("\n")) {
|
||||
const isFence = FENCE_LINE.test(line);
|
||||
if (inFence) {
|
||||
block.push(line);
|
||||
if (isFence) {
|
||||
inFence = false;
|
||||
blocks.push(block);
|
||||
block = [];
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (isFence) {
|
||||
if (block.length) blocks.push(block);
|
||||
block = [line];
|
||||
inFence = true;
|
||||
continue;
|
||||
}
|
||||
if (/^##+ /.test(line) && block.length) {
|
||||
blocks.push(block);
|
||||
block = [];
|
||||
}
|
||||
block.push(line);
|
||||
if (line.trim() === "") {
|
||||
blocks.push(block);
|
||||
block = [];
|
||||
}
|
||||
}
|
||||
if (block.length) blocks.push(block);
|
||||
|
||||
const chunks = [];
|
||||
let current = [];
|
||||
let size = 0;
|
||||
for (const lines of blocks) {
|
||||
const length = lines.join("\n").length + 1;
|
||||
if (size > 0 && size + length > maxChars) {
|
||||
chunks.push(current.join("\n"));
|
||||
current = [];
|
||||
size = 0;
|
||||
}
|
||||
current.push(...lines);
|
||||
size += length;
|
||||
}
|
||||
if (current.length) chunks.push(current.join("\n"));
|
||||
return chunks;
|
||||
}
|
||||
|
||||
|
||||
65
tests/unit/i18n-run-translation-chunking.test.ts
Normal file
65
tests/unit/i18n-run-translation-chunking.test.ts
Normal file
@@ -0,0 +1,65 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { chunkMarkdown } from "../../scripts/i18n/run-translation.mjs";
|
||||
|
||||
// The docs translator splits a page into chunks so one upstream call stays
|
||||
// short. It used to break only on `## ` headings, so a single long section
|
||||
// (README.md carries a 16 KB one, USER_GUIDE.md a 20 KB one) became one
|
||||
// oversized request that the slow fallback model could not answer inside the
|
||||
// backend's 10-minute fetch timeout — the three biggest docs of every locale
|
||||
// then failed with "fetch failed" on every retry.
|
||||
|
||||
const para = (label: string, n = 12) =>
|
||||
Array.from({ length: n }, (_, i) => `${label} sentence ${i + 1} with some filler text.`).join(
|
||||
" "
|
||||
);
|
||||
|
||||
test("a section longer than maxChars is split on sub-headings and paragraphs", () => {
|
||||
const body = [
|
||||
"## Big",
|
||||
para("a"),
|
||||
"",
|
||||
"### Part one",
|
||||
para("b"),
|
||||
"",
|
||||
para("c"),
|
||||
"",
|
||||
"### Part two",
|
||||
para("d"),
|
||||
].join("\n");
|
||||
const chunks = chunkMarkdown(body, 700);
|
||||
assert.ok(chunks.length > 1, "must split an oversized section");
|
||||
for (const chunk of chunks) assert.ok(chunk.length <= 700, `chunk too big: ${chunk.length}`);
|
||||
// Nothing lost: re-joining with blank lines reproduces every line of the source.
|
||||
const lines = (s: string) => s.split("\n").filter((l) => l.trim() !== "");
|
||||
assert.deepEqual(lines(chunks.join("\n\n")), lines(body));
|
||||
});
|
||||
|
||||
test("never splits inside a fenced code block", () => {
|
||||
const code = [
|
||||
"```ts",
|
||||
...Array.from({ length: 30 }, (_, i) => `const v${i} = ${i};`),
|
||||
"```",
|
||||
].join("\n");
|
||||
const body = ["## Code", para("x"), "", code, "", para("y")].join("\n");
|
||||
const chunks = chunkMarkdown(body, 500);
|
||||
const withFence = chunks.filter((c) => c.includes("```"));
|
||||
for (const chunk of withFence) {
|
||||
assert.equal(
|
||||
(chunk.match(/```/g) ?? []).length % 2,
|
||||
0,
|
||||
"fence must open and close in the same chunk"
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
test("keeps the old behaviour for short pages and for normal ## sections", () => {
|
||||
assert.deepEqual(chunkMarkdown("# Title\n\nshort", 6000), ["# Title\n\nshort"]);
|
||||
const body = ["## A", para("a", 4), "", "## B", para("b", 4), "", "## C", para("c", 4)].join(
|
||||
"\n"
|
||||
);
|
||||
const chunks = chunkMarkdown(body, 400);
|
||||
assert.ok(chunks.every((c) => c.length <= 400));
|
||||
assert.ok(chunks.length > 1);
|
||||
for (const chunk of chunks.slice(1)) assert.match(chunk, /^## /, "cuts land on ## boundaries");
|
||||
});
|
||||
43
tests/unit/i18n-translate-backend-icu-quotes.test.ts
Normal file
43
tests/unit/i18n-translate-backend-icu-quotes.test.ts
Normal file
@@ -0,0 +1,43 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { preserveIcuLiteralQuotes } from "../../scripts/i18n/lib/translate-backend.mjs";
|
||||
|
||||
// The English catalog escapes angle placeholders for ICU: the single quotes in
|
||||
// '<name>' make the span literal text. Translation backends drop them, and the
|
||||
// message then parses as an unclosed ICU tag — every locale added in batch 1
|
||||
// shipped two such strings before CI caught them.
|
||||
test("re-escapes an angle span the translation left bare", () => {
|
||||
assert.equal(
|
||||
preserveIcuLiteralQuotes(
|
||||
"Regenerate ~/.claude/profiles/'<name>'/settings.json",
|
||||
"Regenerar ~/.claude/profiles/<nome>/settings.json"
|
||||
),
|
||||
"Regenerar ~/.claude/profiles/'<nome>'/settings.json"
|
||||
);
|
||||
});
|
||||
|
||||
test("doubles an apostrophe inside the span, which would close the literal early", () => {
|
||||
assert.equal(
|
||||
preserveIcuLiteralQuotes("'<your OmniRoute API key>'", "<teie OmniRoute'i API-võti>"),
|
||||
"'<teie OmniRoute''i API-võti>'"
|
||||
);
|
||||
});
|
||||
|
||||
test("leaves a translation that already carries the quotes untouched", () => {
|
||||
const already = "'<seu token>'";
|
||||
assert.equal(preserveIcuLiteralQuotes("'<your token>'", already), already);
|
||||
});
|
||||
|
||||
test("does not touch messages whose English has a real unquoted tag", () => {
|
||||
// <b> here is markup the translation must keep as markup, not literal text.
|
||||
const translated = "Clique em <b>Salvar</b> e em '<nome>'";
|
||||
assert.equal(preserveIcuLiteralQuotes("Click <b>Save</b> and '<name>'", translated), translated);
|
||||
});
|
||||
|
||||
test("leaves messages without angle spans alone", () => {
|
||||
assert.equal(preserveIcuLiteralQuotes("Settings", "Nastavitve"), "Nastavitve");
|
||||
assert.equal(
|
||||
preserveIcuLiteralQuotes("Restricted to {count} endpoints", "Omejeno na {count} točk"),
|
||||
"Omejeno na {count} točk"
|
||||
);
|
||||
});
|
||||
Reference in New Issue
Block a user