Improve placeholder handling in HTML utils

Changed placeholder format to a unique string less likely to be altered by translation services and added zero-padding for consistency. Updated restoration logic to handle case-insensitive matching of placeholders. Added tests to verify correct restoration when placeholders are converted to lowercase.
This commit is contained in:
xixu-me committed 2025-07-27 14:55:50 +08:00
1 parent eb7b449508
commit 49b976acf3
2 files changed
+31 -10

No files matched your search

+13 -7
View File
@@ -17,9 +17,10 @@ const HTML_PATTERNS = {
/**
* Placeholder patterns for temporary replacement
* Using a format that's less likely to be modified by translation services
*/
const PLACEHOLDER_PREFIX = "__DEEPLX_PRESERVE_";
const PLACEHOLDER_SUFFIX = "__";
const PLACEHOLDER_PREFIX = "ĦĐŁXĦ";
const PLACEHOLDER_SUFFIX = "ĦĐŁXĦ";
/**
* Storage for preserved content during translation
@@ -39,7 +40,10 @@ function createPlaceholder(
content: string,
preserved: PreservedContent
): string {
const placeholder = `${PLACEHOLDER_PREFIX}${preserved.counter}${PLACEHOLDER_SUFFIX}`;
// Use a more unique format that's less likely to be translated
const placeholder = `${PLACEHOLDER_PREFIX}${preserved.counter
.toString()
.padStart(3, "0")}${PLACEHOLDER_SUFFIX}`;
preserved.placeholders.set(placeholder, content);
preserved.counter++;
return placeholder;
@@ -93,12 +97,14 @@ export function restoreHtmlContent(
): string {
let restoredText = translatedText;
// Restore all preserved content
// Restore all preserved content - handle case-insensitive matching
for (const [placeholder, originalContent] of preserved.placeholders) {
restoredText = restoredText.replace(
new RegExp(placeholder, "g"),
originalContent
// Create case-insensitive regex to handle lowercase conversion
const regex = new RegExp(
placeholder.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"),
"gi"
);
restoredText = restoredText.replace(regex, originalContent);
}
// Additional cleanup: fix any remaining Chinese colons that might have been introduced
+18 -3
View File
@@ -36,7 +36,7 @@ describe("HTML Utils", () => {
expect(processedText).not.toContain("</code>");
// Should contain placeholders
expect(processedText).toContain("__DEEPLX_PRESERVE_");
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back original content
const restoredText = restoreHtmlContent(processedText, preserved);
@@ -49,7 +49,7 @@ describe("HTML Utils", () => {
// The processed text should not contain colons
expect(processedText).not.toContain(":");
expect(processedText).toContain("__DEEPLX_PRESERVE_");
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back original colon
const restoredText = restoreHtmlContent(processedText, preserved);
@@ -66,7 +66,7 @@ describe("HTML Utils", () => {
expect(processedText).not.toContain(":");
// Should contain placeholders
expect(processedText).toContain("__DEEPLX_PRESERVE_");
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back all original content
const restoredText = restoreHtmlContent(processedText, preserved);
@@ -92,6 +92,21 @@ describe("HTML Utils", () => {
expect(restoredText).toContain("Hello: 世界:extra");
expect(restoredText).not.toContain(":");
});
it("should handle placeholders being converted to lowercase", () => {
const originalText = "<strong>A B </strong>";
const { processedText, preserved } = preserveHtmlContent(originalText);
// Simulate what happens when placeholders get converted to lowercase
const lowercasePlaceholders = processedText.toLowerCase();
const restoredText = restoreHtmlContent(lowercasePlaceholders, preserved);
// Should restore the original HTML tags correctly
expect(restoredText).toBe("<strong>a b </strong>");
expect(restoredText).toContain("<strong>");
expect(restoredText).toContain("</strong>");
});
});
describe("sanitizeHtmlContent", () => {