Remove HTML content preservation utilities

Deleted htmlUtils module and its related tests. Removed all references to HTML preservation functions from index.ts and query.ts, simplifying the query logic and eliminating HTML-specific handling.
This commit is contained in:
xixu-me committed 2025-07-27 15:07:02 +08:00
1 parent 49b976acf3
commit ce22782652
4 files changed
+3 -298

No files matched your search

-142
View File
@@ -1,142 +0,0 @@
/**
* HTML content preservation utilities for DeepLX API
* Handles HTML tags and entities to prevent corruption during translation
*/
/**
* HTML tag and entity patterns for preservation
*/
const HTML_PATTERNS = {
// HTML tags (opening and closing)
TAGS: /<\/?[a-zA-Z][a-zA-Z0-9]*(?:\s+[^>]*)?>/g,
// HTML entities
ENTITIES: /&(?:[a-zA-Z][a-zA-Z0-9]*|#(?:\d+|x[0-9a-fA-F]+));/g,
// Common punctuation that might be converted
PUNCTUATION: /[::]/g,
};
/**
* Placeholder patterns for temporary replacement
* Using a format that's less likely to be modified by translation services
*/
const PLACEHOLDER_PREFIX = "ĦĐŁXĦ";
const PLACEHOLDER_SUFFIX = "ĦĐŁXĦ";
/**
* Storage for preserved content during translation
*/
interface PreservedContent {
placeholders: Map<string, string>;
counter: number;
}
/**
* Create a unique placeholder for preserved content
* @param content The content to preserve
* @param preserved The preservation storage object
* @returns The placeholder string
*/
function createPlaceholder(
content: string,
preserved: PreservedContent
): string {
// Use a more unique format that's less likely to be translated
const placeholder = `${PLACEHOLDER_PREFIX}${preserved.counter
.toString()
.padStart(3, "0")}${PLACEHOLDER_SUFFIX}`;
preserved.placeholders.set(placeholder, content);
preserved.counter++;
return placeholder;
}
/**
* Preserve HTML tags and entities before translation
* Replaces HTML content with placeholders to prevent corruption
* @param text The text containing HTML content
* @returns Object with processed text and preservation data
*/
export function preserveHtmlContent(text: string): {
processedText: string;
preserved: PreservedContent;
} {
const preserved: PreservedContent = {
placeholders: new Map(),
counter: 0,
};
let processedText = text;
// Preserve HTML tags
processedText = processedText.replace(HTML_PATTERNS.TAGS, (match) => {
return createPlaceholder(match, preserved);
});
// Preserve HTML entities
processedText = processedText.replace(HTML_PATTERNS.ENTITIES, (match) => {
return createPlaceholder(match, preserved);
});
// Preserve colons to prevent conversion to Chinese colons
processedText = processedText.replace(/:/g, (match) => {
return createPlaceholder(match, preserved);
});
return { processedText, preserved };
}
/**
* Restore preserved HTML content after translation
* Replaces placeholders with original HTML content
* @param translatedText The translated text with placeholders
* @param preserved The preservation data from preserveHtmlContent
* @returns The text with restored HTML content
*/
export function restoreHtmlContent(
translatedText: string,
preserved: PreservedContent
): string {
let restoredText = translatedText;
// Restore all preserved content - handle case-insensitive matching
for (const [placeholder, originalContent] of preserved.placeholders) {
// Create case-insensitive regex to handle lowercase conversion
const regex = new RegExp(
placeholder.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"),
"gi"
);
restoredText = restoredText.replace(regex, originalContent);
}
// Additional cleanup: fix any remaining Chinese colons that might have been introduced
restoredText = restoredText.replace(/:/g, ":");
return restoredText;
}
/**
* Check if text contains HTML content that should be preserved
* @param text The text to check
* @returns True if the text contains HTML content
*/
export function containsHtmlContent(text: string): boolean {
return HTML_PATTERNS.TAGS.test(text) || HTML_PATTERNS.ENTITIES.test(text);
}
/**
* Sanitize HTML content while preserving structure
* Removes potentially dangerous HTML while keeping formatting tags
* @param text The text to sanitize
* @returns Sanitized text with safe HTML preserved
*/
export function sanitizeHtmlContent(text: string): string {
// Allow only safe formatting tags
const SAFE_TAGS =
/^<\/?(?:strong|b|em|i|u|code|pre|span|div|p|br|hr)(?:\s+[^>]*)?>/i;
return text.replace(HTML_PATTERNS.TAGS, (match) => {
if (SAFE_TAGS.test(match)) {
return match; // Keep safe tags
}
return ""; // Remove unsafe tags
});
}
-1
View File
@@ -9,7 +9,6 @@ import { query } from "./query";
export * from "./cache";
export * from "./circuitBreaker";
export * from "./errorHandler";
export * from "./htmlUtils";
export * from "./proxyManager";
export * from "./rateLimit";
export * from "./retryLogic";
+3 -27
View File
@@ -10,11 +10,6 @@ import {
} from "./config";
import { API_URL, REQUEST_ALTERNATIVES } from "./const";
import { createErrorResponse } from "./errorHandler";
import {
containsHtmlContent,
preserveHtmlContent,
restoreHtmlContent,
} from "./htmlUtils";
import { generateBrowserFingerprint, selectProxy } from "./proxyManager";
import { checkCombinedRateLimit } from "./rateLimit";
import { isRetryableError, RetryOptions, retryWithBackoff } from "./retryLogic";
@@ -279,7 +274,7 @@ function buildRequestBody(data: RequestParams) {
*/
async function query(
params: RequestParams,
config?: Config & { env?: Env }
config?: Config & { env?: any }
): Promise<ResponseParams> {
if (!params?.text) {
return createStandardResponse(
@@ -291,20 +286,6 @@ async function query(
);
}
// Check if text contains HTML content and preserve it
const hasHtmlContent = containsHtmlContent(params.text);
let processedParams = params;
let htmlPreservation: any = null;
if (hasHtmlContent) {
const preservation = preserveHtmlContent(params.text);
htmlPreservation = preservation.preserved;
processedParams = {
...params,
text: preservation.processedText,
};
}
const retryOptions: RetryOptions = {
...DEFAULT_RETRY_CONFIG,
isRetryable: isRetryableError,
@@ -339,7 +320,7 @@ async function query(
const timeoutId = setTimeout(() => controller.abort(), REQUEST_TIMEOUT);
try {
const requestBody = buildRequestBody(processedParams);
const requestBody = buildRequestBody(params);
const response = await fetch(endpoint, {
headers: {
@@ -455,12 +436,7 @@ async function query(
throw error;
}
let translatedText = result.result.texts[0].text;
// Restore HTML content if it was preserved
if (hasHtmlContent && htmlPreservation) {
translatedText = restoreHtmlContent(translatedText, htmlPreservation);
}
const translatedText = result.result.texts[0].text;
return createStandardResponse(
200,
-128
View File
@@ -1,128 +0,0 @@
/**
* Tests for HTML utilities - HTML content preservation functionality
*/
import {
containsHtmlContent,
preserveHtmlContent,
restoreHtmlContent,
sanitizeHtmlContent,
} from "../../src/lib/htmlUtils";
describe("HTML Utils", () => {
describe("containsHtmlContent", () => {
it("should detect HTML tags", () => {
expect(containsHtmlContent("<strong>Hello</strong>")).toBe(true);
expect(containsHtmlContent("<code>test</code>")).toBe(true);
expect(containsHtmlContent("Hello world")).toBe(false);
});
it("should detect HTML entities", () => {
expect(containsHtmlContent("&lt;test&gt;")).toBe(true);
expect(containsHtmlContent("&amp;")).toBe(true);
expect(containsHtmlContent("Hello world")).toBe(false);
});
});
describe("preserveHtmlContent and restoreHtmlContent", () => {
it("should preserve and restore HTML tags", () => {
const originalText = "<strong>A</strong>: <code>B</code>";
const { processedText, preserved } = preserveHtmlContent(originalText);
// The processed text should not contain HTML tags
expect(processedText).not.toContain("<strong>");
expect(processedText).not.toContain("</strong>");
expect(processedText).not.toContain("<code>");
expect(processedText).not.toContain("</code>");
// Should contain placeholders
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back original content
const restoredText = restoreHtmlContent(processedText, preserved);
expect(restoredText).toBe(originalText);
});
it("should preserve colons to prevent Chinese colon conversion", () => {
const originalText = "Hello: World";
const { processedText, preserved } = preserveHtmlContent(originalText);
// The processed text should not contain colons
expect(processedText).not.toContain(":");
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back original colon
const restoredText = restoreHtmlContent(processedText, preserved);
expect(restoredText).toBe(originalText);
});
it("should handle complex HTML with multiple tags", () => {
const originalText =
"<strong>Bold</strong> and <em>italic</em>: <code>code here</code>";
const { processedText, preserved } = preserveHtmlContent(originalText);
// Should not contain any HTML tags
expect(processedText).not.toMatch(/<[^>]+>/);
expect(processedText).not.toContain(":");
// Should contain placeholders
expect(processedText).toContain("ĦĐŁXĦ");
// Restore should bring back all original content
const restoredText = restoreHtmlContent(processedText, preserved);
expect(restoredText).toBe(originalText);
});
it("should fix Chinese colons in translated text", () => {
const originalText = "Hello: World";
const { processedText, preserved } = preserveHtmlContent(originalText);
// Simulate translation that introduces Chinese colon
const translatedWithChineseColon = processedText.replace(
"World",
"世界:extra"
);
const restoredText = restoreHtmlContent(
translatedWithChineseColon,
preserved
);
// Should restore original colon and fix any Chinese colons
expect(restoredText).toContain("Hello: 世界:extra");
expect(restoredText).not.toContain(":");
});
it("should handle placeholders being converted to lowercase", () => {
const originalText = "<strong>A B </strong>";
const { processedText, preserved } = preserveHtmlContent(originalText);
// Simulate what happens when placeholders get converted to lowercase
const lowercasePlaceholders = processedText.toLowerCase();
const restoredText = restoreHtmlContent(lowercasePlaceholders, preserved);
// Should restore the original HTML tags correctly
expect(restoredText).toBe("<strong>a b </strong>");
expect(restoredText).toContain("<strong>");
expect(restoredText).toContain("</strong>");
});
});
describe("sanitizeHtmlContent", () => {
it("should keep safe HTML tags", () => {
const safeHtml =
"<strong>Bold</strong> <code>code</code> <em>italic</em>";
const sanitized = sanitizeHtmlContent(safeHtml);
expect(sanitized).toBe(safeHtml);
});
it("should remove unsafe HTML tags", () => {
const unsafeHtml = "<script>alert('xss')</script><strong>Safe</strong>";
const sanitized = sanitizeHtmlContent(unsafeHtml);
expect(sanitized).not.toContain("<script>");
expect(sanitized).not.toContain("</script>");
expect(sanitized).toContain("<strong>Safe</strong>");
});
});
});