Remove HTML content preservation utilities
Deleted htmlUtils module and its related tests. Removed all references to HTML preservation functions from index.ts and query.ts, simplifying the query logic and eliminating HTML-specific handling.
This commit is contained in:
1 parent
49b976acf3
commit
ce22782652
4 files changed
+3
-298
No files matched your search
@@ -1,142 +0,0 @@
|
||||
/**
|
||||
* HTML content preservation utilities for DeepLX API
|
||||
* Handles HTML tags and entities to prevent corruption during translation
|
||||
*/
|
||||
|
||||
/**
|
||||
* HTML tag and entity patterns for preservation
|
||||
*/
|
||||
const HTML_PATTERNS = {
|
||||
// HTML tags (opening and closing)
|
||||
TAGS: /<\/?[a-zA-Z][a-zA-Z0-9]*(?:\s+[^>]*)?>/g,
|
||||
// HTML entities
|
||||
ENTITIES: /&(?:[a-zA-Z][a-zA-Z0-9]*|#(?:\d+|x[0-9a-fA-F]+));/g,
|
||||
// Common punctuation that might be converted
|
||||
PUNCTUATION: /[::]/g,
|
||||
};
|
||||
|
||||
/**
|
||||
* Placeholder patterns for temporary replacement
|
||||
* Using a format that's less likely to be modified by translation services
|
||||
*/
|
||||
const PLACEHOLDER_PREFIX = "ĦĐŁXĦ";
|
||||
const PLACEHOLDER_SUFFIX = "ĦĐŁXĦ";
|
||||
|
||||
/**
|
||||
* Storage for preserved content during translation
|
||||
*/
|
||||
interface PreservedContent {
|
||||
placeholders: Map<string, string>;
|
||||
counter: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a unique placeholder for preserved content
|
||||
* @param content The content to preserve
|
||||
* @param preserved The preservation storage object
|
||||
* @returns The placeholder string
|
||||
*/
|
||||
function createPlaceholder(
|
||||
content: string,
|
||||
preserved: PreservedContent
|
||||
): string {
|
||||
// Use a more unique format that's less likely to be translated
|
||||
const placeholder = `${PLACEHOLDER_PREFIX}${preserved.counter
|
||||
.toString()
|
||||
.padStart(3, "0")}${PLACEHOLDER_SUFFIX}`;
|
||||
preserved.placeholders.set(placeholder, content);
|
||||
preserved.counter++;
|
||||
return placeholder;
|
||||
}
|
||||
|
||||
/**
|
||||
* Preserve HTML tags and entities before translation
|
||||
* Replaces HTML content with placeholders to prevent corruption
|
||||
* @param text The text containing HTML content
|
||||
* @returns Object with processed text and preservation data
|
||||
*/
|
||||
export function preserveHtmlContent(text: string): {
|
||||
processedText: string;
|
||||
preserved: PreservedContent;
|
||||
} {
|
||||
const preserved: PreservedContent = {
|
||||
placeholders: new Map(),
|
||||
counter: 0,
|
||||
};
|
||||
|
||||
let processedText = text;
|
||||
|
||||
// Preserve HTML tags
|
||||
processedText = processedText.replace(HTML_PATTERNS.TAGS, (match) => {
|
||||
return createPlaceholder(match, preserved);
|
||||
});
|
||||
|
||||
// Preserve HTML entities
|
||||
processedText = processedText.replace(HTML_PATTERNS.ENTITIES, (match) => {
|
||||
return createPlaceholder(match, preserved);
|
||||
});
|
||||
|
||||
// Preserve colons to prevent conversion to Chinese colons
|
||||
processedText = processedText.replace(/:/g, (match) => {
|
||||
return createPlaceholder(match, preserved);
|
||||
});
|
||||
|
||||
return { processedText, preserved };
|
||||
}
|
||||
|
||||
/**
|
||||
* Restore preserved HTML content after translation
|
||||
* Replaces placeholders with original HTML content
|
||||
* @param translatedText The translated text with placeholders
|
||||
* @param preserved The preservation data from preserveHtmlContent
|
||||
* @returns The text with restored HTML content
|
||||
*/
|
||||
export function restoreHtmlContent(
|
||||
translatedText: string,
|
||||
preserved: PreservedContent
|
||||
): string {
|
||||
let restoredText = translatedText;
|
||||
|
||||
// Restore all preserved content - handle case-insensitive matching
|
||||
for (const [placeholder, originalContent] of preserved.placeholders) {
|
||||
// Create case-insensitive regex to handle lowercase conversion
|
||||
const regex = new RegExp(
|
||||
placeholder.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"),
|
||||
"gi"
|
||||
);
|
||||
restoredText = restoredText.replace(regex, originalContent);
|
||||
}
|
||||
|
||||
// Additional cleanup: fix any remaining Chinese colons that might have been introduced
|
||||
restoredText = restoredText.replace(/:/g, ":");
|
||||
|
||||
return restoredText;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if text contains HTML content that should be preserved
|
||||
* @param text The text to check
|
||||
* @returns True if the text contains HTML content
|
||||
*/
|
||||
export function containsHtmlContent(text: string): boolean {
|
||||
return HTML_PATTERNS.TAGS.test(text) || HTML_PATTERNS.ENTITIES.test(text);
|
||||
}
|
||||
|
||||
/**
|
||||
* Sanitize HTML content while preserving structure
|
||||
* Removes potentially dangerous HTML while keeping formatting tags
|
||||
* @param text The text to sanitize
|
||||
* @returns Sanitized text with safe HTML preserved
|
||||
*/
|
||||
export function sanitizeHtmlContent(text: string): string {
|
||||
// Allow only safe formatting tags
|
||||
const SAFE_TAGS =
|
||||
/^<\/?(?:strong|b|em|i|u|code|pre|span|div|p|br|hr)(?:\s+[^>]*)?>/i;
|
||||
|
||||
return text.replace(HTML_PATTERNS.TAGS, (match) => {
|
||||
if (SAFE_TAGS.test(match)) {
|
||||
return match; // Keep safe tags
|
||||
}
|
||||
return ""; // Remove unsafe tags
|
||||
});
|
||||
}
|
||||
@@ -9,7 +9,6 @@ import { query } from "./query";
|
||||
export * from "./cache";
|
||||
export * from "./circuitBreaker";
|
||||
export * from "./errorHandler";
|
||||
export * from "./htmlUtils";
|
||||
export * from "./proxyManager";
|
||||
export * from "./rateLimit";
|
||||
export * from "./retryLogic";
|
||||
|
||||
+3
-27
@@ -10,11 +10,6 @@ import {
|
||||
} from "./config";
|
||||
import { API_URL, REQUEST_ALTERNATIVES } from "./const";
|
||||
import { createErrorResponse } from "./errorHandler";
|
||||
import {
|
||||
containsHtmlContent,
|
||||
preserveHtmlContent,
|
||||
restoreHtmlContent,
|
||||
} from "./htmlUtils";
|
||||
import { generateBrowserFingerprint, selectProxy } from "./proxyManager";
|
||||
import { checkCombinedRateLimit } from "./rateLimit";
|
||||
import { isRetryableError, RetryOptions, retryWithBackoff } from "./retryLogic";
|
||||
@@ -279,7 +274,7 @@ function buildRequestBody(data: RequestParams) {
|
||||
*/
|
||||
async function query(
|
||||
params: RequestParams,
|
||||
config?: Config & { env?: Env }
|
||||
config?: Config & { env?: any }
|
||||
): Promise<ResponseParams> {
|
||||
if (!params?.text) {
|
||||
return createStandardResponse(
|
||||
@@ -291,20 +286,6 @@ async function query(
|
||||
);
|
||||
}
|
||||
|
||||
// Check if text contains HTML content and preserve it
|
||||
const hasHtmlContent = containsHtmlContent(params.text);
|
||||
let processedParams = params;
|
||||
let htmlPreservation: any = null;
|
||||
|
||||
if (hasHtmlContent) {
|
||||
const preservation = preserveHtmlContent(params.text);
|
||||
htmlPreservation = preservation.preserved;
|
||||
processedParams = {
|
||||
...params,
|
||||
text: preservation.processedText,
|
||||
};
|
||||
}
|
||||
|
||||
const retryOptions: RetryOptions = {
|
||||
...DEFAULT_RETRY_CONFIG,
|
||||
isRetryable: isRetryableError,
|
||||
@@ -339,7 +320,7 @@ async function query(
|
||||
const timeoutId = setTimeout(() => controller.abort(), REQUEST_TIMEOUT);
|
||||
|
||||
try {
|
||||
const requestBody = buildRequestBody(processedParams);
|
||||
const requestBody = buildRequestBody(params);
|
||||
|
||||
const response = await fetch(endpoint, {
|
||||
headers: {
|
||||
@@ -455,12 +436,7 @@ async function query(
|
||||
throw error;
|
||||
}
|
||||
|
||||
let translatedText = result.result.texts[0].text;
|
||||
|
||||
// Restore HTML content if it was preserved
|
||||
if (hasHtmlContent && htmlPreservation) {
|
||||
translatedText = restoreHtmlContent(translatedText, htmlPreservation);
|
||||
}
|
||||
const translatedText = result.result.texts[0].text;
|
||||
|
||||
return createStandardResponse(
|
||||
200,
|
||||
|
||||
@@ -1,128 +0,0 @@
|
||||
/**
|
||||
* Tests for HTML utilities - HTML content preservation functionality
|
||||
*/
|
||||
|
||||
import {
|
||||
containsHtmlContent,
|
||||
preserveHtmlContent,
|
||||
restoreHtmlContent,
|
||||
sanitizeHtmlContent,
|
||||
} from "../../src/lib/htmlUtils";
|
||||
|
||||
describe("HTML Utils", () => {
|
||||
describe("containsHtmlContent", () => {
|
||||
it("should detect HTML tags", () => {
|
||||
expect(containsHtmlContent("<strong>Hello</strong>")).toBe(true);
|
||||
expect(containsHtmlContent("<code>test</code>")).toBe(true);
|
||||
expect(containsHtmlContent("Hello world")).toBe(false);
|
||||
});
|
||||
|
||||
it("should detect HTML entities", () => {
|
||||
expect(containsHtmlContent("<test>")).toBe(true);
|
||||
expect(containsHtmlContent("&")).toBe(true);
|
||||
expect(containsHtmlContent("Hello world")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("preserveHtmlContent and restoreHtmlContent", () => {
|
||||
it("should preserve and restore HTML tags", () => {
|
||||
const originalText = "<strong>A</strong>: <code>B</code>";
|
||||
const { processedText, preserved } = preserveHtmlContent(originalText);
|
||||
|
||||
// The processed text should not contain HTML tags
|
||||
expect(processedText).not.toContain("<strong>");
|
||||
expect(processedText).not.toContain("</strong>");
|
||||
expect(processedText).not.toContain("<code>");
|
||||
expect(processedText).not.toContain("</code>");
|
||||
|
||||
// Should contain placeholders
|
||||
expect(processedText).toContain("ĦĐŁXĦ");
|
||||
|
||||
// Restore should bring back original content
|
||||
const restoredText = restoreHtmlContent(processedText, preserved);
|
||||
expect(restoredText).toBe(originalText);
|
||||
});
|
||||
|
||||
it("should preserve colons to prevent Chinese colon conversion", () => {
|
||||
const originalText = "Hello: World";
|
||||
const { processedText, preserved } = preserveHtmlContent(originalText);
|
||||
|
||||
// The processed text should not contain colons
|
||||
expect(processedText).not.toContain(":");
|
||||
expect(processedText).toContain("ĦĐŁXĦ");
|
||||
|
||||
// Restore should bring back original colon
|
||||
const restoredText = restoreHtmlContent(processedText, preserved);
|
||||
expect(restoredText).toBe(originalText);
|
||||
});
|
||||
|
||||
it("should handle complex HTML with multiple tags", () => {
|
||||
const originalText =
|
||||
"<strong>Bold</strong> and <em>italic</em>: <code>code here</code>";
|
||||
const { processedText, preserved } = preserveHtmlContent(originalText);
|
||||
|
||||
// Should not contain any HTML tags
|
||||
expect(processedText).not.toMatch(/<[^>]+>/);
|
||||
expect(processedText).not.toContain(":");
|
||||
|
||||
// Should contain placeholders
|
||||
expect(processedText).toContain("ĦĐŁXĦ");
|
||||
|
||||
// Restore should bring back all original content
|
||||
const restoredText = restoreHtmlContent(processedText, preserved);
|
||||
expect(restoredText).toBe(originalText);
|
||||
});
|
||||
|
||||
it("should fix Chinese colons in translated text", () => {
|
||||
const originalText = "Hello: World";
|
||||
const { processedText, preserved } = preserveHtmlContent(originalText);
|
||||
|
||||
// Simulate translation that introduces Chinese colon
|
||||
const translatedWithChineseColon = processedText.replace(
|
||||
"World",
|
||||
"世界:extra"
|
||||
);
|
||||
|
||||
const restoredText = restoreHtmlContent(
|
||||
translatedWithChineseColon,
|
||||
preserved
|
||||
);
|
||||
|
||||
// Should restore original colon and fix any Chinese colons
|
||||
expect(restoredText).toContain("Hello: 世界:extra");
|
||||
expect(restoredText).not.toContain(":");
|
||||
});
|
||||
|
||||
it("should handle placeholders being converted to lowercase", () => {
|
||||
const originalText = "<strong>A B </strong>";
|
||||
const { processedText, preserved } = preserveHtmlContent(originalText);
|
||||
|
||||
// Simulate what happens when placeholders get converted to lowercase
|
||||
const lowercasePlaceholders = processedText.toLowerCase();
|
||||
|
||||
const restoredText = restoreHtmlContent(lowercasePlaceholders, preserved);
|
||||
|
||||
// Should restore the original HTML tags correctly
|
||||
expect(restoredText).toBe("<strong>a b </strong>");
|
||||
expect(restoredText).toContain("<strong>");
|
||||
expect(restoredText).toContain("</strong>");
|
||||
});
|
||||
});
|
||||
|
||||
describe("sanitizeHtmlContent", () => {
|
||||
it("should keep safe HTML tags", () => {
|
||||
const safeHtml =
|
||||
"<strong>Bold</strong> <code>code</code> <em>italic</em>";
|
||||
const sanitized = sanitizeHtmlContent(safeHtml);
|
||||
expect(sanitized).toBe(safeHtml);
|
||||
});
|
||||
|
||||
it("should remove unsafe HTML tags", () => {
|
||||
const unsafeHtml = "<script>alert('xss')</script><strong>Safe</strong>";
|
||||
const sanitized = sanitizeHtmlContent(unsafeHtml);
|
||||
expect(sanitized).not.toContain("<script>");
|
||||
expect(sanitized).not.toContain("</script>");
|
||||
expect(sanitized).toContain("<strong>Safe</strong>");
|
||||
});
|
||||
});
|
||||
});
|
||||
Reference in new issue
Block a user