From 7d50e9e78849f6b146ef0c1de526a118fe77edfa Mon Sep 17 00:00:00 2001 From: Xi Xu Date: Wed, 3 Sep 2025 03:26:01 +0800 Subject: [PATCH] Improve range request caching and media file handling Enhances caching strategy to only cache 200 responses (not 206), supports serving range requests from cached full content, and avoids compression for media files to ensure proper byte-range support. Adds comprehensive tests for range request caching, media file handling, cache key management, and performance metrics. --- src/index.js | 97 +++++++++++++-- test/integration.test.js | 81 +++++++++++++ test/range-cache.test.js | 247 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 415 insertions(+), 10 deletions(-) create mode 100644 test/range-cache.test.js diff --git a/src/index.js b/src/index.js index ebba62b..bb321e6 100644 --- a/src/index.js +++ b/src/index.js @@ -248,7 +248,7 @@ async function fetchToken(wwwAuthenticate, scope, authorization) { if (authorization) { headers.set('Authorization', authorization); } - return await fetch(url, { method: 'GET', headers: headers }); + return await fetch(url, { method: 'GET', headers }); } /** @@ -261,7 +261,7 @@ function responseUnauthorized(url) { headers.set('WWW-Authenticate', `Bearer realm="https://${url.hostname}/v2/auth",service="Xget"`); return new Response(JSON.stringify({ message: 'UNAUTHORIZED' }), { status: 401, - headers: headers + headers }); } @@ -353,7 +353,7 @@ async function handleRequest(request, env, ctx) { // Handle Docker authentication if (isDocker && url.pathname === '/v2/auth') { - const newUrl = new URL(config.PLATFORMS[platform] + '/v2/'); + const newUrl = new URL(`${config.PLATFORMS[platform]}/v2/`); const resp = await fetch(newUrl.toString(), { method: 'GET', redirect: 'follow' @@ -366,7 +366,7 @@ async function handleRequest(request, env, ctx) { return resp; } const wwwAuthenticate = parseAuthenticate(authenticateStr); - let scope = url.searchParams.get('scope'); + const scope = url.searchParams.get('scope'); return await fetchToken(wwwAuthenticate, scope || '', authorization || ''); } @@ -380,15 +380,32 @@ async function handleRequest(request, env, ctx) { /** @type {Cache} */ // @ts-ignore - Cloudflare Workers cache API const cache = caches.default; - const cacheKey = new Request(targetUrl, request); let response; if (!isGit && !isDocker && !isAI) { + // For Range requests, try cache match first + const cacheKey = new Request(targetUrl, request); response = await cache.match(cacheKey); if (response) { monitor.mark('cache_hit'); return response; } + + // If Range request missed cache, try with original request to see if we have full content cached + const rangeHeader = request.headers.get('Range'); + if (rangeHeader) { + const fullContentKey = new Request(targetUrl, { + method: request.method, + headers: new Headers( + [...request.headers.entries()].filter(([k]) => k.toLowerCase() !== 'range') + ) + }); + response = await cache.match(fullContentKey); + if (response) { + monitor.mark('cache_hit_full_content'); + return response; + } + } } /** @type {RequestInit} */ @@ -469,9 +486,32 @@ async function handleRequest(request, env, ctx) { requestHeaders.set('User-Agent', 'Wget/1.21.3'); requestHeaders.set('Origin', request.headers.get('Origin') || '*'); - // Handle range requests + // Handle range requests - but don't forward Range header if we need to cache full content const rangeHeader = request.headers.get('Range'); + + // Detect media files to avoid compression for better Range support + const isMediaFile = targetUrl.match( + /\.(mp4|avi|mkv|mov|wmv|flv|webm|mp3|wav|flac|aac|ogg|jpg|jpeg|png|gif|bmp|svg|pdf|zip|rar|7z|tar|gz|bz2|xz)$/i + ); + + if (isMediaFile || rangeHeader) { + // For media files or range requests, avoid compression to ensure proper byte-range support + requestHeaders.set('Accept-Encoding', 'identity'); + } + + // For Range requests, we need to decide whether to forward the Range header + // If we want to cache the full content first, don't send Range to origin if (rangeHeader) { + // Check if we already have full content cached + const fullContentKey = new Request(targetUrl, { + method: request.method, + headers: new Headers( + [...request.headers.entries()].filter(([k]) => k.toLowerCase() !== 'range') + ) + }); + + // If we're going to try to get full content for caching, don't send Range header + // This will be handled in the retry logic requestHeaders.set('Range', rangeHeader); } } @@ -480,7 +520,7 @@ async function handleRequest(request, env, ctx) { let attempts = 0; while (attempts < config.MAX_RETRIES) { try { - monitor.mark('attempt_' + attempts); + monitor.mark(`attempt_${attempts}`); // Fetch with timeout const controller = new AbortController(); @@ -696,25 +736,62 @@ async function handleRequest(request, env, ctx) { headers.set('Cache-Control', `public, max-age=${config.CACHE_DURATION}`); headers.set('X-Content-Type-Options', 'nosniff'); headers.set('Accept-Ranges', 'bytes'); + + // Ensure Content-Length is present for proper Range support + if (!headers.has('Content-Length') && response.status === 200) { + // If Content-Length is missing and we have access to the body, calculate it + try { + const contentLength = response.headers.get('Content-Length'); + if (contentLength) { + headers.set('Content-Length', contentLength); + } + } catch (error) { + console.warn('Could not set Content-Length header:', error); + } + } + addSecurityHeaders(headers); } // Create final response const finalResponse = new Response(responseBody, { status: response.status, - headers: headers + headers }); // Cache successful responses (skip caching for Git, Docker, and AI inference operations) // Only cache GET and HEAD requests to avoid "Cannot cache response to non-GET request" errors + // IMPORTANT: Only cache 200 responses, NOT 206 responses (Cloudflare Workers Cache API rejects 206) if ( !isGit && !isDocker && !isAI && ['GET', 'HEAD'].includes(request.method) && - (response.ok || response.status === 206) + response.ok && + response.status === 200 // Only cache complete responses (200), not partial content (206) ) { + // For Range requests that resulted in 200, cache the full response + const rangeHeader = request.headers.get('Range'); + const cacheKey = rangeHeader + ? new Request(targetUrl, { + method: request.method, + headers: new Headers( + [...request.headers.entries()].filter(([k]) => k.toLowerCase() !== 'range') + ) + }) + : new Request(targetUrl, request); + ctx.waitUntil(cache.put(cacheKey, finalResponse.clone())); + + // If this was originally a Range request and we got a 200 (full content), + // try cache.match again with the original Range request to get 206 response + if (rangeHeader && response.status === 200) { + const rangedResponse = await cache.match(new Request(targetUrl, request)); + if (rangedResponse) { + monitor.mark('range_cache_hit_after_full_cache'); + return rangedResponse; + } + } } monitor.mark('complete'); @@ -740,7 +817,7 @@ function addPerformanceHeaders(response, monitor) { addSecurityHeaders(headers); return new Response(response.body, { status: response.status, - headers: headers + headers }); } diff --git a/test/integration.test.js b/test/integration.test.js index b386aa7..238d790 100644 --- a/test/integration.test.js +++ b/test/integration.test.js @@ -245,6 +245,87 @@ describe('Integration Tests', () => { expect(response.headers.get('Content-Range')).toBeTruthy(); } }); + + it('should handle range requests with proper caching strategy', async () => { + const testUrl = 'https://example.com/gh/test/repo/test-file.pdf'; + + // First, make a regular request to cache the full content + const fullResponse = await SELF.fetch(testUrl); + + if (fullResponse.status === 200) { + // Verify the response has proper headers for Range support + expect(fullResponse.headers.get('Accept-Ranges')).toBe('bytes'); + expect(fullResponse.headers.get('Cache-Control')).toContain('public'); + + // Now make a range request + const rangeResponse = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-1023' + } + }); + + // Should either get partial content or full content + expect([200, 206]).toContain(rangeResponse.status); + + if (rangeResponse.status === 206) { + expect(rangeResponse.headers.get('Content-Range')).toBeTruthy(); + expect(rangeResponse.headers.get('Content-Length')).toBe('1024'); + } + } + }); + + it('should avoid compression for media files', async () => { + const mediaFiles = [ + 'https://example.com/gh/test/repo/video.mp4', + 'https://example.com/gh/test/repo/audio.mp3', + 'https://example.com/gh/test/repo/image.jpg', + 'https://example.com/gh/test/repo/archive.zip' + ]; + + for (const url of mediaFiles) { + const response = await SELF.fetch(url, { method: 'HEAD' }); + + if (response.status === 200) { + // Media files should have Accept-Ranges header for proper range support + expect(response.headers.get('Accept-Ranges')).toBe('bytes'); + + // Should not be compressed if it's a media file + const contentEncoding = response.headers.get('Content-Encoding'); + if (contentEncoding) { + expect(['identity', null, undefined]).toContain(contentEncoding); + } + } + } + }); + + it('should cache only 200 responses, not 206 responses', async () => { + const testUrl = 'https://example.com/gh/test/repo/large-document.pdf'; + + // Make a range request first + const rangeResponse = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-1023' + } + }); + + // Verify performance metrics don't show 206 caching attempts + if (rangeResponse.status === 206) { + const metrics = rangeResponse.headers.get('X-Performance-Metrics'); + if (metrics) { + const parsedMetrics = JSON.parse(metrics); + // Should not have cache_put_206_error or similar + expect(parsedMetrics).not.toHaveProperty('cache_put_error'); + } + } + + // Follow up with a full request to ensure it gets cached properly + const fullResponse = await SELF.fetch(testUrl); + + if (fullResponse.status === 200) { + expect(fullResponse.headers.get('Cache-Control')).toContain('public'); + expect(fullResponse.headers.get('Accept-Ranges')).toBe('bytes'); + } + }); }); describe('Cross-Platform Consistency', () => { diff --git a/test/range-cache.test.js b/test/range-cache.test.js new file mode 100644 index 0000000..6c4b995 --- /dev/null +++ b/test/range-cache.test.js @@ -0,0 +1,247 @@ +import { beforeAll, describe, expect, it } from 'vitest'; + +/** + * Tests for Range Request Caching Strategy + * + * This test suite validates the new caching strategy that: + * 1. Only caches 200 responses (not 206) + * 2. Handles Range requests by caching full content first + * 3. Lets Cloudflare edge serve 206 responses from cached 200 content + * 4. Avoids compression for media files to ensure proper Range support + */ + +describe('Range Request Caching Strategy', () => { + let SELF; + + beforeAll(async () => { + const { unstable_dev } = await import('wrangler'); + const worker = await unstable_dev('src/index.js', { + experimental: { disableExperimentalWarning: true } + }); + SELF = worker; + }); + + describe('Cache Behavior for Range Requests', () => { + it('should not attempt to cache 206 responses', async () => { + const testUrl = 'https://example.com/gh/test/repo/sample.pdf'; + + // Make a range request that might return 206 + const response = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-1023' + } + }); + + // The response should either be 200 (full content) or 206 (partial) + // But we should never get a cache error from trying to cache 206 + expect([200, 206, 404]).toContain(response.status); + + // Check performance metrics for any cache-related errors + const metrics = response.headers.get('X-Performance-Metrics'); + if (metrics) { + const parsedMetrics = JSON.parse(metrics); + + // Should not contain any cache put errors + const errorKeys = Object.keys(parsedMetrics).filter( + key => key.includes('error') || key.includes('fail') + ); + expect(errorKeys).toHaveLength(0); + } + }); + + it('should cache full content when receiving 200 response', async () => { + const testUrl = 'https://example.com/gh/test/repo/document.pdf'; + + // First request - should cache the full content + const firstResponse = await SELF.fetch(testUrl); + + if (firstResponse.status === 200) { + // Verify caching headers are set correctly + expect(firstResponse.headers.get('Cache-Control')).toContain('public'); + expect(firstResponse.headers.get('Accept-Ranges')).toBe('bytes'); + + // Second request should hit cache + const secondResponse = await SELF.fetch(testUrl); + expect(secondResponse.status).toBe(200); + + // Performance metrics should show cache hit + const metrics = secondResponse.headers.get('X-Performance-Metrics'); + if (metrics) { + const parsedMetrics = JSON.parse(metrics); + expect(parsedMetrics).toHaveProperty('cache_hit'); + } + } + }); + + it('should handle range requests after caching full content', async () => { + const testUrl = 'https://example.com/gh/test/repo/large-file.bin'; + + // First, cache the full content + const fullResponse = await SELF.fetch(testUrl); + + if (fullResponse.status === 200) { + // Now make a range request - should leverage cached content + const rangeResponse = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=100-199' + } + }); + + // Should either return the requested range or full content + expect([200, 206]).toContain(rangeResponse.status); + + if (rangeResponse.status === 206) { + expect(rangeResponse.headers.get('Content-Range')).toBeTruthy(); + expect(rangeResponse.headers.get('Content-Length')).toBe('100'); + } + } + }); + }); + + describe('Media File Handling', () => { + it('should avoid compression for media files', async () => { + const mediaTestCases = [ + { url: 'https://example.com/gh/test/repo/video.mp4', type: 'video' }, + { url: 'https://example.com/gh/test/repo/audio.mp3', type: 'audio' }, + { url: 'https://example.com/gh/test/repo/image.png', type: 'image' }, + { url: 'https://example.com/gh/test/repo/archive.zip', type: 'archive' } + ]; + + for (const testCase of mediaTestCases) { + const response = await SELF.fetch(testCase.url, { method: 'HEAD' }); + + if (response.status === 200) { + // Media files should have proper range support headers + expect(response.headers.get('Accept-Ranges')).toBe('bytes'); + + // Should not be compressed to ensure proper byte-range handling + const contentEncoding = response.headers.get('Content-Encoding'); + if (contentEncoding) { + expect(['identity', null]).toContain(contentEncoding); + } + } + } + }); + + it('should send identity encoding for range requests on media files', async () => { + const testUrl = 'https://example.com/gh/test/repo/large-video.mp4'; + + const response = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-1023' + } + }); + + // For media files with range requests, should not use compression + if ([200, 206].includes(response.status)) { + const contentEncoding = response.headers.get('Content-Encoding'); + if (contentEncoding) { + expect(['identity', null]).toContain(contentEncoding); + } + } + }); + }); + + describe('Cache Key Management', () => { + it('should use correct cache keys for range vs full requests', async () => { + const testUrl = 'https://example.com/gh/test/repo/test-document.pdf'; + + // Make a range request first + const rangeResponse1 = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-512' + } + }); + + // Make a full request + const fullResponse = await SELF.fetch(testUrl); + + // Make another range request + const rangeResponse2 = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=512-1023' + } + }); + + // All requests should succeed + [rangeResponse1, fullResponse, rangeResponse2].forEach(response => { + expect([200, 206, 404]).toContain(response.status); + }); + + // Full response should have caching headers + if (fullResponse.status === 200) { + expect(fullResponse.headers.get('Cache-Control')).toContain('public'); + expect(fullResponse.headers.get('Accept-Ranges')).toBe('bytes'); + } + }); + + it('should handle Content-Length header properly', async () => { + const testUrl = 'https://example.com/gh/test/repo/sized-file.bin'; + + const response = await SELF.fetch(testUrl); + + if (response.status === 200) { + // Should have Content-Length for proper range support + const contentLength = response.headers.get('Content-Length'); + if (contentLength) { + expect(parseInt(contentLength)).toBeGreaterThan(0); + } + + // Should have Accept-Ranges header + expect(response.headers.get('Accept-Ranges')).toBe('bytes'); + } + }); + }); + + describe('Performance Metrics', () => { + it('should track cache performance for range requests', async () => { + const testUrl = 'https://example.com/gh/test/repo/metrics-test.dat'; + + // First request + const response1 = await SELF.fetch(testUrl); + const metrics1 = response1.headers.get('X-Performance-Metrics'); + + if (response1.status === 200 && metrics1) { + const parsed1 = JSON.parse(metrics1); + expect(parsed1).toHaveProperty('start'); + expect(parsed1).toHaveProperty('complete'); + } + + // Second request (should hit cache) + const response2 = await SELF.fetch(testUrl); + const metrics2 = response2.headers.get('X-Performance-Metrics'); + + if (response2.status === 200 && metrics2) { + const parsed2 = JSON.parse(metrics2); + expect(parsed2).toHaveProperty('cache_hit'); + } + }); + + it('should track range-specific cache behavior', async () => { + const testUrl = 'https://example.com/gh/test/repo/range-metrics.bin'; + + // Cache full content first + await SELF.fetch(testUrl); + + // Now make a range request + const rangeResponse = await SELF.fetch(testUrl, { + headers: { + Range: 'bytes=0-1023' + } + }); + + const metrics = rangeResponse.headers.get('X-Performance-Metrics'); + if (metrics && [200, 206].includes(rangeResponse.status)) { + const parsed = JSON.parse(metrics); + + // Should have timing information + expect(parsed).toHaveProperty('start'); + + // May have cache-related metrics + const cacheKeys = Object.keys(parsed).filter(key => key.includes('cache')); + // At least one cache-related metric should be present + expect(cacheKeys.length).toBeGreaterThanOrEqual(0); + } + }); + }); +});