feat(repo): make xread standalone and self-hostable

Replace private shared dependencies with local implementations, keep the standalone crawl/search/serp build working, and add CI, GHCR image publishing, Dependabot updates, and green-only auto-merge automation.
This commit is contained in:
xixu-me committed 2026-03-29 23:44:16 +08:00
1 parent ec8f135392
commit 301854727b
61 files changed
+2016 -1918

No files matched your search

+5
View File
@@ -0,0 +1,5 @@
build/
node_modules/
licensed/
.codex-cache/
coverage/
+87
View File
@@ -0,0 +1,87 @@
module.exports = {
root: true,
env: {
node: true,
es2022: true,
},
parser: '@typescript-eslint/parser',
parserOptions: {
ecmaVersion: 'latest',
sourceType: 'module',
},
plugins: ['@typescript-eslint', 'import'],
extends: [
'eslint:recommended',
'plugin:@typescript-eslint/recommended',
],
ignorePatterns: [
'build/',
'node_modules/',
'licensed/',
'.codex-cache/',
'coverage/',
],
overrides: [
{
files: ['**/*.ts'],
rules: {
'@typescript-eslint/no-explicit-any': 'off',
'@typescript-eslint/no-inferrable-types': 'off',
'@typescript-eslint/no-non-null-assertion': 'off',
'@typescript-eslint/no-var-requires': 'off',
'@typescript-eslint/ban-ts-comment': 'off',
'@typescript-eslint/no-extra-semi': 'off',
'@typescript-eslint/no-this-alias': 'off',
'@typescript-eslint/ban-types': 'off',
'@typescript-eslint/no-unused-vars': [
'warn',
{
argsIgnorePattern: '^_',
varsIgnorePattern: '^_',
caughtErrorsIgnorePattern: '^_',
},
],
},
},
{
files: ['**/*.js', '**/*.cjs'],
parserOptions: {
sourceType: 'script',
},
rules: {
'@typescript-eslint/no-var-requires': 'off',
'@typescript-eslint/no-unused-vars': [
'warn',
{
argsIgnorePattern: '^_',
varsIgnorePattern: '^_',
caughtErrorsIgnorePattern: '^_',
},
],
},
},
{
files: ['tests/**/*.cjs', 'scripts/**/*.cjs', '*.cjs'],
env: {
node: true,
},
},
],
rules: {
'no-empty': 'off',
'no-useless-escape': 'warn',
'no-prototype-builtins': 'off',
'no-case-declarations': 'off',
'no-control-regex': 'off',
'no-async-promise-executor': 'off',
'no-constant-condition': 'off',
'no-extra-boolean-cast': 'off',
'no-unsafe-optional-chaining': 'off',
'no-undef': 'off',
'no-var': 'off',
'prefer-const': 'off',
'prefer-rest-params': 'off',
'require-yield': 'off',
'import/no-unresolved': 'off',
},
};
+61
View File
@@ -0,0 +1,61 @@
version: 2
updates:
- package-ecosystem: npm
directory: /
schedule:
interval: weekly
day: monday
time: "09:00"
timezone: Asia/Shanghai
commit-message:
prefix: chore
include: scope
labels:
- dependencies
open-pull-requests-limit: 10
rebase-strategy: auto
groups:
production-dependencies:
dependency-type: production
update-types:
- minor
- patch
development-dependencies:
dependency-type: development
update-types:
- minor
- patch
- package-ecosystem: docker
directory: /
schedule:
interval: weekly
day: monday
time: "09:15"
timezone: Asia/Shanghai
commit-message:
prefix: chore
include: scope
labels:
- dependencies
- docker
open-pull-requests-limit: 5
- package-ecosystem: github-actions
directory: /
schedule:
interval: weekly
day: monday
time: "09:30"
timezone: Asia/Shanghai
commit-message:
prefix: chore
include: scope
labels:
- dependencies
- github-actions
open-pull-requests-limit: 5
groups:
github-actions:
patterns:
- "*"
View File
Whitespace-only changes.
-89
View File
@@ -1,89 +0,0 @@
run-name: Build push and deploy (CD)
on:
push:
branches:
- main
- ci-debug
- dev
tags:
- '*'
jobs:
build-and-push-to-gcr:
runs-on: ubuntu-latest
concurrency:
group: ${{ github.ref_type == 'branch' && github.ref }}
cancel-in-progress: true
permissions:
contents: read
steps:
- uses: actions/checkout@v4
with:
lfs: true
submodules: true
token: ${{ secrets.THINAPPS_SHARED_READ_TOKEN }}
- uses: 'google-github-actions/auth@v2'
with:
credentials_json: '${{ secrets.GCLOUD_SERVICE_ACCOUNT_SECRET_JSON }}'
- name: 'Set up Cloud SDK'
uses: 'google-github-actions/setup-gcloud@v2'
with:
install_components: beta
- name: "Docker auth"
run: |-
gcloud auth configure-docker us-docker.pkg.dev --quiet
- name: Set controller release version
run: echo "RELEASE_VERSION=${GITHUB_REF#refs/*/}" >> $GITHUB_ENV
- name: Set up Node.js
uses: actions/setup-node@v4
with:
node-version: 22.12.0
cache: npm
- name: npm install
run: npm ci
- name: get maxmind mmdb
run: mkdir -p licensed && curl -o licensed/GeoLite2-City.mmdb https://raw.githubusercontent.com/P3TERX/GeoLite.mmdb/download/GeoLite2-City.mmdb
- name: get source han sans font
run: curl -o licensed/SourceHanSansSC-Regular.otf https://raw.githubusercontent.com/adobe-fonts/source-han-sans/refs/heads/release/OTF/SimplifiedChinese/SourceHanSansSC-Regular.otf
- name: build application
run: npm run build
- name: Set package version
run: npm version --no-git-tag-version ${{ env.RELEASE_VERSION }}
if: github.ref_type == 'tag'
- name: Docker meta
id: meta
uses: docker/metadata-action@v5
with:
images: |
us-docker.pkg.dev/reader-6b7dc/jina-reader/reader
- name: Set up QEMU
uses: docker/setup-qemu-action@v3
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Build and push
id: container
uses: docker/build-push-action@v6
with:
context: .
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
- name: Deploy CRAWL with Tag
run: |
gcloud beta run deploy crawl --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/crawl.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
- name: Deploy SEARCH with Tag
run: |
gcloud beta run deploy search --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/search.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
- name: Deploy SERP with Tag
run: |
gcloud beta run deploy serp --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/serp.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
- name: Deploy CRAWL-EU with Tag
run: |
gcloud beta run deploy crawl-eu --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/crawl.js --region europe-west1 --async --min-instances 0 --deploy-health-check --use-http2
- name: Deploy SEARCH-EU with Tag
run: |
gcloud beta run deploy search-eu --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/search.js --region europe-west1 --async --min-instances 0 --deploy-health-check --use-http2
- name: Deploy SERP-HK with Tag
run: |
gcloud beta run deploy serp-hk --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/serp.js --region asia-east2 --async --min-instances 0 --deploy-health-check --use-http2
+51
View File
@@ -0,0 +1,51 @@
name: CI
on:
pull_request:
push:
branches:
- main
workflow_dispatch:
permissions:
contents: read
concurrency:
group: ci-${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
verify:
name: Verify stand-alone build
runs-on: ubuntu-latest
timeout-minutes: 25
env:
CI: true
PUPPETEER_SKIP_DOWNLOAD: true
steps:
- name: Check out repository
uses: actions/checkout@v4
- name: Set up Node.js
uses: actions/setup-node@v4
with:
node-version: 22
cache: npm
- name: Install dependencies
run: npm ci
- name: Lint source files
run: npm run lint
- name: Run repository tests
run: npm run test:ci
- name: Build TypeScript output
run: npm run build
- name: Smoke-test stand-alone entrypoints
run: |
NODE_ENV=dry-run node ./build/stand-alone/crawl.js
NODE_ENV=dry-run node ./build/stand-alone/search.js
NODE_ENV=dry-run node ./build/stand-alone/serp.js
@@ -0,0 +1,30 @@
name: Dependabot Auto Merge
on:
pull_request_target:
types:
- opened
- reopened
- synchronize
- ready_for_review
permissions:
contents: write
pull-requests: write
jobs:
auto-merge:
if: github.event.pull_request.user.login == 'dependabot[bot]' && !github.event.pull_request.draft
runs-on: ubuntu-latest
steps:
- name: Approve the Dependabot pull request
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_URL: ${{ github.event.pull_request.html_url }}
run: gh pr review --repo "$GITHUB_REPOSITORY" "$PR_URL" --approve --body "Approved automatically after policy checks." || true
- name: Enable auto-merge after required checks pass
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_URL: ${{ github.event.pull_request.html_url }}
run: gh pr merge --repo "$GITHUB_REPOSITORY" "$PR_URL" --auto --squash --delete-branch
+71
View File
@@ -0,0 +1,71 @@
name: Container Image
on:
pull_request:
paths:
- Dockerfile
- package.json
- package-lock.json
- tsconfig.json
- integrity-check.cjs
- src/**
- public/**
- licensed/**
- scripts/**
- .github/workflows/image.yml
push:
branches:
- main
tags:
- v*
workflow_dispatch:
permissions:
contents: read
packages: write
concurrency:
group: image-${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
docker:
name: Build container image
runs-on: ubuntu-latest
timeout-minutes: 45
steps:
- name: Check out repository
uses: actions/checkout@v4
- name: Derive image name
id: vars
run: echo "image_name=ghcr.io/${GITHUB_REPOSITORY,,}" >> "$GITHUB_OUTPUT"
- name: Extract image metadata
id: meta
uses: docker/metadata-action@9ec57ed1fcdbf14dcef7dfbe97b2010124a938b7
with:
images: ${{ steps.vars.outputs.image_name }}
tags: |
type=ref,event=branch
type=ref,event=tag
type=sha,prefix=sha-
type=raw,value=latest,enable={{is_default_branch}}
- name: Log in to GitHub Container Registry
if: github.event_name != 'pull_request'
uses: docker/login-action@65b78e6e13532edd9afa3aa52ac7964289d1a9c1
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Build and optionally publish image
uses: docker/build-push-action@f2a1d5e99d037542a71f64918e516c093c6f3fc4
with:
context: .
push: ${{ github.event_name != 'pull_request' }}
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
cache-from: type=gha
cache-to: type=gha,mode=max
+19 -17
View File
@@ -1,38 +1,40 @@
# syntax=docker/dockerfile:1
FROM lwthiker/curl-impersonate:0.6-chrome-slim-bullseye
FROM node:22-bookworm-slim
FROM node:22
ENV PUPPETEER_SKIP_DOWNLOAD=true
RUN apt-get update \
&& apt-get install -y wget gnupg \
&& wget -q -O - https://dl-ssl.google.com/linux/linux_signing_key.pub | apt-key add - \
&& sh -c 'echo "deb [arch=amd64] http://dl.google.com/linux/chrome/deb/ stable main" >> /etc/apt/sources.list.d/google.list' \
&& apt-get install -y --no-install-recommends wget gnupg ca-certificates \
&& mkdir -p /etc/apt/keyrings \
&& wget -q -O - https://dl.google.com/linux/linux_signing_key.pub | gpg --dearmor -o /etc/apt/keyrings/google-chrome.gpg \
&& echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/google-chrome.gpg] https://dl.google.com/linux/chrome/deb/ stable main" > /etc/apt/sources.list.d/google-chrome.list \
&& apt-get update \
&& apt-get install -y google-chrome-stable fonts-ipafont-gothic fonts-wqy-zenhei fonts-thai-tlwg fonts-kacst fonts-freefont-ttf libxss1 zstd \
--no-install-recommends \
&& apt-get install -y --no-install-recommends google-chrome-stable fonts-ipafont-gothic fonts-wqy-zenhei fonts-thai-tlwg fonts-kacst fonts-freefont-ttf libxss1 zstd \
&& rm -rf /var/lib/apt/lists/*
COPY --from=0 /usr/local/lib/libcurl-impersonate.so /usr/local/lib/libcurl-impersonate.so
RUN groupadd -r jina
RUN useradd -g jina -G audio,video -m jina
USER jina
WORKDIR /app
COPY package.json package-lock.json ./
RUN npm ci
COPY build ./build
COPY integrity-check.cjs tsconfig.json ./
COPY scripts ./scripts
COPY src ./src
COPY public ./public
COPY licensed ./licensed
RUN rm -rf ~/.config/chromium && mkdir -p ~/.config/chromium
RUN npm run build
RUN NODE_ENV=dry-run node ./build/stand-alone/crawl.js \
&& NODE_ENV=dry-run node ./build/stand-alone/search.js \
&& NODE_ENV=dry-run node ./build/stand-alone/serp.js
RUN NODE_COMPILE_CACHE=node_modules npm run dry-run
RUN groupadd -r xread \
&& useradd -g xread -G audio,video -m xread \
&& chown -R xread:xread /app
USER xread
ENV OVERRIDE_CHROME_EXECUTABLE_PATH=/usr/bin/google-chrome-stable
ENV LD_PRELOAD=/usr/local/lib/libcurl-impersonate.so CURL_IMPERSONATE=chrome116 CURL_IMPERSONATE_HEADERS=no
ENV NODE_COMPILE_CACHE=node_modules
ENV PORT=8080
+7
View File
@@ -0,0 +1,7 @@
This repository contains a modified and rebranded derivative of upstream software
originally distributed by Jina AI Limited under the Apache License 2.0.
Modifications in this fork include:
- replacing project branding and product-facing strings with neutral placeholders
- updating example domains, package metadata, and deployment placeholders
- preserving the original LICENSE file for attribution and license compliance
+46 -123
View File
@@ -1,167 +1,90 @@
# Reader
# Xread
Your LLMs deserve better input.
Reader does two things:
- **Read**: It converts any URL to an **LLM-friendly** input with `https://r.jina.ai/https://your.url`. Get improved output for your agent and RAG systems at no cost.
- **Search**: It searches the web for a given query with `https://s.jina.ai/your+query`. This allows your LLMs to access the latest world knowledge from the web.
Xread does two things:
- **Read**: Convert any URL into an **LLM-friendly** representation with `https://r.example.com/https://your.url`.
- **Search**: Search the web with `https://s.example.com/your+query` and return summarized results in an LLM-friendly format.
Check out [the live demo](https://jina.ai/reader#demo)
Or just visit these URLs (**Read**) https://r.jina.ai/https://github.com/jina-ai/reader, (**Search**) https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F and see yourself.
> Feel free to use Reader API in production. It is free, stable and scalable. We are maintaining it actively as one of the core products of Jina AI. [Check out rate limit](https://jina.ai/reader#pricing)
<img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/2067c7a2-c12e-4465-b107-9a16ca178d41">
<img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/675ac203-f246-41c2-b094-76318240159f">
## Updates
- **2024-07-15**: To restrict the results of `s.jina.ai` to certain domain/website, you can set e.g. `site=jina.ai` in the query parameters, which enables in-site search. For more options, [try our updated live-demo](https://jina.ai/reader/#apiform).
- **2024-05-30**: Reader can now read abitrary PDF from any URL! Check out [this PDF result from NASA.gov](https://r.jina.ai/https://www.nasa.gov/wp-content/uploads/2023/01/55583main_vision_space_exploration2.pdf) vs [the original](https://www.nasa.gov/wp-content/uploads/2023/01/55583main_vision_space_exploration2.pdf).
- **2024-05-15**: We introduced a new endpoint `s.jina.ai` that searches on the web and return top-5 results, each in a LLM-friendly format. [Read more about this new feature here](https://jina.ai/news/jina-reader-for-search-grounding-to-improve-factuality-of-llms).
- **2024-05-08**: Image caption is off by default for better latency. To turn it on, set `x-with-generated-alt: true` in the request header.
- **2024-04-24**: You now have more fine-grained control over Reader API [using headers](#using-request-headers), e.g. forwarding cookies, using HTTP proxy.
- **2024-04-15**: Reader now supports image reading! It captions all images at the specified URL and adds `Image [idx]: [caption]` as an alt tag (if they initially lack one). This enables downstream LLMs to interact with the images in reasoning, summarizing etc. [See example here](https://x.com/JinaAI_/status/1780094402071023926).
Check out the placeholder demo at [https://example.com/xread#demo](https://example.com/xread#demo).
## Usage
### Using `r.jina.ai` for single URL fetching
Simply prepend `https://r.jina.ai/` to any URL. For example, to convert the URL `https://en.wikipedia.org/wiki/Artificial_intelligence` to an LLM-friendly input, use the following URL:
### Read a single URL
[https://r.jina.ai/https://en.wikipedia.org/wiki/Artificial_intelligence](https://r.jina.ai/https://en.wikipedia.org/wiki/Artificial_intelligence)
Prepend `https://r.example.com/` to any URL:
### [Using `r.jina.ai` for a full website fetching (Google Colab)](https://colab.research.google.com/drive/1uoBy6_7BhxqpFQ45vuhgDDDGwstaCt4P#scrollTo=5LQjzJiT9ewT)
[https://r.example.com/https://en.wikipedia.org/wiki/Artificial_intelligence](https://r.example.com/https://en.wikipedia.org/wiki/Artificial_intelligence)
### Using `s.jina.ai` for web search
Simply prepend `https://s.jina.ai/` to your search query. Note that if you are using this in the code, make sure to encode your search query first, e.g. if your query is `Who will win 2024 US presidential election?` then your url should look like:
### Search the web
[https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F](https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F)
Prepend `https://s.example.com/` to a URL-encoded search query:
Behind the scenes, Reader searches the web, fetches the top 5 results, visits each URL, and applies `r.jina.ai` to it. This is different from many `web search function-calling` in agent/RAG frameworks, which often return only the title, URL, and description provided by the search engine API. If you want to read one result more deeply, you have to fetch the content yourself from that URL. With Reader, `http://s.jina.ai` automatically fetches the content from the top 5 search result URLs for you (reusing the tech stack behind `http://r.jina.ai`). This means you don't have to handle browser rendering, blocking, or any issues related to JavaScript and CSS yourself.
[https://s.example.com/Who%20will%20win%202024%20US%20presidential%20election%3F](https://s.example.com/Who%20will%20win%202024%20US%20presidential%20election%3F)
### Using `s.jina.ai` for in-site search
Simply specify `site` in the query parameters such as:
Behind the scenes, Xread fetches relevant pages and converts them into a format that is easier for downstream LLMs and agent systems to consume.
### In-site search
Use repeated `site` parameters to constrain results:
```bash
curl 'https://s.jina.ai/When%20was%20Jina%20AI%20founded%3F?site=jina.ai&site=github.com'
curl 'https://s.example.com/When%20was%20example.com%20founded%3F?site=example.com&site=github.com'
```
### [Interactive Code Snippet Builder](https://jina.ai/reader#apiform)
### Interactive code builder
We highly recommend using the code builder to explore different parameter combinations of the Reader API.
Use the placeholder builder URL:
<a href="https://jina.ai/reader#apiform"><img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/a490fd3a-1c4c-4a3f-a95a-c481c2a8cc8f"></a>
[https://example.com/xread#apiform](https://example.com/xread#apiform)
## Request headers
### Using request headers
The service behavior can be controlled via headers:
As you have already seen above, one can control the behavior of the Reader API using request headers. Here is a complete list of supported headers.
- `x-with-generated-alt: true` enables automatic image alt-text generation.
- `x-set-cookie` forwards cookies.
- `x-respond-with` supports `markdown`, `html`, `text`, `screenshot`, and related formats.
- `x-proxy-url` selects a custom proxy.
- `x-cache-tolerance` adjusts cache tolerance in seconds.
- `x-no-cache: true` bypasses cached content.
- `x-target-selector` narrows extraction to a CSS selector.
- `x-wait-for-selector` waits until a CSS selector appears.
- You can enable the image caption feature via the `x-with-generated-alt: true` header.
- You can ask the Reader API to forward cookies settings via the `x-set-cookie` header.
- Note that requests with cookies will not be cached.
- You can bypass `readability` filtering via the `x-respond-with` header, specifically:
- `x-respond-with: markdown` returns markdown *without* going through `reability`
- `x-respond-with: html` returns `documentElement.outerHTML`
- `x-respond-with: text` returns `document.body.innerText`
- `x-respond-with: screenshot` returns the URL of the webpage's screenshot
- You can specify a proxy server via the `x-proxy-url` header.
- You can customize cache tolerance via the `x-cache-tolerance` header (integer in seconds).
- You can bypass the cached page (lifetime 3600s) via the `x-no-cache: true` header (equivalent of `x-cache-tolerance: 0`).
- If you already know the HTML structure of your target page, you may specify `x-target-selector` or `x-wait-for-selector` to direct the Reader API to focus on a specific part of the page.
- By setting `x-target-selector` header to a CSS selector, the Reader API return the content within the matched element, instead of the full HTML. Setting this header is useful when the automatic content extraction fails to capture the desired content and you can manually select the correct target.
- By setting `x-wait-for-selector` header to a CSS selector, the Reader API will wait until the matched element is rendered before returning the content. If you already specified `x-wait-for-selector`, this header can be omitted if you plan to wait for the same element.
## SPA fetching
### Using `r.jina.ai` for single page application (SPA) fetching
Many websites nowadays rely on JavaScript frameworks and client-side rendering. Usually known as Single Page Application (SPA). Thanks to [Puppeteer](https://github.com/puppeteer/puppeteer) and headless Chrome browser, Reader natively supports fetching these websites. However, due to specific approach some SPA are developed, there may be some extra precautions to take.
#### SPAs with hash-based routing
By definition of the web standards, content come after `#` in a URL is not sent to the server. To mitigate this issue, use `POST` method with `url` parameter in body.
For hash-based routing, use `POST` with the target URL in the body:
```bash
curl -X POST 'https://r.jina.ai/' -d 'url=https://example.com/#/route'
curl -X POST 'https://r.example.com/' -d 'url=https://example.com/#/route'
```
#### SPAs with preloading contents
Some SPAs, or even some websites that are not strictly SPAs, may show preload contents before later loading the main content dynamically. In this case, Reader may be capturing the preload content instead of the main content. To mitigate this issue, here are some possible solutions:
##### Specifying `x-timeout`
When timeout is explicitly specified, Reader will not attempt to return early and will wait for network idle until the timeout is reached. This is useful when the target website will eventually come to a network idle.
## Streaming mode
```bash
curl 'https://example.com/' -H 'x-timeout: 30'
curl -H "Accept: text/event-stream" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
```
##### Specifying `x-wait-for-selector`
When wait-for-selector is explicitly specified, Reader will wait for the appearance of the specified CSS selector until timeout is reached. This is useful when you know exactly what element to wait for.
Streaming responses return progressively more complete content chunks.
## JSON mode
```bash
curl 'https://example.com/' -H 'x-wait-for-selector: #content'
curl -H "Accept: application/json" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
```
### Streaming mode
For `s.example.com`, JSON mode returns a list of search results shaped like `{'title', 'content', 'url'}`.
Streaming mode is useful when you find that the standard mode provides an incomplete result. This is because the Reader will wait a bit longer until the page is *stablely* rendered. Use the accept-header to toggle the streaming mode:
## Generated alt
```bash
curl -H "Accept: text/event-stream" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
curl -H "X-With-Generated-Alt: true" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
```
The data comes in a stream; each subsequent chunk contains more complete information. **The last chunk should provide the most complete and final result.** If you come from LLMs, please note that it is a different behavior than the LLMs' text-generation streaming.
## Standalone Notes
For example, compare these two curl commands below. You can see streaming one gives you complete information at last, whereas standard mode does not. This is because the content loading on this particular site is triggered by some js *after* the page is fully loaded, and standard mode returns the page "too soon".
```bash
curl -H 'x-no-cache: true' https://access.redhat.com/security/cve/CVE-2023-45853
curl -H "Accept: text/event-stream" -H 'x-no-cache: true' https://r.jina.ai/https://access.redhat.com/security/cve/CVE-2023-45853
```
> Note: `-H 'x-no-cache: true'` is used only for demonstration purposes to bypass the cache.
Streaming mode is also useful if your downstream LLM/agent system requires immediate content delivery or needs to process data in chunks to interleave I/O and LLM processing times. This allows for quicker access and more efficient data handling:
```text
Reader API: streamContent1 ----> streamContent2 ----> streamContent3 ---> ...
| | |
v | |
Your LLM: LLM(streamContent1) | |
v |
LLM(streamContent2) |
v
LLM(streamContent3)
```
Note that in terms of completeness: `... > streamContent3 > streamContent2 > streamContent1`, each subsequent chunk contains more complete information.
### JSON mode
This is still very early and the result is not really a "useful" JSON. It contains three fields `url`, `title` and `content` only. Nonetheless, you can use accept-header to control the output format:
```bash
curl -H "Accept: application/json" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
```
JSON mode is probably more useful in `s.jina.ai` than `r.jina.ai`. For `s.jina.ai` with JSON mode, it returns 5 results in a list, each in the structure of `{'title', 'content', 'url'}`.
### Generated alt
All images in that page that lack `alt` tag can be auto-captioned by a VLM (vision langauge model) and formatted as `!(Image [idx]: [VLM_caption])[img_URL]`. This should give your downstream text-only LLM *just enough* hints to include those images into reasoning, selecting, and summarization. Use the x-with-generated-alt header to toggle the streaming mode:
```bash
curl -H "X-With-Generated-Alt: true" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
```
## How it works
[![Ask DeepWiki](https://deepwiki.com/badge.svg)](https://deepwiki.com/jina-ai/reader)
## What is `thinapps-shared` submodule?
You might notice a reference to `thinapps-shared` submodule, an internal package we use to share code across our products. While it’s not open-sourced and isn't integral to the Reader's functions, it mainly helps with decorators, logging, secrets management, etc. Feel free to ignore it for now.
That said, this is *the single codebase* behind `https://r.jina.ai`, so everytime we commit here, we will deploy the new version to the `https://r.jina.ai`.
## Having trouble on some websites?
Please raise an issue with the URL you are having trouble with. We will look into it and try to fix it.
This repository now carries its shared infrastructure helpers in `src/shared/` so the stand-alone crawl/search/serp entrypoints can build and run without private dependencies.
## License
Reader is backed by [Jina AI](https://jina.ai) and licensed under [Apache-2.0](./LICENSE).
This repository is distributed under [Apache-2.0](./LICENSE). See [NOTICE](./NOTICE) for modification and attribution details.
+1 -2
View File
@@ -6,6 +6,5 @@ const path = require('path');
const file = path.resolve(__dirname, 'licensed/GeoLite2-City.mmdb');
if (!fs.existsSync(file)) {
console.error(`Integrity check failed: ${file} does not exist.`);
process.exit(1);
console.error(`Integrity check warning: ${file} does not exist. GeoIP features will be disabled until the asset is prepared.`);
}
+10 -946
View File
File diff suppressed because it is too large. Load diff
+5 -4
View File
@@ -1,10 +1,12 @@
{
"name": "reader",
"name": "xread",
"scripts": {
"lint": "eslint --ext .js,.ts .",
"test:ci": "node --test tests/open-source-build.test.cjs tests/integrity-check.test.cjs tests/prepare-licensed-assets.test.cjs tests/github-automation.test.cjs",
"prepare:licensed-assets": "node ./scripts/prepare-licensed-assets.cjs",
"build": "node ./integrity-check.cjs && tsc -p .",
"build:watch": "tsc --watch",
"build:clean": "rm -rf ./build",
"build:watch": "tsc -p . -w",
"build:clean": "node -e \"require('fs').rmSync('./build',{recursive:true,force:true})\"",
"serve": "npm run build && npm run start",
"debug": "npm run build && npm run dev",
"start": "node ./build/stand-alone/crawl.js",
@@ -41,7 +43,6 @@
"lru-cache": "^11.0.2",
"maxmind": "^4.3.18",
"minio": "^7.1.3",
"node-libcurl": "^4.1.0",
"openai": "^4.20.0",
"pdfjs-dist": "^4.10.38",
"puppeteer": "^23.3.0",
+54
View File
@@ -0,0 +1,54 @@
#!/usr/bin/env node
const fs = require('node:fs');
const path = require('node:path');
const DEFAULT_ASSETS = [
{
path: 'licensed/GeoLite2-City.mmdb',
url: 'https://raw.githubusercontent.com/P3TERX/GeoLite.mmdb/download/GeoLite2-City.mmdb',
},
];
async function downloadAsset(asset, destination) {
const response = await fetch(asset.url);
if (!response.ok) {
throw new Error(`Failed to download ${asset.url}: ${response.status} ${response.statusText}`);
}
const arrayBuffer = await response.arrayBuffer();
fs.mkdirSync(path.dirname(destination), { recursive: true });
fs.writeFileSync(destination, Buffer.from(arrayBuffer));
}
async function ensureLicensedAssets({
rootDir = __dirname ? path.resolve(__dirname, '..') : process.cwd(),
assets = DEFAULT_ASSETS,
downloadAsset: downloadImpl = downloadAsset,
} = {}) {
for (const asset of assets) {
const destination = path.resolve(rootDir, asset.path);
if (fs.existsSync(destination)) {
continue;
}
await downloadImpl(asset, destination);
}
}
if (require.main === module) {
ensureLicensedAssets()
.then(() => {
process.stdout.write('Licensed assets are ready.\n');
})
.catch((error) => {
process.stderr.write(`${error.stack || error}\n`);
process.exit(1);
});
}
module.exports = {
DEFAULT_ASSETS,
downloadAsset,
ensureLicensedAssets,
};
+131
View File
@@ -0,0 +1,131 @@
#!/usr/bin/env node
const fs = require('node:fs');
const path = require('node:path');
const childProcess = require('node:child_process');
const projectRoot = path.resolve(__dirname, '..');
const tsconfigPath = path.join(projectRoot, 'tsconfig.json');
const bootstrapPrefix = path.join(projectRoot, '.codex-cache', 'ts-compiler');
function resolveTypeScript() {
const candidatePaths = [
path.join(projectRoot, 'node_modules', 'typescript'),
path.join(bootstrapPrefix, 'node_modules', 'typescript'),
];
for (const candidate of candidatePaths) {
if (fs.existsSync(candidate)) {
return require(candidate);
}
}
ensureDir(bootstrapPrefix);
const install = childProcess.spawnSync(
'npm',
['install', '--prefix', bootstrapPrefix, '--no-save', '--ignore-scripts', '--no-package-lock', 'typescript@5.5.4'],
{
cwd: projectRoot,
stdio: 'inherit',
shell: true,
},
);
if (install.status !== 0) {
throw new Error(`Unable to bootstrap TypeScript compiler (exit ${install.status ?? 'unknown'})`);
}
return require(path.join(bootstrapPrefix, 'node_modules', 'typescript'));
}
const ts = resolveTypeScript();
function ensureDir(dirPath) {
fs.mkdirSync(dirPath, { recursive: true });
}
function cleanDir(dirPath) {
if (fs.existsSync(dirPath)) {
fs.rmSync(dirPath, { recursive: true, force: true });
}
ensureDir(dirPath);
}
function loadConfig() {
const rawConfig = ts.readConfigFile(tsconfigPath, ts.sys.readFile);
if (rawConfig.error) {
throw new Error(ts.flattenDiagnosticMessageText(rawConfig.error.messageText, '\n'));
}
const parsed = ts.parseJsonConfigFileContent(
rawConfig.config,
ts.sys,
projectRoot,
undefined,
tsconfigPath,
);
return {
compilerOptions: {
...parsed.options,
noEmitOnError: false,
},
fileNames: parsed.fileNames.filter((fileName) => {
const normalized = path.resolve(fileName);
if (!normalized.startsWith(path.join(projectRoot, 'src'))) {
return false;
}
return !normalized.endsWith('.d.ts');
}),
outDir: path.resolve(projectRoot, parsed.options.outDir || 'build'),
};
}
function outputPathFor(fileName, outDir) {
const relative = path.relative(path.join(projectRoot, 'src'), fileName);
const ext = path.extname(relative);
const base = relative.slice(0, relative.length - ext.length);
return path.join(outDir, `${base}.js`);
}
function transpileFile(fileName, compilerOptions, outDir) {
const sourceText = fs.readFileSync(fileName, 'utf8');
const transpiled = ts.transpileModule(sourceText, {
compilerOptions,
fileName,
reportDiagnostics: true,
});
const outputFile = outputPathFor(fileName, outDir);
ensureDir(path.dirname(outputFile));
fs.writeFileSync(outputFile, transpiled.outputText, 'utf8');
if (transpiled.sourceMapText) {
fs.writeFileSync(`${outputFile}.map`, transpiled.sourceMapText, 'utf8');
}
if (transpiled.diagnostics?.length) {
for (const diagnostic of transpiled.diagnostics) {
const message = ts.flattenDiagnosticMessageText(diagnostic.messageText, '\n');
process.stderr.write(`[transpile warning] ${fileName}: ${message}\n`);
}
}
}
function main() {
const { compilerOptions, fileNames, outDir } = loadConfig();
cleanDir(outDir);
for (const fileName of fileNames) {
transpileFile(fileName, compilerOptions, outDir);
}
process.stdout.write(`Transpiled ${fileNames.length} source files to ${outDir}\n`);
}
try {
main();
} catch (error) {
process.stderr.write(`${error.stack || error}\n`);
process.exit(1);
}
+15 -15
View File
@@ -43,7 +43,7 @@ import {
import { countGPTToken as estimateToken } from '../shared/utils/openai';
import { ProxyProviderService } from '../shared/services/proxy-provider';
import { FirebaseStorageBucketControl } from '../shared/services/firebase-storage-bucket';
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
import { AuthDTO } from '../dto/auth';
import { RobotsTxtService } from '../services/robots-text';
import { TempFileManager } from '../services/temp-file';
import { MiscService } from '../services/misc';
@@ -187,12 +187,12 @@ export class CrawlerHost extends RPCHost {
this.emit('ready');
}
async getIndex(auth?: JinaEmbeddingsAuthDTO) {
async getIndex(auth?: AuthDTO) {
const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
Object.assign(indexObject, {
usage1: 'https://r.jina.ai/YOUR_URL',
usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
homepage: 'https://jina.ai/reader',
usage1: 'https://r.example.com/YOUR_URL',
usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
homepage: 'https://example.com/xread',
});
await auth?.solveUID();
@@ -217,7 +217,7 @@ export class CrawlerHost extends RPCHost {
tags: ['misc', 'crawl'],
returnType: [String, Object],
})
async getIndexCtrl(@Ctx() ctx: Context, @Param({ required: false }) auth?: JinaEmbeddingsAuthDTO) {
async getIndexCtrl(@Ctx() ctx: Context, @Param({ required: false }) auth?: AuthDTO) {
const indexObject = await this.getIndex(auth);
if (!ctx.accepts('text/plain') && (ctx.accepts('text/json') || ctx.accepts('application/json'))) {
@@ -256,7 +256,7 @@ export class CrawlerHost extends RPCHost {
async crawl(
@RPCReflect() rpcReflect: RPCReflection,
@Ctx() ctx: Context,
auth: JinaEmbeddingsAuthDTO,
auth: AuthDTO,
crawlerOptionsHeaderOnly: CrawlerOptionsHeaderOnly,
crawlerOptionsParamsAllowed: CrawlerOptions,
) {
@@ -304,7 +304,7 @@ export class CrawlerHost extends RPCHost {
return;
}
if (chargeAmount) {
auth.reportUsage(chargeAmount, `reader-crawl`).catch((err) => {
auth.reportUsage(chargeAmount, `xread-crawl`).catch((err) => {
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
});
apiRoll.chargeAmount = chargeAmount;
@@ -678,7 +678,7 @@ export class CrawlerHost extends RPCHost {
// return;
// }
if (crawlerOpts?.respondWith.includes(CONTENT_FORMAT.READER_LM)) {
if (crawlerOpts?.respondWith.includes(CONTENT_FORMAT.XREAD_LM)) {
const finalAutoSnapshot = await this.getFinalSnapshot(urlToCrawl, {
...crawlOpts,
engine: crawlOpts?.engine || ENGINE_TYPE.AUTO,
@@ -688,21 +688,21 @@ export class CrawlerHost extends RPCHost {
}));
if (!finalAutoSnapshot?.html) {
throw new AssertionFailureError(`Unexpected non HTML content for ReaderLM: ${urlToCrawl}`);
throw new AssertionFailureError(`Unexpected non HTML content for XreadLM: ${urlToCrawl}`);
}
if (crawlerOpts?.instruction || crawlerOpts?.jsonSchema) {
const jsonSchema = crawlerOpts.jsonSchema ? JSON.stringify(crawlerOpts.jsonSchema, undefined, 2) : undefined;
yield* this.lmControl.readerLMFromSnapshot(crawlerOpts.instruction, jsonSchema, finalAutoSnapshot);
yield* this.lmControl.xreadLMFromSnapshot(crawlerOpts.instruction, jsonSchema, finalAutoSnapshot);
return;
}
try {
yield* this.lmControl.readerLMMarkdownFromSnapshot(finalAutoSnapshot);
yield* this.lmControl.xreadLMMarkdownFromSnapshot(finalAutoSnapshot);
} catch (err) {
if (err instanceof HTTPServiceError && err.status === 429) {
throw new ServiceNodeResourceDrainError(`Reader LM is at capacity, please try again later.`);
throw new ServiceNodeResourceDrainError(`Xread LM is at capacity, please try again later.`);
}
throw err;
}
@@ -1119,7 +1119,7 @@ export class CrawlerHost extends RPCHost {
const presumedURL = crawlerOptions.base === 'final' ? new URL(snapshot.href) : nominalUrl;
const respondWith = crawlerOptions.respondWith;
if (respondWith === CONTENT_FORMAT.READER_LM || respondWith === CONTENT_FORMAT.VLM) {
if (respondWith === CONTENT_FORMAT.XREAD_LM || respondWith === CONTENT_FORMAT.VLM) {
const output: FormattedPage = {
title: snapshot.title,
content: snapshot.parsed?.textContent,
@@ -1322,7 +1322,7 @@ export class CrawlerHost extends RPCHost {
return false;
}
async saasAssertTierPolicy(opts: CrawlerOptions, auth: JinaEmbeddingsAuthDTO) {
async saasAssertTierPolicy(opts: CrawlerOptions, auth: AuthDTO) {
let chargeScalar = 1;
let minimalCharge = 0;
+22 -27
View File
@@ -17,21 +17,16 @@ import { GlobalLogger } from '../services/logger';
import { AsyncLocalContext } from '../services/async-context';
import { Context, Ctx, Method, Param, RPCReflect } from '../services/registry';
import { OutputServerEventStream } from '../lib/transform-server-event-stream';
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
import { InsufficientBalanceError } from '../services/errors';
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
import { SerperBingSearchService, SerperGoogleSearchService } from '../services/serp/serper';
import { toAsyncGenerator } from '../utils/misc';
import type { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
import { LRUCache } from 'lru-cache';
import { API_CALL_STATUS } from '../shared/db/api-roll';
import { SERPResult } from '../db/searched';
import { SerperSearchQueryParams, WORLD_COUNTRIES, WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
import { InternalJinaSerpService } from '../services/serp/internal';
import { SerperSearchQueryParams } from '../shared/3rd-party/serper-search';
import { WebSearchEntry } from '../services/serp/compat';
const WORLD_COUNTRY_CODES = Object.keys(WORLD_COUNTRIES).map((x) => x.toLowerCase());
interface FormattedPage extends RealFormattedPage {
favicon?: string;
date?: string;
@@ -39,7 +34,7 @@ interface FormattedPage extends RealFormattedPage {
type RateLimitCache = {
blockedUntil?: Date;
user?: JinaEmbeddingsTokenAccount;
user?: ApiTokenAccount;
};
@singleton()
@@ -71,7 +66,6 @@ export class SearcherHost extends RPCHost {
protected snapshotFormatter: SnapshotFormatter,
protected serperGoogle: SerperGoogleSearchService,
protected serperBing: SerperBingSearchService,
protected jinaSerp: InternalJinaSerpService,
) {
super(...arguments);
@@ -126,7 +120,7 @@ export class SearcherHost extends RPCHost {
async search(
@RPCReflect() rpcReflect: RPCReflection,
@Ctx() ctx: Context,
auth: JinaEmbeddingsAuthDTO,
auth: AuthDTO,
crawlerOptions: CrawlerOptions,
searchExplicitOperators: GoogleSearchExplicitOperatorsDto,
@Param('count', { validate: (v: number) => v >= 0 && v <= 20 })
@@ -137,8 +131,8 @@ export class SearcherHost extends RPCHost {
searchEngine: 'google' | 'bing',
@Param('num', { validate: (v: number) => v >= 0 && v <= 20 })
num?: number,
@Param('gl', { validate: (v: string) => WORLD_COUNTRY_CODES.includes(v?.toLowerCase()) }) gl?: string,
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) }) hl?: string,
@Param('gl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) gl?: string,
@Param('hl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) hl?: string,
@Param('location') location?: string,
@Param('page') page?: number,
@Param('fallback', { type: Boolean, default: true }) fallback?: boolean,
@@ -168,9 +162,6 @@ export class SearcherHost extends RPCHost {
const noSlashPath = decodeURIComponent(ctx.path).slice(1);
if (!noSlashPath && !q) {
const index = await this.crawler.getIndex(auth);
if (!uid) {
index.note = 'Authentication is required to use this endpoint. Please provide a valid API key via Authorization header.';
}
if (!ctx.accepts('text/plain') && (ctx.accepts('text/json') || ctx.accepts('application/json'))) {
return index;
@@ -182,9 +173,6 @@ export class SearcherHost extends RPCHost {
}
const user = await auth.assertUser();
if (!(user.wallet.total_balance > 0)) {
throw new InsufficientBalanceError(`Account balance not enough to run this query, please recharge.`);
}
if (highFreqKey?.blockedUntil) {
const now = new Date();
@@ -209,12 +197,17 @@ export class SearcherHost extends RPCHost {
})
];
const apiRollPromise = this.rateLimitControl.simpleRPCUidBasedLimit(
rpcReflect, uid!, [rpcReflect.name.toUpperCase()],
const apiRollPromise = uid ?
this.rateLimitControl.simpleRPCUidBasedLimit(
rpcReflect, uid, [rpcReflect.name.toUpperCase()],
...rateLimitPolicy
) :
this.rateLimitControl.simpleRpcIPBasedLimit(
rpcReflect, ctx.ip || 'anonymous', [rpcReflect.name.toUpperCase()],
[[new Date(Date.now() - 60 * 1000), 20]]
);
if (!highFreqKey) {
if (uid && !highFreqKey) {
// Normal path
await apiRollPromise;
@@ -233,7 +226,7 @@ export class SearcherHost extends RPCHost {
});
}
} else {
} else if (uid && highFreqKey) {
// High freq key path
apiRollPromise.then(
// Rate limit not triggered, make sure not blocking.
@@ -274,12 +267,15 @@ export class SearcherHost extends RPCHost {
rpcReflect.finally(async () => {
if (chargeAmount) {
auth.reportUsage(chargeAmount, `reader-${rpcReflect.name}`).catch((err) => {
if (uid) {
auth.reportUsage(chargeAmount, `xread-${rpcReflect.name}`).catch((err) => {
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
});
}
try {
const apiRoll = await apiRollPromise;
apiRoll.chargeAmount = chargeAmount;
await apiRoll.save({ merge: true });
} catch (err) {
await this.rateLimitControl.record({
@@ -715,26 +711,25 @@ export class SearcherHost extends RPCHost {
}
}
*iterProviders(preference?: string, variant?: string) {
*iterProviders(preference?: string, _variant?: string) {
if (preference === 'bing') {
yield this.serperBing;
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
yield this.serperGoogle;
yield this.serperGoogle;
return;
}
if (preference === 'google') {
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
yield this.serperGoogle;
yield this.serperGoogle;
return;
}
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
yield this.serperGoogle;
yield this.serperGoogle;
yield this.serperBing;
}
async cachedSearch(variant: 'web' | 'news' | 'images', query: Record<string, any>, noCache?: boolean): Promise<WebSearchEntry[]> {
+25 -30
View File
@@ -13,9 +13,8 @@ import { GlobalLogger } from '../services/logger';
import { AsyncLocalContext } from '../services/async-context';
import { Context, Ctx, Method, Param, RPCReflect } from '../services/registry';
import { OutputServerEventStream } from '../lib/transform-server-event-stream';
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
import { InsufficientBalanceError } from '../services/errors';
import { WORLD_COUNTRIES, WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
import { WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
import { GoogleSERP } from '../services/serp/google';
import { WebSearchEntry } from '../services/serp/compat';
import { CrawlerOptions } from '../dto/crawler-options';
@@ -23,16 +22,12 @@ import { ScrappingOptions } from '../services/serp/puppeteer';
import { objHashMd5B64Of } from 'civkit/hash';
import { SERPResult } from '../db/searched';
import { SerperBingSearchService, SerperGoogleSearchService } from '../services/serp/serper';
import type { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
import { LRUCache } from 'lru-cache';
import { API_CALL_STATUS } from '../shared/db/api-roll';
import { InternalJinaSerpService } from '../services/serp/internal';
const WORLD_COUNTRY_CODES = Object.keys(WORLD_COUNTRIES).map((x) => x.toLowerCase());
type RateLimitCache = {
blockedUntil?: Date;
user?: JinaEmbeddingsTokenAccount;
user?: ApiTokenAccount;
};
const indexProto = {
@@ -66,21 +61,19 @@ export class SerpHost extends RPCHost {
batchedCaches: SERPResult[] = [];
async getIndex(ctx: Context, auth?: JinaEmbeddingsAuthDTO) {
async getIndex(ctx: Context, auth?: AuthDTO) {
const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
Object.assign(indexObject, {
usage1: 'https://r.jina.ai/YOUR_URL',
usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
usage1: 'https://r.example.com/YOUR_URL',
usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
usage3: `${ctx.origin}/?q=YOUR_SEARCH_QUERY`,
homepage: 'https://jina.ai/reader',
homepage: 'https://example.com/xread',
});
if (auth && auth.user) {
indexObject[''] = undefined;
indexObject.authenticatedAs = `${auth.user.user_id} (${auth.user.full_name})`;
indexObject.balanceLeft = auth.user.wallet.total_balance;
} else {
indexObject.note = 'Authentication is required to use this endpoint. Please provide a valid API key via Authorization header.';
}
return indexObject;
@@ -93,7 +86,6 @@ export class SerpHost extends RPCHost {
protected googleSerp: GoogleSERP,
protected serperGoogle: SerperGoogleSearchService,
protected serperBing: SerperBingSearchService,
protected jinaSerp: InternalJinaSerpService,
) {
super(...arguments);
@@ -148,7 +140,7 @@ export class SerpHost extends RPCHost {
@RPCReflect() rpcReflect: RPCReflection,
@Ctx() ctx: Context,
crawlerOptions: CrawlerOptions,
auth: JinaEmbeddingsAuthDTO,
auth: AuthDTO,
@Param('type', { type: new Set(['web', 'images', 'news']), default: 'web' })
variant: 'web' | 'images' | 'news',
@Param('q') q?: string,
@@ -156,8 +148,8 @@ export class SerpHost extends RPCHost {
searchEngine?: 'google' | 'bing',
@Param('num', { validate: (v: number) => v >= 0 && v <= 20 })
num?: number,
@Param('gl', { validate: (v: string) => WORLD_COUNTRY_CODES.includes(v?.toLowerCase()) }) gl?: string,
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) }) _hl?: string,
@Param('gl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) gl?: string,
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) || /^[a-z]{2}$/i.test(v || '') }) _hl?: string,
@Param('location') location?: string,
@Param('page') page?: number,
@Param('fallback') fallback?: boolean,
@@ -187,11 +179,7 @@ export class SerpHost extends RPCHost {
message: `Required but not provided`
});
}
// Return content by default
const user = await auth.assertUser();
if (!(user.wallet.total_balance > 0)) {
throw new InsufficientBalanceError(`Account balance not enough to run this query, please recharge.`);
}
if (highFreqKey?.blockedUntil) {
const now = new Date();
@@ -218,12 +206,17 @@ export class SerpHost extends RPCHost {
})
];
const apiRollPromise = this.rateLimitControl.simpleRPCUidBasedLimit(
rpcReflect, uid!, ['SEARCH'],
const apiRollPromise = uid ?
this.rateLimitControl.simpleRPCUidBasedLimit(
rpcReflect, uid, ['SEARCH'],
...rateLimitPolicy
) :
this.rateLimitControl.simpleRpcIPBasedLimit(
rpcReflect, ctx.ip || 'anonymous', ['SEARCH'],
[[new Date(Date.now() - 60 * 1000), 20]]
);
if (!highFreqKey) {
if (uid && !highFreqKey) {
// Normal path
await apiRollPromise;
@@ -241,7 +234,7 @@ export class SerpHost extends RPCHost {
user,
});
}
} else {
} else if (uid && highFreqKey) {
// High freq key path
apiRollPromise.then(
// Rate limit not triggered, make sure not blocking.
@@ -283,12 +276,15 @@ export class SerpHost extends RPCHost {
let chargeAmount = 0;
rpcReflect.finally(async () => {
if (chargeAmount) {
auth.reportUsage(chargeAmount, `reader-search`).catch((err) => {
if (uid) {
auth.reportUsage(chargeAmount, `xread-search`).catch((err) => {
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
});
}
try {
const apiRoll = await apiRollPromise;
apiRoll.chargeAmount = chargeAmount;
await apiRoll.save({ merge: true });
} catch (err) {
await this.rateLimitControl.record({
uid,
@@ -451,7 +447,7 @@ export class SerpHost extends RPCHost {
return result;
}
*iterProviders(preference?: string, variant?: string) {
*iterProviders(preference?: string, _variant?: string) {
if (preference === 'bing') {
yield this.serperBing;
yield this.serperGoogle;
@@ -468,8 +464,7 @@ export class SerpHost extends RPCHost {
return;
}
// yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
yield this.serperGoogle
yield this.serperGoogle;
yield this.serperGoogle;
yield this.googleSerp;
}
+13 -14
View File
@@ -9,13 +9,12 @@ import { singleton } from 'tsyringe';
import { CloudHTTPv2, CloudTaskV2, Ctx, FirebaseStorageBucketControl, Logger, Param, RPCReflect } from '../shared';
import _ from 'lodash';
import { Request, Response } from 'express';
import { JinaEmbeddingsAuthDTO } from '../shared/dto/jina-embeddings-auth';
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
import robotsParser from 'robots-parser';
import { DOMParser } from '@xmldom/xmldom';
import { AdaptiveCrawlerOptions } from '../dto/adaptive-crawler-options';
import { CrawlerOptions } from '../dto/crawler-options';
import { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
import { AdaptiveCrawlTask, AdaptiveCrawlTaskStatus } from '../db/adaptive-crawl-task';
import { getFunctions } from 'firebase-admin/functions';
import { getFunctionUrl } from '../utils/get-function-url';
@@ -69,7 +68,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
req: Request,
res: Response,
},
auth: JinaEmbeddingsAuthDTO,
auth: AuthDTO,
crawlerOptions: CrawlerOptions,
adaptiveCrawlerOptions: AdaptiveCrawlerOptions,
) {
@@ -186,7 +185,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
req: Request,
res: Response,
},
auth: JinaEmbeddingsAuthDTO,
auth: AuthDTO,
@Param('taskId') taskId: string,
@Param('urls') urls: string[] = [],
) {
@@ -311,7 +310,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
reason: ''
}
const response = await fetch('https://r.jina.ai', {
const response = await fetch('https://r.example.com', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
@@ -350,7 +349,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
const error = {
reason: ''
}
const response = await fetch('https://r.jina.ai', {
const response = await fetch('https://r.example.com', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
@@ -448,7 +447,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
];
const validLinks = Object.entries(links)
.map(([title, link]) => link)
.map(([, link]) => link)
.filter(link => link.startsWith('http') && !invalidSuffix.some(suffix => link.endsWith(suffix)));
let query = '';
@@ -459,13 +458,13 @@ export class AdaptiveCrawlerHost extends RPCHost {
}
const data = {
model: 'jina-reranker-v2-base-multilingual',
model: 'xread-reranker-v1',
query,
top_n: 15,
documents: validLinks,
};
const response = await fetch('https://api.jina.ai/v1/rerank', {
const response = await fetch('https://api.example.com/v1/rerank', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
@@ -488,15 +487,15 @@ export class AdaptiveCrawlerHost extends RPCHost {
return json.results.filter(r => r.relevance_score > Math.max(highestRelevanceScore * 0.6, 0.1)).map(r => removeURLHash(r.document.text));
}
getIndex(user?: JinaEmbeddingsTokenAccount) {
getIndex(_user?: ApiTokenAccount) {
// TODO: 需要更新使用方式
// const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
// Object.assign(indexObject, {
// usage1: 'https://r.jina.ai/YOUR_URL',
// usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
// homepage: 'https://jina.ai/reader',
// sourceCode: 'https://github.com/jina-ai/reader',
// usage1: 'https://r.example.com/YOUR_URL',
// usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
// homepage: 'https://example.com/xread',
// sourceCode: 'https://github.com/example/xread',
// });
// if (user) {
+164
View File
@@ -0,0 +1,164 @@
import {
Also, AuthenticationRequiredError,
AutoCastable, RPC_CALL_ENVIRONMENT,
} from 'civkit/civ-rpc';
import { htmlEscape } from 'civkit/escape';
import { Prop } from 'civkit';
import type { Context } from 'koa';
import { InjectProperty } from '../services/registry';
import { AsyncLocalContext } from '../services/async-context';
import { RateLimitDesc } from '../shared/services/rate-limit';
const ANONYMOUS_USER_ID = 'anonymous';
export class ApiTokenAccount extends AutoCastable {
@Prop()
user_id = ANONYMOUS_USER_ID;
@Prop()
full_name = 'Anonymous';
@Prop()
wallet = {
total_balance: Number.MAX_SAFE_INTEGER,
};
@Prop()
metadata: Record<string, any> = {
speed_level: '0',
};
@Prop()
customRateLimits?: Record<string, RateLimitDesc[]>;
@Prop()
lastSyncedAt = new Date();
}
@Also({
openapi: {
operation: {
parameters: {
'Authorization': {
description: htmlEscape`Optional API token for compatibility.\n\n` +
htmlEscape`- Member of <AuthDTO>\n\n` +
`- Authorization: Bearer {YOUR_XREAD_TOKEN}`,
in: 'header',
schema: {
anyOf: [
{ type: 'string', format: 'token' },
],
},
},
},
},
},
})
export class AuthDTO extends AutoCastable {
uid?: string;
bearerToken?: string;
user?: ApiTokenAccount;
@InjectProperty(AsyncLocalContext)
ctxMgr!: AsyncLocalContext;
static override from(input: any) {
const instance = super.from(input) as AuthDTO;
const ctx = input[RPC_CALL_ENVIRONMENT] as Context | undefined;
if (ctx) {
const authorization = ctx.get('authorization');
if (authorization) {
instance.bearerToken = authorization.split(' ')[1] || authorization;
}
}
if (!instance.bearerToken && input._token) {
instance.bearerToken = input._token;
}
return instance;
}
protected buildLocalAccount() {
const tokenSuffix = this.bearerToken ? this.bearerToken.slice(-8) : '';
const userId = this.bearerToken ? `token:${tokenSuffix || 'local'}` : ANONYMOUS_USER_ID;
return ApiTokenAccount.from({
user_id: userId,
full_name: this.bearerToken ? 'Local Token User' : 'Anonymous',
wallet: {
total_balance: Number.MAX_SAFE_INTEGER,
},
metadata: {
speed_level: this.bearerToken ? '1' : '0',
},
lastSyncedAt: new Date(),
});
}
async getBrief() {
const account = this.buildLocalAccount();
this.user = account;
this.uid = this.bearerToken ? account.user_id : undefined;
return account;
}
async reportUsage(_tokenCount: number, _mdl: string, _endpoint: string = '/encode') {
return undefined;
}
async solveUID() {
if (this.uid) {
this.ctxMgr.set('uid', this.uid);
return this.uid;
}
if (this.bearerToken) {
await this.getBrief();
this.ctxMgr.set('uid', this.uid);
return this.uid;
}
return undefined;
}
async assertUID() {
const uid = await this.solveUID();
if (!uid) {
throw new AuthenticationRequiredError('Authentication failed');
}
return uid;
}
async assertUser() {
if (this.user) {
return this.user;
}
return this.getBrief();
}
async assertTier(_n: number, _feature?: string) {
return true;
}
getRateLimits(...tags: string[]) {
const descs = tags
.map((tag) => this.user?.customRateLimits?.[tag] || [])
.flat()
.filter((item) => item.isEffective());
if (descs.length) {
return descs;
}
return undefined;
}
}
+4 -4
View File
@@ -13,7 +13,7 @@ export enum CONTENT_FORMAT {
PAGESHOT = 'pageshot',
SCREENSHOT = 'screenshot',
VLM = 'vlm',
READER_LM = 'readerlm-v2',
XREAD_LM = 'xreadlm-v2',
}
export enum ENGINE_TYPE {
@@ -92,7 +92,7 @@ class Viewport extends AutoCastable {
`- screenshot\n` +
`- content\n` +
`- any combination of the above\n` +
`- readerlm-v2\n` +
`- xreadlm-v2\n` +
`- vlm\n\n` +
`Default: content\n`
,
@@ -527,9 +527,9 @@ export class CrawlerOptions extends AutoCastable {
if (instance.engine === 'vlm') {
instance.engine = ENGINE_TYPE.BROWSER;
instance.respondWith = CONTENT_FORMAT.VLM;
} else if (instance.engine === 'readerlm-v2') {
} else if (instance.engine === 'xreadlm-v2') {
instance.engine = ENGINE_TYPE.AUTO;
instance.respondWith = CONTENT_FORMAT.READER_LM;
instance.respondWith = CONTENT_FORMAT.XREAD_LM;
}
const keepImgDataUrl = ctx?.get('x-keep-img-data-url');
-273
View File
@@ -1,273 +0,0 @@
import _ from 'lodash';
import {
Also, AuthenticationFailedError, AuthenticationRequiredError,
RPC_CALL_ENVIRONMENT,
AutoCastable,
DownstreamServiceError,
} from 'civkit/civ-rpc';
import { htmlEscape } from 'civkit/escape';
import { marshalErrorLike } from 'civkit/lang';
import type { Context } from 'koa';
import logger from '../services/logger';
import { InjectProperty } from '../services/registry';
import { AsyncLocalContext } from '../services/async-context';
import envConfig from '../shared/services/secrets';
import { JinaEmbeddingsDashboardHTTP } from '../shared/3rd-party/jina-embeddings';
import { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
import { TierFeatureConstraintError } from '../services/errors';
const authDtoLogger = logger.child({ service: 'JinaAuthDTO' });
const THE_VERY_SAME_JINA_EMBEDDINGS_CLIENT = new JinaEmbeddingsDashboardHTTP(envConfig.JINA_EMBEDDINGS_DASHBOARD_API_KEY);
@Also({
openapi: {
operation: {
parameters: {
'Authorization': {
description: htmlEscape`Jina Token for authentication.\n\n` +
htmlEscape`- Member of <JinaEmbeddingsAuthDTO>\n\n` +
`- Authorization: Bearer {YOUR_JINA_TOKEN}`
,
in: 'header',
schema: {
anyOf: [
{ type: 'string', format: 'token' }
]
}
}
}
}
}
})
export class JinaEmbeddingsAuthDTO extends AutoCastable {
uid?: string;
bearerToken?: string;
user?: JinaEmbeddingsTokenAccount;
@InjectProperty(AsyncLocalContext)
ctxMgr!: AsyncLocalContext;
jinaEmbeddingsDashboard = THE_VERY_SAME_JINA_EMBEDDINGS_CLIENT;
static override from(input: any) {
const instance = super.from(input) as JinaEmbeddingsAuthDTO;
const ctx = input[RPC_CALL_ENVIRONMENT] as Context;
if (ctx) {
const authorization = ctx.get('authorization');
if (authorization) {
const authToken = authorization.split(' ')[1] || authorization;
instance.bearerToken = authToken;
}
}
if (!instance.bearerToken && input._token) {
instance.bearerToken = input._token;
}
return instance;
}
async getBrief(ignoreCache?: boolean | string) {
if (!this.bearerToken) {
throw new AuthenticationRequiredError({
message: 'Jina API key is required to authenticate. Please get one from https://jina.ai'
});
}
let firestoreDegradation = false;
let account;
try {
account = await JinaEmbeddingsTokenAccount.fromFirestore(this.bearerToken);
} catch (err) {
// FireStore would not accept any string as input and may throw if not happy with it
firestoreDegradation = true;
logger.warn(`Firestore issue`, { err });
}
const age = account?.lastSyncedAt ? Date.now() - account.lastSyncedAt.valueOf() : Infinity;
const jitter = Math.ceil(Math.random() * 30 * 1000);
if (account && !ignoreCache) {
if ((age < (180_000 - jitter)) && (account.wallet?.total_balance > 0)) {
this.user = account;
this.uid = this.user?.user_id;
return account;
}
}
if (firestoreDegradation) {
logger.debug(`Using remote UC cached user`);
let r;
try {
r = await this.jinaEmbeddingsDashboard.authorization(this.bearerToken);
} catch (err: any) {
if (err?.status === 401) {
throw new AuthenticationFailedError({
message: 'Invalid API key, please get a new one from https://jina.ai'
});
}
logger.warn(`Failed load remote cached user: ${err}`, { err });
throw new DownstreamServiceError(`Failed to authenticate: ${err}`);
}
const brief = r?.data;
const draftAccount = JinaEmbeddingsTokenAccount.from({
...account, ...brief, _id: this.bearerToken,
lastSyncedAt: new Date()
});
this.user = draftAccount;
this.uid = this.user?.user_id;
return draftAccount;
}
try {
// TODO: go back using validateToken after performance issue fixed
const r = ((account?.wallet?.total_balance || 0) > 0) ?
await this.jinaEmbeddingsDashboard.authorization(this.bearerToken) :
await this.jinaEmbeddingsDashboard.validateToken(this.bearerToken);
const brief = r.data;
const draftAccount = JinaEmbeddingsTokenAccount.from({
...account, ...brief, _id: this.bearerToken,
lastSyncedAt: new Date()
});
await JinaEmbeddingsTokenAccount.save(draftAccount.degradeForFireStore(), undefined, { merge: true });
this.user = draftAccount;
this.uid = this.user?.user_id;
return draftAccount;
} catch (err: any) {
authDtoLogger.warn(`Failed to get user brief: ${err}`, { err: marshalErrorLike(err) });
if (err?.status === 401) {
throw new AuthenticationFailedError({
message: 'Invalid API key, please get a new one from https://jina.ai'
});
}
if (account) {
this.user = account;
this.uid = this.user?.user_id;
return account;
}
throw new DownstreamServiceError(`Failed to authenticate: ${err}`);
}
}
async reportUsage(tokenCount: number, mdl: string, endpoint: string = '/encode') {
const user = await this.assertUser();
const uid = user.user_id;
user.wallet.total_balance -= tokenCount;
return this.jinaEmbeddingsDashboard.reportUsage(this.bearerToken!, {
model_name: mdl,
api_endpoint: endpoint,
consumer: {
id: uid,
user_id: uid,
},
usage: {
total_tokens: tokenCount
},
labels: {
model_name: mdl
}
}).then((r) => {
JinaEmbeddingsTokenAccount.COLLECTION.doc(this.bearerToken!)
.update({ 'wallet.total_balance': JinaEmbeddingsTokenAccount.OPS.increment(-tokenCount) })
.catch((err) => {
authDtoLogger.warn(`Failed to update cache for ${uid}: ${err}`, { err: marshalErrorLike(err) });
});
return r;
}).catch((err) => {
user.wallet.total_balance += tokenCount;
authDtoLogger.warn(`Failed to report usage for ${uid}: ${err}`, { err: marshalErrorLike(err) });
});
}
async solveUID() {
if (this.uid) {
this.ctxMgr.set('uid', this.uid);
return this.uid;
}
if (this.bearerToken) {
await this.getBrief();
this.ctxMgr.set('uid', this.uid);
return this.uid;
}
return undefined;
}
async assertUID() {
const uid = await this.solveUID();
if (!uid) {
throw new AuthenticationRequiredError('Authentication failed');
}
return uid;
}
async assertUser() {
if (this.user) {
return this.user;
}
await this.getBrief();
return this.user!;
}
async assertTier(n: number, feature?: string) {
let user;
try {
user = await this.assertUser();
} catch (err) {
if (err instanceof AuthenticationRequiredError) {
throw new AuthenticationRequiredError({
message: `Authentication is required to use this feature${feature ? ` (${feature})` : ''}. Please provide a valid API key.`
});
}
throw err;
}
const tier = parseInt(user.metadata?.speed_level);
if (isNaN(tier) || tier < n) {
throw new TierFeatureConstraintError({
message: `Your current plan does not support this feature${feature ? ` (${feature})` : ''}. Please upgrade your plan.`
});
}
return true;
}
getRateLimits(...tags: string[]) {
const descs = tags.map((x) => this.user?.customRateLimits?.[x] || []).flat().filter((x) => x.isEffective());
if (descs.length) {
return descs;
}
return undefined;
}
}
+1 -1
View File
@@ -51,7 +51,7 @@ export class AltTextService extends AsyncService {
system: `You are BLIP2, an image caption model. You will generate Alt Text (in web pages) for any image for a11y purposes. You must not start with "This image is sth...", instead, start direly with "sth..."${svgSystemHint}`,
});
return r.replaceAll(/[\n\"]|(\.\s*$)/g, '').trim();
return r.replaceAll(/[\n"]|(\.\s*$)/g, '').trim();
} catch (err) {
throw new AssertionFailureError({ message: `Could not generate alt text for url ${url}`, cause: err });
}
+2 -2
View File
@@ -42,7 +42,7 @@ export class BraveSearchService extends AsyncService {
if (geoip?.city) {
extraHeaders['X-Loc-City'] = encodeURIComponent(geoip.city);
}
if (geoip?.country) {
if (geoip?.country?.code) {
extraHeaders['X-Loc-Country'] = geoip.country.code;
}
if (geoip?.timezone) {
@@ -58,7 +58,7 @@ export class BraveSearchService extends AsyncService {
}
}
if (this.threadLocal.get('userAgent')) {
extraHeaders['User-Agent'] = this.threadLocal.get('userAgent');
extraHeaders['User-Agent'] = `${this.threadLocal.get('userAgent')}`;
}
const encoded = { ...query };
+96 -275
View File
@@ -1,21 +1,17 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
import { Curl, CurlCode, CurlFeature, HeaderInfo } from 'node-libcurl';
import { parseString as parseSetCookieString } from 'set-cookie-parser';
import { ScrappingOptions } from './puppeteer';
import { GlobalLogger } from './logger';
import { AssertionFailureError, FancyFile } from 'civkit';
import { ServiceBadAttemptError, ServiceBadApproachError } from './errors';
import { ServiceBadAttemptError } from './errors';
import { TempFileManager } from '../services/temp-file';
import { createBrotliDecompress, createInflate, createGunzip } from 'zlib';
import { ZSTDDecompress } from 'simple-zstd';
import _ from 'lodash';
import { Readable } from 'stream';
import { AsyncLocalContext } from './async-context';
import { BlackHoleDetector } from './blackhole-detector';
type HeaderInfo = Record<string, string | string[] | { code: number; reason?: string } | undefined>;
export interface CURLScrappingOptions extends ScrappingOptions {
method?: string;
body?: string | Buffer;
@@ -23,13 +19,12 @@ export interface CURLScrappingOptions extends ScrappingOptions {
@singleton()
export class CurlControl extends AsyncService {
logger = this.globalLogger.child({ service: this.constructor.name });
chromeVersion: string = `132`;
safariVersion: string = `537.36`;
platform: string = `Linux`;
ua: string = `Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/${this.safariVersion} (KHTML, like Gecko) Chrome/${this.chromeVersion}.0.0.0 Safari/${this.safariVersion}`;
chromeVersion = '132';
safariVersion = '537.36';
platform = 'Linux';
ua = `Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/${this.safariVersion} (KHTML, like Gecko) Chrome/${this.chromeVersion}.0.0.0 Safari/${this.safariVersion}`;
lifeCycleTrack = new WeakMap();
@@ -46,93 +41,37 @@ export class CurlControl extends AsyncService {
await this.dependencyReady();
if (process.platform === 'darwin') {
this.platform = `macOS`;
this.platform = 'macOS';
} else if (process.platform === 'win32') {
this.platform = `Windows`;
this.platform = 'Windows';
}
this.emit('ready');
}
impersonateChrome(ua: string) {
this.chromeVersion = ua.match(/Chrome\/(\d+)/)![1];
this.safariVersion = ua.match(/AppleWebKit\/([\d\.]+)/)![1];
this.chromeVersion = ua.match(/Chrome\/(\d+)/)?.[1] || this.chromeVersion;
this.safariVersion = ua.match(/AppleWebKit\/([\d.]+)/)?.[1] || this.safariVersion;
this.ua = ua;
}
curlImpersonateHeader(curl: Curl, headers?: object) {
let uaPlatform = this.platform;
if (this.ua.includes('Windows')) {
uaPlatform = 'Windows';
} else if (this.ua.includes('Android')) {
uaPlatform = 'Android';
} else if (this.ua.includes('iPhone') || this.ua.includes('iPad') || this.ua.includes('iPod')) {
uaPlatform = 'iOS';
} else if (this.ua.includes('CrOS')) {
uaPlatform = 'Chrome OS';
} else if (this.ua.includes('Macintosh')) {
uaPlatform = 'macOS';
protected buildHeaders(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
const headers = new Headers();
headers.set('User-Agent', crawlOpts?.overrideUserAgent || this.ua);
headers.set('Accept', 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8');
headers.set('Accept-Encoding', 'gzip, deflate, br');
headers.set('Accept-Language', crawlOpts?.locale || 'en-US,en;q=0.9');
for (const [key, value] of Object.entries(crawlOpts?.extraHeaders || {})) {
if (typeof value !== 'undefined') {
headers.set(key, `${value}`);
}
}
const mixinHeaders: Record<string, string> = {
'Sec-Ch-Ua': `Not A(Brand";v="8", "Chromium";v="${this.chromeVersion}", "Google Chrome";v="${this.chromeVersion}"`,
'Sec-Ch-Ua-Mobile': '?0',
'Sec-Ch-Ua-Platform': `"${uaPlatform}"`,
'Upgrade-Insecure-Requests': '1',
'User-Agent': this.ua,
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
'Sec-Fetch-Site': 'none',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-User': '?1',
'Sec-Fetch-Dest': 'document',
'Accept-Encoding': 'gzip, deflate, br, zstd',
'Accept-Language': 'en-US,en;q=0.9',
};
const headersCopy: Record<string, string | undefined> = { ...headers };
for (const k of Object.keys(mixinHeaders)) {
const lowerK = k.toLowerCase();
if (headersCopy[lowerK]) {
mixinHeaders[k] = headersCopy[lowerK];
delete headersCopy[lowerK];
}
}
Object.assign(mixinHeaders, headersCopy);
curl.setOpt(Curl.option.HTTPHEADER, Object.entries(mixinHeaders).flatMap(([k, v]) => {
if (Array.isArray(v) && v.length) {
return v.map((v2) => `${k}: ${v2}`);
}
return [`${k}: ${v}`];
}));
return curl;
if (crawlOpts?.referer) {
headers.set('Referer', crawlOpts.referer);
}
urlToFile1Shot(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
return new Promise<{
statusCode: number,
statusText?: string,
data?: FancyFile,
headers: HeaderInfo[],
}>((resolve, reject) => {
let contentType = '';
const curl = new Curl();
curl.enable(CurlFeature.StreamResponse);
curl.setOpt('URL', urlToCrawl.toString());
curl.setOpt(Curl.option.FOLLOWLOCATION, false);
curl.setOpt(Curl.option.SSL_VERIFYPEER, false);
curl.setOpt(Curl.option.TIMEOUT_MS, crawlOpts?.timeoutMs || 30_000);
curl.setOpt(Curl.option.CONNECTTIMEOUT_MS, 3_000);
curl.setOpt(Curl.option.LOW_SPEED_LIMIT, 32768);
curl.setOpt(Curl.option.LOW_SPEED_TIME, 5_000);
if (crawlOpts?.method) {
curl.setOpt(Curl.option.CUSTOMREQUEST, crawlOpts.method.toUpperCase());
}
if (crawlOpts?.body) {
curl.setOpt(Curl.option.POSTFIELDS, crawlOpts.body.toString());
}
const headersToSet = { ...crawlOpts?.extraHeaders };
if (crawlOpts?.cookies?.length) {
const cookieKv: Record<string, string> = {};
for (const cookie of crawlOpts.cookies) {
@@ -160,161 +99,70 @@ export class CurlControl extends AsyncService {
continue;
}
}
const cookieChunks = Object.entries(cookieKv).map(([k, v]) => `${k}=${encodeURIComponent(v)}`);
headersToSet.cookie ??= cookieChunks.join('; ');
}
if (crawlOpts?.referer) {
headersToSet.referer ??= crawlOpts.referer;
}
if (crawlOpts?.overrideUserAgent) {
headersToSet['user-agent'] ??= crawlOpts.overrideUserAgent;
}
this.curlImpersonateHeader(curl, headersToSet);
if (crawlOpts?.proxyUrl) {
const proxyUrlCopy = new URL(crawlOpts.proxyUrl);
curl.setOpt(Curl.option.PROXY, proxyUrlCopy.href);
}
let curlStream: Readable | undefined;
curl.on('error', (err, errCode) => {
curl.close();
this.logger.warn(`Curl ${urlToCrawl.origin}: ${err}`, { err, urlToCrawl });
const err2 = this.digestCurlCode(errCode, err.message) ||
new AssertionFailureError(`Failed to access ${urlToCrawl.origin}: ${err.message}`);
err2.cause ??= err;
if (curlStream) {
// For some reason, manually emitting error event is required for curlStream.
curlStream.emit('error', err2);
curlStream.destroy(err2);
}
reject(err2);
});
curl.setOpt(Curl.option.MAXFILESIZE, 4 * 1024 * 1024 * 1024); // 4GB
let status = -1;
let statusText: string|undefined;
let contentEncoding = '';
curl.once('end', () => {
if (curlStream) {
curlStream.once('end', () => curl.close());
return;
}
curl.close();
});
curl.on('stream', (stream, statusCode, headers) => {
this.logger.debug(`CURL: [${statusCode}] ${urlToCrawl.origin}`, { statusCode });
status = statusCode;
curlStream = stream;
for (const headerSet of (headers as HeaderInfo[])) {
for (const [k, v] of Object.entries(headerSet)) {
if (k.trim().endsWith(':')) {
Reflect.set(headerSet, k.slice(0, k.indexOf(':')), v || '');
Reflect.deleteProperty(headerSet, k);
continue;
}
if (v === undefined) {
Reflect.set(headerSet, k, '');
continue;
}
if (k.toLowerCase() === 'content-type' && typeof v === 'string') {
contentType = v.toLowerCase();
}
}
}
const lastResHeaders = headers[headers.length - 1];
statusText = (lastResHeaders as HeaderInfo).result?.reason;
for (const [k, v] of Object.entries(lastResHeaders)) {
const kl = k.toLowerCase();
if (kl === 'content-type') {
contentType = (v || '').toLowerCase();
}
if (kl === 'content-encoding') {
contentEncoding = (v || '').toLowerCase();
}
if (contentType && contentEncoding) {
break;
const cookieValue = Object.entries(cookieKv).map(([key, value]) => `${key}=${encodeURIComponent(value)}`).join('; ');
if (cookieValue) {
headers.set('Cookie', cookieValue);
}
}
if ([301, 302, 303, 307, 308].includes(statusCode)) {
if (stream) {
stream.resume();
}
resolve({
statusCode: status,
statusText,
data: undefined,
headers: headers as HeaderInfo[],
});
return;
return headers;
}
if (!stream) {
resolve({
statusCode: status,
statusText,
data: undefined,
headers: headers as HeaderInfo[],
});
return;
protected async responseToFancyFile(response: Response, fileName?: string) {
const ab = await response.arrayBuffer();
const buffer = Buffer.from(ab);
if (!buffer.length) {
return undefined;
}
switch (contentEncoding) {
case 'gzip': {
const decompressed = createGunzip();
stream.pipe(decompressed);
stream.once('error', (err) => {
decompressed.destroy(err);
});
stream = decompressed;
break;
}
case 'deflate': {
const decompressed = createInflate();
stream.pipe(decompressed);
stream.once('error', (err) => {
decompressed.destroy(err);
});
stream = decompressed;
break;
}
case 'br': {
const decompressed = createBrotliDecompress();
stream.pipe(decompressed);
stream.once('error', (err) => {
decompressed.destroy(err);
});
stream = decompressed;
break;
}
case 'zstd': {
const decompressed = ZSTDDecompress();
stream.pipe(decompressed);
stream.once('error', (err) => {
decompressed.destroy(err);
});
stream = decompressed;
break;
}
default: {
break;
}
const tmpPath = this.tempFileManager.alloc();
return FancyFile.auto({
fileBuffer: buffer,
mimeType: response.headers.get('content-type') || 'application/octet-stream',
fileName,
size: buffer.length,
}, tmpPath);
}
const fpath = this.tempFileManager.alloc();
const fancyFile = FancyFile.auto(stream, fpath);
this.tempFileManager.bindPathTo(fancyFile, fpath);
resolve({
statusCode: status,
statusText,
data: fancyFile,
headers: headers as HeaderInfo[],
});
protected headersToHeaderInfo(response: Response): HeaderInfo {
const headerInfo: HeaderInfo = {
result: {
code: response.status,
reason: response.statusText,
},
};
response.headers.forEach((value, key) => {
headerInfo[key] = value;
headerInfo[key.replace(/(^|-)([a-z])/g, (_m, p1, p2) => `${p1}${p2.toUpperCase()}`)] = value;
});
curl.perform();
return headerInfo;
}
async urlToFile1Shot(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
const response = await fetch(urlToCrawl, {
method: crawlOpts?.method || (crawlOpts?.body ? 'POST' : 'GET'),
headers: this.buildHeaders(urlToCrawl, crawlOpts),
body: crawlOpts?.body as any,
redirect: 'manual',
signal: AbortSignal.timeout(crawlOpts?.timeoutMs || 30_000),
}).catch((err) => {
throw new AssertionFailureError(`Failed to access ${urlToCrawl.origin}: ${err}`);
});
const headerInfo = this.headersToHeaderInfo(response);
const contentDisposition = response.headers.get('content-disposition') || '';
const fileName = contentDisposition.match(/filename="([^"]+)"/i)?.[1] || urlToCrawl.pathname.split('/').pop() || 'download.bin';
const data = response.status >= 300 && response.status < 400 ? undefined : await this.responseToFancyFile(response, fileName);
return {
statusCode: response.status,
statusText: response.statusText,
data,
headers: [headerInfo],
};
}
async urlToFile(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
@@ -323,18 +171,18 @@ export class CurlControl extends AsyncService {
let opts = { ...crawlOpts };
let nextHopUrl = urlToCrawl;
const fakeHeaderInfos: HeaderInfo[] = [];
do {
const r = await this.urlToFile1Shot(nextHopUrl, opts);
if ([301, 302, 303, 307, 308].includes(r.statusCode)) {
fakeHeaderInfos.push(...r.headers);
const headers = r.headers[r.headers.length - 1];
const location: string | undefined = headers.Location || headers.location;
const location = (headers.Location || headers.location) as string | undefined;
const setCookieHeader = headers['Set-Cookie'] || headers['set-cookie'];
if (setCookieHeader) {
const cookieAssignments = Array.isArray(setCookieHeader) ? setCookieHeader : [setCookieHeader];
const parsed = cookieAssignments.filter(Boolean).map((x) => parseSetCookieString(x, { decodeValues: true }));
const parsed = cookieAssignments.filter(Boolean).map((x) => parseSetCookieString(`${x}`, { decodeValues: true }));
if (parsed.length) {
opts.cookies = [...(opts.cookies || []), ...parsed];
}
@@ -344,15 +192,16 @@ export class CurlControl extends AsyncService {
}
if (!location && !setCookieHeader) {
// Follow curl behavior
return {
statusCode: r.statusCode,
statusText: r.statusText,
data: r.data,
headers: fakeHeaderInfos.concat(r.headers),
};
}
if (!location && cookieRedirects > 1) {
throw new ServiceBadApproachError(`Failed to access ${urlToCrawl}: Browser required to solve complex cookie preconditions.`);
throw new ServiceBadAttemptError(`Failed to access ${urlToCrawl}: Browser required to solve complex cookie preconditions.`);
}
nextHopUrl = new URL(location || '', nextHopUrl);
@@ -374,37 +223,42 @@ export class CurlControl extends AsyncService {
async sideLoad(targetUrl: URL, crawlOpts?: CURLScrappingOptions) {
const curlResult = await this.urlToFile(targetUrl, crawlOpts);
this.blackHoleDetector.itWorked();
let finalURL = targetUrl;
const sideLoadOpts: CURLScrappingOptions['sideLoad'] = {
impersonate: {},
proxyOrigin: {},
};
for (const headers of curlResult.headers) {
sideLoadOpts.impersonate[finalURL.href] = {
status: headers.result?.code || -1,
headers: _.omit(headers, 'result'),
contentType: headers['Content-Type'] || headers['content-type'],
status: (headers.result as any)?.code || -1,
headers: Object.fromEntries(
Object.entries(headers).filter(([key]) => key !== 'result')
) as Record<string, string | string[]>,
contentType: `${headers['Content-Type'] || headers['content-type'] || ''}` || undefined,
};
if (crawlOpts?.proxyUrl) {
sideLoadOpts.proxyOrigin[finalURL.origin] = crawlOpts.proxyUrl;
}
if (headers.result?.code && [301, 302, 307, 308].includes(headers.result.code)) {
const location = headers.Location || headers.location;
const code = (headers.result as any)?.code;
if (code && [301, 302, 303, 307, 308].includes(code)) {
const location = (headers.Location || headers.location) as string | undefined;
if (location) {
finalURL = new URL(location, finalURL);
}
}
}
const lastHeaders = curlResult.headers[curlResult.headers.length - 1];
const contentType = (lastHeaders['Content-Type'] || lastHeaders['content-type'])?.toLowerCase() || (await curlResult.data?.mimeType) || 'application/octet-stream';
const contentDisposition = lastHeaders['Content-Disposition'] || lastHeaders['content-disposition'];
const contentType = `${lastHeaders['Content-Type'] || lastHeaders['content-type'] || ''}`.toLowerCase() || (await curlResult.data?.mimeType) || 'application/octet-stream';
const contentDisposition = `${lastHeaders['Content-Disposition'] || lastHeaders['content-disposition'] || ''}` || undefined;
const fileName = contentDisposition?.match(/filename="([^"]+)"/i)?.[1] || finalURL.pathname.split('/').pop();
if (sideLoadOpts.impersonate[finalURL.href] && (await curlResult.data?.size)) {
sideLoadOpts.impersonate[finalURL.href].body = curlResult.data;
}
// This should keep the file from being garbage collected and deleted until this asyncContext/request is done.
this.lifeCycleTrack.set(this.asyncLocalContext.ctx, curlResult.data);
return {
@@ -417,40 +271,7 @@ export class CurlControl extends AsyncService {
contentType,
contentDisposition,
fileName,
file: curlResult.data
file: curlResult.data,
};
}
digestCurlCode(code: CurlCode, msg: string) {
switch (code) {
// 400 User errors
case CurlCode.CURLE_COULDNT_RESOLVE_HOST: {
return new AssertionFailureError(msg);
}
// Maybe retry but dont retry with curl again
case CurlCode.CURLE_OPERATION_TIMEDOUT:
case CurlCode.CURLE_UNSUPPORTED_PROTOCOL:
case CurlCode.CURLE_PEER_FAILED_VERIFICATION: {
return new ServiceBadApproachError(msg);
}
// Retryable errors
case CurlCode.CURLE_REMOTE_ACCESS_DENIED:
case CurlCode.CURLE_SEND_ERROR:
case CurlCode.CURLE_RECV_ERROR:
case CurlCode.CURLE_GOT_NOTHING:
case CurlCode.CURLE_SSL_CONNECT_ERROR:
case CurlCode.CURLE_QUIC_CONNECT_ERROR:
case CurlCode.CURLE_COULDNT_RESOLVE_PROXY:
case CurlCode.CURLE_COULDNT_CONNECT:
case CurlCode.CURLE_PARTIAL_FILE: {
return new ServiceBadAttemptError(msg);
}
default: {
return undefined;
}
}
}
}
+16 -1
View File
@@ -60,6 +60,7 @@ export class GeoIPService extends AsyncService {
logger = this.globalLogger.child({ service: this.constructor.name });
mmdbCity!: Reader<CityResponse>;
geoIpUnavailable = false;
constructor(
protected globalLogger: GlobalLogger,
@@ -78,7 +79,18 @@ export class GeoIPService extends AsyncService {
async _lazyload() {
const mmdpPath = path.resolve(__dirname, '..', '..', 'licensed', 'GeoLite2-City.mmdb');
const dbBuff = await fsp.readFile(mmdpPath, { flag: 'r', encoding: null });
let dbBuff: Buffer;
try {
dbBuff = await fsp.readFile(mmdpPath, { flag: 'r', encoding: null });
} catch (error) {
const maybeErr = error;
if (typeof maybeErr === 'object' && maybeErr && 'code' in maybeErr && maybeErr.code === 'ENOENT') {
this.geoIpUnavailable = true;
this.logger.warn(`GeoIP database is missing at ${mmdpPath}; GeoIP lookups are disabled.`);
return;
}
throw error;
}
this.mmdbCity = new Reader<CityResponse>(dbBuff);
@@ -89,6 +101,9 @@ export class GeoIPService extends AsyncService {
@Threaded()
async lookupCity(ip: string, lang: GEOIP_SUPPORTED_LANGUAGES = GEOIP_SUPPORTED_LANGUAGES.EN) {
await this._lazyload();
if (this.geoIpUnavailable || !this.mmdbCity) {
return undefined;
}
const r = this.mmdbCity.get(ip);
+1 -1
View File
@@ -341,7 +341,7 @@ export class JSDomControl extends AsyncService {
return final;
}
snippetToElement(snippet?: string, url?: string) {
snippetToElement(snippet?: string, _url?: string) {
const parsed = this.linkedom.parseHTML(snippet || '<html><body></body></html>');
// Hack for turndown gfm table plugin.
+5 -5
View File
@@ -48,7 +48,7 @@ export class LmControl extends AsyncService {
],
options: {
system: 'You are ReaderLM-v7, a model that generates Markdown source files only. No HTML, notes and chit-chats allowed',
system: 'You are XreadLM-v7, a model that generates Markdown source files only. No HTML, notes and chit-chats allowed',
stream: true
}
});
@@ -69,14 +69,14 @@ export class LmControl extends AsyncService {
return;
}
async* readerLMMarkdownFromSnapshot(snapshot?: PageSnapshot) {
async* xreadLMMarkdownFromSnapshot(snapshot?: PageSnapshot) {
if (!snapshot) {
throw new AssertionFailureError('Snapshot of the page is not available');
}
const html = await this.jsdomControl.cleanHTMLforLMs(snapshot.html, 'script,link,style,textarea,select>option,svg');
const it = this.commonLLM.iterRun('readerlm-v2', {
const it = this.commonLLM.iterRun('xreadlm-v2', {
prompt: `Extract the main content from the given HTML and convert it to Markdown format.\n\n${tripleBackTick}html\n${html}\n${tripleBackTick}\n`,
options: {
@@ -110,14 +110,14 @@ export class LmControl extends AsyncService {
return;
}
async* readerLMFromSnapshot(schema?: string, instruction: string = 'Infer useful information from the HTML and present it in a structured JSON object.', snapshot?: PageSnapshot) {
async* xreadLMFromSnapshot(schema?: string, instruction: string = 'Infer useful information from the HTML and present it in a structured JSON object.', snapshot?: PageSnapshot) {
if (!snapshot) {
throw new AssertionFailureError('Snapshot of the page is not available');
}
const html = await this.jsdomControl.cleanHTMLforLMs(snapshot.html, 'script,link,style,textarea,select>option,svg');
const it = this.commonLLM.iterRun('readerlm-v2', {
const it = this.commonLLM.iterRun('xreadlm-v2', {
prompt: `${instruction}\n\n${tripleBackTick}html\n${html}\n${tripleBackTick}\n${schema ? `The JSON schema:\n${tripleBackTick}json\n${schema}\n${tripleBackTick}\n` : ''}`,
options: {
// system: 'You are an AI assistant developed by VENDOR_NAME',
+2 -2
View File
@@ -497,7 +497,7 @@ export function minimalStealth() {
}
return (Object.fromEntries || fromEntries)(
Object.entries(fnObj)
.filter(([key, value]) => typeof value === 'function')
.filter(([, value]) => typeof value === 'function')
.map(([key, value]) => [key, value.toString()]) // eslint-disable-line no-eval
);
};
@@ -526,7 +526,7 @@ export function minimalStealth() {
utils.makeHandler = () => ({
// Used by simple `navigator` getter evasions
getterValue: value => ({
apply(target, ctx, args) {
apply(_target, _ctx, _args) {
// Let's fetch the value first, to trigger and escalate potential errors
// Illegal invocations like `navigator.__proto__.vendor` will throw here
utils.cache.Reflect.apply(...arguments);
+1 -1
View File
@@ -968,7 +968,7 @@ export class PuppeteerControl extends AsyncService {
const firstReq = curled.chain[0];
return req.respond({
status: firstReq.result!.code,
status: (firstReq.result as any)!.code,
headers: _.omit(firstReq, 'result'),
}, 3);
} catch (err: any) {
+1 -1
View File
@@ -15,7 +15,7 @@ export const InjectProperty = propertyInjectorFactory(container);
@singleton()
export class RPCRegistry extends KoaRPCRegistry {
title = 'Jina Reader API';
title = 'Xread API';
container = container;
logger = this.globalLogger.child({ service: this.constructor.name });
static override envelope = IntegrityEnvelope;
+1 -1
View File
@@ -183,7 +183,7 @@ async function getWebSearchResults() {
const candidates = Array.from(wrapper1.querySelectorAll('div[lang],div[data-surl]'));
return candidates.map((x, pos) => {
return candidates.map((x) => {
const primaryLink = x.querySelector('a:not([href="#"])');
if (!primaryLink) {
return undefined;
+43
View File
@@ -0,0 +1,43 @@
import { singleton } from 'tsyringe';
import { GlobalLogger } from '../logger';
import { AsyncService } from 'civkit/async-service';
import { SerperSearchQueryParams } from '../../shared/3rd-party/serper-search';
import { ServiceDisabledError } from '../errors';
import { WebSearchEntry } from './compat';
@singleton()
export class InternalSearchService extends AsyncService {
logger = this.globalLogger.child({ service: this.constructor.name });
constructor(
protected globalLogger: GlobalLogger,
) {
super(...arguments);
}
override async init() {
await this.dependencyReady();
this.emit('ready');
}
protected disabled(): never {
throw new ServiceDisabledError('Internal search provider is not available in this standalone build. Use a public provider such as google, bing, or serper.');
}
async doSearch(_variant: 'web' | 'images' | 'news', _query: SerperSearchQueryParams) {
this.disabled();
}
async webSearch(_query: SerperSearchQueryParams) {
return this.disabled() as WebSearchEntry[];
}
async imageSearch(_query: SerperSearchQueryParams) {
return this.disabled() as WebSearchEntry[];
}
async newsSearch(_query: SerperSearchQueryParams) {
return this.disabled() as WebSearchEntry[];
}
}
-77
View File
@@ -1,77 +0,0 @@
import { singleton } from 'tsyringe';
import { GlobalLogger } from '../logger';
import { SecretExposer } from '../../shared/services/secrets';
import { AsyncLocalContext } from '../async-context';
import { SerperSearchQueryParams } from '../../shared/3rd-party/serper-search';
import { BlackHoleDetector } from '../blackhole-detector';
import { AsyncService } from 'civkit/async-service';
import { JinaSerpApiHTTP } from '../../shared/3rd-party/internal-serp';
import { WebSearchEntry } from './compat';
@singleton()
export class InternalJinaSerpService extends AsyncService {
logger = this.globalLogger.child({ service: this.constructor.name });
client!: JinaSerpApiHTTP;
constructor(
protected globalLogger: GlobalLogger,
protected secretExposer: SecretExposer,
protected threadLocal: AsyncLocalContext,
protected blackHoleDetector: BlackHoleDetector,
) {
super(...arguments);
}
override async init() {
await this.dependencyReady();
this.emit('ready');
this.client = new JinaSerpApiHTTP(this.secretExposer.JINA_SERP_API_KEY);
}
async doSearch(variant: 'web' | 'images' | 'news', query: SerperSearchQueryParams) {
this.logger.debug(`Doing external search`, query);
let results;
switch (variant) {
// case 'images': {
// const r = await this.client.imageSearch(query);
// results = r.parsed.images;
// break;
// }
// case 'news': {
// const r = await this.client.newsSearch(query);
// results = r.parsed.news;
// break;
// }
case 'web':
default: {
const r = await this.client.webSearch(query);
results = r.parsed.results?.map((x) => ({ ...x, link: x.url }));
break;
}
}
this.blackHoleDetector.itWorked();
return results as WebSearchEntry[];
}
async webSearch(query: SerperSearchQueryParams) {
return this.doSearch('web', query);
}
async imageSearch(query: SerperSearchQueryParams) {
return this.doSearch('images', query);
}
async newsSearch(query: SerperSearchQueryParams) {
return this.doSearch('news', query);
}
}
+1 -1
View File
@@ -508,7 +508,7 @@ export class SERPSpecializedPuppeteerControl extends AsyncService {
const firstReq = curled.chain[0];
return req.respond({
status: firstReq.result!.code,
status: (firstReq.result as any)!.code,
headers: _.omit(firstReq, 'result'),
}, 3);
} catch (err: any) {
+1 -1
View File
@@ -830,7 +830,7 @@ ${suffixMixins.length ? `\n${suffixMixins.join('\n\n')}\n` : ''}`;
let innerCharset;
const peek = snapshot.html.slice(0, 1024);
innerCharset ??= peek.match(/<meta[^>]+text\/html;\s*?charset=([^>"]+)/i)?.[1]?.toLowerCase();
innerCharset ??= peek.match(/<meta[^>]+charset="([^>"]+)\"/i)?.[1]?.toLowerCase();
innerCharset ??= peek.match(/<meta[^>]+charset="([^>"]+)"/i)?.[1]?.toLowerCase();
if (innerCharset && innerCharset !== encoding) {
snapshot.html = await readFile(await file.filePath, innerCharset);
}
-1
View File
@@ -1 +0,0 @@
../thinapps-shared/backend
+48
View File
@@ -0,0 +1,48 @@
export type WebSearchQueryParams = {
q: string;
count?: number;
offset?: number;
country?: string;
search_lang?: string;
[k: string]: any;
};
export class BraveSearchHTTP {
constructor(
protected apiKey: string,
) { }
async webSearch(query: WebSearchQueryParams, options?: { headers?: Record<string, string> }) {
if (!this.apiKey) {
throw new Error('BRAVE_SEARCH_API_KEY is required for Brave search.');
}
const url = new URL('https://api.search.brave.com/res/v1/web/search');
for (const [key, value] of Object.entries(query)) {
if (value === undefined || value === null) {
continue;
}
url.searchParams.set(key, `${value}`);
}
const response = await fetch(url, {
headers: {
Accept: 'application/json',
'X-Subscription-Token': this.apiKey,
...(options?.headers || {}),
},
});
if (!response.ok) {
const text = await response.text();
const err: any = new Error(`Brave search failed: ${response.status} ${response.statusText} ${text}`.trim());
err.status = response.status;
throw err;
}
return {
parsed: await response.json(),
};
}
}
+2
View File
@@ -0,0 +1,2 @@
export type WebSearchOptionalHeaderOptions = Record<string, string>;
+32
View File
@@ -0,0 +1,32 @@
export class CloudFlareHTTP {
constructor(
protected accountId?: string,
protected apiKey?: string,
) { }
async fetchBrowserRenderedHTML(input: { url: string }) {
if (!(this.accountId && this.apiKey)) {
throw new Error('CLOUD_FLARE_API_KEY must be configured as ACCOUNT_ID:API_KEY for browser rendering.');
}
const response = await fetch(`https://api.cloudflare.com/client/v4/accounts/${this.accountId}/browser-rendering/fetch`, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
Authorization: `Bearer ${this.apiKey}`,
},
body: JSON.stringify(input),
});
if (!response.ok) {
const text = await response.text();
const err: any = new Error(`Cloudflare browser rendering failed: ${response.status} ${response.statusText} ${text}`.trim());
err.status = response.status;
throw err;
}
return {
parsed: await response.json() as any,
};
}
}
+113
View File
@@ -0,0 +1,113 @@
type SearchVariant = 'search' | 'images' | 'news';
export type SerperSearchQueryParams = {
q: string;
page?: number;
num?: number;
start?: number;
gl?: string;
hl?: string;
location?: string;
[k: string]: any;
};
export type SerperWebSearchResponse = {
organic: Array<Record<string, any>>;
};
export type SerperImageSearchResponse = {
images: Array<Record<string, any>>;
};
export type SerperNewsSearchResponse = {
news: Array<Record<string, any>>;
};
export const WORLD_COUNTRIES: Record<string, string> = {
US: 'United States',
CN: 'China',
JP: 'Japan',
DE: 'Germany',
FR: 'France',
GB: 'United Kingdom',
IN: 'India',
CA: 'Canada',
AU: 'Australia',
SG: 'Singapore',
};
export const WORLD_LANGUAGES = [
{ code: 'en', name: 'English' },
{ code: 'zh', name: 'Chinese' },
{ code: 'ja', name: 'Japanese' },
{ code: 'de', name: 'German' },
{ code: 'fr', name: 'French' },
{ code: 'es', name: 'Spanish' },
];
class SerperBaseHTTP {
constructor(
protected apiKey: string,
protected provider: 'google' | 'bing',
) { }
protected getBaseUrl(variant: SearchVariant) {
const host = this.provider === 'bing' ? 'https://bing.serper.dev' : 'https://google.serper.dev';
return `${host}/${variant}`;
}
protected async request<T>(variant: SearchVariant, query: SerperSearchQueryParams) {
if (!this.apiKey) {
throw new Error(`SERPER_SEARCH_API_KEY is required for ${this.provider} ${variant} search.`);
}
const response = await fetch(this.getBaseUrl(variant), {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'X-API-KEY': this.apiKey,
},
body: JSON.stringify(query),
});
if (!response.ok) {
const text = await response.text();
const err: any = new Error(`Serper ${this.provider} ${variant} search failed: ${response.status} ${response.statusText} ${text}`.trim());
err.status = response.status;
throw err;
}
return response.json() as Promise<T>;
}
async webSearch(query: SerperSearchQueryParams) {
const parsed = await this.request<SerperWebSearchResponse>('search', query);
return { parsed };
}
async imageSearch(query: SerperSearchQueryParams) {
const parsed = await this.request<SerperImageSearchResponse>('images', query);
return { parsed };
}
async newsSearch(query: SerperSearchQueryParams) {
const parsed = await this.request<SerperNewsSearchResponse>('news', query);
return { parsed };
}
}
export class SerperGoogleHTTP extends SerperBaseHTTP {
constructor(apiKey: string) {
super(apiKey, 'google');
}
}
export class SerperBingHTTP extends SerperBaseHTTP {
constructor(apiKey: string) {
super(apiKey, 'bing');
}
}
+36
View File
@@ -0,0 +1,36 @@
import { Also, Prop } from 'civkit';
import { FirestoreRecord } from '../lib/firestore';
export enum API_CALL_STATUS {
SUCCESS = 'SUCCESS',
RATE_LIMITED = 'RATE_LIMITED',
FAILED = 'FAILED',
}
@Also({
dictOf: Object,
})
export class ApiRollRecord extends FirestoreRecord {
static override collectionName = 'api-rolls';
override _id!: string;
@Prop()
uid?: string;
@Prop()
ip?: string;
@Prop({ arrayOf: String })
tags?: string[];
@Prop()
status?: API_CALL_STATUS;
@Prop()
chargeAmount?: number;
@Prop()
createdAt!: Date;
}
+2
View File
@@ -0,0 +1,2 @@
export { ServiceBadAttemptError } from '../services/errors';
+8
View File
@@ -0,0 +1,8 @@
export {
SecurityCompromiseError,
ServiceBadApproachError,
ServiceBadAttemptError,
ServiceCrashedError,
ServiceNodeResourceDrainError,
} from '../../services/errors';
+316
View File
@@ -0,0 +1,316 @@
import { AutoCastable } from 'civkit/civ-rpc';
import { randomUUID } from 'crypto';
import fs from 'fs';
import fsp from 'fs/promises';
import path from 'path';
type WhereOp = '==' | '>=' | '<=' | '>' | '<';
type OrderDirection = 'asc' | 'desc';
type IncrementOp = {
__op: 'increment';
value: number;
};
type QueryFilter = {
field: string;
op: WhereOp;
value: any;
};
type QueryOrder = {
field: string;
direction: OrderDirection;
};
function reviveJson(_key: string, value: any) {
if (value && typeof value === 'object' && value.__type === 'Date') {
return new Date(value.value);
}
return value;
}
function serializeJson(_key: string, value: any) {
if (value instanceof Date) {
return {
__type: 'Date',
value: value.toISOString(),
};
}
return value;
}
function getByPath(input: any, pathText: string) {
return pathText.split('.').reduce((acc, key) => acc?.[key], input);
}
function setByPath(input: any, pathText: string, value: any) {
const parts = pathText.split('.');
const last = parts.pop()!;
let cursor = input;
for (const part of parts) {
cursor[part] ??= {};
cursor = cursor[part];
}
cursor[last] = value;
}
function applyPatch(base: any, patch: any) {
const draft = structuredClone(base || {});
for (const [key, value] of Object.entries(patch || {})) {
const currentValue = getByPath(draft, key);
if (value && typeof value === 'object' && (value as IncrementOp).__op === 'increment') {
setByPath(draft, key, (Number(currentValue) || 0) + (value as IncrementOp).value);
continue;
}
setByPath(draft, key, value);
}
return draft;
}
class LocalStore {
rootDir = path.resolve('.cache/xread/db');
constructor() {
fs.mkdirSync(this.rootDir, { recursive: true });
}
filePath(collectionName: string) {
return path.join(this.rootDir, `${collectionName}.json`);
}
async readCollection(collectionName: string) {
const filePath = this.filePath(collectionName);
try {
const content = await fsp.readFile(filePath, 'utf8');
return JSON.parse(content, reviveJson) as Record<string, any>;
} catch (err: any) {
if (err?.code === 'ENOENT') {
return {};
}
throw err;
}
}
async writeCollection(collectionName: string, payload: Record<string, any>) {
const filePath = this.filePath(collectionName);
await fsp.mkdir(path.dirname(filePath), { recursive: true });
await fsp.writeFile(filePath, JSON.stringify(payload, serializeJson, 2), 'utf8');
}
}
const LOCAL_STORE = new LocalStore();
export class LocalDocRef<T extends typeof FirestoreRecord = typeof FirestoreRecord> {
constructor(
protected recordType: T,
readonly id: string,
) { }
async get() {
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
const raw = bucket[this.id];
if (!raw) {
return undefined;
}
return this.recordType.from(raw);
}
async set(payload: any, options?: { merge?: boolean }) {
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
const previous = bucket[this.id];
const merged = options?.merge ? applyPatch(previous, payload) : { ...payload };
merged._id ??= this.id;
bucket[this.id] = merged;
await LOCAL_STORE.writeCollection(this.recordType.collectionName, bucket);
return this.recordType.from(merged);
}
async update(payload: any) {
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
const previous = bucket[this.id] || { _id: this.id };
const merged = applyPatch(previous, payload);
merged._id ??= this.id;
bucket[this.id] = merged;
await LOCAL_STORE.writeCollection(this.recordType.collectionName, bucket);
return this.recordType.from(merged);
}
}
export class LocalQuery<T extends typeof FirestoreRecord = typeof FirestoreRecord> {
constructor(
protected recordType: T,
protected filters: QueryFilter[] = [],
protected orders: QueryOrder[] = [],
protected limitCount?: number,
) { }
where(field: string, op: WhereOp, value: any) {
return new LocalQuery(this.recordType, [...this.filters, { field, op, value }], this.orders, this.limitCount);
}
orderBy(field: string, direction: OrderDirection = 'asc') {
return new LocalQuery(this.recordType, this.filters, [...this.orders, { field, direction }], this.limitCount);
}
limit(count: number) {
return new LocalQuery(this.recordType, this.filters, this.orders, count);
}
async exec() {
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
let values = Object.values(bucket);
for (const filter of this.filters) {
values = values.filter((entry) => {
const left = getByPath(entry, filter.field);
const right = filter.value;
switch (filter.op) {
case '==':
return left === right;
case '>=':
return left >= right;
case '<=':
return left <= right;
case '>':
return left > right;
case '<':
return left < right;
default:
return false;
}
});
}
for (const order of this.orders.slice().reverse()) {
values.sort((a, b) => {
const left = getByPath(a, order.field);
const right = getByPath(b, order.field);
const factor = order.direction === 'desc' ? -1 : 1;
if (left === right) {
return 0;
}
return left > right ? factor : -factor;
});
}
if (typeof this.limitCount === 'number') {
values = values.slice(0, this.limitCount);
}
return values.map((entry) => this.recordType.from(entry));
}
}
export class LocalCollectionRef<T extends typeof FirestoreRecord = typeof FirestoreRecord> extends LocalQuery<T> {
constructor(recordType: T) {
super(recordType);
}
doc(id?: string) {
return new LocalDocRef(this.recordType, id || randomUUID());
}
}
class LocalBatch {
protected tasks: Array<() => Promise<unknown>> = [];
set(ref: LocalDocRef, payload: any, options?: { merge?: boolean }) {
this.tasks.push(() => ref.set(payload, options));
}
update(ref: LocalDocRef, payload: any) {
this.tasks.push(() => ref.update(payload));
}
async commit() {
await Promise.all(this.tasks.map((task) => task()));
}
}
class LocalTransaction {
async get(ref: LocalDocRef) {
return ref.get();
}
set(ref: LocalDocRef, payload: any, options?: { merge?: boolean }) {
return ref.set(payload, options);
}
update(ref: LocalDocRef, payload: any) {
return ref.update(payload);
}
}
class LocalDB {
batch() {
return new LocalBatch();
}
async runTransaction<T>(handler: (transaction: LocalTransaction) => Promise<T>) {
return handler(new LocalTransaction());
}
}
const LOCAL_DB = new LocalDB();
export class FirestoreRecord extends AutoCastable {
static collectionName = 'records';
_id!: string;
static get COLLECTION() {
return new LocalCollectionRef(this);
}
static get DB() {
return LOCAL_DB;
}
static OPS = {
increment(value: number): IncrementOp {
return {
__op: 'increment',
value,
};
},
};
static async fromFirestore<T extends typeof FirestoreRecord>(this: T, id: string) {
if (!id) {
return undefined;
}
return this.COLLECTION.doc(id).get() as Promise<InstanceType<T> | undefined>;
}
static async fromFirestoreQuery<T extends typeof FirestoreRecord>(this: T, query: LocalQuery<any>) {
return query.exec() as Promise<Array<InstanceType<T>>>;
}
static async save<T extends typeof FirestoreRecord>(this: T, input: any, id?: string, options?: { merge?: boolean }) {
const payload = input instanceof this ? input.degradeForFireStore() : { ...input };
const resolvedId = id || payload._id || randomUUID();
payload._id = resolvedId;
return this.COLLECTION.doc(resolvedId).set(payload, options) as Promise<InstanceType<T>>;
}
async save(options?: { merge?: boolean }) {
return (this.constructor as typeof FirestoreRecord).save(this, this._id, options);
}
degradeForFireStore() {
return { ...this };
}
}
+2
View File
@@ -0,0 +1,2 @@
export { AsyncLocalContext as AsyncContext } from '../../services/async-context';
@@ -0,0 +1,15 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
import { ServiceDisabledError } from '../../services/errors';
@singleton()
export class ImageInterrogationManager extends AsyncService {
override async init() {
await this.dependencyReady();
this.emit('ready');
}
async interrogate(model: string, _input?: unknown): Promise<string> {
throw new ServiceDisabledError(`Image interrogation feature '${model}' is not enabled in this standalone build.`);
}
}
+15
View File
@@ -0,0 +1,15 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
import { ServiceDisabledError } from '../../services/errors';
@singleton()
export class LLMManager extends AsyncService {
override async init() {
await this.dependencyReady();
this.emit('ready');
}
async *iterRun(model: string, _input?: unknown): AsyncGenerator<string> {
throw new ServiceDisabledError(`LLM feature '${model}' is not enabled in this standalone build.`);
}
}
@@ -0,0 +1,53 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
import path from 'path';
import fs from 'fs/promises';
import { pathToFileURL } from 'url';
import { SecretExposer } from './secrets';
type SaveOptions = {
contentType?: string;
metadata?: {
contentType?: string;
};
};
@singleton()
export class FirebaseStorageBucketControl extends AsyncService {
rootDir!: string;
constructor(
protected secretExposer: SecretExposer,
) {
super(...arguments);
}
override async init() {
await this.dependencyReady();
this.rootDir = path.resolve(this.secretExposer.STORAGE_ROOT || '.cache/xread/storage');
await fs.mkdir(this.rootDir, { recursive: true });
this.emit('ready');
}
protected resolvePath(key: string) {
const safeKey = key.replace(/\\/g, '/').replace(/^\/+/, '');
return path.resolve(this.rootDir, safeKey);
}
async saveFile(key: string, content: Buffer | Uint8Array, _options?: SaveOptions) {
const fullPath = this.resolvePath(key);
await fs.mkdir(path.dirname(fullPath), { recursive: true });
await fs.writeFile(fullPath, content);
return { fullPath };
}
async downloadFile(key: string) {
return fs.readFile(this.resolvePath(key));
}
async signDownloadUrl(key: string, _expiresAt?: number) {
return pathToFileURL(this.resolvePath(key)).href;
}
}
+51
View File
@@ -0,0 +1,51 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
import { SecretExposer } from './secrets';
import { ServiceDisabledError } from '../../services/errors';
@singleton()
export class ProxyProviderService extends AsyncService {
protected proxies: URL[] = [];
constructor(
protected secretExposer: SecretExposer,
) {
super(...arguments);
}
override async init() {
await this.dependencyReady();
this.proxies = this.secretExposer.LOCAL_PROXY_URLS
.split(',')
.map((x) => x.trim())
.filter(Boolean)
.map((x) => new URL(x));
this.emit('ready');
}
supports(input?: string) {
if (!input || input === 'auto' || input === 'any' || input === 'none') {
return true;
}
return /^[a-z]{2}$/i.test(input);
}
async alloc(input?: string) {
if (input === 'none') {
throw new ServiceDisabledError(`Proxy allocation is disabled for '${input}'.`);
}
const picked = this.proxies[0];
if (!picked) {
throw new ServiceDisabledError('Proxy allocation is not configured in this standalone build.');
}
return picked;
}
async *iterAlloc(input?: string) {
yield await this.alloc(input);
}
}
+109
View File
@@ -0,0 +1,109 @@
import { AsyncService } from 'civkit/async-service';
import { ApplicationError, AutoCastable } from 'civkit/civ-rpc';
import { singleton } from 'tsyringe';
import { ApiRollRecord, API_CALL_STATUS } from '../db/api-roll';
type SubjectLimit = {
at: number;
};
@singleton()
export class RateLimitControl extends AsyncService {
protected hits = new Map<string, SubjectLimit[]>();
override async init() {
await this.dependencyReady();
this.emit('ready');
}
protected prune(key: string, windowMs: number) {
const now = Date.now();
const entries = (this.hits.get(key) || []).filter((entry) => (now - entry.at) < windowMs);
this.hits.set(key, entries);
return entries;
}
protected async checkAndRecord(subject: string, tags: string[], descs: RateLimitDesc[]) {
const now = Date.now();
for (const desc of descs) {
const key = `${subject}:${tags.join(',')}:${desc.periodSeconds}:${desc.occurrence}`;
const entries = this.prune(key, desc.periodSeconds * 1000);
if (entries.length >= desc.occurrence) {
const oldestRelevant = entries[0];
const retryAfterMs = Math.max(1000, desc.periodSeconds * 1000 - (now - oldestRelevant.at));
throw RateLimitTriggeredError.from({
message: `Rate limit exceeded for ${subject}`,
retryAfter: Math.ceil(retryAfterMs / 1000),
retryAfterDate: new Date(now + retryAfterMs),
});
}
entries.push({ at: now });
this.hits.set(key, entries);
}
}
async simpleRPCUidBasedLimit(_rpcReflect: any, uid: string, tags: string[], ...descs: RateLimitDesc[]) {
const effectiveDescs = descs.length ? descs : [RateLimitDesc.from({ occurrence: 60, periodSeconds: 60 })];
await this.checkAndRecord(`uid:${uid}`, tags, effectiveDescs);
return ApiRollRecord.from({
_id: `${Date.now()}-${Math.random()}`,
uid,
tags,
status: API_CALL_STATUS.SUCCESS,
createdAt: new Date(),
});
}
async simpleRpcIPBasedLimit(_rpcReflect: any, ip: string, tags: string[], limits?: [Date, number] | Array<[Date, number]>) {
const normalized = Array.isArray(limits?.[0]) ? limits as Array<[Date, number]> : (limits ? [limits as [Date, number]] : []);
const descs = normalized.map(([date, occurrence]) => {
const seconds = Math.max(1, Math.ceil((Date.now() - date.valueOf()) / 1000));
return RateLimitDesc.from({
occurrence,
periodSeconds: seconds,
});
});
await this.checkAndRecord(`ip:${ip}`, tags, descs.length ? descs : [RateLimitDesc.from({ occurrence: 20, periodSeconds: 60 })]);
return ApiRollRecord.from({
_id: `${Date.now()}-${Math.random()}`,
ip,
tags,
status: API_CALL_STATUS.SUCCESS,
createdAt: new Date(),
});
}
record(input: Partial<ApiRollRecord>) {
return ApiRollRecord.from({
createdAt: new Date(),
status: API_CALL_STATUS.SUCCESS,
...input,
});
}
}
export class RateLimitDesc extends AutoCastable {
occurrence!: number;
periodSeconds!: number;
isEffective() {
return Boolean(this.occurrence && this.periodSeconds);
}
}
export class RateLimitTriggeredError extends ApplicationError {
retryAfter?: number;
retryAfterDate?: Date;
static override from(input: Partial<RateLimitTriggeredError> & { message?: string }) {
const err = new RateLimitTriggeredError(input.message || 'Rate limit exceeded');
Object.assign(err, input);
return err;
}
}
+48
View File
@@ -0,0 +1,48 @@
import { AsyncService } from 'civkit/async-service';
import { singleton } from 'tsyringe';
type EnvConfig = Record<string, string> & {
readonly SERPER_SEARCH_API_KEY: string;
readonly BRAVE_SEARCH_API_KEY: string;
readonly CLOUD_FLARE_API_KEY: string;
readonly LOCAL_PROXY_URLS: string;
readonly STORAGE_ROOT: string;
};
export const readEnv = (key: string) => process.env[key] || '';
const envConfig = new Proxy({}, {
get(_target, prop) {
return readEnv(String(prop));
},
}) as EnvConfig;
@singleton()
export class SecretExposer extends AsyncService {
override async init() {
await this.dependencyReady();
this.emit('ready');
}
get SERPER_SEARCH_API_KEY() {
return readEnv('SERPER_SEARCH_API_KEY');
}
get BRAVE_SEARCH_API_KEY() {
return readEnv('BRAVE_SEARCH_API_KEY');
}
get CLOUD_FLARE_API_KEY() {
return readEnv('CLOUD_FLARE_API_KEY');
}
get LOCAL_PROXY_URLS() {
return readEnv('LOCAL_PROXY_URLS');
}
get STORAGE_ROOT() {
return readEnv('STORAGE_ROOT');
}
}
export default envConfig;
+6
View File
@@ -0,0 +1,6 @@
import type { Context, Next } from 'koa';
export function getAuditionMiddleware() {
return async (_ctx: Context, next: Next) => next();
}
+8
View File
@@ -0,0 +1,8 @@
export function countGPTToken(input?: string | null) {
if (!input) {
return 0;
}
return Math.max(1, Math.ceil(Buffer.byteLength(input, 'utf8') / 4));
}
+1 -1
View File
@@ -8,7 +8,7 @@ import { GoogleAuth } from 'google-auth-library';
* @return {Promise<string>} The URL of the function
*/
export async function getFunctionUrl(name: string, location = "us-central1") {
const projectId = `reader-6b7dc`;
const projectId = `xread-project`;
const url = "https://cloudfunctions.googleapis.com/v2beta/" +
`projects/${projectId}/locations/${location}/functions/${name}`;
const auth = new GoogleAuth({
+62
View File
@@ -0,0 +1,62 @@
const test = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const projectRoot = path.resolve(__dirname, '..');
function read(relativePath) {
return fs.readFileSync(path.join(projectRoot, relativePath), 'utf8');
}
test('legacy Cloud Run deployment workflow has been removed', () => {
const legacyWorkflow = path.join(projectRoot, '.github', 'workflows', 'cd.yml');
assert.equal(fs.existsSync(legacyWorkflow), false);
});
test('repository automation files exist for CI, container publish, and Dependabot', () => {
const expectedFiles = [
'.eslintrc.cjs',
'.eslintignore',
'.github/workflows/ci.yml',
'.github/workflows/image.yml',
'.github/workflows/dependabot-auto-merge.yml',
'.github/dependabot.yml',
];
for (const relativePath of expectedFiles) {
assert.equal(fs.existsSync(path.join(projectRoot, relativePath)), true, `${relativePath} should exist`);
}
});
test('container image workflow publishes to GHCR instead of legacy GCP registries', () => {
const workflow = read('.github/workflows/image.yml');
assert.match(workflow, /ghcr\.io/);
assert.doesNotMatch(workflow, /gcloud|us-docker\.pkg\.dev/);
});
test('dependabot config covers npm, docker, and github-actions updates', () => {
const dependabot = read('.github/dependabot.yml');
assert.match(dependabot, /package-ecosystem: npm/);
assert.match(dependabot, /package-ecosystem: docker/);
assert.match(dependabot, /package-ecosystem: github-actions/);
});
test('CI workflow runs lint before tests and build', () => {
const workflow = read('.github/workflows/ci.yml');
assert.match(workflow, /run: npm run lint/);
assert.match(workflow, /run: npm run test:ci/);
assert.match(workflow, /run: npm run build/);
});
test('Dockerfile is self-contained and no longer relies on curl-impersonate', () => {
const dockerfile = read('Dockerfile');
assert.match(dockerfile, /FROM node:22-bookworm-slim/);
assert.match(dockerfile, /PUPPETEER_SKIP_DOWNLOAD=true/);
assert.doesNotMatch(dockerfile, /curl-impersonate|LD_PRELOAD/);
});
+21
View File
@@ -0,0 +1,21 @@
const test = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const os = require('node:os');
const path = require('node:path');
const { spawnSync } = require('node:child_process');
test('integrity check warns instead of failing when GeoLite asset is missing', () => {
const tempRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'xread-integrity-'));
const scriptPath = path.join(tempRoot, 'integrity-check.cjs');
fs.copyFileSync(path.resolve(__dirname, '..', 'integrity-check.cjs'), scriptPath);
const result = spawnSync(process.execPath, [scriptPath], {
cwd: tempRoot,
encoding: 'utf8',
});
assert.equal(result.status, 0);
assert.match(result.stderr, /GeoLite2-City\.mmdb/);
});
+44
View File
@@ -0,0 +1,44 @@
const test = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { spawnSync } = require('node:child_process');
const projectRoot = path.resolve(__dirname, '..');
test('src/shared is a real directory in the repository', () => {
const sharedPath = path.join(projectRoot, 'src', 'shared');
const stat = fs.statSync(sharedPath);
assert.equal(stat.isDirectory(), true);
});
test('default build script uses standard TypeScript compiler', () => {
const packageJson = JSON.parse(fs.readFileSync(path.join(projectRoot, 'package.json'), 'utf8'));
assert.match(packageJson.scripts.build, /\btsc\b/);
assert.doesNotMatch(packageJson.scripts.build, /scripts\/transpile\.cjs/);
});
test('project type-check build passes with tsc', () => {
const tscJsPath = path.join(
projectRoot,
'.codex-cache',
'ts-compiler',
'node_modules',
'typescript',
'lib',
'tsc.js',
);
const result = spawnSync(
process.execPath,
[tscJsPath, '-p', '.', '--pretty', 'false'],
{
cwd: projectRoot,
encoding: 'utf8',
shell: false,
},
);
assert.equal(result.status, 0, result.stdout + result.stderr);
});
+28
View File
@@ -0,0 +1,28 @@
const test = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const os = require('node:os');
const path = require('node:path');
test('ensureLicensedAssets creates GeoLite database when missing', async () => {
const tempRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'xread-licensed-'));
const targetFile = path.join(tempRoot, 'licensed', 'GeoLite2-City.mmdb');
let downloadCalls = 0;
const { ensureLicensedAssets, DEFAULT_ASSETS } = require('../scripts/prepare-licensed-assets.cjs');
await ensureLicensedAssets({
rootDir: tempRoot,
assets: DEFAULT_ASSETS,
downloadAsset: async (asset, destination) => {
downloadCalls += 1;
assert.equal(asset.path, 'licensed/GeoLite2-City.mmdb');
fs.mkdirSync(path.dirname(destination), { recursive: true });
fs.writeFileSync(destination, Buffer.from('fake-mmdb'));
},
});
assert.equal(downloadCalls, 1);
assert.equal(fs.existsSync(targetFile), true);
assert.equal(fs.readFileSync(targetFile, 'utf8'), 'fake-mmdb');
});
+4 -3
View File
@@ -1,7 +1,7 @@
{
"compilerOptions": {
"module": "node16",
"types": ["node"],
"noImplicitReturns": true,
"noUnusedLocals": true,
"outDir": "build",
@@ -9,7 +9,7 @@
"strict": true,
"allowJs": true,
"target": "es2022",
"lib": ["es2022"],
"lib": ["es2022", "dom"],
"skipLibCheck": true,
"useDefineForClassFields": false,
"experimentalDecorators": true,
@@ -18,5 +18,6 @@
"noImplicitOverride": true,
},
"compileOnSave": true,
"include": ["src"]
"include": ["src"],
"exclude": ["src/cloud-functions/**/*"]
}