feat(repo): make xread standalone and self-hostable
Replace private shared dependencies with local implementations, keep the standalone crawl/search/serp build working, and add CI, GHCR image publishing, Dependabot updates, and green-only auto-merge automation.
This commit is contained in:
1 parent
ec8f135392
commit
301854727b
61 files changed
+2058
-1960
No files matched your search
@@ -0,0 +1,5 @@
|
||||
build/
|
||||
node_modules/
|
||||
licensed/
|
||||
.codex-cache/
|
||||
coverage/
|
||||
@@ -0,0 +1,87 @@
|
||||
module.exports = {
|
||||
root: true,
|
||||
env: {
|
||||
node: true,
|
||||
es2022: true,
|
||||
},
|
||||
parser: '@typescript-eslint/parser',
|
||||
parserOptions: {
|
||||
ecmaVersion: 'latest',
|
||||
sourceType: 'module',
|
||||
},
|
||||
plugins: ['@typescript-eslint', 'import'],
|
||||
extends: [
|
||||
'eslint:recommended',
|
||||
'plugin:@typescript-eslint/recommended',
|
||||
],
|
||||
ignorePatterns: [
|
||||
'build/',
|
||||
'node_modules/',
|
||||
'licensed/',
|
||||
'.codex-cache/',
|
||||
'coverage/',
|
||||
],
|
||||
overrides: [
|
||||
{
|
||||
files: ['**/*.ts'],
|
||||
rules: {
|
||||
'@typescript-eslint/no-explicit-any': 'off',
|
||||
'@typescript-eslint/no-inferrable-types': 'off',
|
||||
'@typescript-eslint/no-non-null-assertion': 'off',
|
||||
'@typescript-eslint/no-var-requires': 'off',
|
||||
'@typescript-eslint/ban-ts-comment': 'off',
|
||||
'@typescript-eslint/no-extra-semi': 'off',
|
||||
'@typescript-eslint/no-this-alias': 'off',
|
||||
'@typescript-eslint/ban-types': 'off',
|
||||
'@typescript-eslint/no-unused-vars': [
|
||||
'warn',
|
||||
{
|
||||
argsIgnorePattern: '^_',
|
||||
varsIgnorePattern: '^_',
|
||||
caughtErrorsIgnorePattern: '^_',
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
{
|
||||
files: ['**/*.js', '**/*.cjs'],
|
||||
parserOptions: {
|
||||
sourceType: 'script',
|
||||
},
|
||||
rules: {
|
||||
'@typescript-eslint/no-var-requires': 'off',
|
||||
'@typescript-eslint/no-unused-vars': [
|
||||
'warn',
|
||||
{
|
||||
argsIgnorePattern: '^_',
|
||||
varsIgnorePattern: '^_',
|
||||
caughtErrorsIgnorePattern: '^_',
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
{
|
||||
files: ['tests/**/*.cjs', 'scripts/**/*.cjs', '*.cjs'],
|
||||
env: {
|
||||
node: true,
|
||||
},
|
||||
},
|
||||
],
|
||||
rules: {
|
||||
'no-empty': 'off',
|
||||
'no-useless-escape': 'warn',
|
||||
'no-prototype-builtins': 'off',
|
||||
'no-case-declarations': 'off',
|
||||
'no-control-regex': 'off',
|
||||
'no-async-promise-executor': 'off',
|
||||
'no-constant-condition': 'off',
|
||||
'no-extra-boolean-cast': 'off',
|
||||
'no-unsafe-optional-chaining': 'off',
|
||||
'no-undef': 'off',
|
||||
'no-var': 'off',
|
||||
'prefer-const': 'off',
|
||||
'prefer-rest-params': 'off',
|
||||
'require-yield': 'off',
|
||||
'import/no-unresolved': 'off',
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,61 @@
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: npm
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: monday
|
||||
time: "09:00"
|
||||
timezone: Asia/Shanghai
|
||||
commit-message:
|
||||
prefix: chore
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
open-pull-requests-limit: 10
|
||||
rebase-strategy: auto
|
||||
groups:
|
||||
production-dependencies:
|
||||
dependency-type: production
|
||||
update-types:
|
||||
- minor
|
||||
- patch
|
||||
development-dependencies:
|
||||
dependency-type: development
|
||||
update-types:
|
||||
- minor
|
||||
- patch
|
||||
|
||||
- package-ecosystem: docker
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: monday
|
||||
time: "09:15"
|
||||
timezone: Asia/Shanghai
|
||||
commit-message:
|
||||
prefix: chore
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
- docker
|
||||
open-pull-requests-limit: 5
|
||||
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
day: monday
|
||||
time: "09:30"
|
||||
timezone: Asia/Shanghai
|
||||
commit-message:
|
||||
prefix: chore
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
- github-actions
|
||||
open-pull-requests-limit: 5
|
||||
groups:
|
||||
github-actions:
|
||||
patterns:
|
||||
- "*"
|
||||
Whitespace-only changes.
@@ -1,89 +0,0 @@
|
||||
run-name: Build push and deploy (CD)
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
- ci-debug
|
||||
- dev
|
||||
tags:
|
||||
- '*'
|
||||
|
||||
jobs:
|
||||
build-and-push-to-gcr:
|
||||
runs-on: ubuntu-latest
|
||||
concurrency:
|
||||
group: ${{ github.ref_type == 'branch' && github.ref }}
|
||||
cancel-in-progress: true
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
lfs: true
|
||||
submodules: true
|
||||
token: ${{ secrets.THINAPPS_SHARED_READ_TOKEN }}
|
||||
- uses: 'google-github-actions/auth@v2'
|
||||
with:
|
||||
credentials_json: '${{ secrets.GCLOUD_SERVICE_ACCOUNT_SECRET_JSON }}'
|
||||
- name: 'Set up Cloud SDK'
|
||||
uses: 'google-github-actions/setup-gcloud@v2'
|
||||
with:
|
||||
install_components: beta
|
||||
- name: "Docker auth"
|
||||
run: |-
|
||||
gcloud auth configure-docker us-docker.pkg.dev --quiet
|
||||
- name: Set controller release version
|
||||
run: echo "RELEASE_VERSION=${GITHUB_REF#refs/*/}" >> $GITHUB_ENV
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: 22.12.0
|
||||
cache: npm
|
||||
|
||||
- name: npm install
|
||||
run: npm ci
|
||||
- name: get maxmind mmdb
|
||||
run: mkdir -p licensed && curl -o licensed/GeoLite2-City.mmdb https://raw.githubusercontent.com/P3TERX/GeoLite.mmdb/download/GeoLite2-City.mmdb
|
||||
- name: get source han sans font
|
||||
run: curl -o licensed/SourceHanSansSC-Regular.otf https://raw.githubusercontent.com/adobe-fonts/source-han-sans/refs/heads/release/OTF/SimplifiedChinese/SourceHanSansSC-Regular.otf
|
||||
- name: build application
|
||||
run: npm run build
|
||||
- name: Set package version
|
||||
run: npm version --no-git-tag-version ${{ env.RELEASE_VERSION }}
|
||||
if: github.ref_type == 'tag'
|
||||
- name: Docker meta
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
us-docker.pkg.dev/reader-6b7dc/jina-reader/reader
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
- name: Build and push
|
||||
id: container
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
- name: Deploy CRAWL with Tag
|
||||
run: |
|
||||
gcloud beta run deploy crawl --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/crawl.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
- name: Deploy SEARCH with Tag
|
||||
run: |
|
||||
gcloud beta run deploy search --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/search.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
- name: Deploy SERP with Tag
|
||||
run: |
|
||||
gcloud beta run deploy serp --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/serp.js --region us-central1 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
- name: Deploy CRAWL-EU with Tag
|
||||
run: |
|
||||
gcloud beta run deploy crawl-eu --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/crawl.js --region europe-west1 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
- name: Deploy SEARCH-EU with Tag
|
||||
run: |
|
||||
gcloud beta run deploy search-eu --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/search.js --region europe-west1 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
- name: Deploy SERP-HK with Tag
|
||||
run: |
|
||||
gcloud beta run deploy serp-hk --image us-docker.pkg.dev/reader-6b7dc/jina-reader/reader@${{steps.container.outputs.imageid}} --tag ${{ env.RELEASE_VERSION }} --command '' --args build/stand-alone/serp.js --region asia-east2 --async --min-instances 0 --deploy-health-check --use-http2
|
||||
@@ -0,0 +1,51 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
name: Verify stand-alone build
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
env:
|
||||
CI: true
|
||||
PUPPETEER_SKIP_DOWNLOAD: true
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
|
||||
- name: Install dependencies
|
||||
run: npm ci
|
||||
|
||||
- name: Lint source files
|
||||
run: npm run lint
|
||||
|
||||
- name: Run repository tests
|
||||
run: npm run test:ci
|
||||
|
||||
- name: Build TypeScript output
|
||||
run: npm run build
|
||||
|
||||
- name: Smoke-test stand-alone entrypoints
|
||||
run: |
|
||||
NODE_ENV=dry-run node ./build/stand-alone/crawl.js
|
||||
NODE_ENV=dry-run node ./build/stand-alone/search.js
|
||||
NODE_ENV=dry-run node ./build/stand-alone/serp.js
|
||||
@@ -0,0 +1,30 @@
|
||||
name: Dependabot Auto Merge
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
types:
|
||||
- opened
|
||||
- reopened
|
||||
- synchronize
|
||||
- ready_for_review
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
auto-merge:
|
||||
if: github.event.pull_request.user.login == 'dependabot[bot]' && !github.event.pull_request.draft
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Approve the Dependabot pull request
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
PR_URL: ${{ github.event.pull_request.html_url }}
|
||||
run: gh pr review --repo "$GITHUB_REPOSITORY" "$PR_URL" --approve --body "Approved automatically after policy checks." || true
|
||||
|
||||
- name: Enable auto-merge after required checks pass
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
PR_URL: ${{ github.event.pull_request.html_url }}
|
||||
run: gh pr merge --repo "$GITHUB_REPOSITORY" "$PR_URL" --auto --squash --delete-branch
|
||||
@@ -0,0 +1,71 @@
|
||||
name: Container Image
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- Dockerfile
|
||||
- package.json
|
||||
- package-lock.json
|
||||
- tsconfig.json
|
||||
- integrity-check.cjs
|
||||
- src/**
|
||||
- public/**
|
||||
- licensed/**
|
||||
- scripts/**
|
||||
- .github/workflows/image.yml
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
tags:
|
||||
- v*
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
concurrency:
|
||||
group: image-${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
docker:
|
||||
name: Build container image
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 45
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Derive image name
|
||||
id: vars
|
||||
run: echo "image_name=ghcr.io/${GITHUB_REPOSITORY,,}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Extract image metadata
|
||||
id: meta
|
||||
uses: docker/metadata-action@9ec57ed1fcdbf14dcef7dfbe97b2010124a938b7
|
||||
with:
|
||||
images: ${{ steps.vars.outputs.image_name }}
|
||||
tags: |
|
||||
type=ref,event=branch
|
||||
type=ref,event=tag
|
||||
type=sha,prefix=sha-
|
||||
type=raw,value=latest,enable={{is_default_branch}}
|
||||
|
||||
- name: Log in to GitHub Container Registry
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@65b78e6e13532edd9afa3aa52ac7964289d1a9c1
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Build and optionally publish image
|
||||
uses: docker/build-push-action@f2a1d5e99d037542a71f64918e516c093c6f3fc4
|
||||
with:
|
||||
context: .
|
||||
push: ${{ github.event_name != 'pull_request' }}
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=gha
|
||||
cache-to: type=gha,mode=max
|
||||
+20
-18
@@ -1,41 +1,43 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
FROM lwthiker/curl-impersonate:0.6-chrome-slim-bullseye
|
||||
FROM node:22-bookworm-slim
|
||||
|
||||
FROM node:22
|
||||
ENV PUPPETEER_SKIP_DOWNLOAD=true
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y wget gnupg \
|
||||
&& wget -q -O - https://dl-ssl.google.com/linux/linux_signing_key.pub | apt-key add - \
|
||||
&& sh -c 'echo "deb [arch=amd64] http://dl.google.com/linux/chrome/deb/ stable main" >> /etc/apt/sources.list.d/google.list' \
|
||||
&& apt-get install -y --no-install-recommends wget gnupg ca-certificates \
|
||||
&& mkdir -p /etc/apt/keyrings \
|
||||
&& wget -q -O - https://dl.google.com/linux/linux_signing_key.pub | gpg --dearmor -o /etc/apt/keyrings/google-chrome.gpg \
|
||||
&& echo "deb [arch=amd64 signed-by=/etc/apt/keyrings/google-chrome.gpg] https://dl.google.com/linux/chrome/deb/ stable main" > /etc/apt/sources.list.d/google-chrome.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y google-chrome-stable fonts-ipafont-gothic fonts-wqy-zenhei fonts-thai-tlwg fonts-kacst fonts-freefont-ttf libxss1 zstd \
|
||||
--no-install-recommends \
|
||||
&& apt-get install -y --no-install-recommends google-chrome-stable fonts-ipafont-gothic fonts-wqy-zenhei fonts-thai-tlwg fonts-kacst fonts-freefont-ttf libxss1 zstd \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY --from=0 /usr/local/lib/libcurl-impersonate.so /usr/local/lib/libcurl-impersonate.so
|
||||
|
||||
RUN groupadd -r jina
|
||||
RUN useradd -g jina -G audio,video -m jina
|
||||
USER jina
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY package.json package-lock.json ./
|
||||
RUN npm ci
|
||||
|
||||
COPY build ./build
|
||||
COPY integrity-check.cjs tsconfig.json ./
|
||||
COPY scripts ./scripts
|
||||
COPY src ./src
|
||||
COPY public ./public
|
||||
COPY licensed ./licensed
|
||||
|
||||
RUN rm -rf ~/.config/chromium && mkdir -p ~/.config/chromium
|
||||
RUN npm run build
|
||||
RUN NODE_ENV=dry-run node ./build/stand-alone/crawl.js \
|
||||
&& NODE_ENV=dry-run node ./build/stand-alone/search.js \
|
||||
&& NODE_ENV=dry-run node ./build/stand-alone/serp.js
|
||||
|
||||
RUN NODE_COMPILE_CACHE=node_modules npm run dry-run
|
||||
RUN groupadd -r xread \
|
||||
&& useradd -g xread -G audio,video -m xread \
|
||||
&& chown -R xread:xread /app
|
||||
|
||||
USER xread
|
||||
|
||||
ENV OVERRIDE_CHROME_EXECUTABLE_PATH=/usr/bin/google-chrome-stable
|
||||
ENV LD_PRELOAD=/usr/local/lib/libcurl-impersonate.so CURL_IMPERSONATE=chrome116 CURL_IMPERSONATE_HEADERS=no
|
||||
ENV NODE_COMPILE_CACHE=node_modules
|
||||
ENV PORT=8080
|
||||
|
||||
EXPOSE 3000 3001 8080 8081
|
||||
ENTRYPOINT ["node"]
|
||||
CMD [ "build/stand-alone/crawl.js" ]
|
||||
CMD ["build/stand-alone/crawl.js"]
|
||||
@@ -0,0 +1,7 @@
|
||||
This repository contains a modified and rebranded derivative of upstream software
|
||||
originally distributed by Jina AI Limited under the Apache License 2.0.
|
||||
|
||||
Modifications in this fork include:
|
||||
- replacing project branding and product-facing strings with neutral placeholders
|
||||
- updating example domains, package metadata, and deployment placeholders
|
||||
- preserving the original LICENSE file for attribution and license compliance
|
||||
@@ -1,167 +1,90 @@
|
||||
# Reader
|
||||
# Xread
|
||||
|
||||
Your LLMs deserve better input.
|
||||
|
||||
Reader does two things:
|
||||
- **Read**: It converts any URL to an **LLM-friendly** input with `https://r.jina.ai/https://your.url`. Get improved output for your agent and RAG systems at no cost.
|
||||
- **Search**: It searches the web for a given query with `https://s.jina.ai/your+query`. This allows your LLMs to access the latest world knowledge from the web.
|
||||
Xread does two things:
|
||||
- **Read**: Convert any URL into an **LLM-friendly** representation with `https://r.example.com/https://your.url`.
|
||||
- **Search**: Search the web with `https://s.example.com/your+query` and return summarized results in an LLM-friendly format.
|
||||
|
||||
Check out [the live demo](https://jina.ai/reader#demo)
|
||||
|
||||
Or just visit these URLs (**Read**) https://r.jina.ai/https://github.com/jina-ai/reader, (**Search**) https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F and see yourself.
|
||||
|
||||
> Feel free to use Reader API in production. It is free, stable and scalable. We are maintaining it actively as one of the core products of Jina AI. [Check out rate limit](https://jina.ai/reader#pricing)
|
||||
|
||||
<img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/2067c7a2-c12e-4465-b107-9a16ca178d41">
|
||||
<img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/675ac203-f246-41c2-b094-76318240159f">
|
||||
|
||||
|
||||
## Updates
|
||||
|
||||
- **2024-07-15**: To restrict the results of `s.jina.ai` to certain domain/website, you can set e.g. `site=jina.ai` in the query parameters, which enables in-site search. For more options, [try our updated live-demo](https://jina.ai/reader/#apiform).
|
||||
- **2024-05-30**: Reader can now read abitrary PDF from any URL! Check out [this PDF result from NASA.gov](https://r.jina.ai/https://www.nasa.gov/wp-content/uploads/2023/01/55583main_vision_space_exploration2.pdf) vs [the original](https://www.nasa.gov/wp-content/uploads/2023/01/55583main_vision_space_exploration2.pdf).
|
||||
- **2024-05-15**: We introduced a new endpoint `s.jina.ai` that searches on the web and return top-5 results, each in a LLM-friendly format. [Read more about this new feature here](https://jina.ai/news/jina-reader-for-search-grounding-to-improve-factuality-of-llms).
|
||||
- **2024-05-08**: Image caption is off by default for better latency. To turn it on, set `x-with-generated-alt: true` in the request header.
|
||||
- **2024-04-24**: You now have more fine-grained control over Reader API [using headers](#using-request-headers), e.g. forwarding cookies, using HTTP proxy.
|
||||
- **2024-04-15**: Reader now supports image reading! It captions all images at the specified URL and adds `Image [idx]: [caption]` as an alt tag (if they initially lack one). This enables downstream LLMs to interact with the images in reasoning, summarizing etc. [See example here](https://x.com/JinaAI_/status/1780094402071023926).
|
||||
Check out the placeholder demo at [https://example.com/xread#demo](https://example.com/xread#demo).
|
||||
|
||||
## Usage
|
||||
|
||||
### Using `r.jina.ai` for single URL fetching
|
||||
Simply prepend `https://r.jina.ai/` to any URL. For example, to convert the URL `https://en.wikipedia.org/wiki/Artificial_intelligence` to an LLM-friendly input, use the following URL:
|
||||
### Read a single URL
|
||||
|
||||
[https://r.jina.ai/https://en.wikipedia.org/wiki/Artificial_intelligence](https://r.jina.ai/https://en.wikipedia.org/wiki/Artificial_intelligence)
|
||||
Prepend `https://r.example.com/` to any URL:
|
||||
|
||||
### [Using `r.jina.ai` for a full website fetching (Google Colab)](https://colab.research.google.com/drive/1uoBy6_7BhxqpFQ45vuhgDDDGwstaCt4P#scrollTo=5LQjzJiT9ewT)
|
||||
[https://r.example.com/https://en.wikipedia.org/wiki/Artificial_intelligence](https://r.example.com/https://en.wikipedia.org/wiki/Artificial_intelligence)
|
||||
|
||||
### Using `s.jina.ai` for web search
|
||||
Simply prepend `https://s.jina.ai/` to your search query. Note that if you are using this in the code, make sure to encode your search query first, e.g. if your query is `Who will win 2024 US presidential election?` then your url should look like:
|
||||
### Search the web
|
||||
|
||||
[https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F](https://s.jina.ai/Who%20will%20win%202024%20US%20presidential%20election%3F)
|
||||
Prepend `https://s.example.com/` to a URL-encoded search query:
|
||||
|
||||
Behind the scenes, Reader searches the web, fetches the top 5 results, visits each URL, and applies `r.jina.ai` to it. This is different from many `web search function-calling` in agent/RAG frameworks, which often return only the title, URL, and description provided by the search engine API. If you want to read one result more deeply, you have to fetch the content yourself from that URL. With Reader, `http://s.jina.ai` automatically fetches the content from the top 5 search result URLs for you (reusing the tech stack behind `http://r.jina.ai`). This means you don't have to handle browser rendering, blocking, or any issues related to JavaScript and CSS yourself.
|
||||
[https://s.example.com/Who%20will%20win%202024%20US%20presidential%20election%3F](https://s.example.com/Who%20will%20win%202024%20US%20presidential%20election%3F)
|
||||
|
||||
### Using `s.jina.ai` for in-site search
|
||||
Simply specify `site` in the query parameters such as:
|
||||
Behind the scenes, Xread fetches relevant pages and converts them into a format that is easier for downstream LLMs and agent systems to consume.
|
||||
|
||||
### In-site search
|
||||
|
||||
Use repeated `site` parameters to constrain results:
|
||||
|
||||
```bash
|
||||
curl 'https://s.jina.ai/When%20was%20Jina%20AI%20founded%3F?site=jina.ai&site=github.com'
|
||||
curl 'https://s.example.com/When%20was%20example.com%20founded%3F?site=example.com&site=github.com'
|
||||
```
|
||||
|
||||
### [Interactive Code Snippet Builder](https://jina.ai/reader#apiform)
|
||||
### Interactive code builder
|
||||
|
||||
We highly recommend using the code builder to explore different parameter combinations of the Reader API.
|
||||
Use the placeholder builder URL:
|
||||
|
||||
<a href="https://jina.ai/reader#apiform"><img width="973" alt="image" src="https://github.com/jina-ai/reader/assets/2041322/a490fd3a-1c4c-4a3f-a95a-c481c2a8cc8f"></a>
|
||||
[https://example.com/xread#apiform](https://example.com/xread#apiform)
|
||||
|
||||
## Request headers
|
||||
|
||||
### Using request headers
|
||||
The service behavior can be controlled via headers:
|
||||
|
||||
As you have already seen above, one can control the behavior of the Reader API using request headers. Here is a complete list of supported headers.
|
||||
- `x-with-generated-alt: true` enables automatic image alt-text generation.
|
||||
- `x-set-cookie` forwards cookies.
|
||||
- `x-respond-with` supports `markdown`, `html`, `text`, `screenshot`, and related formats.
|
||||
- `x-proxy-url` selects a custom proxy.
|
||||
- `x-cache-tolerance` adjusts cache tolerance in seconds.
|
||||
- `x-no-cache: true` bypasses cached content.
|
||||
- `x-target-selector` narrows extraction to a CSS selector.
|
||||
- `x-wait-for-selector` waits until a CSS selector appears.
|
||||
|
||||
- You can enable the image caption feature via the `x-with-generated-alt: true` header.
|
||||
- You can ask the Reader API to forward cookies settings via the `x-set-cookie` header.
|
||||
- Note that requests with cookies will not be cached.
|
||||
- You can bypass `readability` filtering via the `x-respond-with` header, specifically:
|
||||
- `x-respond-with: markdown` returns markdown *without* going through `reability`
|
||||
- `x-respond-with: html` returns `documentElement.outerHTML`
|
||||
- `x-respond-with: text` returns `document.body.innerText`
|
||||
- `x-respond-with: screenshot` returns the URL of the webpage's screenshot
|
||||
- You can specify a proxy server via the `x-proxy-url` header.
|
||||
- You can customize cache tolerance via the `x-cache-tolerance` header (integer in seconds).
|
||||
- You can bypass the cached page (lifetime 3600s) via the `x-no-cache: true` header (equivalent of `x-cache-tolerance: 0`).
|
||||
- If you already know the HTML structure of your target page, you may specify `x-target-selector` or `x-wait-for-selector` to direct the Reader API to focus on a specific part of the page.
|
||||
- By setting `x-target-selector` header to a CSS selector, the Reader API return the content within the matched element, instead of the full HTML. Setting this header is useful when the automatic content extraction fails to capture the desired content and you can manually select the correct target.
|
||||
- By setting `x-wait-for-selector` header to a CSS selector, the Reader API will wait until the matched element is rendered before returning the content. If you already specified `x-wait-for-selector`, this header can be omitted if you plan to wait for the same element.
|
||||
## SPA fetching
|
||||
|
||||
### Using `r.jina.ai` for single page application (SPA) fetching
|
||||
Many websites nowadays rely on JavaScript frameworks and client-side rendering. Usually known as Single Page Application (SPA). Thanks to [Puppeteer](https://github.com/puppeteer/puppeteer) and headless Chrome browser, Reader natively supports fetching these websites. However, due to specific approach some SPA are developed, there may be some extra precautions to take.
|
||||
|
||||
#### SPAs with hash-based routing
|
||||
By definition of the web standards, content come after `#` in a URL is not sent to the server. To mitigate this issue, use `POST` method with `url` parameter in body.
|
||||
For hash-based routing, use `POST` with the target URL in the body:
|
||||
|
||||
```bash
|
||||
curl -X POST 'https://r.jina.ai/' -d 'url=https://example.com/#/route'
|
||||
curl -X POST 'https://r.example.com/' -d 'url=https://example.com/#/route'
|
||||
```
|
||||
|
||||
#### SPAs with preloading contents
|
||||
Some SPAs, or even some websites that are not strictly SPAs, may show preload contents before later loading the main content dynamically. In this case, Reader may be capturing the preload content instead of the main content. To mitigate this issue, here are some possible solutions:
|
||||
|
||||
##### Specifying `x-timeout`
|
||||
When timeout is explicitly specified, Reader will not attempt to return early and will wait for network idle until the timeout is reached. This is useful when the target website will eventually come to a network idle.
|
||||
## Streaming mode
|
||||
|
||||
```bash
|
||||
curl 'https://example.com/' -H 'x-timeout: 30'
|
||||
curl -H "Accept: text/event-stream" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
```
|
||||
|
||||
##### Specifying `x-wait-for-selector`
|
||||
When wait-for-selector is explicitly specified, Reader will wait for the appearance of the specified CSS selector until timeout is reached. This is useful when you know exactly what element to wait for.
|
||||
Streaming responses return progressively more complete content chunks.
|
||||
|
||||
## JSON mode
|
||||
|
||||
```bash
|
||||
curl 'https://example.com/' -H 'x-wait-for-selector: #content'
|
||||
curl -H "Accept: application/json" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
```
|
||||
|
||||
### Streaming mode
|
||||
For `s.example.com`, JSON mode returns a list of search results shaped like `{'title', 'content', 'url'}`.
|
||||
|
||||
Streaming mode is useful when you find that the standard mode provides an incomplete result. This is because the Reader will wait a bit longer until the page is *stablely* rendered. Use the accept-header to toggle the streaming mode:
|
||||
## Generated alt
|
||||
|
||||
```bash
|
||||
curl -H "Accept: text/event-stream" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
curl -H "X-With-Generated-Alt: true" https://r.example.com/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
```
|
||||
|
||||
The data comes in a stream; each subsequent chunk contains more complete information. **The last chunk should provide the most complete and final result.** If you come from LLMs, please note that it is a different behavior than the LLMs' text-generation streaming.
|
||||
## Standalone Notes
|
||||
|
||||
For example, compare these two curl commands below. You can see streaming one gives you complete information at last, whereas standard mode does not. This is because the content loading on this particular site is triggered by some js *after* the page is fully loaded, and standard mode returns the page "too soon".
|
||||
```bash
|
||||
curl -H 'x-no-cache: true' https://access.redhat.com/security/cve/CVE-2023-45853
|
||||
curl -H "Accept: text/event-stream" -H 'x-no-cache: true' https://r.jina.ai/https://access.redhat.com/security/cve/CVE-2023-45853
|
||||
```
|
||||
|
||||
> Note: `-H 'x-no-cache: true'` is used only for demonstration purposes to bypass the cache.
|
||||
|
||||
Streaming mode is also useful if your downstream LLM/agent system requires immediate content delivery or needs to process data in chunks to interleave I/O and LLM processing times. This allows for quicker access and more efficient data handling:
|
||||
|
||||
```text
|
||||
Reader API: streamContent1 ----> streamContent2 ----> streamContent3 ---> ...
|
||||
| | |
|
||||
v | |
|
||||
Your LLM: LLM(streamContent1) | |
|
||||
v |
|
||||
LLM(streamContent2) |
|
||||
v
|
||||
LLM(streamContent3)
|
||||
```
|
||||
|
||||
Note that in terms of completeness: `... > streamContent3 > streamContent2 > streamContent1`, each subsequent chunk contains more complete information.
|
||||
|
||||
### JSON mode
|
||||
|
||||
This is still very early and the result is not really a "useful" JSON. It contains three fields `url`, `title` and `content` only. Nonetheless, you can use accept-header to control the output format:
|
||||
```bash
|
||||
curl -H "Accept: application/json" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
```
|
||||
|
||||
JSON mode is probably more useful in `s.jina.ai` than `r.jina.ai`. For `s.jina.ai` with JSON mode, it returns 5 results in a list, each in the structure of `{'title', 'content', 'url'}`.
|
||||
|
||||
### Generated alt
|
||||
|
||||
All images in that page that lack `alt` tag can be auto-captioned by a VLM (vision langauge model) and formatted as `!(Image [idx]: [VLM_caption])[img_URL]`. This should give your downstream text-only LLM *just enough* hints to include those images into reasoning, selecting, and summarization. Use the x-with-generated-alt header to toggle the streaming mode:
|
||||
|
||||
```bash
|
||||
curl -H "X-With-Generated-Alt: true" https://r.jina.ai/https://en.m.wikipedia.org/wiki/Main_Page
|
||||
```
|
||||
|
||||
## How it works
|
||||
[](https://deepwiki.com/jina-ai/reader)
|
||||
|
||||
## What is `thinapps-shared` submodule?
|
||||
|
||||
You might notice a reference to `thinapps-shared` submodule, an internal package we use to share code across our products. While it’s not open-sourced and isn't integral to the Reader's functions, it mainly helps with decorators, logging, secrets management, etc. Feel free to ignore it for now.
|
||||
|
||||
That said, this is *the single codebase* behind `https://r.jina.ai`, so everytime we commit here, we will deploy the new version to the `https://r.jina.ai`.
|
||||
|
||||
## Having trouble on some websites?
|
||||
Please raise an issue with the URL you are having trouble with. We will look into it and try to fix it.
|
||||
This repository now carries its shared infrastructure helpers in `src/shared/` so the stand-alone crawl/search/serp entrypoints can build and run without private dependencies.
|
||||
|
||||
## License
|
||||
Reader is backed by [Jina AI](https://jina.ai) and licensed under [Apache-2.0](./LICENSE).
|
||||
|
||||
This repository is distributed under [Apache-2.0](./LICENSE). See [NOTICE](./NOTICE) for modification and attribution details.
|
||||
+1
-2
@@ -6,6 +6,5 @@ const path = require('path');
|
||||
const file = path.resolve(__dirname, 'licensed/GeoLite2-City.mmdb');
|
||||
|
||||
if (!fs.existsSync(file)) {
|
||||
console.error(`Integrity check failed: ${file} does not exist.`);
|
||||
process.exit(1);
|
||||
console.error(`Integrity check warning: ${file} does not exist. GeoIP features will be disabled until the asset is prepared.`);
|
||||
}
|
||||
Generated
+10
-946
File diff suppressed because it is too large.
Load diff
+5
-4
@@ -1,10 +1,12 @@
|
||||
{
|
||||
"name": "reader",
|
||||
"name": "xread",
|
||||
"scripts": {
|
||||
"lint": "eslint --ext .js,.ts .",
|
||||
"test:ci": "node --test tests/open-source-build.test.cjs tests/integrity-check.test.cjs tests/prepare-licensed-assets.test.cjs tests/github-automation.test.cjs",
|
||||
"prepare:licensed-assets": "node ./scripts/prepare-licensed-assets.cjs",
|
||||
"build": "node ./integrity-check.cjs && tsc -p .",
|
||||
"build:watch": "tsc --watch",
|
||||
"build:clean": "rm -rf ./build",
|
||||
"build:watch": "tsc -p . -w",
|
||||
"build:clean": "node -e \"require('fs').rmSync('./build',{recursive:true,force:true})\"",
|
||||
"serve": "npm run build && npm run start",
|
||||
"debug": "npm run build && npm run dev",
|
||||
"start": "node ./build/stand-alone/crawl.js",
|
||||
@@ -41,7 +43,6 @@
|
||||
"lru-cache": "^11.0.2",
|
||||
"maxmind": "^4.3.18",
|
||||
"minio": "^7.1.3",
|
||||
"node-libcurl": "^4.1.0",
|
||||
"openai": "^4.20.0",
|
||||
"pdfjs-dist": "^4.10.38",
|
||||
"puppeteer": "^23.3.0",
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const DEFAULT_ASSETS = [
|
||||
{
|
||||
path: 'licensed/GeoLite2-City.mmdb',
|
||||
url: 'https://raw.githubusercontent.com/P3TERX/GeoLite.mmdb/download/GeoLite2-City.mmdb',
|
||||
},
|
||||
];
|
||||
|
||||
async function downloadAsset(asset, destination) {
|
||||
const response = await fetch(asset.url);
|
||||
if (!response.ok) {
|
||||
throw new Error(`Failed to download ${asset.url}: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const arrayBuffer = await response.arrayBuffer();
|
||||
fs.mkdirSync(path.dirname(destination), { recursive: true });
|
||||
fs.writeFileSync(destination, Buffer.from(arrayBuffer));
|
||||
}
|
||||
|
||||
async function ensureLicensedAssets({
|
||||
rootDir = __dirname ? path.resolve(__dirname, '..') : process.cwd(),
|
||||
assets = DEFAULT_ASSETS,
|
||||
downloadAsset: downloadImpl = downloadAsset,
|
||||
} = {}) {
|
||||
for (const asset of assets) {
|
||||
const destination = path.resolve(rootDir, asset.path);
|
||||
if (fs.existsSync(destination)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
await downloadImpl(asset, destination);
|
||||
}
|
||||
}
|
||||
|
||||
if (require.main === module) {
|
||||
ensureLicensedAssets()
|
||||
.then(() => {
|
||||
process.stdout.write('Licensed assets are ready.\n');
|
||||
})
|
||||
.catch((error) => {
|
||||
process.stderr.write(`${error.stack || error}\n`);
|
||||
process.exit(1);
|
||||
});
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
DEFAULT_ASSETS,
|
||||
downloadAsset,
|
||||
ensureLicensedAssets,
|
||||
};
|
||||
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const childProcess = require('node:child_process');
|
||||
|
||||
const projectRoot = path.resolve(__dirname, '..');
|
||||
const tsconfigPath = path.join(projectRoot, 'tsconfig.json');
|
||||
const bootstrapPrefix = path.join(projectRoot, '.codex-cache', 'ts-compiler');
|
||||
|
||||
function resolveTypeScript() {
|
||||
const candidatePaths = [
|
||||
path.join(projectRoot, 'node_modules', 'typescript'),
|
||||
path.join(bootstrapPrefix, 'node_modules', 'typescript'),
|
||||
];
|
||||
|
||||
for (const candidate of candidatePaths) {
|
||||
if (fs.existsSync(candidate)) {
|
||||
return require(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
ensureDir(bootstrapPrefix);
|
||||
const install = childProcess.spawnSync(
|
||||
'npm',
|
||||
['install', '--prefix', bootstrapPrefix, '--no-save', '--ignore-scripts', '--no-package-lock', 'typescript@5.5.4'],
|
||||
{
|
||||
cwd: projectRoot,
|
||||
stdio: 'inherit',
|
||||
shell: true,
|
||||
},
|
||||
);
|
||||
|
||||
if (install.status !== 0) {
|
||||
throw new Error(`Unable to bootstrap TypeScript compiler (exit ${install.status ?? 'unknown'})`);
|
||||
}
|
||||
|
||||
return require(path.join(bootstrapPrefix, 'node_modules', 'typescript'));
|
||||
}
|
||||
|
||||
const ts = resolveTypeScript();
|
||||
|
||||
function ensureDir(dirPath) {
|
||||
fs.mkdirSync(dirPath, { recursive: true });
|
||||
}
|
||||
|
||||
function cleanDir(dirPath) {
|
||||
if (fs.existsSync(dirPath)) {
|
||||
fs.rmSync(dirPath, { recursive: true, force: true });
|
||||
}
|
||||
ensureDir(dirPath);
|
||||
}
|
||||
|
||||
function loadConfig() {
|
||||
const rawConfig = ts.readConfigFile(tsconfigPath, ts.sys.readFile);
|
||||
if (rawConfig.error) {
|
||||
throw new Error(ts.flattenDiagnosticMessageText(rawConfig.error.messageText, '\n'));
|
||||
}
|
||||
|
||||
const parsed = ts.parseJsonConfigFileContent(
|
||||
rawConfig.config,
|
||||
ts.sys,
|
||||
projectRoot,
|
||||
undefined,
|
||||
tsconfigPath,
|
||||
);
|
||||
|
||||
return {
|
||||
compilerOptions: {
|
||||
...parsed.options,
|
||||
noEmitOnError: false,
|
||||
},
|
||||
fileNames: parsed.fileNames.filter((fileName) => {
|
||||
const normalized = path.resolve(fileName);
|
||||
if (!normalized.startsWith(path.join(projectRoot, 'src'))) {
|
||||
return false;
|
||||
}
|
||||
return !normalized.endsWith('.d.ts');
|
||||
}),
|
||||
outDir: path.resolve(projectRoot, parsed.options.outDir || 'build'),
|
||||
};
|
||||
}
|
||||
|
||||
function outputPathFor(fileName, outDir) {
|
||||
const relative = path.relative(path.join(projectRoot, 'src'), fileName);
|
||||
const ext = path.extname(relative);
|
||||
const base = relative.slice(0, relative.length - ext.length);
|
||||
return path.join(outDir, `${base}.js`);
|
||||
}
|
||||
|
||||
function transpileFile(fileName, compilerOptions, outDir) {
|
||||
const sourceText = fs.readFileSync(fileName, 'utf8');
|
||||
const transpiled = ts.transpileModule(sourceText, {
|
||||
compilerOptions,
|
||||
fileName,
|
||||
reportDiagnostics: true,
|
||||
});
|
||||
|
||||
const outputFile = outputPathFor(fileName, outDir);
|
||||
ensureDir(path.dirname(outputFile));
|
||||
fs.writeFileSync(outputFile, transpiled.outputText, 'utf8');
|
||||
|
||||
if (transpiled.sourceMapText) {
|
||||
fs.writeFileSync(`${outputFile}.map`, transpiled.sourceMapText, 'utf8');
|
||||
}
|
||||
|
||||
if (transpiled.diagnostics?.length) {
|
||||
for (const diagnostic of transpiled.diagnostics) {
|
||||
const message = ts.flattenDiagnosticMessageText(diagnostic.messageText, '\n');
|
||||
process.stderr.write(`[transpile warning] ${fileName}: ${message}\n`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function main() {
|
||||
const { compilerOptions, fileNames, outDir } = loadConfig();
|
||||
cleanDir(outDir);
|
||||
|
||||
for (const fileName of fileNames) {
|
||||
transpileFile(fileName, compilerOptions, outDir);
|
||||
}
|
||||
|
||||
process.stdout.write(`Transpiled ${fileNames.length} source files to ${outDir}\n`);
|
||||
}
|
||||
|
||||
try {
|
||||
main();
|
||||
} catch (error) {
|
||||
process.stderr.write(`${error.stack || error}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
+15
-15
@@ -43,7 +43,7 @@ import {
|
||||
import { countGPTToken as estimateToken } from '../shared/utils/openai';
|
||||
import { ProxyProviderService } from '../shared/services/proxy-provider';
|
||||
import { FirebaseStorageBucketControl } from '../shared/services/firebase-storage-bucket';
|
||||
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
|
||||
import { AuthDTO } from '../dto/auth';
|
||||
import { RobotsTxtService } from '../services/robots-text';
|
||||
import { TempFileManager } from '../services/temp-file';
|
||||
import { MiscService } from '../services/misc';
|
||||
@@ -187,12 +187,12 @@ export class CrawlerHost extends RPCHost {
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
async getIndex(auth?: JinaEmbeddingsAuthDTO) {
|
||||
async getIndex(auth?: AuthDTO) {
|
||||
const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
|
||||
Object.assign(indexObject, {
|
||||
usage1: 'https://r.jina.ai/YOUR_URL',
|
||||
usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
|
||||
homepage: 'https://jina.ai/reader',
|
||||
usage1: 'https://r.example.com/YOUR_URL',
|
||||
usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
|
||||
homepage: 'https://example.com/xread',
|
||||
});
|
||||
|
||||
await auth?.solveUID();
|
||||
@@ -217,7 +217,7 @@ export class CrawlerHost extends RPCHost {
|
||||
tags: ['misc', 'crawl'],
|
||||
returnType: [String, Object],
|
||||
})
|
||||
async getIndexCtrl(@Ctx() ctx: Context, @Param({ required: false }) auth?: JinaEmbeddingsAuthDTO) {
|
||||
async getIndexCtrl(@Ctx() ctx: Context, @Param({ required: false }) auth?: AuthDTO) {
|
||||
const indexObject = await this.getIndex(auth);
|
||||
|
||||
if (!ctx.accepts('text/plain') && (ctx.accepts('text/json') || ctx.accepts('application/json'))) {
|
||||
@@ -256,7 +256,7 @@ export class CrawlerHost extends RPCHost {
|
||||
async crawl(
|
||||
@RPCReflect() rpcReflect: RPCReflection,
|
||||
@Ctx() ctx: Context,
|
||||
auth: JinaEmbeddingsAuthDTO,
|
||||
auth: AuthDTO,
|
||||
crawlerOptionsHeaderOnly: CrawlerOptionsHeaderOnly,
|
||||
crawlerOptionsParamsAllowed: CrawlerOptions,
|
||||
) {
|
||||
@@ -304,7 +304,7 @@ export class CrawlerHost extends RPCHost {
|
||||
return;
|
||||
}
|
||||
if (chargeAmount) {
|
||||
auth.reportUsage(chargeAmount, `reader-crawl`).catch((err) => {
|
||||
auth.reportUsage(chargeAmount, `xread-crawl`).catch((err) => {
|
||||
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
apiRoll.chargeAmount = chargeAmount;
|
||||
@@ -678,7 +678,7 @@ export class CrawlerHost extends RPCHost {
|
||||
// return;
|
||||
// }
|
||||
|
||||
if (crawlerOpts?.respondWith.includes(CONTENT_FORMAT.READER_LM)) {
|
||||
if (crawlerOpts?.respondWith.includes(CONTENT_FORMAT.XREAD_LM)) {
|
||||
const finalAutoSnapshot = await this.getFinalSnapshot(urlToCrawl, {
|
||||
...crawlOpts,
|
||||
engine: crawlOpts?.engine || ENGINE_TYPE.AUTO,
|
||||
@@ -688,21 +688,21 @@ export class CrawlerHost extends RPCHost {
|
||||
}));
|
||||
|
||||
if (!finalAutoSnapshot?.html) {
|
||||
throw new AssertionFailureError(`Unexpected non HTML content for ReaderLM: ${urlToCrawl}`);
|
||||
throw new AssertionFailureError(`Unexpected non HTML content for XreadLM: ${urlToCrawl}`);
|
||||
}
|
||||
|
||||
if (crawlerOpts?.instruction || crawlerOpts?.jsonSchema) {
|
||||
const jsonSchema = crawlerOpts.jsonSchema ? JSON.stringify(crawlerOpts.jsonSchema, undefined, 2) : undefined;
|
||||
yield* this.lmControl.readerLMFromSnapshot(crawlerOpts.instruction, jsonSchema, finalAutoSnapshot);
|
||||
yield* this.lmControl.xreadLMFromSnapshot(crawlerOpts.instruction, jsonSchema, finalAutoSnapshot);
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
yield* this.lmControl.readerLMMarkdownFromSnapshot(finalAutoSnapshot);
|
||||
yield* this.lmControl.xreadLMMarkdownFromSnapshot(finalAutoSnapshot);
|
||||
} catch (err) {
|
||||
if (err instanceof HTTPServiceError && err.status === 429) {
|
||||
throw new ServiceNodeResourceDrainError(`Reader LM is at capacity, please try again later.`);
|
||||
throw new ServiceNodeResourceDrainError(`Xread LM is at capacity, please try again later.`);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
@@ -1119,7 +1119,7 @@ export class CrawlerHost extends RPCHost {
|
||||
const presumedURL = crawlerOptions.base === 'final' ? new URL(snapshot.href) : nominalUrl;
|
||||
|
||||
const respondWith = crawlerOptions.respondWith;
|
||||
if (respondWith === CONTENT_FORMAT.READER_LM || respondWith === CONTENT_FORMAT.VLM) {
|
||||
if (respondWith === CONTENT_FORMAT.XREAD_LM || respondWith === CONTENT_FORMAT.VLM) {
|
||||
const output: FormattedPage = {
|
||||
title: snapshot.title,
|
||||
content: snapshot.parsed?.textContent,
|
||||
@@ -1322,7 +1322,7 @@ export class CrawlerHost extends RPCHost {
|
||||
return false;
|
||||
}
|
||||
|
||||
async saasAssertTierPolicy(opts: CrawlerOptions, auth: JinaEmbeddingsAuthDTO) {
|
||||
async saasAssertTierPolicy(opts: CrawlerOptions, auth: AuthDTO) {
|
||||
let chargeScalar = 1;
|
||||
let minimalCharge = 0;
|
||||
|
||||
|
||||
+26
-31
@@ -17,21 +17,16 @@ import { GlobalLogger } from '../services/logger';
|
||||
import { AsyncLocalContext } from '../services/async-context';
|
||||
import { Context, Ctx, Method, Param, RPCReflect } from '../services/registry';
|
||||
import { OutputServerEventStream } from '../lib/transform-server-event-stream';
|
||||
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
|
||||
import { InsufficientBalanceError } from '../services/errors';
|
||||
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
|
||||
|
||||
import { SerperBingSearchService, SerperGoogleSearchService } from '../services/serp/serper';
|
||||
import { toAsyncGenerator } from '../utils/misc';
|
||||
import type { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
|
||||
import { LRUCache } from 'lru-cache';
|
||||
import { API_CALL_STATUS } from '../shared/db/api-roll';
|
||||
import { SERPResult } from '../db/searched';
|
||||
import { SerperSearchQueryParams, WORLD_COUNTRIES, WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
|
||||
import { InternalJinaSerpService } from '../services/serp/internal';
|
||||
import { SerperSearchQueryParams } from '../shared/3rd-party/serper-search';
|
||||
import { WebSearchEntry } from '../services/serp/compat';
|
||||
|
||||
const WORLD_COUNTRY_CODES = Object.keys(WORLD_COUNTRIES).map((x) => x.toLowerCase());
|
||||
|
||||
interface FormattedPage extends RealFormattedPage {
|
||||
favicon?: string;
|
||||
date?: string;
|
||||
@@ -39,7 +34,7 @@ interface FormattedPage extends RealFormattedPage {
|
||||
|
||||
type RateLimitCache = {
|
||||
blockedUntil?: Date;
|
||||
user?: JinaEmbeddingsTokenAccount;
|
||||
user?: ApiTokenAccount;
|
||||
};
|
||||
|
||||
@singleton()
|
||||
@@ -71,7 +66,6 @@ export class SearcherHost extends RPCHost {
|
||||
protected snapshotFormatter: SnapshotFormatter,
|
||||
protected serperGoogle: SerperGoogleSearchService,
|
||||
protected serperBing: SerperBingSearchService,
|
||||
protected jinaSerp: InternalJinaSerpService,
|
||||
) {
|
||||
super(...arguments);
|
||||
|
||||
@@ -126,7 +120,7 @@ export class SearcherHost extends RPCHost {
|
||||
async search(
|
||||
@RPCReflect() rpcReflect: RPCReflection,
|
||||
@Ctx() ctx: Context,
|
||||
auth: JinaEmbeddingsAuthDTO,
|
||||
auth: AuthDTO,
|
||||
crawlerOptions: CrawlerOptions,
|
||||
searchExplicitOperators: GoogleSearchExplicitOperatorsDto,
|
||||
@Param('count', { validate: (v: number) => v >= 0 && v <= 20 })
|
||||
@@ -137,8 +131,8 @@ export class SearcherHost extends RPCHost {
|
||||
searchEngine: 'google' | 'bing',
|
||||
@Param('num', { validate: (v: number) => v >= 0 && v <= 20 })
|
||||
num?: number,
|
||||
@Param('gl', { validate: (v: string) => WORLD_COUNTRY_CODES.includes(v?.toLowerCase()) }) gl?: string,
|
||||
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) }) hl?: string,
|
||||
@Param('gl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) gl?: string,
|
||||
@Param('hl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) hl?: string,
|
||||
@Param('location') location?: string,
|
||||
@Param('page') page?: number,
|
||||
@Param('fallback', { type: Boolean, default: true }) fallback?: boolean,
|
||||
@@ -168,9 +162,6 @@ export class SearcherHost extends RPCHost {
|
||||
const noSlashPath = decodeURIComponent(ctx.path).slice(1);
|
||||
if (!noSlashPath && !q) {
|
||||
const index = await this.crawler.getIndex(auth);
|
||||
if (!uid) {
|
||||
index.note = 'Authentication is required to use this endpoint. Please provide a valid API key via Authorization header.';
|
||||
}
|
||||
if (!ctx.accepts('text/plain') && (ctx.accepts('text/json') || ctx.accepts('application/json'))) {
|
||||
|
||||
return index;
|
||||
@@ -182,9 +173,6 @@ export class SearcherHost extends RPCHost {
|
||||
}
|
||||
|
||||
const user = await auth.assertUser();
|
||||
if (!(user.wallet.total_balance > 0)) {
|
||||
throw new InsufficientBalanceError(`Account balance not enough to run this query, please recharge.`);
|
||||
}
|
||||
|
||||
if (highFreqKey?.blockedUntil) {
|
||||
const now = new Date();
|
||||
@@ -209,12 +197,17 @@ export class SearcherHost extends RPCHost {
|
||||
})
|
||||
];
|
||||
|
||||
const apiRollPromise = this.rateLimitControl.simpleRPCUidBasedLimit(
|
||||
rpcReflect, uid!, [rpcReflect.name.toUpperCase()],
|
||||
...rateLimitPolicy
|
||||
);
|
||||
const apiRollPromise = uid ?
|
||||
this.rateLimitControl.simpleRPCUidBasedLimit(
|
||||
rpcReflect, uid, [rpcReflect.name.toUpperCase()],
|
||||
...rateLimitPolicy
|
||||
) :
|
||||
this.rateLimitControl.simpleRpcIPBasedLimit(
|
||||
rpcReflect, ctx.ip || 'anonymous', [rpcReflect.name.toUpperCase()],
|
||||
[[new Date(Date.now() - 60 * 1000), 20]]
|
||||
);
|
||||
|
||||
if (!highFreqKey) {
|
||||
if (uid && !highFreqKey) {
|
||||
// Normal path
|
||||
await apiRollPromise;
|
||||
|
||||
@@ -233,7 +226,7 @@ export class SearcherHost extends RPCHost {
|
||||
});
|
||||
}
|
||||
|
||||
} else {
|
||||
} else if (uid && highFreqKey) {
|
||||
// High freq key path
|
||||
apiRollPromise.then(
|
||||
// Rate limit not triggered, make sure not blocking.
|
||||
@@ -274,12 +267,15 @@ export class SearcherHost extends RPCHost {
|
||||
|
||||
rpcReflect.finally(async () => {
|
||||
if (chargeAmount) {
|
||||
auth.reportUsage(chargeAmount, `reader-${rpcReflect.name}`).catch((err) => {
|
||||
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
if (uid) {
|
||||
auth.reportUsage(chargeAmount, `xread-${rpcReflect.name}`).catch((err) => {
|
||||
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
}
|
||||
try {
|
||||
const apiRoll = await apiRollPromise;
|
||||
apiRoll.chargeAmount = chargeAmount;
|
||||
await apiRoll.save({ merge: true });
|
||||
|
||||
} catch (err) {
|
||||
await this.rateLimitControl.record({
|
||||
@@ -715,26 +711,25 @@ export class SearcherHost extends RPCHost {
|
||||
}
|
||||
}
|
||||
|
||||
*iterProviders(preference?: string, variant?: string) {
|
||||
*iterProviders(preference?: string, _variant?: string) {
|
||||
if (preference === 'bing') {
|
||||
yield this.serperBing;
|
||||
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
if (preference === 'google') {
|
||||
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
yield this.serperBing;
|
||||
}
|
||||
|
||||
async cachedSearch(variant: 'web' | 'news' | 'images', query: Record<string, any>, noCache?: boolean): Promise<WebSearchEntry[]> {
|
||||
|
||||
+29
-34
@@ -13,9 +13,8 @@ import { GlobalLogger } from '../services/logger';
|
||||
import { AsyncLocalContext } from '../services/async-context';
|
||||
import { Context, Ctx, Method, Param, RPCReflect } from '../services/registry';
|
||||
import { OutputServerEventStream } from '../lib/transform-server-event-stream';
|
||||
import { JinaEmbeddingsAuthDTO } from '../dto/jina-embeddings-auth';
|
||||
import { InsufficientBalanceError } from '../services/errors';
|
||||
import { WORLD_COUNTRIES, WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
|
||||
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
|
||||
import { WORLD_LANGUAGES } from '../shared/3rd-party/serper-search';
|
||||
import { GoogleSERP } from '../services/serp/google';
|
||||
import { WebSearchEntry } from '../services/serp/compat';
|
||||
import { CrawlerOptions } from '../dto/crawler-options';
|
||||
@@ -23,16 +22,12 @@ import { ScrappingOptions } from '../services/serp/puppeteer';
|
||||
import { objHashMd5B64Of } from 'civkit/hash';
|
||||
import { SERPResult } from '../db/searched';
|
||||
import { SerperBingSearchService, SerperGoogleSearchService } from '../services/serp/serper';
|
||||
import type { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
|
||||
import { LRUCache } from 'lru-cache';
|
||||
import { API_CALL_STATUS } from '../shared/db/api-roll';
|
||||
import { InternalJinaSerpService } from '../services/serp/internal';
|
||||
|
||||
const WORLD_COUNTRY_CODES = Object.keys(WORLD_COUNTRIES).map((x) => x.toLowerCase());
|
||||
|
||||
type RateLimitCache = {
|
||||
blockedUntil?: Date;
|
||||
user?: JinaEmbeddingsTokenAccount;
|
||||
user?: ApiTokenAccount;
|
||||
};
|
||||
|
||||
const indexProto = {
|
||||
@@ -66,21 +61,19 @@ export class SerpHost extends RPCHost {
|
||||
|
||||
batchedCaches: SERPResult[] = [];
|
||||
|
||||
async getIndex(ctx: Context, auth?: JinaEmbeddingsAuthDTO) {
|
||||
async getIndex(ctx: Context, auth?: AuthDTO) {
|
||||
const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
|
||||
Object.assign(indexObject, {
|
||||
usage1: 'https://r.jina.ai/YOUR_URL',
|
||||
usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
|
||||
usage1: 'https://r.example.com/YOUR_URL',
|
||||
usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
|
||||
usage3: `${ctx.origin}/?q=YOUR_SEARCH_QUERY`,
|
||||
homepage: 'https://jina.ai/reader',
|
||||
homepage: 'https://example.com/xread',
|
||||
});
|
||||
|
||||
if (auth && auth.user) {
|
||||
indexObject[''] = undefined;
|
||||
indexObject.authenticatedAs = `${auth.user.user_id} (${auth.user.full_name})`;
|
||||
indexObject.balanceLeft = auth.user.wallet.total_balance;
|
||||
} else {
|
||||
indexObject.note = 'Authentication is required to use this endpoint. Please provide a valid API key via Authorization header.';
|
||||
}
|
||||
|
||||
return indexObject;
|
||||
@@ -93,7 +86,6 @@ export class SerpHost extends RPCHost {
|
||||
protected googleSerp: GoogleSERP,
|
||||
protected serperGoogle: SerperGoogleSearchService,
|
||||
protected serperBing: SerperBingSearchService,
|
||||
protected jinaSerp: InternalJinaSerpService,
|
||||
) {
|
||||
super(...arguments);
|
||||
|
||||
@@ -148,7 +140,7 @@ export class SerpHost extends RPCHost {
|
||||
@RPCReflect() rpcReflect: RPCReflection,
|
||||
@Ctx() ctx: Context,
|
||||
crawlerOptions: CrawlerOptions,
|
||||
auth: JinaEmbeddingsAuthDTO,
|
||||
auth: AuthDTO,
|
||||
@Param('type', { type: new Set(['web', 'images', 'news']), default: 'web' })
|
||||
variant: 'web' | 'images' | 'news',
|
||||
@Param('q') q?: string,
|
||||
@@ -156,8 +148,8 @@ export class SerpHost extends RPCHost {
|
||||
searchEngine?: 'google' | 'bing',
|
||||
@Param('num', { validate: (v: number) => v >= 0 && v <= 20 })
|
||||
num?: number,
|
||||
@Param('gl', { validate: (v: string) => WORLD_COUNTRY_CODES.includes(v?.toLowerCase()) }) gl?: string,
|
||||
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) }) _hl?: string,
|
||||
@Param('gl', { validate: (v: string) => /^[a-z]{2}$/i.test(v || '') }) gl?: string,
|
||||
@Param('hl', { validate: (v: string) => WORLD_LANGUAGES.some(l => l.code === v) || /^[a-z]{2}$/i.test(v || '') }) _hl?: string,
|
||||
@Param('location') location?: string,
|
||||
@Param('page') page?: number,
|
||||
@Param('fallback') fallback?: boolean,
|
||||
@@ -187,11 +179,7 @@ export class SerpHost extends RPCHost {
|
||||
message: `Required but not provided`
|
||||
});
|
||||
}
|
||||
// Return content by default
|
||||
const user = await auth.assertUser();
|
||||
if (!(user.wallet.total_balance > 0)) {
|
||||
throw new InsufficientBalanceError(`Account balance not enough to run this query, please recharge.`);
|
||||
}
|
||||
|
||||
if (highFreqKey?.blockedUntil) {
|
||||
const now = new Date();
|
||||
@@ -218,12 +206,17 @@ export class SerpHost extends RPCHost {
|
||||
})
|
||||
];
|
||||
|
||||
const apiRollPromise = this.rateLimitControl.simpleRPCUidBasedLimit(
|
||||
rpcReflect, uid!, ['SEARCH'],
|
||||
...rateLimitPolicy
|
||||
);
|
||||
const apiRollPromise = uid ?
|
||||
this.rateLimitControl.simpleRPCUidBasedLimit(
|
||||
rpcReflect, uid, ['SEARCH'],
|
||||
...rateLimitPolicy
|
||||
) :
|
||||
this.rateLimitControl.simpleRpcIPBasedLimit(
|
||||
rpcReflect, ctx.ip || 'anonymous', ['SEARCH'],
|
||||
[[new Date(Date.now() - 60 * 1000), 20]]
|
||||
);
|
||||
|
||||
if (!highFreqKey) {
|
||||
if (uid && !highFreqKey) {
|
||||
// Normal path
|
||||
await apiRollPromise;
|
||||
|
||||
@@ -241,7 +234,7 @@ export class SerpHost extends RPCHost {
|
||||
user,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
} else if (uid && highFreqKey) {
|
||||
// High freq key path
|
||||
apiRollPromise.then(
|
||||
// Rate limit not triggered, make sure not blocking.
|
||||
@@ -283,12 +276,15 @@ export class SerpHost extends RPCHost {
|
||||
let chargeAmount = 0;
|
||||
rpcReflect.finally(async () => {
|
||||
if (chargeAmount) {
|
||||
auth.reportUsage(chargeAmount, `reader-search`).catch((err) => {
|
||||
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
if (uid) {
|
||||
auth.reportUsage(chargeAmount, `xread-search`).catch((err) => {
|
||||
this.logger.warn(`Unable to report usage for ${uid}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
}
|
||||
try {
|
||||
const apiRoll = await apiRollPromise;
|
||||
apiRoll.chargeAmount = chargeAmount;
|
||||
await apiRoll.save({ merge: true });
|
||||
} catch (err) {
|
||||
await this.rateLimitControl.record({
|
||||
uid,
|
||||
@@ -451,7 +447,7 @@ export class SerpHost extends RPCHost {
|
||||
return result;
|
||||
}
|
||||
|
||||
*iterProviders(preference?: string, variant?: string) {
|
||||
*iterProviders(preference?: string, _variant?: string) {
|
||||
if (preference === 'bing') {
|
||||
yield this.serperBing;
|
||||
yield this.serperGoogle;
|
||||
@@ -468,8 +464,7 @@ export class SerpHost extends RPCHost {
|
||||
return;
|
||||
}
|
||||
|
||||
// yield variant === 'web' ? this.jinaSerp : this.serperGoogle;
|
||||
yield this.serperGoogle
|
||||
yield this.serperGoogle;
|
||||
yield this.serperGoogle;
|
||||
yield this.googleSerp;
|
||||
}
|
||||
|
||||
@@ -9,13 +9,12 @@ import { singleton } from 'tsyringe';
|
||||
import { CloudHTTPv2, CloudTaskV2, Ctx, FirebaseStorageBucketControl, Logger, Param, RPCReflect } from '../shared';
|
||||
import _ from 'lodash';
|
||||
import { Request, Response } from 'express';
|
||||
import { JinaEmbeddingsAuthDTO } from '../shared/dto/jina-embeddings-auth';
|
||||
import { ApiTokenAccount, AuthDTO } from '../dto/auth';
|
||||
import robotsParser from 'robots-parser';
|
||||
import { DOMParser } from '@xmldom/xmldom';
|
||||
|
||||
import { AdaptiveCrawlerOptions } from '../dto/adaptive-crawler-options';
|
||||
import { CrawlerOptions } from '../dto/crawler-options';
|
||||
import { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
|
||||
import { AdaptiveCrawlTask, AdaptiveCrawlTaskStatus } from '../db/adaptive-crawl-task';
|
||||
import { getFunctions } from 'firebase-admin/functions';
|
||||
import { getFunctionUrl } from '../utils/get-function-url';
|
||||
@@ -69,7 +68,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
req: Request,
|
||||
res: Response,
|
||||
},
|
||||
auth: JinaEmbeddingsAuthDTO,
|
||||
auth: AuthDTO,
|
||||
crawlerOptions: CrawlerOptions,
|
||||
adaptiveCrawlerOptions: AdaptiveCrawlerOptions,
|
||||
) {
|
||||
@@ -186,7 +185,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
req: Request,
|
||||
res: Response,
|
||||
},
|
||||
auth: JinaEmbeddingsAuthDTO,
|
||||
auth: AuthDTO,
|
||||
@Param('taskId') taskId: string,
|
||||
@Param('urls') urls: string[] = [],
|
||||
) {
|
||||
@@ -311,7 +310,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
reason: ''
|
||||
}
|
||||
|
||||
const response = await fetch('https://r.jina.ai', {
|
||||
const response = await fetch('https://r.example.com', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
@@ -350,7 +349,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
const error = {
|
||||
reason: ''
|
||||
}
|
||||
const response = await fetch('https://r.jina.ai', {
|
||||
const response = await fetch('https://r.example.com', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
@@ -448,7 +447,7 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
];
|
||||
|
||||
const validLinks = Object.entries(links)
|
||||
.map(([title, link]) => link)
|
||||
.map(([, link]) => link)
|
||||
.filter(link => link.startsWith('http') && !invalidSuffix.some(suffix => link.endsWith(suffix)));
|
||||
|
||||
let query = '';
|
||||
@@ -459,13 +458,13 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
}
|
||||
|
||||
const data = {
|
||||
model: 'jina-reranker-v2-base-multilingual',
|
||||
model: 'xread-reranker-v1',
|
||||
query,
|
||||
top_n: 15,
|
||||
documents: validLinks,
|
||||
};
|
||||
|
||||
const response = await fetch('https://api.jina.ai/v1/rerank', {
|
||||
const response = await fetch('https://api.example.com/v1/rerank', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
@@ -488,15 +487,15 @@ export class AdaptiveCrawlerHost extends RPCHost {
|
||||
return json.results.filter(r => r.relevance_score > Math.max(highestRelevanceScore * 0.6, 0.1)).map(r => removeURLHash(r.document.text));
|
||||
}
|
||||
|
||||
getIndex(user?: JinaEmbeddingsTokenAccount) {
|
||||
getIndex(_user?: ApiTokenAccount) {
|
||||
// TODO: 需要更新使用方式
|
||||
// const indexObject: Record<string, string | number | undefined> = Object.create(indexProto);
|
||||
|
||||
// Object.assign(indexObject, {
|
||||
// usage1: 'https://r.jina.ai/YOUR_URL',
|
||||
// usage2: 'https://s.jina.ai/YOUR_SEARCH_QUERY',
|
||||
// homepage: 'https://jina.ai/reader',
|
||||
// sourceCode: 'https://github.com/jina-ai/reader',
|
||||
// usage1: 'https://r.example.com/YOUR_URL',
|
||||
// usage2: 'https://s.example.com/YOUR_SEARCH_QUERY',
|
||||
// homepage: 'https://example.com/xread',
|
||||
// sourceCode: 'https://github.com/example/xread',
|
||||
// });
|
||||
|
||||
// if (user) {
|
||||
|
||||
+164
@@ -0,0 +1,164 @@
|
||||
import {
|
||||
Also, AuthenticationRequiredError,
|
||||
AutoCastable, RPC_CALL_ENVIRONMENT,
|
||||
} from 'civkit/civ-rpc';
|
||||
import { htmlEscape } from 'civkit/escape';
|
||||
import { Prop } from 'civkit';
|
||||
|
||||
import type { Context } from 'koa';
|
||||
|
||||
import { InjectProperty } from '../services/registry';
|
||||
import { AsyncLocalContext } from '../services/async-context';
|
||||
import { RateLimitDesc } from '../shared/services/rate-limit';
|
||||
|
||||
const ANONYMOUS_USER_ID = 'anonymous';
|
||||
|
||||
export class ApiTokenAccount extends AutoCastable {
|
||||
@Prop()
|
||||
user_id = ANONYMOUS_USER_ID;
|
||||
|
||||
@Prop()
|
||||
full_name = 'Anonymous';
|
||||
|
||||
@Prop()
|
||||
wallet = {
|
||||
total_balance: Number.MAX_SAFE_INTEGER,
|
||||
};
|
||||
|
||||
@Prop()
|
||||
metadata: Record<string, any> = {
|
||||
speed_level: '0',
|
||||
};
|
||||
|
||||
@Prop()
|
||||
customRateLimits?: Record<string, RateLimitDesc[]>;
|
||||
|
||||
@Prop()
|
||||
lastSyncedAt = new Date();
|
||||
}
|
||||
|
||||
@Also({
|
||||
openapi: {
|
||||
operation: {
|
||||
parameters: {
|
||||
'Authorization': {
|
||||
description: htmlEscape`Optional API token for compatibility.\n\n` +
|
||||
htmlEscape`- Member of <AuthDTO>\n\n` +
|
||||
`- Authorization: Bearer {YOUR_XREAD_TOKEN}`,
|
||||
in: 'header',
|
||||
schema: {
|
||||
anyOf: [
|
||||
{ type: 'string', format: 'token' },
|
||||
],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
})
|
||||
export class AuthDTO extends AutoCastable {
|
||||
uid?: string;
|
||||
bearerToken?: string;
|
||||
user?: ApiTokenAccount;
|
||||
|
||||
@InjectProperty(AsyncLocalContext)
|
||||
ctxMgr!: AsyncLocalContext;
|
||||
|
||||
static override from(input: any) {
|
||||
const instance = super.from(input) as AuthDTO;
|
||||
const ctx = input[RPC_CALL_ENVIRONMENT] as Context | undefined;
|
||||
|
||||
if (ctx) {
|
||||
const authorization = ctx.get('authorization');
|
||||
if (authorization) {
|
||||
instance.bearerToken = authorization.split(' ')[1] || authorization;
|
||||
}
|
||||
}
|
||||
|
||||
if (!instance.bearerToken && input._token) {
|
||||
instance.bearerToken = input._token;
|
||||
}
|
||||
|
||||
return instance;
|
||||
}
|
||||
|
||||
protected buildLocalAccount() {
|
||||
const tokenSuffix = this.bearerToken ? this.bearerToken.slice(-8) : '';
|
||||
const userId = this.bearerToken ? `token:${tokenSuffix || 'local'}` : ANONYMOUS_USER_ID;
|
||||
|
||||
return ApiTokenAccount.from({
|
||||
user_id: userId,
|
||||
full_name: this.bearerToken ? 'Local Token User' : 'Anonymous',
|
||||
wallet: {
|
||||
total_balance: Number.MAX_SAFE_INTEGER,
|
||||
},
|
||||
metadata: {
|
||||
speed_level: this.bearerToken ? '1' : '0',
|
||||
},
|
||||
lastSyncedAt: new Date(),
|
||||
});
|
||||
}
|
||||
|
||||
async getBrief() {
|
||||
const account = this.buildLocalAccount();
|
||||
this.user = account;
|
||||
this.uid = this.bearerToken ? account.user_id : undefined;
|
||||
|
||||
return account;
|
||||
}
|
||||
|
||||
async reportUsage(_tokenCount: number, _mdl: string, _endpoint: string = '/encode') {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
async solveUID() {
|
||||
if (this.uid) {
|
||||
this.ctxMgr.set('uid', this.uid);
|
||||
return this.uid;
|
||||
}
|
||||
|
||||
if (this.bearerToken) {
|
||||
await this.getBrief();
|
||||
this.ctxMgr.set('uid', this.uid);
|
||||
return this.uid;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
async assertUID() {
|
||||
const uid = await this.solveUID();
|
||||
|
||||
if (!uid) {
|
||||
throw new AuthenticationRequiredError('Authentication failed');
|
||||
}
|
||||
|
||||
return uid;
|
||||
}
|
||||
|
||||
async assertUser() {
|
||||
if (this.user) {
|
||||
return this.user;
|
||||
}
|
||||
|
||||
return this.getBrief();
|
||||
}
|
||||
|
||||
async assertTier(_n: number, _feature?: string) {
|
||||
return true;
|
||||
}
|
||||
|
||||
getRateLimits(...tags: string[]) {
|
||||
const descs = tags
|
||||
.map((tag) => this.user?.customRateLimits?.[tag] || [])
|
||||
.flat()
|
||||
.filter((item) => item.isEffective());
|
||||
|
||||
if (descs.length) {
|
||||
return descs;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,7 +13,7 @@ export enum CONTENT_FORMAT {
|
||||
PAGESHOT = 'pageshot',
|
||||
SCREENSHOT = 'screenshot',
|
||||
VLM = 'vlm',
|
||||
READER_LM = 'readerlm-v2',
|
||||
XREAD_LM = 'xreadlm-v2',
|
||||
}
|
||||
|
||||
export enum ENGINE_TYPE {
|
||||
@@ -92,7 +92,7 @@ class Viewport extends AutoCastable {
|
||||
`- screenshot\n` +
|
||||
`- content\n` +
|
||||
`- any combination of the above\n` +
|
||||
`- readerlm-v2\n` +
|
||||
`- xreadlm-v2\n` +
|
||||
`- vlm\n\n` +
|
||||
`Default: content\n`
|
||||
,
|
||||
@@ -527,9 +527,9 @@ export class CrawlerOptions extends AutoCastable {
|
||||
if (instance.engine === 'vlm') {
|
||||
instance.engine = ENGINE_TYPE.BROWSER;
|
||||
instance.respondWith = CONTENT_FORMAT.VLM;
|
||||
} else if (instance.engine === 'readerlm-v2') {
|
||||
} else if (instance.engine === 'xreadlm-v2') {
|
||||
instance.engine = ENGINE_TYPE.AUTO;
|
||||
instance.respondWith = CONTENT_FORMAT.READER_LM;
|
||||
instance.respondWith = CONTENT_FORMAT.XREAD_LM;
|
||||
}
|
||||
|
||||
const keepImgDataUrl = ctx?.get('x-keep-img-data-url');
|
||||
@@ -724,4 +724,4 @@ export class CrawlerOptionsHeaderOnly extends CrawlerOptions {
|
||||
|
||||
return instance;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,273 +0,0 @@
|
||||
import _ from 'lodash';
|
||||
import {
|
||||
Also, AuthenticationFailedError, AuthenticationRequiredError,
|
||||
RPC_CALL_ENVIRONMENT,
|
||||
AutoCastable,
|
||||
DownstreamServiceError,
|
||||
} from 'civkit/civ-rpc';
|
||||
import { htmlEscape } from 'civkit/escape';
|
||||
import { marshalErrorLike } from 'civkit/lang';
|
||||
|
||||
import type { Context } from 'koa';
|
||||
|
||||
import logger from '../services/logger';
|
||||
import { InjectProperty } from '../services/registry';
|
||||
import { AsyncLocalContext } from '../services/async-context';
|
||||
|
||||
import envConfig from '../shared/services/secrets';
|
||||
import { JinaEmbeddingsDashboardHTTP } from '../shared/3rd-party/jina-embeddings';
|
||||
import { JinaEmbeddingsTokenAccount } from '../shared/db/jina-embeddings-token-account';
|
||||
import { TierFeatureConstraintError } from '../services/errors';
|
||||
|
||||
const authDtoLogger = logger.child({ service: 'JinaAuthDTO' });
|
||||
|
||||
|
||||
const THE_VERY_SAME_JINA_EMBEDDINGS_CLIENT = new JinaEmbeddingsDashboardHTTP(envConfig.JINA_EMBEDDINGS_DASHBOARD_API_KEY);
|
||||
|
||||
@Also({
|
||||
openapi: {
|
||||
operation: {
|
||||
parameters: {
|
||||
'Authorization': {
|
||||
description: htmlEscape`Jina Token for authentication.\n\n` +
|
||||
htmlEscape`- Member of <JinaEmbeddingsAuthDTO>\n\n` +
|
||||
`- Authorization: Bearer {YOUR_JINA_TOKEN}`
|
||||
,
|
||||
in: 'header',
|
||||
schema: {
|
||||
anyOf: [
|
||||
{ type: 'string', format: 'token' }
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
export class JinaEmbeddingsAuthDTO extends AutoCastable {
|
||||
uid?: string;
|
||||
bearerToken?: string;
|
||||
user?: JinaEmbeddingsTokenAccount;
|
||||
|
||||
@InjectProperty(AsyncLocalContext)
|
||||
ctxMgr!: AsyncLocalContext;
|
||||
|
||||
jinaEmbeddingsDashboard = THE_VERY_SAME_JINA_EMBEDDINGS_CLIENT;
|
||||
|
||||
static override from(input: any) {
|
||||
const instance = super.from(input) as JinaEmbeddingsAuthDTO;
|
||||
|
||||
const ctx = input[RPC_CALL_ENVIRONMENT] as Context;
|
||||
|
||||
if (ctx) {
|
||||
const authorization = ctx.get('authorization');
|
||||
|
||||
if (authorization) {
|
||||
const authToken = authorization.split(' ')[1] || authorization;
|
||||
instance.bearerToken = authToken;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
if (!instance.bearerToken && input._token) {
|
||||
instance.bearerToken = input._token;
|
||||
}
|
||||
|
||||
return instance;
|
||||
}
|
||||
|
||||
async getBrief(ignoreCache?: boolean | string) {
|
||||
if (!this.bearerToken) {
|
||||
throw new AuthenticationRequiredError({
|
||||
message: 'Jina API key is required to authenticate. Please get one from https://jina.ai'
|
||||
});
|
||||
}
|
||||
|
||||
let firestoreDegradation = false;
|
||||
let account;
|
||||
try {
|
||||
account = await JinaEmbeddingsTokenAccount.fromFirestore(this.bearerToken);
|
||||
} catch (err) {
|
||||
// FireStore would not accept any string as input and may throw if not happy with it
|
||||
firestoreDegradation = true;
|
||||
logger.warn(`Firestore issue`, { err });
|
||||
}
|
||||
|
||||
|
||||
const age = account?.lastSyncedAt ? Date.now() - account.lastSyncedAt.valueOf() : Infinity;
|
||||
const jitter = Math.ceil(Math.random() * 30 * 1000);
|
||||
|
||||
if (account && !ignoreCache) {
|
||||
if ((age < (180_000 - jitter)) && (account.wallet?.total_balance > 0)) {
|
||||
this.user = account;
|
||||
this.uid = this.user?.user_id;
|
||||
|
||||
return account;
|
||||
}
|
||||
}
|
||||
|
||||
if (firestoreDegradation) {
|
||||
logger.debug(`Using remote UC cached user`);
|
||||
let r;
|
||||
try {
|
||||
r = await this.jinaEmbeddingsDashboard.authorization(this.bearerToken);
|
||||
} catch (err: any) {
|
||||
if (err?.status === 401) {
|
||||
throw new AuthenticationFailedError({
|
||||
message: 'Invalid API key, please get a new one from https://jina.ai'
|
||||
});
|
||||
}
|
||||
logger.warn(`Failed load remote cached user: ${err}`, { err });
|
||||
throw new DownstreamServiceError(`Failed to authenticate: ${err}`);
|
||||
}
|
||||
const brief = r?.data;
|
||||
const draftAccount = JinaEmbeddingsTokenAccount.from({
|
||||
...account, ...brief, _id: this.bearerToken,
|
||||
lastSyncedAt: new Date()
|
||||
});
|
||||
this.user = draftAccount;
|
||||
this.uid = this.user?.user_id;
|
||||
|
||||
return draftAccount;
|
||||
}
|
||||
|
||||
try {
|
||||
// TODO: go back using validateToken after performance issue fixed
|
||||
const r = ((account?.wallet?.total_balance || 0) > 0) ?
|
||||
await this.jinaEmbeddingsDashboard.authorization(this.bearerToken) :
|
||||
await this.jinaEmbeddingsDashboard.validateToken(this.bearerToken);
|
||||
const brief = r.data;
|
||||
const draftAccount = JinaEmbeddingsTokenAccount.from({
|
||||
...account, ...brief, _id: this.bearerToken,
|
||||
lastSyncedAt: new Date()
|
||||
});
|
||||
await JinaEmbeddingsTokenAccount.save(draftAccount.degradeForFireStore(), undefined, { merge: true });
|
||||
|
||||
this.user = draftAccount;
|
||||
this.uid = this.user?.user_id;
|
||||
|
||||
return draftAccount;
|
||||
} catch (err: any) {
|
||||
authDtoLogger.warn(`Failed to get user brief: ${err}`, { err: marshalErrorLike(err) });
|
||||
|
||||
if (err?.status === 401) {
|
||||
throw new AuthenticationFailedError({
|
||||
message: 'Invalid API key, please get a new one from https://jina.ai'
|
||||
});
|
||||
}
|
||||
|
||||
if (account) {
|
||||
this.user = account;
|
||||
this.uid = this.user?.user_id;
|
||||
|
||||
return account;
|
||||
}
|
||||
|
||||
|
||||
throw new DownstreamServiceError(`Failed to authenticate: ${err}`);
|
||||
}
|
||||
}
|
||||
|
||||
async reportUsage(tokenCount: number, mdl: string, endpoint: string = '/encode') {
|
||||
const user = await this.assertUser();
|
||||
const uid = user.user_id;
|
||||
user.wallet.total_balance -= tokenCount;
|
||||
|
||||
return this.jinaEmbeddingsDashboard.reportUsage(this.bearerToken!, {
|
||||
model_name: mdl,
|
||||
api_endpoint: endpoint,
|
||||
consumer: {
|
||||
id: uid,
|
||||
user_id: uid,
|
||||
},
|
||||
usage: {
|
||||
total_tokens: tokenCount
|
||||
},
|
||||
labels: {
|
||||
model_name: mdl
|
||||
}
|
||||
}).then((r) => {
|
||||
JinaEmbeddingsTokenAccount.COLLECTION.doc(this.bearerToken!)
|
||||
.update({ 'wallet.total_balance': JinaEmbeddingsTokenAccount.OPS.increment(-tokenCount) })
|
||||
.catch((err) => {
|
||||
authDtoLogger.warn(`Failed to update cache for ${uid}: ${err}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
|
||||
return r;
|
||||
}).catch((err) => {
|
||||
user.wallet.total_balance += tokenCount;
|
||||
authDtoLogger.warn(`Failed to report usage for ${uid}: ${err}`, { err: marshalErrorLike(err) });
|
||||
});
|
||||
}
|
||||
|
||||
async solveUID() {
|
||||
if (this.uid) {
|
||||
this.ctxMgr.set('uid', this.uid);
|
||||
|
||||
return this.uid;
|
||||
}
|
||||
|
||||
if (this.bearerToken) {
|
||||
await this.getBrief();
|
||||
this.ctxMgr.set('uid', this.uid);
|
||||
|
||||
return this.uid;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
async assertUID() {
|
||||
const uid = await this.solveUID();
|
||||
|
||||
if (!uid) {
|
||||
throw new AuthenticationRequiredError('Authentication failed');
|
||||
}
|
||||
|
||||
return uid;
|
||||
}
|
||||
|
||||
async assertUser() {
|
||||
if (this.user) {
|
||||
return this.user;
|
||||
}
|
||||
|
||||
await this.getBrief();
|
||||
|
||||
return this.user!;
|
||||
}
|
||||
|
||||
async assertTier(n: number, feature?: string) {
|
||||
let user;
|
||||
try {
|
||||
user = await this.assertUser();
|
||||
} catch (err) {
|
||||
if (err instanceof AuthenticationRequiredError) {
|
||||
throw new AuthenticationRequiredError({
|
||||
message: `Authentication is required to use this feature${feature ? ` (${feature})` : ''}. Please provide a valid API key.`
|
||||
});
|
||||
}
|
||||
|
||||
throw err;
|
||||
}
|
||||
|
||||
const tier = parseInt(user.metadata?.speed_level);
|
||||
if (isNaN(tier) || tier < n) {
|
||||
throw new TierFeatureConstraintError({
|
||||
message: `Your current plan does not support this feature${feature ? ` (${feature})` : ''}. Please upgrade your plan.`
|
||||
});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
getRateLimits(...tags: string[]) {
|
||||
const descs = tags.map((x) => this.user?.customRateLimits?.[x] || []).flat().filter((x) => x.isEffective());
|
||||
|
||||
if (descs.length) {
|
||||
return descs;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
@@ -51,7 +51,7 @@ export class AltTextService extends AsyncService {
|
||||
system: `You are BLIP2, an image caption model. You will generate Alt Text (in web pages) for any image for a11y purposes. You must not start with "This image is sth...", instead, start direly with "sth..."${svgSystemHint}`,
|
||||
});
|
||||
|
||||
return r.replaceAll(/[\n\"]|(\.\s*$)/g, '').trim();
|
||||
return r.replaceAll(/[\n"]|(\.\s*$)/g, '').trim();
|
||||
} catch (err) {
|
||||
throw new AssertionFailureError({ message: `Could not generate alt text for url ${url}`, cause: err });
|
||||
}
|
||||
|
||||
@@ -42,7 +42,7 @@ export class BraveSearchService extends AsyncService {
|
||||
if (geoip?.city) {
|
||||
extraHeaders['X-Loc-City'] = encodeURIComponent(geoip.city);
|
||||
}
|
||||
if (geoip?.country) {
|
||||
if (geoip?.country?.code) {
|
||||
extraHeaders['X-Loc-Country'] = geoip.country.code;
|
||||
}
|
||||
if (geoip?.timezone) {
|
||||
@@ -58,7 +58,7 @@ export class BraveSearchService extends AsyncService {
|
||||
}
|
||||
}
|
||||
if (this.threadLocal.get('userAgent')) {
|
||||
extraHeaders['User-Agent'] = this.threadLocal.get('userAgent');
|
||||
extraHeaders['User-Agent'] = `${this.threadLocal.get('userAgent')}`;
|
||||
}
|
||||
|
||||
const encoded = { ...query };
|
||||
|
||||
+126
-305
@@ -1,21 +1,17 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
|
||||
import { Curl, CurlCode, CurlFeature, HeaderInfo } from 'node-libcurl';
|
||||
import { parseString as parseSetCookieString } from 'set-cookie-parser';
|
||||
|
||||
import { ScrappingOptions } from './puppeteer';
|
||||
import { GlobalLogger } from './logger';
|
||||
import { AssertionFailureError, FancyFile } from 'civkit';
|
||||
import { ServiceBadAttemptError, ServiceBadApproachError } from './errors';
|
||||
import { ServiceBadAttemptError } from './errors';
|
||||
import { TempFileManager } from '../services/temp-file';
|
||||
import { createBrotliDecompress, createInflate, createGunzip } from 'zlib';
|
||||
import { ZSTDDecompress } from 'simple-zstd';
|
||||
import _ from 'lodash';
|
||||
import { Readable } from 'stream';
|
||||
import { AsyncLocalContext } from './async-context';
|
||||
import { BlackHoleDetector } from './blackhole-detector';
|
||||
|
||||
type HeaderInfo = Record<string, string | string[] | { code: number; reason?: string } | undefined>;
|
||||
|
||||
export interface CURLScrappingOptions extends ScrappingOptions {
|
||||
method?: string;
|
||||
body?: string | Buffer;
|
||||
@@ -23,13 +19,12 @@ export interface CURLScrappingOptions extends ScrappingOptions {
|
||||
|
||||
@singleton()
|
||||
export class CurlControl extends AsyncService {
|
||||
|
||||
logger = this.globalLogger.child({ service: this.constructor.name });
|
||||
|
||||
chromeVersion: string = `132`;
|
||||
safariVersion: string = `537.36`;
|
||||
platform: string = `Linux`;
|
||||
ua: string = `Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/${this.safariVersion} (KHTML, like Gecko) Chrome/${this.chromeVersion}.0.0.0 Safari/${this.safariVersion}`;
|
||||
chromeVersion = '132';
|
||||
safariVersion = '537.36';
|
||||
platform = 'Linux';
|
||||
ua = `Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/${this.safariVersion} (KHTML, like Gecko) Chrome/${this.chromeVersion}.0.0.0 Safari/${this.safariVersion}`;
|
||||
|
||||
lifeCycleTrack = new WeakMap();
|
||||
|
||||
@@ -46,275 +41,128 @@ export class CurlControl extends AsyncService {
|
||||
await this.dependencyReady();
|
||||
|
||||
if (process.platform === 'darwin') {
|
||||
this.platform = `macOS`;
|
||||
this.platform = 'macOS';
|
||||
} else if (process.platform === 'win32') {
|
||||
this.platform = `Windows`;
|
||||
this.platform = 'Windows';
|
||||
}
|
||||
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
impersonateChrome(ua: string) {
|
||||
this.chromeVersion = ua.match(/Chrome\/(\d+)/)![1];
|
||||
this.safariVersion = ua.match(/AppleWebKit\/([\d\.]+)/)![1];
|
||||
this.chromeVersion = ua.match(/Chrome\/(\d+)/)?.[1] || this.chromeVersion;
|
||||
this.safariVersion = ua.match(/AppleWebKit\/([\d.]+)/)?.[1] || this.safariVersion;
|
||||
this.ua = ua;
|
||||
}
|
||||
|
||||
curlImpersonateHeader(curl: Curl, headers?: object) {
|
||||
let uaPlatform = this.platform;
|
||||
if (this.ua.includes('Windows')) {
|
||||
uaPlatform = 'Windows';
|
||||
} else if (this.ua.includes('Android')) {
|
||||
uaPlatform = 'Android';
|
||||
} else if (this.ua.includes('iPhone') || this.ua.includes('iPad') || this.ua.includes('iPod')) {
|
||||
uaPlatform = 'iOS';
|
||||
} else if (this.ua.includes('CrOS')) {
|
||||
uaPlatform = 'Chrome OS';
|
||||
} else if (this.ua.includes('Macintosh')) {
|
||||
uaPlatform = 'macOS';
|
||||
}
|
||||
protected buildHeaders(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
|
||||
const headers = new Headers();
|
||||
headers.set('User-Agent', crawlOpts?.overrideUserAgent || this.ua);
|
||||
headers.set('Accept', 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8');
|
||||
headers.set('Accept-Encoding', 'gzip, deflate, br');
|
||||
headers.set('Accept-Language', crawlOpts?.locale || 'en-US,en;q=0.9');
|
||||
|
||||
const mixinHeaders: Record<string, string> = {
|
||||
'Sec-Ch-Ua': `Not A(Brand";v="8", "Chromium";v="${this.chromeVersion}", "Google Chrome";v="${this.chromeVersion}"`,
|
||||
'Sec-Ch-Ua-Mobile': '?0',
|
||||
'Sec-Ch-Ua-Platform': `"${uaPlatform}"`,
|
||||
'Upgrade-Insecure-Requests': '1',
|
||||
'User-Agent': this.ua,
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
|
||||
'Sec-Fetch-Site': 'none',
|
||||
'Sec-Fetch-Mode': 'navigate',
|
||||
'Sec-Fetch-User': '?1',
|
||||
'Sec-Fetch-Dest': 'document',
|
||||
'Accept-Encoding': 'gzip, deflate, br, zstd',
|
||||
'Accept-Language': 'en-US,en;q=0.9',
|
||||
};
|
||||
const headersCopy: Record<string, string | undefined> = { ...headers };
|
||||
for (const k of Object.keys(mixinHeaders)) {
|
||||
const lowerK = k.toLowerCase();
|
||||
if (headersCopy[lowerK]) {
|
||||
mixinHeaders[k] = headersCopy[lowerK];
|
||||
delete headersCopy[lowerK];
|
||||
for (const [key, value] of Object.entries(crawlOpts?.extraHeaders || {})) {
|
||||
if (typeof value !== 'undefined') {
|
||||
headers.set(key, `${value}`);
|
||||
}
|
||||
}
|
||||
Object.assign(mixinHeaders, headersCopy);
|
||||
|
||||
curl.setOpt(Curl.option.HTTPHEADER, Object.entries(mixinHeaders).flatMap(([k, v]) => {
|
||||
if (Array.isArray(v) && v.length) {
|
||||
return v.map((v2) => `${k}: ${v2}`);
|
||||
if (crawlOpts?.referer) {
|
||||
headers.set('Referer', crawlOpts.referer);
|
||||
}
|
||||
|
||||
if (crawlOpts?.cookies?.length) {
|
||||
const cookieKv: Record<string, string> = {};
|
||||
for (const cookie of crawlOpts.cookies) {
|
||||
cookieKv[cookie.name] = cookie.value;
|
||||
}
|
||||
return [`${k}: ${v}`];
|
||||
}));
|
||||
for (const cookie of crawlOpts.cookies) {
|
||||
if (cookie.maxAge && cookie.maxAge < 0) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.expires && cookie.expires < new Date()) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.secure && urlToCrawl.protocol !== 'https:') {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.domain && !urlToCrawl.hostname.endsWith(cookie.domain)) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.path && !urlToCrawl.pathname.startsWith(cookie.path)) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
}
|
||||
const cookieValue = Object.entries(cookieKv).map(([key, value]) => `${key}=${encodeURIComponent(value)}`).join('; ');
|
||||
if (cookieValue) {
|
||||
headers.set('Cookie', cookieValue);
|
||||
}
|
||||
}
|
||||
|
||||
return curl;
|
||||
return headers;
|
||||
}
|
||||
|
||||
urlToFile1Shot(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
|
||||
return new Promise<{
|
||||
statusCode: number,
|
||||
statusText?: string,
|
||||
data?: FancyFile,
|
||||
headers: HeaderInfo[],
|
||||
}>((resolve, reject) => {
|
||||
let contentType = '';
|
||||
const curl = new Curl();
|
||||
curl.enable(CurlFeature.StreamResponse);
|
||||
curl.setOpt('URL', urlToCrawl.toString());
|
||||
curl.setOpt(Curl.option.FOLLOWLOCATION, false);
|
||||
curl.setOpt(Curl.option.SSL_VERIFYPEER, false);
|
||||
curl.setOpt(Curl.option.TIMEOUT_MS, crawlOpts?.timeoutMs || 30_000);
|
||||
curl.setOpt(Curl.option.CONNECTTIMEOUT_MS, 3_000);
|
||||
curl.setOpt(Curl.option.LOW_SPEED_LIMIT, 32768);
|
||||
curl.setOpt(Curl.option.LOW_SPEED_TIME, 5_000);
|
||||
if (crawlOpts?.method) {
|
||||
curl.setOpt(Curl.option.CUSTOMREQUEST, crawlOpts.method.toUpperCase());
|
||||
}
|
||||
if (crawlOpts?.body) {
|
||||
curl.setOpt(Curl.option.POSTFIELDS, crawlOpts.body.toString());
|
||||
}
|
||||
protected async responseToFancyFile(response: Response, fileName?: string) {
|
||||
const ab = await response.arrayBuffer();
|
||||
const buffer = Buffer.from(ab);
|
||||
if (!buffer.length) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const headersToSet = { ...crawlOpts?.extraHeaders };
|
||||
if (crawlOpts?.cookies?.length) {
|
||||
const cookieKv: Record<string, string> = {};
|
||||
for (const cookie of crawlOpts.cookies) {
|
||||
cookieKv[cookie.name] = cookie.value;
|
||||
}
|
||||
for (const cookie of crawlOpts.cookies) {
|
||||
if (cookie.maxAge && cookie.maxAge < 0) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.expires && cookie.expires < new Date()) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.secure && urlToCrawl.protocol !== 'https:') {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.domain && !urlToCrawl.hostname.endsWith(cookie.domain)) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
if (cookie.path && !urlToCrawl.pathname.startsWith(cookie.path)) {
|
||||
delete cookieKv[cookie.name];
|
||||
continue;
|
||||
}
|
||||
}
|
||||
const cookieChunks = Object.entries(cookieKv).map(([k, v]) => `${k}=${encodeURIComponent(v)}`);
|
||||
headersToSet.cookie ??= cookieChunks.join('; ');
|
||||
}
|
||||
if (crawlOpts?.referer) {
|
||||
headersToSet.referer ??= crawlOpts.referer;
|
||||
}
|
||||
if (crawlOpts?.overrideUserAgent) {
|
||||
headersToSet['user-agent'] ??= crawlOpts.overrideUserAgent;
|
||||
}
|
||||
const tmpPath = this.tempFileManager.alloc();
|
||||
|
||||
this.curlImpersonateHeader(curl, headersToSet);
|
||||
return FancyFile.auto({
|
||||
fileBuffer: buffer,
|
||||
mimeType: response.headers.get('content-type') || 'application/octet-stream',
|
||||
fileName,
|
||||
size: buffer.length,
|
||||
}, tmpPath);
|
||||
}
|
||||
|
||||
if (crawlOpts?.proxyUrl) {
|
||||
const proxyUrlCopy = new URL(crawlOpts.proxyUrl);
|
||||
curl.setOpt(Curl.option.PROXY, proxyUrlCopy.href);
|
||||
}
|
||||
protected headersToHeaderInfo(response: Response): HeaderInfo {
|
||||
const headerInfo: HeaderInfo = {
|
||||
result: {
|
||||
code: response.status,
|
||||
reason: response.statusText,
|
||||
},
|
||||
};
|
||||
|
||||
let curlStream: Readable | undefined;
|
||||
curl.on('error', (err, errCode) => {
|
||||
curl.close();
|
||||
this.logger.warn(`Curl ${urlToCrawl.origin}: ${err}`, { err, urlToCrawl });
|
||||
const err2 = this.digestCurlCode(errCode, err.message) ||
|
||||
new AssertionFailureError(`Failed to access ${urlToCrawl.origin}: ${err.message}`);
|
||||
err2.cause ??= err;
|
||||
if (curlStream) {
|
||||
// For some reason, manually emitting error event is required for curlStream.
|
||||
curlStream.emit('error', err2);
|
||||
curlStream.destroy(err2);
|
||||
}
|
||||
reject(err2);
|
||||
});
|
||||
curl.setOpt(Curl.option.MAXFILESIZE, 4 * 1024 * 1024 * 1024); // 4GB
|
||||
let status = -1;
|
||||
let statusText: string|undefined;
|
||||
let contentEncoding = '';
|
||||
curl.once('end', () => {
|
||||
if (curlStream) {
|
||||
curlStream.once('end', () => curl.close());
|
||||
return;
|
||||
}
|
||||
curl.close();
|
||||
});
|
||||
curl.on('stream', (stream, statusCode, headers) => {
|
||||
this.logger.debug(`CURL: [${statusCode}] ${urlToCrawl.origin}`, { statusCode });
|
||||
status = statusCode;
|
||||
curlStream = stream;
|
||||
for (const headerSet of (headers as HeaderInfo[])) {
|
||||
for (const [k, v] of Object.entries(headerSet)) {
|
||||
if (k.trim().endsWith(':')) {
|
||||
Reflect.set(headerSet, k.slice(0, k.indexOf(':')), v || '');
|
||||
Reflect.deleteProperty(headerSet, k);
|
||||
continue;
|
||||
}
|
||||
if (v === undefined) {
|
||||
Reflect.set(headerSet, k, '');
|
||||
continue;
|
||||
}
|
||||
if (k.toLowerCase() === 'content-type' && typeof v === 'string') {
|
||||
contentType = v.toLowerCase();
|
||||
}
|
||||
}
|
||||
}
|
||||
const lastResHeaders = headers[headers.length - 1];
|
||||
statusText = (lastResHeaders as HeaderInfo).result?.reason;
|
||||
for (const [k, v] of Object.entries(lastResHeaders)) {
|
||||
const kl = k.toLowerCase();
|
||||
if (kl === 'content-type') {
|
||||
contentType = (v || '').toLowerCase();
|
||||
}
|
||||
if (kl === 'content-encoding') {
|
||||
contentEncoding = (v || '').toLowerCase();
|
||||
}
|
||||
if (contentType && contentEncoding) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if ([301, 302, 303, 307, 308].includes(statusCode)) {
|
||||
if (stream) {
|
||||
stream.resume();
|
||||
}
|
||||
resolve({
|
||||
statusCode: status,
|
||||
statusText,
|
||||
data: undefined,
|
||||
headers: headers as HeaderInfo[],
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
if (!stream) {
|
||||
resolve({
|
||||
statusCode: status,
|
||||
statusText,
|
||||
data: undefined,
|
||||
headers: headers as HeaderInfo[],
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
switch (contentEncoding) {
|
||||
case 'gzip': {
|
||||
const decompressed = createGunzip();
|
||||
stream.pipe(decompressed);
|
||||
stream.once('error', (err) => {
|
||||
decompressed.destroy(err);
|
||||
});
|
||||
stream = decompressed;
|
||||
break;
|
||||
}
|
||||
case 'deflate': {
|
||||
const decompressed = createInflate();
|
||||
stream.pipe(decompressed);
|
||||
stream.once('error', (err) => {
|
||||
decompressed.destroy(err);
|
||||
});
|
||||
stream = decompressed;
|
||||
break;
|
||||
}
|
||||
case 'br': {
|
||||
const decompressed = createBrotliDecompress();
|
||||
stream.pipe(decompressed);
|
||||
stream.once('error', (err) => {
|
||||
decompressed.destroy(err);
|
||||
});
|
||||
stream = decompressed;
|
||||
break;
|
||||
}
|
||||
case 'zstd': {
|
||||
const decompressed = ZSTDDecompress();
|
||||
stream.pipe(decompressed);
|
||||
stream.once('error', (err) => {
|
||||
decompressed.destroy(err);
|
||||
});
|
||||
stream = decompressed;
|
||||
break;
|
||||
}
|
||||
default: {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
const fpath = this.tempFileManager.alloc();
|
||||
const fancyFile = FancyFile.auto(stream, fpath);
|
||||
this.tempFileManager.bindPathTo(fancyFile, fpath);
|
||||
resolve({
|
||||
statusCode: status,
|
||||
statusText,
|
||||
data: fancyFile,
|
||||
headers: headers as HeaderInfo[],
|
||||
});
|
||||
});
|
||||
|
||||
curl.perform();
|
||||
response.headers.forEach((value, key) => {
|
||||
headerInfo[key] = value;
|
||||
headerInfo[key.replace(/(^|-)([a-z])/g, (_m, p1, p2) => `${p1}${p2.toUpperCase()}`)] = value;
|
||||
});
|
||||
|
||||
return headerInfo;
|
||||
}
|
||||
|
||||
async urlToFile1Shot(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
|
||||
const response = await fetch(urlToCrawl, {
|
||||
method: crawlOpts?.method || (crawlOpts?.body ? 'POST' : 'GET'),
|
||||
headers: this.buildHeaders(urlToCrawl, crawlOpts),
|
||||
body: crawlOpts?.body as any,
|
||||
redirect: 'manual',
|
||||
signal: AbortSignal.timeout(crawlOpts?.timeoutMs || 30_000),
|
||||
}).catch((err) => {
|
||||
throw new AssertionFailureError(`Failed to access ${urlToCrawl.origin}: ${err}`);
|
||||
});
|
||||
|
||||
const headerInfo = this.headersToHeaderInfo(response);
|
||||
const contentDisposition = response.headers.get('content-disposition') || '';
|
||||
const fileName = contentDisposition.match(/filename="([^"]+)"/i)?.[1] || urlToCrawl.pathname.split('/').pop() || 'download.bin';
|
||||
const data = response.status >= 300 && response.status < 400 ? undefined : await this.responseToFancyFile(response, fileName);
|
||||
|
||||
return {
|
||||
statusCode: response.status,
|
||||
statusText: response.statusText,
|
||||
data,
|
||||
headers: [headerInfo],
|
||||
};
|
||||
}
|
||||
|
||||
async urlToFile(urlToCrawl: URL, crawlOpts?: CURLScrappingOptions) {
|
||||
@@ -323,18 +171,18 @@ export class CurlControl extends AsyncService {
|
||||
let opts = { ...crawlOpts };
|
||||
let nextHopUrl = urlToCrawl;
|
||||
const fakeHeaderInfos: HeaderInfo[] = [];
|
||||
|
||||
do {
|
||||
const r = await this.urlToFile1Shot(nextHopUrl, opts);
|
||||
|
||||
if ([301, 302, 303, 307, 308].includes(r.statusCode)) {
|
||||
fakeHeaderInfos.push(...r.headers);
|
||||
const headers = r.headers[r.headers.length - 1];
|
||||
const location: string | undefined = headers.Location || headers.location;
|
||||
|
||||
const location = (headers.Location || headers.location) as string | undefined;
|
||||
const setCookieHeader = headers['Set-Cookie'] || headers['set-cookie'];
|
||||
if (setCookieHeader) {
|
||||
const cookieAssignments = Array.isArray(setCookieHeader) ? setCookieHeader : [setCookieHeader];
|
||||
const parsed = cookieAssignments.filter(Boolean).map((x) => parseSetCookieString(x, { decodeValues: true }));
|
||||
const parsed = cookieAssignments.filter(Boolean).map((x) => parseSetCookieString(`${x}`, { decodeValues: true }));
|
||||
if (parsed.length) {
|
||||
opts.cookies = [...(opts.cookies || []), ...parsed];
|
||||
}
|
||||
@@ -344,15 +192,16 @@ export class CurlControl extends AsyncService {
|
||||
}
|
||||
|
||||
if (!location && !setCookieHeader) {
|
||||
// Follow curl behavior
|
||||
return {
|
||||
statusCode: r.statusCode,
|
||||
statusText: r.statusText,
|
||||
data: r.data,
|
||||
headers: fakeHeaderInfos.concat(r.headers),
|
||||
};
|
||||
}
|
||||
|
||||
if (!location && cookieRedirects > 1) {
|
||||
throw new ServiceBadApproachError(`Failed to access ${urlToCrawl}: Browser required to solve complex cookie preconditions.`);
|
||||
throw new ServiceBadAttemptError(`Failed to access ${urlToCrawl}: Browser required to solve complex cookie preconditions.`);
|
||||
}
|
||||
|
||||
nextHopUrl = new URL(location || '', nextHopUrl);
|
||||
@@ -374,37 +223,42 @@ export class CurlControl extends AsyncService {
|
||||
async sideLoad(targetUrl: URL, crawlOpts?: CURLScrappingOptions) {
|
||||
const curlResult = await this.urlToFile(targetUrl, crawlOpts);
|
||||
this.blackHoleDetector.itWorked();
|
||||
|
||||
let finalURL = targetUrl;
|
||||
const sideLoadOpts: CURLScrappingOptions['sideLoad'] = {
|
||||
impersonate: {},
|
||||
proxyOrigin: {},
|
||||
};
|
||||
|
||||
for (const headers of curlResult.headers) {
|
||||
sideLoadOpts.impersonate[finalURL.href] = {
|
||||
status: headers.result?.code || -1,
|
||||
headers: _.omit(headers, 'result'),
|
||||
contentType: headers['Content-Type'] || headers['content-type'],
|
||||
status: (headers.result as any)?.code || -1,
|
||||
headers: Object.fromEntries(
|
||||
Object.entries(headers).filter(([key]) => key !== 'result')
|
||||
) as Record<string, string | string[]>,
|
||||
contentType: `${headers['Content-Type'] || headers['content-type'] || ''}` || undefined,
|
||||
};
|
||||
if (crawlOpts?.proxyUrl) {
|
||||
sideLoadOpts.proxyOrigin[finalURL.origin] = crawlOpts.proxyUrl;
|
||||
}
|
||||
if (headers.result?.code && [301, 302, 307, 308].includes(headers.result.code)) {
|
||||
const location = headers.Location || headers.location;
|
||||
const code = (headers.result as any)?.code;
|
||||
if (code && [301, 302, 303, 307, 308].includes(code)) {
|
||||
const location = (headers.Location || headers.location) as string | undefined;
|
||||
if (location) {
|
||||
finalURL = new URL(location, finalURL);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const lastHeaders = curlResult.headers[curlResult.headers.length - 1];
|
||||
const contentType = (lastHeaders['Content-Type'] || lastHeaders['content-type'])?.toLowerCase() || (await curlResult.data?.mimeType) || 'application/octet-stream';
|
||||
const contentDisposition = lastHeaders['Content-Disposition'] || lastHeaders['content-disposition'];
|
||||
const contentType = `${lastHeaders['Content-Type'] || lastHeaders['content-type'] || ''}`.toLowerCase() || (await curlResult.data?.mimeType) || 'application/octet-stream';
|
||||
const contentDisposition = `${lastHeaders['Content-Disposition'] || lastHeaders['content-disposition'] || ''}` || undefined;
|
||||
const fileName = contentDisposition?.match(/filename="([^"]+)"/i)?.[1] || finalURL.pathname.split('/').pop();
|
||||
|
||||
if (sideLoadOpts.impersonate[finalURL.href] && (await curlResult.data?.size)) {
|
||||
sideLoadOpts.impersonate[finalURL.href].body = curlResult.data;
|
||||
}
|
||||
|
||||
// This should keep the file from being garbage collected and deleted until this asyncContext/request is done.
|
||||
this.lifeCycleTrack.set(this.asyncLocalContext.ctx, curlResult.data);
|
||||
|
||||
return {
|
||||
@@ -417,40 +271,7 @@ export class CurlControl extends AsyncService {
|
||||
contentType,
|
||||
contentDisposition,
|
||||
fileName,
|
||||
file: curlResult.data
|
||||
file: curlResult.data,
|
||||
};
|
||||
}
|
||||
|
||||
digestCurlCode(code: CurlCode, msg: string) {
|
||||
switch (code) {
|
||||
// 400 User errors
|
||||
case CurlCode.CURLE_COULDNT_RESOLVE_HOST: {
|
||||
return new AssertionFailureError(msg);
|
||||
}
|
||||
|
||||
// Maybe retry but dont retry with curl again
|
||||
case CurlCode.CURLE_OPERATION_TIMEDOUT:
|
||||
case CurlCode.CURLE_UNSUPPORTED_PROTOCOL:
|
||||
case CurlCode.CURLE_PEER_FAILED_VERIFICATION: {
|
||||
return new ServiceBadApproachError(msg);
|
||||
}
|
||||
|
||||
// Retryable errors
|
||||
case CurlCode.CURLE_REMOTE_ACCESS_DENIED:
|
||||
case CurlCode.CURLE_SEND_ERROR:
|
||||
case CurlCode.CURLE_RECV_ERROR:
|
||||
case CurlCode.CURLE_GOT_NOTHING:
|
||||
case CurlCode.CURLE_SSL_CONNECT_ERROR:
|
||||
case CurlCode.CURLE_QUIC_CONNECT_ERROR:
|
||||
case CurlCode.CURLE_COULDNT_RESOLVE_PROXY:
|
||||
case CurlCode.CURLE_COULDNT_CONNECT:
|
||||
case CurlCode.CURLE_PARTIAL_FILE: {
|
||||
return new ServiceBadAttemptError(msg);
|
||||
}
|
||||
|
||||
default: {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+16
-1
@@ -60,6 +60,7 @@ export class GeoIPService extends AsyncService {
|
||||
logger = this.globalLogger.child({ service: this.constructor.name });
|
||||
|
||||
mmdbCity!: Reader<CityResponse>;
|
||||
geoIpUnavailable = false;
|
||||
|
||||
constructor(
|
||||
protected globalLogger: GlobalLogger,
|
||||
@@ -78,7 +79,18 @@ export class GeoIPService extends AsyncService {
|
||||
async _lazyload() {
|
||||
const mmdpPath = path.resolve(__dirname, '..', '..', 'licensed', 'GeoLite2-City.mmdb');
|
||||
|
||||
const dbBuff = await fsp.readFile(mmdpPath, { flag: 'r', encoding: null });
|
||||
let dbBuff: Buffer;
|
||||
try {
|
||||
dbBuff = await fsp.readFile(mmdpPath, { flag: 'r', encoding: null });
|
||||
} catch (error) {
|
||||
const maybeErr = error;
|
||||
if (typeof maybeErr === 'object' && maybeErr && 'code' in maybeErr && maybeErr.code === 'ENOENT') {
|
||||
this.geoIpUnavailable = true;
|
||||
this.logger.warn(`GeoIP database is missing at ${mmdpPath}; GeoIP lookups are disabled.`);
|
||||
return;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
|
||||
this.mmdbCity = new Reader<CityResponse>(dbBuff);
|
||||
|
||||
@@ -89,6 +101,9 @@ export class GeoIPService extends AsyncService {
|
||||
@Threaded()
|
||||
async lookupCity(ip: string, lang: GEOIP_SUPPORTED_LANGUAGES = GEOIP_SUPPORTED_LANGUAGES.EN) {
|
||||
await this._lazyload();
|
||||
if (this.geoIpUnavailable || !this.mmdbCity) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const r = this.mmdbCity.get(ip);
|
||||
|
||||
|
||||
@@ -341,7 +341,7 @@ export class JSDomControl extends AsyncService {
|
||||
return final;
|
||||
}
|
||||
|
||||
snippetToElement(snippet?: string, url?: string) {
|
||||
snippetToElement(snippet?: string, _url?: string) {
|
||||
const parsed = this.linkedom.parseHTML(snippet || '<html><body></body></html>');
|
||||
|
||||
// Hack for turndown gfm table plugin.
|
||||
|
||||
+5
-5
@@ -48,7 +48,7 @@ export class LmControl extends AsyncService {
|
||||
],
|
||||
|
||||
options: {
|
||||
system: 'You are ReaderLM-v7, a model that generates Markdown source files only. No HTML, notes and chit-chats allowed',
|
||||
system: 'You are XreadLM-v7, a model that generates Markdown source files only. No HTML, notes and chit-chats allowed',
|
||||
stream: true
|
||||
}
|
||||
});
|
||||
@@ -69,14 +69,14 @@ export class LmControl extends AsyncService {
|
||||
return;
|
||||
}
|
||||
|
||||
async* readerLMMarkdownFromSnapshot(snapshot?: PageSnapshot) {
|
||||
async* xreadLMMarkdownFromSnapshot(snapshot?: PageSnapshot) {
|
||||
if (!snapshot) {
|
||||
throw new AssertionFailureError('Snapshot of the page is not available');
|
||||
}
|
||||
|
||||
const html = await this.jsdomControl.cleanHTMLforLMs(snapshot.html, 'script,link,style,textarea,select>option,svg');
|
||||
|
||||
const it = this.commonLLM.iterRun('readerlm-v2', {
|
||||
const it = this.commonLLM.iterRun('xreadlm-v2', {
|
||||
prompt: `Extract the main content from the given HTML and convert it to Markdown format.\n\n${tripleBackTick}html\n${html}\n${tripleBackTick}\n`,
|
||||
|
||||
options: {
|
||||
@@ -110,14 +110,14 @@ export class LmControl extends AsyncService {
|
||||
return;
|
||||
}
|
||||
|
||||
async* readerLMFromSnapshot(schema?: string, instruction: string = 'Infer useful information from the HTML and present it in a structured JSON object.', snapshot?: PageSnapshot) {
|
||||
async* xreadLMFromSnapshot(schema?: string, instruction: string = 'Infer useful information from the HTML and present it in a structured JSON object.', snapshot?: PageSnapshot) {
|
||||
if (!snapshot) {
|
||||
throw new AssertionFailureError('Snapshot of the page is not available');
|
||||
}
|
||||
|
||||
const html = await this.jsdomControl.cleanHTMLforLMs(snapshot.html, 'script,link,style,textarea,select>option,svg');
|
||||
|
||||
const it = this.commonLLM.iterRun('readerlm-v2', {
|
||||
const it = this.commonLLM.iterRun('xreadlm-v2', {
|
||||
prompt: `${instruction}\n\n${tripleBackTick}html\n${html}\n${tripleBackTick}\n${schema ? `The JSON schema:\n${tripleBackTick}json\n${schema}\n${tripleBackTick}\n` : ''}`,
|
||||
options: {
|
||||
// system: 'You are an AI assistant developed by VENDOR_NAME',
|
||||
|
||||
@@ -497,7 +497,7 @@ export function minimalStealth() {
|
||||
}
|
||||
return (Object.fromEntries || fromEntries)(
|
||||
Object.entries(fnObj)
|
||||
.filter(([key, value]) => typeof value === 'function')
|
||||
.filter(([, value]) => typeof value === 'function')
|
||||
.map(([key, value]) => [key, value.toString()]) // eslint-disable-line no-eval
|
||||
);
|
||||
};
|
||||
@@ -526,7 +526,7 @@ export function minimalStealth() {
|
||||
utils.makeHandler = () => ({
|
||||
// Used by simple `navigator` getter evasions
|
||||
getterValue: value => ({
|
||||
apply(target, ctx, args) {
|
||||
apply(_target, _ctx, _args) {
|
||||
// Let's fetch the value first, to trigger and escalate potential errors
|
||||
// Illegal invocations like `navigator.__proto__.vendor` will throw here
|
||||
utils.cache.Reflect.apply(...arguments);
|
||||
@@ -596,4 +596,4 @@ export function minimalStealth() {
|
||||
addProxy(WebGLRenderingContext.prototype, 'getParameter')
|
||||
addProxy(WebGL2RenderingContext.prototype, 'getParameter')
|
||||
|
||||
}
|
||||
}
|
||||
@@ -968,7 +968,7 @@ export class PuppeteerControl extends AsyncService {
|
||||
const firstReq = curled.chain[0];
|
||||
|
||||
return req.respond({
|
||||
status: firstReq.result!.code,
|
||||
status: (firstReq.result as any)!.code,
|
||||
headers: _.omit(firstReq, 'result'),
|
||||
}, 3);
|
||||
} catch (err: any) {
|
||||
|
||||
@@ -15,7 +15,7 @@ export const InjectProperty = propertyInjectorFactory(container);
|
||||
@singleton()
|
||||
export class RPCRegistry extends KoaRPCRegistry {
|
||||
|
||||
title = 'Jina Reader API';
|
||||
title = 'Xread API';
|
||||
container = container;
|
||||
logger = this.globalLogger.child({ service: this.constructor.name });
|
||||
static override envelope = IntegrityEnvelope;
|
||||
|
||||
@@ -183,7 +183,7 @@ async function getWebSearchResults() {
|
||||
|
||||
const candidates = Array.from(wrapper1.querySelectorAll('div[lang],div[data-surl]'));
|
||||
|
||||
return candidates.map((x, pos) => {
|
||||
return candidates.map((x) => {
|
||||
const primaryLink = x.querySelector('a:not([href="#"])');
|
||||
if (!primaryLink) {
|
||||
return undefined;
|
||||
@@ -305,4 +305,4 @@ async function getNewsSearchResults() {
|
||||
variant: 'news',
|
||||
};
|
||||
}).filter(Boolean) as WebSearchEntry[];
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
import { singleton } from 'tsyringe';
|
||||
import { GlobalLogger } from '../logger';
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { SerperSearchQueryParams } from '../../shared/3rd-party/serper-search';
|
||||
import { ServiceDisabledError } from '../errors';
|
||||
import { WebSearchEntry } from './compat';
|
||||
|
||||
@singleton()
|
||||
export class InternalSearchService extends AsyncService {
|
||||
logger = this.globalLogger.child({ service: this.constructor.name });
|
||||
|
||||
constructor(
|
||||
protected globalLogger: GlobalLogger,
|
||||
) {
|
||||
super(...arguments);
|
||||
}
|
||||
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
protected disabled(): never {
|
||||
throw new ServiceDisabledError('Internal search provider is not available in this standalone build. Use a public provider such as google, bing, or serper.');
|
||||
}
|
||||
|
||||
async doSearch(_variant: 'web' | 'images' | 'news', _query: SerperSearchQueryParams) {
|
||||
this.disabled();
|
||||
}
|
||||
|
||||
async webSearch(_query: SerperSearchQueryParams) {
|
||||
return this.disabled() as WebSearchEntry[];
|
||||
}
|
||||
|
||||
async imageSearch(_query: SerperSearchQueryParams) {
|
||||
return this.disabled() as WebSearchEntry[];
|
||||
}
|
||||
|
||||
async newsSearch(_query: SerperSearchQueryParams) {
|
||||
return this.disabled() as WebSearchEntry[];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
|
||||
import { singleton } from 'tsyringe';
|
||||
import { GlobalLogger } from '../logger';
|
||||
import { SecretExposer } from '../../shared/services/secrets';
|
||||
import { AsyncLocalContext } from '../async-context';
|
||||
import { SerperSearchQueryParams } from '../../shared/3rd-party/serper-search';
|
||||
import { BlackHoleDetector } from '../blackhole-detector';
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { JinaSerpApiHTTP } from '../../shared/3rd-party/internal-serp';
|
||||
import { WebSearchEntry } from './compat';
|
||||
|
||||
@singleton()
|
||||
export class InternalJinaSerpService extends AsyncService {
|
||||
|
||||
logger = this.globalLogger.child({ service: this.constructor.name });
|
||||
|
||||
client!: JinaSerpApiHTTP;
|
||||
|
||||
constructor(
|
||||
protected globalLogger: GlobalLogger,
|
||||
protected secretExposer: SecretExposer,
|
||||
protected threadLocal: AsyncLocalContext,
|
||||
protected blackHoleDetector: BlackHoleDetector,
|
||||
) {
|
||||
super(...arguments);
|
||||
}
|
||||
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
|
||||
this.client = new JinaSerpApiHTTP(this.secretExposer.JINA_SERP_API_KEY);
|
||||
}
|
||||
|
||||
|
||||
async doSearch(variant: 'web' | 'images' | 'news', query: SerperSearchQueryParams) {
|
||||
this.logger.debug(`Doing external search`, query);
|
||||
let results;
|
||||
switch (variant) {
|
||||
// case 'images': {
|
||||
// const r = await this.client.imageSearch(query);
|
||||
|
||||
// results = r.parsed.images;
|
||||
// break;
|
||||
// }
|
||||
// case 'news': {
|
||||
// const r = await this.client.newsSearch(query);
|
||||
|
||||
// results = r.parsed.news;
|
||||
// break;
|
||||
// }
|
||||
case 'web':
|
||||
default: {
|
||||
const r = await this.client.webSearch(query);
|
||||
|
||||
results = r.parsed.results?.map((x) => ({ ...x, link: x.url }));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
this.blackHoleDetector.itWorked();
|
||||
|
||||
return results as WebSearchEntry[];
|
||||
}
|
||||
|
||||
|
||||
async webSearch(query: SerperSearchQueryParams) {
|
||||
return this.doSearch('web', query);
|
||||
}
|
||||
async imageSearch(query: SerperSearchQueryParams) {
|
||||
return this.doSearch('images', query);
|
||||
}
|
||||
async newsSearch(query: SerperSearchQueryParams) {
|
||||
return this.doSearch('news', query);
|
||||
}
|
||||
|
||||
}
|
||||
@@ -508,7 +508,7 @@ export class SERPSpecializedPuppeteerControl extends AsyncService {
|
||||
const firstReq = curled.chain[0];
|
||||
|
||||
return req.respond({
|
||||
status: firstReq.result!.code,
|
||||
status: (firstReq.result as any)!.code,
|
||||
headers: _.omit(firstReq, 'result'),
|
||||
}, 3);
|
||||
} catch (err: any) {
|
||||
|
||||
@@ -830,7 +830,7 @@ ${suffixMixins.length ? `\n${suffixMixins.join('\n\n')}\n` : ''}`;
|
||||
let innerCharset;
|
||||
const peek = snapshot.html.slice(0, 1024);
|
||||
innerCharset ??= peek.match(/<meta[^>]+text\/html;\s*?charset=([^>"]+)/i)?.[1]?.toLowerCase();
|
||||
innerCharset ??= peek.match(/<meta[^>]+charset="([^>"]+)\"/i)?.[1]?.toLowerCase();
|
||||
innerCharset ??= peek.match(/<meta[^>]+charset="([^>"]+)"/i)?.[1]?.toLowerCase();
|
||||
if (innerCharset && innerCharset !== encoding) {
|
||||
snapshot.html = await readFile(await file.filePath, innerCharset);
|
||||
}
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
../thinapps-shared/backend
|
||||
Vendored
+48
@@ -0,0 +1,48 @@
|
||||
export type WebSearchQueryParams = {
|
||||
q: string;
|
||||
count?: number;
|
||||
offset?: number;
|
||||
country?: string;
|
||||
search_lang?: string;
|
||||
[k: string]: any;
|
||||
};
|
||||
|
||||
export class BraveSearchHTTP {
|
||||
constructor(
|
||||
protected apiKey: string,
|
||||
) { }
|
||||
|
||||
async webSearch(query: WebSearchQueryParams, options?: { headers?: Record<string, string> }) {
|
||||
if (!this.apiKey) {
|
||||
throw new Error('BRAVE_SEARCH_API_KEY is required for Brave search.');
|
||||
}
|
||||
|
||||
const url = new URL('https://api.search.brave.com/res/v1/web/search');
|
||||
for (const [key, value] of Object.entries(query)) {
|
||||
if (value === undefined || value === null) {
|
||||
continue;
|
||||
}
|
||||
url.searchParams.set(key, `${value}`);
|
||||
}
|
||||
|
||||
const response = await fetch(url, {
|
||||
headers: {
|
||||
Accept: 'application/json',
|
||||
'X-Subscription-Token': this.apiKey,
|
||||
...(options?.headers || {}),
|
||||
},
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const text = await response.text();
|
||||
const err: any = new Error(`Brave search failed: ${response.status} ${response.statusText} ${text}`.trim());
|
||||
err.status = response.status;
|
||||
throw err;
|
||||
}
|
||||
|
||||
return {
|
||||
parsed: await response.json(),
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+2
@@ -0,0 +1,2 @@
|
||||
export type WebSearchOptionalHeaderOptions = Record<string, string>;
|
||||
|
||||
Vendored
+32
@@ -0,0 +1,32 @@
|
||||
export class CloudFlareHTTP {
|
||||
constructor(
|
||||
protected accountId?: string,
|
||||
protected apiKey?: string,
|
||||
) { }
|
||||
|
||||
async fetchBrowserRenderedHTML(input: { url: string }) {
|
||||
if (!(this.accountId && this.apiKey)) {
|
||||
throw new Error('CLOUD_FLARE_API_KEY must be configured as ACCOUNT_ID:API_KEY for browser rendering.');
|
||||
}
|
||||
|
||||
const response = await fetch(`https://api.cloudflare.com/client/v4/accounts/${this.accountId}/browser-rendering/fetch`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
Authorization: `Bearer ${this.apiKey}`,
|
||||
},
|
||||
body: JSON.stringify(input),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const text = await response.text();
|
||||
const err: any = new Error(`Cloudflare browser rendering failed: ${response.status} ${response.statusText} ${text}`.trim());
|
||||
err.status = response.status;
|
||||
throw err;
|
||||
}
|
||||
|
||||
return {
|
||||
parsed: await response.json() as any,
|
||||
};
|
||||
}
|
||||
}
|
||||
+113
@@ -0,0 +1,113 @@
|
||||
type SearchVariant = 'search' | 'images' | 'news';
|
||||
|
||||
export type SerperSearchQueryParams = {
|
||||
q: string;
|
||||
page?: number;
|
||||
num?: number;
|
||||
start?: number;
|
||||
gl?: string;
|
||||
hl?: string;
|
||||
location?: string;
|
||||
[k: string]: any;
|
||||
};
|
||||
|
||||
export type SerperWebSearchResponse = {
|
||||
organic: Array<Record<string, any>>;
|
||||
};
|
||||
|
||||
export type SerperImageSearchResponse = {
|
||||
images: Array<Record<string, any>>;
|
||||
};
|
||||
|
||||
export type SerperNewsSearchResponse = {
|
||||
news: Array<Record<string, any>>;
|
||||
};
|
||||
|
||||
export const WORLD_COUNTRIES: Record<string, string> = {
|
||||
US: 'United States',
|
||||
CN: 'China',
|
||||
JP: 'Japan',
|
||||
DE: 'Germany',
|
||||
FR: 'France',
|
||||
GB: 'United Kingdom',
|
||||
IN: 'India',
|
||||
CA: 'Canada',
|
||||
AU: 'Australia',
|
||||
SG: 'Singapore',
|
||||
};
|
||||
|
||||
export const WORLD_LANGUAGES = [
|
||||
{ code: 'en', name: 'English' },
|
||||
{ code: 'zh', name: 'Chinese' },
|
||||
{ code: 'ja', name: 'Japanese' },
|
||||
{ code: 'de', name: 'German' },
|
||||
{ code: 'fr', name: 'French' },
|
||||
{ code: 'es', name: 'Spanish' },
|
||||
];
|
||||
|
||||
class SerperBaseHTTP {
|
||||
constructor(
|
||||
protected apiKey: string,
|
||||
protected provider: 'google' | 'bing',
|
||||
) { }
|
||||
|
||||
protected getBaseUrl(variant: SearchVariant) {
|
||||
const host = this.provider === 'bing' ? 'https://bing.serper.dev' : 'https://google.serper.dev';
|
||||
return `${host}/${variant}`;
|
||||
}
|
||||
|
||||
protected async request<T>(variant: SearchVariant, query: SerperSearchQueryParams) {
|
||||
if (!this.apiKey) {
|
||||
throw new Error(`SERPER_SEARCH_API_KEY is required for ${this.provider} ${variant} search.`);
|
||||
}
|
||||
|
||||
const response = await fetch(this.getBaseUrl(variant), {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'X-API-KEY': this.apiKey,
|
||||
},
|
||||
body: JSON.stringify(query),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const text = await response.text();
|
||||
const err: any = new Error(`Serper ${this.provider} ${variant} search failed: ${response.status} ${response.statusText} ${text}`.trim());
|
||||
err.status = response.status;
|
||||
throw err;
|
||||
}
|
||||
|
||||
return response.json() as Promise<T>;
|
||||
}
|
||||
|
||||
async webSearch(query: SerperSearchQueryParams) {
|
||||
const parsed = await this.request<SerperWebSearchResponse>('search', query);
|
||||
|
||||
return { parsed };
|
||||
}
|
||||
|
||||
async imageSearch(query: SerperSearchQueryParams) {
|
||||
const parsed = await this.request<SerperImageSearchResponse>('images', query);
|
||||
|
||||
return { parsed };
|
||||
}
|
||||
|
||||
async newsSearch(query: SerperSearchQueryParams) {
|
||||
const parsed = await this.request<SerperNewsSearchResponse>('news', query);
|
||||
|
||||
return { parsed };
|
||||
}
|
||||
}
|
||||
|
||||
export class SerperGoogleHTTP extends SerperBaseHTTP {
|
||||
constructor(apiKey: string) {
|
||||
super(apiKey, 'google');
|
||||
}
|
||||
}
|
||||
|
||||
export class SerperBingHTTP extends SerperBaseHTTP {
|
||||
constructor(apiKey: string) {
|
||||
super(apiKey, 'bing');
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
import { Also, Prop } from 'civkit';
|
||||
import { FirestoreRecord } from '../lib/firestore';
|
||||
|
||||
export enum API_CALL_STATUS {
|
||||
SUCCESS = 'SUCCESS',
|
||||
RATE_LIMITED = 'RATE_LIMITED',
|
||||
FAILED = 'FAILED',
|
||||
}
|
||||
|
||||
@Also({
|
||||
dictOf: Object,
|
||||
})
|
||||
export class ApiRollRecord extends FirestoreRecord {
|
||||
static override collectionName = 'api-rolls';
|
||||
|
||||
override _id!: string;
|
||||
|
||||
@Prop()
|
||||
uid?: string;
|
||||
|
||||
@Prop()
|
||||
ip?: string;
|
||||
|
||||
@Prop({ arrayOf: String })
|
||||
tags?: string[];
|
||||
|
||||
@Prop()
|
||||
status?: API_CALL_STATUS;
|
||||
|
||||
@Prop()
|
||||
chargeAmount?: number;
|
||||
|
||||
@Prop()
|
||||
createdAt!: Date;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
export { ServiceBadAttemptError } from '../services/errors';
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
export {
|
||||
SecurityCompromiseError,
|
||||
ServiceBadApproachError,
|
||||
ServiceBadAttemptError,
|
||||
ServiceCrashedError,
|
||||
ServiceNodeResourceDrainError,
|
||||
} from '../../services/errors';
|
||||
|
||||
@@ -0,0 +1,316 @@
|
||||
import { AutoCastable } from 'civkit/civ-rpc';
|
||||
import { randomUUID } from 'crypto';
|
||||
import fs from 'fs';
|
||||
import fsp from 'fs/promises';
|
||||
import path from 'path';
|
||||
|
||||
type WhereOp = '==' | '>=' | '<=' | '>' | '<';
|
||||
type OrderDirection = 'asc' | 'desc';
|
||||
|
||||
type IncrementOp = {
|
||||
__op: 'increment';
|
||||
value: number;
|
||||
};
|
||||
|
||||
type QueryFilter = {
|
||||
field: string;
|
||||
op: WhereOp;
|
||||
value: any;
|
||||
};
|
||||
|
||||
type QueryOrder = {
|
||||
field: string;
|
||||
direction: OrderDirection;
|
||||
};
|
||||
|
||||
function reviveJson(_key: string, value: any) {
|
||||
if (value && typeof value === 'object' && value.__type === 'Date') {
|
||||
return new Date(value.value);
|
||||
}
|
||||
|
||||
return value;
|
||||
}
|
||||
|
||||
function serializeJson(_key: string, value: any) {
|
||||
if (value instanceof Date) {
|
||||
return {
|
||||
__type: 'Date',
|
||||
value: value.toISOString(),
|
||||
};
|
||||
}
|
||||
|
||||
return value;
|
||||
}
|
||||
|
||||
function getByPath(input: any, pathText: string) {
|
||||
return pathText.split('.').reduce((acc, key) => acc?.[key], input);
|
||||
}
|
||||
|
||||
function setByPath(input: any, pathText: string, value: any) {
|
||||
const parts = pathText.split('.');
|
||||
const last = parts.pop()!;
|
||||
let cursor = input;
|
||||
for (const part of parts) {
|
||||
cursor[part] ??= {};
|
||||
cursor = cursor[part];
|
||||
}
|
||||
|
||||
cursor[last] = value;
|
||||
}
|
||||
|
||||
function applyPatch(base: any, patch: any) {
|
||||
const draft = structuredClone(base || {});
|
||||
|
||||
for (const [key, value] of Object.entries(patch || {})) {
|
||||
const currentValue = getByPath(draft, key);
|
||||
if (value && typeof value === 'object' && (value as IncrementOp).__op === 'increment') {
|
||||
setByPath(draft, key, (Number(currentValue) || 0) + (value as IncrementOp).value);
|
||||
continue;
|
||||
}
|
||||
|
||||
setByPath(draft, key, value);
|
||||
}
|
||||
|
||||
return draft;
|
||||
}
|
||||
|
||||
class LocalStore {
|
||||
rootDir = path.resolve('.cache/xread/db');
|
||||
|
||||
constructor() {
|
||||
fs.mkdirSync(this.rootDir, { recursive: true });
|
||||
}
|
||||
|
||||
filePath(collectionName: string) {
|
||||
return path.join(this.rootDir, `${collectionName}.json`);
|
||||
}
|
||||
|
||||
async readCollection(collectionName: string) {
|
||||
const filePath = this.filePath(collectionName);
|
||||
try {
|
||||
const content = await fsp.readFile(filePath, 'utf8');
|
||||
return JSON.parse(content, reviveJson) as Record<string, any>;
|
||||
} catch (err: any) {
|
||||
if (err?.code === 'ENOENT') {
|
||||
return {};
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
async writeCollection(collectionName: string, payload: Record<string, any>) {
|
||||
const filePath = this.filePath(collectionName);
|
||||
await fsp.mkdir(path.dirname(filePath), { recursive: true });
|
||||
await fsp.writeFile(filePath, JSON.stringify(payload, serializeJson, 2), 'utf8');
|
||||
}
|
||||
}
|
||||
|
||||
const LOCAL_STORE = new LocalStore();
|
||||
|
||||
export class LocalDocRef<T extends typeof FirestoreRecord = typeof FirestoreRecord> {
|
||||
constructor(
|
||||
protected recordType: T,
|
||||
readonly id: string,
|
||||
) { }
|
||||
|
||||
async get() {
|
||||
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
|
||||
const raw = bucket[this.id];
|
||||
if (!raw) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return this.recordType.from(raw);
|
||||
}
|
||||
|
||||
async set(payload: any, options?: { merge?: boolean }) {
|
||||
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
|
||||
const previous = bucket[this.id];
|
||||
const merged = options?.merge ? applyPatch(previous, payload) : { ...payload };
|
||||
merged._id ??= this.id;
|
||||
bucket[this.id] = merged;
|
||||
await LOCAL_STORE.writeCollection(this.recordType.collectionName, bucket);
|
||||
|
||||
return this.recordType.from(merged);
|
||||
}
|
||||
|
||||
async update(payload: any) {
|
||||
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
|
||||
const previous = bucket[this.id] || { _id: this.id };
|
||||
const merged = applyPatch(previous, payload);
|
||||
merged._id ??= this.id;
|
||||
bucket[this.id] = merged;
|
||||
await LOCAL_STORE.writeCollection(this.recordType.collectionName, bucket);
|
||||
|
||||
return this.recordType.from(merged);
|
||||
}
|
||||
}
|
||||
|
||||
export class LocalQuery<T extends typeof FirestoreRecord = typeof FirestoreRecord> {
|
||||
constructor(
|
||||
protected recordType: T,
|
||||
protected filters: QueryFilter[] = [],
|
||||
protected orders: QueryOrder[] = [],
|
||||
protected limitCount?: number,
|
||||
) { }
|
||||
|
||||
where(field: string, op: WhereOp, value: any) {
|
||||
return new LocalQuery(this.recordType, [...this.filters, { field, op, value }], this.orders, this.limitCount);
|
||||
}
|
||||
|
||||
orderBy(field: string, direction: OrderDirection = 'asc') {
|
||||
return new LocalQuery(this.recordType, this.filters, [...this.orders, { field, direction }], this.limitCount);
|
||||
}
|
||||
|
||||
limit(count: number) {
|
||||
return new LocalQuery(this.recordType, this.filters, this.orders, count);
|
||||
}
|
||||
|
||||
async exec() {
|
||||
const bucket = await LOCAL_STORE.readCollection(this.recordType.collectionName);
|
||||
let values = Object.values(bucket);
|
||||
|
||||
for (const filter of this.filters) {
|
||||
values = values.filter((entry) => {
|
||||
const left = getByPath(entry, filter.field);
|
||||
const right = filter.value;
|
||||
switch (filter.op) {
|
||||
case '==':
|
||||
return left === right;
|
||||
case '>=':
|
||||
return left >= right;
|
||||
case '<=':
|
||||
return left <= right;
|
||||
case '>':
|
||||
return left > right;
|
||||
case '<':
|
||||
return left < right;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
for (const order of this.orders.slice().reverse()) {
|
||||
values.sort((a, b) => {
|
||||
const left = getByPath(a, order.field);
|
||||
const right = getByPath(b, order.field);
|
||||
const factor = order.direction === 'desc' ? -1 : 1;
|
||||
if (left === right) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return left > right ? factor : -factor;
|
||||
});
|
||||
}
|
||||
|
||||
if (typeof this.limitCount === 'number') {
|
||||
values = values.slice(0, this.limitCount);
|
||||
}
|
||||
|
||||
return values.map((entry) => this.recordType.from(entry));
|
||||
}
|
||||
}
|
||||
|
||||
export class LocalCollectionRef<T extends typeof FirestoreRecord = typeof FirestoreRecord> extends LocalQuery<T> {
|
||||
constructor(recordType: T) {
|
||||
super(recordType);
|
||||
}
|
||||
|
||||
doc(id?: string) {
|
||||
return new LocalDocRef(this.recordType, id || randomUUID());
|
||||
}
|
||||
}
|
||||
|
||||
class LocalBatch {
|
||||
protected tasks: Array<() => Promise<unknown>> = [];
|
||||
|
||||
set(ref: LocalDocRef, payload: any, options?: { merge?: boolean }) {
|
||||
this.tasks.push(() => ref.set(payload, options));
|
||||
}
|
||||
|
||||
update(ref: LocalDocRef, payload: any) {
|
||||
this.tasks.push(() => ref.update(payload));
|
||||
}
|
||||
|
||||
async commit() {
|
||||
await Promise.all(this.tasks.map((task) => task()));
|
||||
}
|
||||
}
|
||||
|
||||
class LocalTransaction {
|
||||
async get(ref: LocalDocRef) {
|
||||
return ref.get();
|
||||
}
|
||||
|
||||
set(ref: LocalDocRef, payload: any, options?: { merge?: boolean }) {
|
||||
return ref.set(payload, options);
|
||||
}
|
||||
|
||||
update(ref: LocalDocRef, payload: any) {
|
||||
return ref.update(payload);
|
||||
}
|
||||
}
|
||||
|
||||
class LocalDB {
|
||||
batch() {
|
||||
return new LocalBatch();
|
||||
}
|
||||
|
||||
async runTransaction<T>(handler: (transaction: LocalTransaction) => Promise<T>) {
|
||||
return handler(new LocalTransaction());
|
||||
}
|
||||
}
|
||||
|
||||
const LOCAL_DB = new LocalDB();
|
||||
|
||||
export class FirestoreRecord extends AutoCastable {
|
||||
static collectionName = 'records';
|
||||
|
||||
_id!: string;
|
||||
|
||||
static get COLLECTION() {
|
||||
return new LocalCollectionRef(this);
|
||||
}
|
||||
|
||||
static get DB() {
|
||||
return LOCAL_DB;
|
||||
}
|
||||
|
||||
static OPS = {
|
||||
increment(value: number): IncrementOp {
|
||||
return {
|
||||
__op: 'increment',
|
||||
value,
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
static async fromFirestore<T extends typeof FirestoreRecord>(this: T, id: string) {
|
||||
if (!id) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return this.COLLECTION.doc(id).get() as Promise<InstanceType<T> | undefined>;
|
||||
}
|
||||
|
||||
static async fromFirestoreQuery<T extends typeof FirestoreRecord>(this: T, query: LocalQuery<any>) {
|
||||
return query.exec() as Promise<Array<InstanceType<T>>>;
|
||||
}
|
||||
|
||||
static async save<T extends typeof FirestoreRecord>(this: T, input: any, id?: string, options?: { merge?: boolean }) {
|
||||
const payload = input instanceof this ? input.degradeForFireStore() : { ...input };
|
||||
const resolvedId = id || payload._id || randomUUID();
|
||||
payload._id = resolvedId;
|
||||
|
||||
return this.COLLECTION.doc(resolvedId).set(payload, options) as Promise<InstanceType<T>>;
|
||||
}
|
||||
|
||||
async save(options?: { merge?: boolean }) {
|
||||
return (this.constructor as typeof FirestoreRecord).save(this, this._id, options);
|
||||
}
|
||||
|
||||
degradeForFireStore() {
|
||||
return { ...this };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
export { AsyncLocalContext as AsyncContext } from '../../services/async-context';
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
import { ServiceDisabledError } from '../../services/errors';
|
||||
|
||||
@singleton()
|
||||
export class ImageInterrogationManager extends AsyncService {
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
async interrogate(model: string, _input?: unknown): Promise<string> {
|
||||
throw new ServiceDisabledError(`Image interrogation feature '${model}' is not enabled in this standalone build.`);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
import { ServiceDisabledError } from '../../services/errors';
|
||||
|
||||
@singleton()
|
||||
export class LLMManager extends AsyncService {
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
async *iterRun(model: string, _input?: unknown): AsyncGenerator<string> {
|
||||
throw new ServiceDisabledError(`LLM feature '${model}' is not enabled in this standalone build.`);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
import path from 'path';
|
||||
import fs from 'fs/promises';
|
||||
import { pathToFileURL } from 'url';
|
||||
import { SecretExposer } from './secrets';
|
||||
|
||||
type SaveOptions = {
|
||||
contentType?: string;
|
||||
metadata?: {
|
||||
contentType?: string;
|
||||
};
|
||||
};
|
||||
|
||||
@singleton()
|
||||
export class FirebaseStorageBucketControl extends AsyncService {
|
||||
rootDir!: string;
|
||||
|
||||
constructor(
|
||||
protected secretExposer: SecretExposer,
|
||||
) {
|
||||
super(...arguments);
|
||||
}
|
||||
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.rootDir = path.resolve(this.secretExposer.STORAGE_ROOT || '.cache/xread/storage');
|
||||
await fs.mkdir(this.rootDir, { recursive: true });
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
protected resolvePath(key: string) {
|
||||
const safeKey = key.replace(/\\/g, '/').replace(/^\/+/, '');
|
||||
return path.resolve(this.rootDir, safeKey);
|
||||
}
|
||||
|
||||
async saveFile(key: string, content: Buffer | Uint8Array, _options?: SaveOptions) {
|
||||
const fullPath = this.resolvePath(key);
|
||||
await fs.mkdir(path.dirname(fullPath), { recursive: true });
|
||||
await fs.writeFile(fullPath, content);
|
||||
|
||||
return { fullPath };
|
||||
}
|
||||
|
||||
async downloadFile(key: string) {
|
||||
return fs.readFile(this.resolvePath(key));
|
||||
}
|
||||
|
||||
async signDownloadUrl(key: string, _expiresAt?: number) {
|
||||
return pathToFileURL(this.resolvePath(key)).href;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
import { SecretExposer } from './secrets';
|
||||
import { ServiceDisabledError } from '../../services/errors';
|
||||
|
||||
@singleton()
|
||||
export class ProxyProviderService extends AsyncService {
|
||||
protected proxies: URL[] = [];
|
||||
|
||||
constructor(
|
||||
protected secretExposer: SecretExposer,
|
||||
) {
|
||||
super(...arguments);
|
||||
}
|
||||
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.proxies = this.secretExposer.LOCAL_PROXY_URLS
|
||||
.split(',')
|
||||
.map((x) => x.trim())
|
||||
.filter(Boolean)
|
||||
.map((x) => new URL(x));
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
supports(input?: string) {
|
||||
if (!input || input === 'auto' || input === 'any' || input === 'none') {
|
||||
return true;
|
||||
}
|
||||
|
||||
return /^[a-z]{2}$/i.test(input);
|
||||
}
|
||||
|
||||
async alloc(input?: string) {
|
||||
if (input === 'none') {
|
||||
throw new ServiceDisabledError(`Proxy allocation is disabled for '${input}'.`);
|
||||
}
|
||||
|
||||
const picked = this.proxies[0];
|
||||
if (!picked) {
|
||||
throw new ServiceDisabledError('Proxy allocation is not configured in this standalone build.');
|
||||
}
|
||||
|
||||
return picked;
|
||||
}
|
||||
|
||||
async *iterAlloc(input?: string) {
|
||||
yield await this.alloc(input);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { ApplicationError, AutoCastable } from 'civkit/civ-rpc';
|
||||
import { singleton } from 'tsyringe';
|
||||
import { ApiRollRecord, API_CALL_STATUS } from '../db/api-roll';
|
||||
|
||||
type SubjectLimit = {
|
||||
at: number;
|
||||
};
|
||||
|
||||
@singleton()
|
||||
export class RateLimitControl extends AsyncService {
|
||||
protected hits = new Map<string, SubjectLimit[]>();
|
||||
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
protected prune(key: string, windowMs: number) {
|
||||
const now = Date.now();
|
||||
const entries = (this.hits.get(key) || []).filter((entry) => (now - entry.at) < windowMs);
|
||||
this.hits.set(key, entries);
|
||||
|
||||
return entries;
|
||||
}
|
||||
|
||||
protected async checkAndRecord(subject: string, tags: string[], descs: RateLimitDesc[]) {
|
||||
const now = Date.now();
|
||||
for (const desc of descs) {
|
||||
const key = `${subject}:${tags.join(',')}:${desc.periodSeconds}:${desc.occurrence}`;
|
||||
const entries = this.prune(key, desc.periodSeconds * 1000);
|
||||
if (entries.length >= desc.occurrence) {
|
||||
const oldestRelevant = entries[0];
|
||||
const retryAfterMs = Math.max(1000, desc.periodSeconds * 1000 - (now - oldestRelevant.at));
|
||||
throw RateLimitTriggeredError.from({
|
||||
message: `Rate limit exceeded for ${subject}`,
|
||||
retryAfter: Math.ceil(retryAfterMs / 1000),
|
||||
retryAfterDate: new Date(now + retryAfterMs),
|
||||
});
|
||||
}
|
||||
|
||||
entries.push({ at: now });
|
||||
this.hits.set(key, entries);
|
||||
}
|
||||
}
|
||||
|
||||
async simpleRPCUidBasedLimit(_rpcReflect: any, uid: string, tags: string[], ...descs: RateLimitDesc[]) {
|
||||
const effectiveDescs = descs.length ? descs : [RateLimitDesc.from({ occurrence: 60, periodSeconds: 60 })];
|
||||
await this.checkAndRecord(`uid:${uid}`, tags, effectiveDescs);
|
||||
|
||||
return ApiRollRecord.from({
|
||||
_id: `${Date.now()}-${Math.random()}`,
|
||||
uid,
|
||||
tags,
|
||||
status: API_CALL_STATUS.SUCCESS,
|
||||
createdAt: new Date(),
|
||||
});
|
||||
}
|
||||
|
||||
async simpleRpcIPBasedLimit(_rpcReflect: any, ip: string, tags: string[], limits?: [Date, number] | Array<[Date, number]>) {
|
||||
const normalized = Array.isArray(limits?.[0]) ? limits as Array<[Date, number]> : (limits ? [limits as [Date, number]] : []);
|
||||
const descs = normalized.map(([date, occurrence]) => {
|
||||
const seconds = Math.max(1, Math.ceil((Date.now() - date.valueOf()) / 1000));
|
||||
return RateLimitDesc.from({
|
||||
occurrence,
|
||||
periodSeconds: seconds,
|
||||
});
|
||||
});
|
||||
|
||||
await this.checkAndRecord(`ip:${ip}`, tags, descs.length ? descs : [RateLimitDesc.from({ occurrence: 20, periodSeconds: 60 })]);
|
||||
|
||||
return ApiRollRecord.from({
|
||||
_id: `${Date.now()}-${Math.random()}`,
|
||||
ip,
|
||||
tags,
|
||||
status: API_CALL_STATUS.SUCCESS,
|
||||
createdAt: new Date(),
|
||||
});
|
||||
}
|
||||
|
||||
record(input: Partial<ApiRollRecord>) {
|
||||
return ApiRollRecord.from({
|
||||
createdAt: new Date(),
|
||||
status: API_CALL_STATUS.SUCCESS,
|
||||
...input,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
export class RateLimitDesc extends AutoCastable {
|
||||
occurrence!: number;
|
||||
periodSeconds!: number;
|
||||
|
||||
isEffective() {
|
||||
return Boolean(this.occurrence && this.periodSeconds);
|
||||
}
|
||||
}
|
||||
|
||||
export class RateLimitTriggeredError extends ApplicationError {
|
||||
retryAfter?: number;
|
||||
retryAfterDate?: Date;
|
||||
|
||||
static override from(input: Partial<RateLimitTriggeredError> & { message?: string }) {
|
||||
const err = new RateLimitTriggeredError(input.message || 'Rate limit exceeded');
|
||||
Object.assign(err, input);
|
||||
|
||||
return err;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
import { AsyncService } from 'civkit/async-service';
|
||||
import { singleton } from 'tsyringe';
|
||||
|
||||
type EnvConfig = Record<string, string> & {
|
||||
readonly SERPER_SEARCH_API_KEY: string;
|
||||
readonly BRAVE_SEARCH_API_KEY: string;
|
||||
readonly CLOUD_FLARE_API_KEY: string;
|
||||
readonly LOCAL_PROXY_URLS: string;
|
||||
readonly STORAGE_ROOT: string;
|
||||
};
|
||||
|
||||
export const readEnv = (key: string) => process.env[key] || '';
|
||||
|
||||
const envConfig = new Proxy({}, {
|
||||
get(_target, prop) {
|
||||
return readEnv(String(prop));
|
||||
},
|
||||
}) as EnvConfig;
|
||||
|
||||
@singleton()
|
||||
export class SecretExposer extends AsyncService {
|
||||
override async init() {
|
||||
await this.dependencyReady();
|
||||
this.emit('ready');
|
||||
}
|
||||
|
||||
get SERPER_SEARCH_API_KEY() {
|
||||
return readEnv('SERPER_SEARCH_API_KEY');
|
||||
}
|
||||
|
||||
get BRAVE_SEARCH_API_KEY() {
|
||||
return readEnv('BRAVE_SEARCH_API_KEY');
|
||||
}
|
||||
|
||||
get CLOUD_FLARE_API_KEY() {
|
||||
return readEnv('CLOUD_FLARE_API_KEY');
|
||||
}
|
||||
|
||||
get LOCAL_PROXY_URLS() {
|
||||
return readEnv('LOCAL_PROXY_URLS');
|
||||
}
|
||||
|
||||
get STORAGE_ROOT() {
|
||||
return readEnv('STORAGE_ROOT');
|
||||
}
|
||||
}
|
||||
|
||||
export default envConfig;
|
||||
@@ -0,0 +1,6 @@
|
||||
import type { Context, Next } from 'koa';
|
||||
|
||||
export function getAuditionMiddleware() {
|
||||
return async (_ctx: Context, next: Next) => next();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
export function countGPTToken(input?: string | null) {
|
||||
if (!input) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
return Math.max(1, Math.ceil(Buffer.byteLength(input, 'utf8') / 4));
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@ import { GoogleAuth } from 'google-auth-library';
|
||||
* @return {Promise<string>} The URL of the function
|
||||
*/
|
||||
export async function getFunctionUrl(name: string, location = "us-central1") {
|
||||
const projectId = `reader-6b7dc`;
|
||||
const projectId = `xread-project`;
|
||||
const url = "https://cloudfunctions.googleapis.com/v2beta/" +
|
||||
`projects/${projectId}/locations/${location}/functions/${name}`;
|
||||
const auth = new GoogleAuth({
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
const test = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const projectRoot = path.resolve(__dirname, '..');
|
||||
|
||||
function read(relativePath) {
|
||||
return fs.readFileSync(path.join(projectRoot, relativePath), 'utf8');
|
||||
}
|
||||
|
||||
test('legacy Cloud Run deployment workflow has been removed', () => {
|
||||
const legacyWorkflow = path.join(projectRoot, '.github', 'workflows', 'cd.yml');
|
||||
|
||||
assert.equal(fs.existsSync(legacyWorkflow), false);
|
||||
});
|
||||
|
||||
test('repository automation files exist for CI, container publish, and Dependabot', () => {
|
||||
const expectedFiles = [
|
||||
'.eslintrc.cjs',
|
||||
'.eslintignore',
|
||||
'.github/workflows/ci.yml',
|
||||
'.github/workflows/image.yml',
|
||||
'.github/workflows/dependabot-auto-merge.yml',
|
||||
'.github/dependabot.yml',
|
||||
];
|
||||
|
||||
for (const relativePath of expectedFiles) {
|
||||
assert.equal(fs.existsSync(path.join(projectRoot, relativePath)), true, `${relativePath} should exist`);
|
||||
}
|
||||
});
|
||||
|
||||
test('container image workflow publishes to GHCR instead of legacy GCP registries', () => {
|
||||
const workflow = read('.github/workflows/image.yml');
|
||||
|
||||
assert.match(workflow, /ghcr\.io/);
|
||||
assert.doesNotMatch(workflow, /gcloud|us-docker\.pkg\.dev/);
|
||||
});
|
||||
|
||||
test('dependabot config covers npm, docker, and github-actions updates', () => {
|
||||
const dependabot = read('.github/dependabot.yml');
|
||||
|
||||
assert.match(dependabot, /package-ecosystem: npm/);
|
||||
assert.match(dependabot, /package-ecosystem: docker/);
|
||||
assert.match(dependabot, /package-ecosystem: github-actions/);
|
||||
});
|
||||
|
||||
test('CI workflow runs lint before tests and build', () => {
|
||||
const workflow = read('.github/workflows/ci.yml');
|
||||
|
||||
assert.match(workflow, /run: npm run lint/);
|
||||
assert.match(workflow, /run: npm run test:ci/);
|
||||
assert.match(workflow, /run: npm run build/);
|
||||
});
|
||||
|
||||
test('Dockerfile is self-contained and no longer relies on curl-impersonate', () => {
|
||||
const dockerfile = read('Dockerfile');
|
||||
|
||||
assert.match(dockerfile, /FROM node:22-bookworm-slim/);
|
||||
assert.match(dockerfile, /PUPPETEER_SKIP_DOWNLOAD=true/);
|
||||
assert.doesNotMatch(dockerfile, /curl-impersonate|LD_PRELOAD/);
|
||||
});
|
||||
@@ -0,0 +1,21 @@
|
||||
const test = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const os = require('node:os');
|
||||
const path = require('node:path');
|
||||
const { spawnSync } = require('node:child_process');
|
||||
|
||||
test('integrity check warns instead of failing when GeoLite asset is missing', () => {
|
||||
const tempRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'xread-integrity-'));
|
||||
const scriptPath = path.join(tempRoot, 'integrity-check.cjs');
|
||||
|
||||
fs.copyFileSync(path.resolve(__dirname, '..', 'integrity-check.cjs'), scriptPath);
|
||||
|
||||
const result = spawnSync(process.execPath, [scriptPath], {
|
||||
cwd: tempRoot,
|
||||
encoding: 'utf8',
|
||||
});
|
||||
|
||||
assert.equal(result.status, 0);
|
||||
assert.match(result.stderr, /GeoLite2-City\.mmdb/);
|
||||
});
|
||||
@@ -0,0 +1,44 @@
|
||||
const test = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const { spawnSync } = require('node:child_process');
|
||||
|
||||
const projectRoot = path.resolve(__dirname, '..');
|
||||
|
||||
test('src/shared is a real directory in the repository', () => {
|
||||
const sharedPath = path.join(projectRoot, 'src', 'shared');
|
||||
const stat = fs.statSync(sharedPath);
|
||||
|
||||
assert.equal(stat.isDirectory(), true);
|
||||
});
|
||||
|
||||
test('default build script uses standard TypeScript compiler', () => {
|
||||
const packageJson = JSON.parse(fs.readFileSync(path.join(projectRoot, 'package.json'), 'utf8'));
|
||||
|
||||
assert.match(packageJson.scripts.build, /\btsc\b/);
|
||||
assert.doesNotMatch(packageJson.scripts.build, /scripts\/transpile\.cjs/);
|
||||
});
|
||||
|
||||
test('project type-check build passes with tsc', () => {
|
||||
const tscJsPath = path.join(
|
||||
projectRoot,
|
||||
'.codex-cache',
|
||||
'ts-compiler',
|
||||
'node_modules',
|
||||
'typescript',
|
||||
'lib',
|
||||
'tsc.js',
|
||||
);
|
||||
const result = spawnSync(
|
||||
process.execPath,
|
||||
[tscJsPath, '-p', '.', '--pretty', 'false'],
|
||||
{
|
||||
cwd: projectRoot,
|
||||
encoding: 'utf8',
|
||||
shell: false,
|
||||
},
|
||||
);
|
||||
|
||||
assert.equal(result.status, 0, result.stdout + result.stderr);
|
||||
});
|
||||
@@ -0,0 +1,28 @@
|
||||
const test = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const os = require('node:os');
|
||||
const path = require('node:path');
|
||||
|
||||
test('ensureLicensedAssets creates GeoLite database when missing', async () => {
|
||||
const tempRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'xread-licensed-'));
|
||||
const targetFile = path.join(tempRoot, 'licensed', 'GeoLite2-City.mmdb');
|
||||
let downloadCalls = 0;
|
||||
|
||||
const { ensureLicensedAssets, DEFAULT_ASSETS } = require('../scripts/prepare-licensed-assets.cjs');
|
||||
|
||||
await ensureLicensedAssets({
|
||||
rootDir: tempRoot,
|
||||
assets: DEFAULT_ASSETS,
|
||||
downloadAsset: async (asset, destination) => {
|
||||
downloadCalls += 1;
|
||||
assert.equal(asset.path, 'licensed/GeoLite2-City.mmdb');
|
||||
fs.mkdirSync(path.dirname(destination), { recursive: true });
|
||||
fs.writeFileSync(destination, Buffer.from('fake-mmdb'));
|
||||
},
|
||||
});
|
||||
|
||||
assert.equal(downloadCalls, 1);
|
||||
assert.equal(fs.existsSync(targetFile), true);
|
||||
assert.equal(fs.readFileSync(targetFile, 'utf8'), 'fake-mmdb');
|
||||
});
|
||||
+4
-3
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"compilerOptions": {
|
||||
"module": "node16",
|
||||
|
||||
"types": ["node"],
|
||||
"noImplicitReturns": true,
|
||||
"noUnusedLocals": true,
|
||||
"outDir": "build",
|
||||
@@ -9,7 +9,7 @@
|
||||
"strict": true,
|
||||
"allowJs": true,
|
||||
"target": "es2022",
|
||||
"lib": ["es2022"],
|
||||
"lib": ["es2022", "dom"],
|
||||
"skipLibCheck": true,
|
||||
"useDefineForClassFields": false,
|
||||
"experimentalDecorators": true,
|
||||
@@ -18,5 +18,6 @@
|
||||
"noImplicitOverride": true,
|
||||
},
|
||||
"compileOnSave": true,
|
||||
"include": ["src"]
|
||||
"include": ["src"],
|
||||
"exclude": ["src/cloud-functions/**/*"]
|
||||
}
|
||||
Reference in new issue
Block a user