mirror of
https://github.com/searxng/searxng.git
synced 2026-07-25 09:21:25 +00:00
Compare commits
30 Commits
fe1848673f
...
update_dat
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c6707dc0b6 | ||
|
|
8605230eb8 | ||
|
|
a54722e692 | ||
|
|
7e9a6c2861 | ||
|
|
357662d86d | ||
|
|
bfa76e4cd9 | ||
|
|
a9d49a3349 | ||
|
|
f8ffbf36f9 | ||
|
|
e3126b89e6 | ||
|
|
f9f3d089ec | ||
|
|
e3713717f2 | ||
|
|
952896d29e | ||
|
|
4cc32b2457 | ||
|
|
cce0957f54 | ||
|
|
9375c0a6b6 | ||
|
|
a702741e4e | ||
|
|
aeced67249 | ||
|
|
199e03de1d | ||
|
|
9cd2439e5e | ||
|
|
9f4d8bca02 | ||
|
|
de76a4a39b | ||
|
|
a85a5e2794 | ||
|
|
92abd98a55 | ||
|
|
93e867c6b1 | ||
|
|
75c1b1dade | ||
|
|
097ab64c70 | ||
|
|
0e9f513efc | ||
|
|
fd42d4fda1 | ||
|
|
5c38d2feab | ||
|
|
38b678c493 |
12
.github/workflows/container.yml
vendored
12
.github/workflows/container.yml
vendored
@@ -73,18 +73,18 @@ jobs:
|
||||
# yamllint enable rule:line-length
|
||||
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
fetch-depth: "0"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
@@ -96,7 +96,7 @@ jobs:
|
||||
run: echo "date=$(date +'%Y%m%d')" >>$GITHUB_OUTPUT
|
||||
|
||||
- name: Setup cache container
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "container-${{ matrix.arch }}-${{ steps.date.outputs.date }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
@@ -141,7 +141,7 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
@@ -175,7 +175,7 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
|
||||
6
.github/workflows/data-update.yml
vendored
6
.github/workflows/data-update.yml
vendored
@@ -41,17 +41,17 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
|
||||
6
.github/workflows/documentation.yml
vendored
6
.github/workflows/documentation.yml
vendored
@@ -32,18 +32,18 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
fetch-depth: "0"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
|
||||
14
.github/workflows/integration.yml
vendored
14
.github/workflows/integration.yml
vendored
@@ -34,17 +34,17 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ matrix.python-version }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ matrix.python-version }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
@@ -62,12 +62,12 @@ jobs:
|
||||
runs-on: ubuntu-24.04-arm
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
@@ -77,13 +77,13 @@ jobs:
|
||||
node-version-file: "./.nvmrc"
|
||||
|
||||
- name: Setup cache Node.js
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "nodejs-${{ runner.arch }}-${{ hashFiles('./.nvmrc', './package.json') }}"
|
||||
path: "./client/simple/node_modules/"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
|
||||
12
.github/workflows/l10n.yml
vendored
12
.github/workflows/l10n.yml
vendored
@@ -35,18 +35,18 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
token: "${{ secrets.WEBLATE_GITHUB_TOKEN }}"
|
||||
fetch-depth: "0"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
@@ -83,18 +83,18 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
with:
|
||||
python-version: "${{ env.PYTHON_VERSION }}"
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
token: "${{ secrets.WEBLATE_GITHUB_TOKEN }}"
|
||||
fetch-depth: "0"
|
||||
|
||||
- name: Setup cache Python
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
uses: actions/cache@2c8a9bd7457de244a408f35966fab2fb45fda9c8 # v6.0.0
|
||||
with:
|
||||
key: "python-${{ env.PYTHON_VERSION }}-${{ runner.arch }}-${{ hashFiles('./requirements*.txt') }}"
|
||||
restore-keys: |
|
||||
|
||||
4
.github/workflows/security.yml
vendored
4
.github/workflows/security.yml
vendored
@@ -24,12 +24,12 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: "false"
|
||||
|
||||
- name: Sync GHCS from Docker Scout
|
||||
uses: docker/scout-action@cd72f264beff1cd72735de31148b9d3244a0234a # v1.21.0
|
||||
uses: docker/scout-action@7520205ff60037fdc436b40b6a1d1e55a839ec2d # v1.22.0
|
||||
with:
|
||||
organization: "searxng"
|
||||
dockerhub-user: "${{ secrets.DOCKER_USER }}"
|
||||
|
||||
63
client/simple/package-lock.json
generated
63
client/simple/package-lock.json
generated
@@ -16,11 +16,11 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@biomejs/biome": "2.5.0",
|
||||
"@types/node": "^25.9.3",
|
||||
"@types/node": "^26.0.0",
|
||||
"browserslist": "^4.28.2",
|
||||
"browserslist-to-esbuild": "^2.1.1",
|
||||
"edge.js": "^6.5.1",
|
||||
"less": "^4.6.4",
|
||||
"less": "^4.6.6",
|
||||
"mathjs": "^15.2.0",
|
||||
"sharp": "~0.35.1",
|
||||
"sort-package-json": "^4.0.0",
|
||||
@@ -1570,13 +1570,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@types/node": {
|
||||
"version": "25.9.3",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.3.tgz",
|
||||
"integrity": "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==",
|
||||
"version": "26.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.0.0.tgz",
|
||||
"integrity": "sha512-vf2YFi1iY9lHGwNJMs01biZFbKJkrZR1T6/MlzjhJLPdntOHLhTrDSnSVcdtvjihi4VQNlrFRIxLsDBlQpAipA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"undici-types": ">=7.24.0 <7.24.7"
|
||||
"undici-types": "~8.3.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/pluralize": {
|
||||
@@ -2890,9 +2890,9 @@
|
||||
"license": "Apache-2.0"
|
||||
},
|
||||
"node_modules/less": {
|
||||
"version": "4.6.4",
|
||||
"resolved": "https://registry.npmjs.org/less/-/less-4.6.4.tgz",
|
||||
"integrity": "sha512-OJmO5+HxZLLw0RLzkqaNHzcgEAQG7C0y3aMbwtCzIUFZsLMNNq/1IdAdHEycQ58CwUO3jPTHmoN+tE5I7FQxNg==",
|
||||
"version": "4.6.6",
|
||||
"resolved": "https://registry.npmjs.org/less/-/less-4.6.6.tgz",
|
||||
"integrity": "sha512-ooPSwQGQ2sVe8Dh1jVsbKKsRR2gd8lFK72BDkeSzjnD1T5aIHL65hCMfO0GVmtriKgDKrQv6xp9UrihUsWuAzA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
@@ -2909,7 +2909,7 @@
|
||||
"errno": "^0.1.1",
|
||||
"graceful-fs": "^4.1.2",
|
||||
"image-size": "~0.5.0",
|
||||
"make-dir": "^2.1.0",
|
||||
"make-dir": "^5.1.0",
|
||||
"mime": "^1.4.1",
|
||||
"needle": "^3.1.0",
|
||||
"source-map": "~0.6.0"
|
||||
@@ -3191,18 +3191,17 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/make-dir": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/make-dir/-/make-dir-2.1.0.tgz",
|
||||
"integrity": "sha512-LS9X+dc8KLxXCb8dni79fLIIUA5VyZoyjSMCwTluaXA0o27cCK0bhXkpgw+sTXVpPy/lSO57ilRixqk0vDmtRA==",
|
||||
"version": "5.1.0",
|
||||
"resolved": "https://registry.npmjs.org/make-dir/-/make-dir-5.1.0.tgz",
|
||||
"integrity": "sha512-IfpFq6UM39dUNiphpA6uDezNx/AvWyhwfICWPR3t1VspkgkMZrL+Rk1RbN1bx+aeNYwOrqGJgEgV3yotk+ZUVw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"pify": "^4.0.1",
|
||||
"semver": "^5.6.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6"
|
||||
"node": ">=18"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/mathjs": {
|
||||
@@ -3491,17 +3490,6 @@
|
||||
"url": "https://github.com/sponsors/jonschlinkert"
|
||||
}
|
||||
},
|
||||
"node_modules/pify": {
|
||||
"version": "4.0.1",
|
||||
"resolved": "https://registry.npmjs.org/pify/-/pify-4.0.1.tgz",
|
||||
"integrity": "sha512-uB80kBFb/tfd68bVleG9T5GGsGPjJrLAUpR5PZIrhBnIaRTQRjqdJSsIKkOP6OAIFbj7GOrcudc5pNjZ+geV2g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": ">=6"
|
||||
}
|
||||
},
|
||||
"node_modules/pluralize": {
|
||||
"version": "8.0.0",
|
||||
"resolved": "https://registry.npmjs.org/pluralize/-/pluralize-8.0.0.tgz",
|
||||
@@ -3861,17 +3849,6 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/semver": {
|
||||
"version": "5.7.2",
|
||||
"resolved": "https://registry.npmjs.org/semver/-/semver-5.7.2.tgz",
|
||||
"integrity": "sha512-cBznnQ9KjJqU67B52RMC65CMarK2600WFnbkcaiwWq3xy/5haFJlshgnpjovMVJ+Hff49d8GEn0b87C5pDQ10g==",
|
||||
"dev": true,
|
||||
"license": "ISC",
|
||||
"optional": true,
|
||||
"bin": {
|
||||
"semver": "bin/semver"
|
||||
}
|
||||
},
|
||||
"node_modules/sharp": {
|
||||
"version": "0.35.1",
|
||||
"resolved": "https://registry.npmjs.org/sharp/-/sharp-0.35.1.tgz",
|
||||
@@ -4515,9 +4492,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/undici-types": {
|
||||
"version": "7.24.6",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz",
|
||||
"integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==",
|
||||
"version": "8.3.0",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz",
|
||||
"integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
|
||||
@@ -30,11 +30,11 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@biomejs/biome": "2.5.0",
|
||||
"@types/node": "^25.9.3",
|
||||
"@types/node": "^26.0.0",
|
||||
"browserslist": "^4.28.2",
|
||||
"browserslist-to-esbuild": "^2.1.1",
|
||||
"edge.js": "^6.5.1",
|
||||
"less": "^4.6.4",
|
||||
"less": "^4.6.6",
|
||||
"mathjs": "^15.2.0",
|
||||
"sharp": "~0.35.1",
|
||||
"sort-package-json": "^4.0.0",
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
.. _aol engine:
|
||||
|
||||
===
|
||||
AOL
|
||||
===
|
||||
|
||||
.. automodule:: searx.engines.aol
|
||||
:members:
|
||||
@@ -2,16 +2,16 @@ mock==5.2.0
|
||||
nose2[coverage_plugin]==0.16.0
|
||||
cov-core==1.15.0
|
||||
black==25.9.0
|
||||
pylint==4.0.5
|
||||
pylint==4.0.6
|
||||
splinter==0.21.0
|
||||
selenium==4.44.0
|
||||
selenium==4.45.0
|
||||
Sphinx==8.2.3;python_version <= "3.11"
|
||||
Sphinx==9.1.0; python_version > "3.11"
|
||||
sphinx-issues==6.0.0
|
||||
sphinx-jinja==2.0.2
|
||||
sphinx-tabs==3.5.0
|
||||
furo==2025.12.19
|
||||
sphinxcontrib-programoutput==0.19
|
||||
sphinxcontrib-programoutput==0.20
|
||||
sphinx-autobuild==2025.8.25
|
||||
sphinx-notfound-page==1.1.0
|
||||
myst-parser==5.0.0
|
||||
@@ -24,5 +24,5 @@ docutils>=0.21.2;python_version <= "3.11"
|
||||
docutils>=0.22.4; python_version > "3.11"
|
||||
parameterized==0.9.0
|
||||
granian[reload]==2.7.6
|
||||
basedpyright==1.39.7
|
||||
basedpyright==1.39.8
|
||||
types-lxml==2026.2.16
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
certifi==2026.5.20
|
||||
certifi==2026.6.17
|
||||
babel==2.18.0
|
||||
flask-babel==4.0.0
|
||||
flask==3.1.3
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -7165,6 +7165,7 @@
|
||||
"IR",
|
||||
"IS",
|
||||
"IT",
|
||||
"JE",
|
||||
"JM",
|
||||
"JO",
|
||||
"JP",
|
||||
@@ -7194,6 +7195,7 @@
|
||||
"MC",
|
||||
"MD",
|
||||
"ME",
|
||||
"MF",
|
||||
"MG",
|
||||
"MH",
|
||||
"MK",
|
||||
@@ -7260,6 +7262,7 @@
|
||||
"SS",
|
||||
"ST",
|
||||
"SV",
|
||||
"SX",
|
||||
"SY",
|
||||
"SZ",
|
||||
"TC",
|
||||
@@ -7407,6 +7410,7 @@
|
||||
"th": "thai",
|
||||
"ti": "tigrinya",
|
||||
"tk": "turkmen",
|
||||
"to": "tonga",
|
||||
"tr": "turkish",
|
||||
"tt": "tatar",
|
||||
"uk": "ukrainian",
|
||||
@@ -7687,460 +7691,22 @@
|
||||
"all_locale": null,
|
||||
"custom": {},
|
||||
"data_type": "traits_v1",
|
||||
"languages": {
|
||||
"af": "afrikaans",
|
||||
"am": "amharic",
|
||||
"ar": "arabic",
|
||||
"az": "azerbaijani",
|
||||
"be": "belarusian",
|
||||
"bg": "bulgarian",
|
||||
"bn": "bengali",
|
||||
"bs": "bosnian",
|
||||
"ca": "catalan",
|
||||
"cs": "czech",
|
||||
"cy": "welsh",
|
||||
"da": "dansk",
|
||||
"de": "deutsch",
|
||||
"el": "greek",
|
||||
"en": "english",
|
||||
"eo": "esperanto",
|
||||
"es": "espanol",
|
||||
"et": "estonian",
|
||||
"eu": "basque",
|
||||
"fa": "persian",
|
||||
"fi": "suomi",
|
||||
"fo": "faroese",
|
||||
"fr": "francais",
|
||||
"fy": "frisian",
|
||||
"ga": "irish",
|
||||
"gd": "gaelic",
|
||||
"gl": "galician",
|
||||
"gu": "gujarati",
|
||||
"he": "hebrew",
|
||||
"hi": "hindi",
|
||||
"hr": "croatian",
|
||||
"hu": "hungarian",
|
||||
"ia": "interlingua",
|
||||
"id": "indonesian",
|
||||
"is": "icelandic",
|
||||
"it": "italiano",
|
||||
"ja": "nihongo",
|
||||
"jv": "javanese",
|
||||
"ka": "georgian",
|
||||
"kn": "kannada",
|
||||
"ko": "hangul",
|
||||
"la": "latin",
|
||||
"lt": "lithuanian",
|
||||
"lv": "latvian",
|
||||
"mai": "bihari",
|
||||
"mk": "macedonian",
|
||||
"ml": "malayalam",
|
||||
"mr": "marathi",
|
||||
"ms": "malay",
|
||||
"mt": "maltese",
|
||||
"nb": "norsk",
|
||||
"ne": "nepali",
|
||||
"nl": "nederlands",
|
||||
"oc": "occitan",
|
||||
"pa": "punjabi",
|
||||
"pl": "polski",
|
||||
"pt": "portugues",
|
||||
"pt_BR": "brazilian",
|
||||
"ro": "romanian",
|
||||
"ru": "russian",
|
||||
"si": "sinhalese",
|
||||
"sk": "slovak",
|
||||
"sl": "slovenian",
|
||||
"sq": "albanian",
|
||||
"sr": "serbian",
|
||||
"su": "sudanese",
|
||||
"sv": "svenska",
|
||||
"sw": "swahili",
|
||||
"ta": "tamil",
|
||||
"te": "telugu",
|
||||
"th": "thai",
|
||||
"ti": "tigrinya",
|
||||
"tl": "tagalog",
|
||||
"tr": "turkce",
|
||||
"uk": "ukrainian",
|
||||
"ur": "urdu",
|
||||
"uz": "uzbek",
|
||||
"vi": "vietnamese",
|
||||
"xh": "xhosa",
|
||||
"zh": "jiantizhongwen",
|
||||
"zh_Hant": "fantizhengwen",
|
||||
"zu": "zulu"
|
||||
},
|
||||
"regions": {
|
||||
"ar-EG": "ar_EG",
|
||||
"bg-BG": "bg_BG",
|
||||
"ca-ES": "ca_ES",
|
||||
"cs-CZ": "cs_CZ",
|
||||
"da-DK": "da_DK",
|
||||
"de-AT": "de_AT",
|
||||
"de-CH": "de_CH",
|
||||
"de-DE": "de_DE",
|
||||
"el-GR": "el_GR",
|
||||
"en-AU": "en_AU",
|
||||
"en-CA": "en_CA",
|
||||
"en-GB": "en-GB_GB",
|
||||
"en-ID": "en_ID",
|
||||
"en-IE": "en_IE",
|
||||
"en-IN": "en_IN",
|
||||
"en-MY": "en_MY",
|
||||
"en-NZ": "en_NZ",
|
||||
"en-PH": "en_PH",
|
||||
"en-SG": "en_SG",
|
||||
"en-US": "en_US",
|
||||
"en-ZA": "en_ZA",
|
||||
"es-AR": "es_AR",
|
||||
"es-CL": "es_CL",
|
||||
"es-CO": "es_CO",
|
||||
"es-ES": "es_ES",
|
||||
"es-MX": "es_MX",
|
||||
"es-PE": "es_PE",
|
||||
"es-US": "es_US",
|
||||
"es-UY": "es_UY",
|
||||
"es-VE": "es_VE",
|
||||
"et-EE": "et_EE",
|
||||
"fi-FI": "fi_FI",
|
||||
"fil-PH": "fil_PH",
|
||||
"fr-BE": "fr_BE",
|
||||
"fr-CA": "fr_CA",
|
||||
"fr-CH": "fr_CH",
|
||||
"fr-FR": "fr_FR",
|
||||
"hi-IN": "hi_IN",
|
||||
"hu-HU": "hu_HU",
|
||||
"id-ID": "id_ID",
|
||||
"it-CH": "it_CH",
|
||||
"it-IT": "it_IT",
|
||||
"ja-JP": "ja_JP",
|
||||
"ko-KR": "ko_KR",
|
||||
"ms-MY": "ms_MY",
|
||||
"ms-SG": "ms_SG",
|
||||
"nb-NO": "no_NO",
|
||||
"nl-BE": "nl_BE",
|
||||
"nl-NL": "nl_NL",
|
||||
"pl-PL": "pl_PL",
|
||||
"pt-BR": "pt-BR_BR",
|
||||
"pt-PT": "pt_PT",
|
||||
"ro-RO": "ro_RO",
|
||||
"ru-BY": "ru_BY",
|
||||
"ru-RU": "ru_RU",
|
||||
"sv-SE": "sv_SE",
|
||||
"tr-TR": "tr_TR",
|
||||
"uk-UA": "uk_UA",
|
||||
"vi-VN": "vi_VN",
|
||||
"zh-CN": "zh-CN_CN",
|
||||
"zh-HK": "zh-TW_HK",
|
||||
"zh-TW": "zh-TW_TW"
|
||||
}
|
||||
"languages": {},
|
||||
"regions": {}
|
||||
},
|
||||
"startpage images": {
|
||||
"all_locale": null,
|
||||
"custom": {},
|
||||
"data_type": "traits_v1",
|
||||
"languages": {
|
||||
"af": "afrikaans",
|
||||
"am": "amharic",
|
||||
"ar": "arabic",
|
||||
"az": "azerbaijani",
|
||||
"be": "belarusian",
|
||||
"bg": "bulgarian",
|
||||
"bn": "bengali",
|
||||
"bs": "bosnian",
|
||||
"ca": "catalan",
|
||||
"cs": "czech",
|
||||
"cy": "welsh",
|
||||
"da": "dansk",
|
||||
"de": "deutsch",
|
||||
"el": "greek",
|
||||
"en": "english",
|
||||
"eo": "esperanto",
|
||||
"es": "espanol",
|
||||
"et": "estonian",
|
||||
"eu": "basque",
|
||||
"fa": "persian",
|
||||
"fi": "suomi",
|
||||
"fo": "faroese",
|
||||
"fr": "francais",
|
||||
"fy": "frisian",
|
||||
"ga": "irish",
|
||||
"gd": "gaelic",
|
||||
"gl": "galician",
|
||||
"gu": "gujarati",
|
||||
"he": "hebrew",
|
||||
"hi": "hindi",
|
||||
"hr": "croatian",
|
||||
"hu": "hungarian",
|
||||
"ia": "interlingua",
|
||||
"id": "indonesian",
|
||||
"is": "icelandic",
|
||||
"it": "italiano",
|
||||
"ja": "nihongo",
|
||||
"jv": "javanese",
|
||||
"ka": "georgian",
|
||||
"kn": "kannada",
|
||||
"ko": "hangul",
|
||||
"la": "latin",
|
||||
"lt": "lithuanian",
|
||||
"lv": "latvian",
|
||||
"mai": "bihari",
|
||||
"mk": "macedonian",
|
||||
"ml": "malayalam",
|
||||
"mr": "marathi",
|
||||
"ms": "malay",
|
||||
"mt": "maltese",
|
||||
"nb": "norsk",
|
||||
"ne": "nepali",
|
||||
"nl": "nederlands",
|
||||
"oc": "occitan",
|
||||
"pa": "punjabi",
|
||||
"pl": "polski",
|
||||
"pt": "portugues",
|
||||
"pt_BR": "brazilian",
|
||||
"ro": "romanian",
|
||||
"ru": "russian",
|
||||
"si": "sinhalese",
|
||||
"sk": "slovak",
|
||||
"sl": "slovenian",
|
||||
"sq": "albanian",
|
||||
"sr": "serbian",
|
||||
"su": "sudanese",
|
||||
"sv": "svenska",
|
||||
"sw": "swahili",
|
||||
"ta": "tamil",
|
||||
"te": "telugu",
|
||||
"th": "thai",
|
||||
"ti": "tigrinya",
|
||||
"tl": "tagalog",
|
||||
"tr": "turkce",
|
||||
"uk": "ukrainian",
|
||||
"ur": "urdu",
|
||||
"uz": "uzbek",
|
||||
"vi": "vietnamese",
|
||||
"xh": "xhosa",
|
||||
"zh": "jiantizhongwen",
|
||||
"zh_Hant": "fantizhengwen",
|
||||
"zu": "zulu"
|
||||
},
|
||||
"regions": {
|
||||
"ar-EG": "ar_EG",
|
||||
"bg-BG": "bg_BG",
|
||||
"ca-ES": "ca_ES",
|
||||
"cs-CZ": "cs_CZ",
|
||||
"da-DK": "da_DK",
|
||||
"de-AT": "de_AT",
|
||||
"de-CH": "de_CH",
|
||||
"de-DE": "de_DE",
|
||||
"el-GR": "el_GR",
|
||||
"en-AU": "en_AU",
|
||||
"en-CA": "en_CA",
|
||||
"en-GB": "en-GB_GB",
|
||||
"en-ID": "en_ID",
|
||||
"en-IE": "en_IE",
|
||||
"en-IN": "en_IN",
|
||||
"en-MY": "en_MY",
|
||||
"en-NZ": "en_NZ",
|
||||
"en-PH": "en_PH",
|
||||
"en-SG": "en_SG",
|
||||
"en-US": "en_US",
|
||||
"en-ZA": "en_ZA",
|
||||
"es-AR": "es_AR",
|
||||
"es-CL": "es_CL",
|
||||
"es-CO": "es_CO",
|
||||
"es-ES": "es_ES",
|
||||
"es-MX": "es_MX",
|
||||
"es-PE": "es_PE",
|
||||
"es-US": "es_US",
|
||||
"es-UY": "es_UY",
|
||||
"es-VE": "es_VE",
|
||||
"et-EE": "et_EE",
|
||||
"fi-FI": "fi_FI",
|
||||
"fil-PH": "fil_PH",
|
||||
"fr-BE": "fr_BE",
|
||||
"fr-CA": "fr_CA",
|
||||
"fr-CH": "fr_CH",
|
||||
"fr-FR": "fr_FR",
|
||||
"hi-IN": "hi_IN",
|
||||
"hu-HU": "hu_HU",
|
||||
"id-ID": "id_ID",
|
||||
"it-CH": "it_CH",
|
||||
"it-IT": "it_IT",
|
||||
"ja-JP": "ja_JP",
|
||||
"ko-KR": "ko_KR",
|
||||
"ms-MY": "ms_MY",
|
||||
"ms-SG": "ms_SG",
|
||||
"nb-NO": "no_NO",
|
||||
"nl-BE": "nl_BE",
|
||||
"nl-NL": "nl_NL",
|
||||
"pl-PL": "pl_PL",
|
||||
"pt-BR": "pt-BR_BR",
|
||||
"pt-PT": "pt_PT",
|
||||
"ro-RO": "ro_RO",
|
||||
"ru-BY": "ru_BY",
|
||||
"ru-RU": "ru_RU",
|
||||
"sv-SE": "sv_SE",
|
||||
"tr-TR": "tr_TR",
|
||||
"uk-UA": "uk_UA",
|
||||
"vi-VN": "vi_VN",
|
||||
"zh-CN": "zh-CN_CN",
|
||||
"zh-HK": "zh-TW_HK",
|
||||
"zh-TW": "zh-TW_TW"
|
||||
}
|
||||
"languages": {},
|
||||
"regions": {}
|
||||
},
|
||||
"startpage news": {
|
||||
"all_locale": null,
|
||||
"custom": {},
|
||||
"data_type": "traits_v1",
|
||||
"languages": {
|
||||
"af": "afrikaans",
|
||||
"am": "amharic",
|
||||
"ar": "arabic",
|
||||
"az": "azerbaijani",
|
||||
"be": "belarusian",
|
||||
"bg": "bulgarian",
|
||||
"bn": "bengali",
|
||||
"bs": "bosnian",
|
||||
"ca": "catalan",
|
||||
"cs": "czech",
|
||||
"cy": "welsh",
|
||||
"da": "dansk",
|
||||
"de": "deutsch",
|
||||
"el": "greek",
|
||||
"en": "english",
|
||||
"eo": "esperanto",
|
||||
"es": "espanol",
|
||||
"et": "estonian",
|
||||
"eu": "basque",
|
||||
"fa": "persian",
|
||||
"fi": "suomi",
|
||||
"fo": "faroese",
|
||||
"fr": "francais",
|
||||
"fy": "frisian",
|
||||
"ga": "irish",
|
||||
"gd": "gaelic",
|
||||
"gl": "galician",
|
||||
"gu": "gujarati",
|
||||
"he": "hebrew",
|
||||
"hi": "hindi",
|
||||
"hr": "croatian",
|
||||
"hu": "hungarian",
|
||||
"ia": "interlingua",
|
||||
"id": "indonesian",
|
||||
"is": "icelandic",
|
||||
"it": "italiano",
|
||||
"ja": "nihongo",
|
||||
"jv": "javanese",
|
||||
"ka": "georgian",
|
||||
"kn": "kannada",
|
||||
"ko": "hangul",
|
||||
"la": "latin",
|
||||
"lt": "lithuanian",
|
||||
"lv": "latvian",
|
||||
"mai": "bihari",
|
||||
"mk": "macedonian",
|
||||
"ml": "malayalam",
|
||||
"mr": "marathi",
|
||||
"ms": "malay",
|
||||
"mt": "maltese",
|
||||
"nb": "norsk",
|
||||
"ne": "nepali",
|
||||
"nl": "nederlands",
|
||||
"oc": "occitan",
|
||||
"pa": "punjabi",
|
||||
"pl": "polski",
|
||||
"pt": "portugues",
|
||||
"pt_BR": "brazilian",
|
||||
"ro": "romanian",
|
||||
"ru": "russian",
|
||||
"si": "sinhalese",
|
||||
"sk": "slovak",
|
||||
"sl": "slovenian",
|
||||
"sq": "albanian",
|
||||
"sr": "serbian",
|
||||
"su": "sudanese",
|
||||
"sv": "svenska",
|
||||
"sw": "swahili",
|
||||
"ta": "tamil",
|
||||
"te": "telugu",
|
||||
"th": "thai",
|
||||
"ti": "tigrinya",
|
||||
"tl": "tagalog",
|
||||
"tr": "turkce",
|
||||
"uk": "ukrainian",
|
||||
"ur": "urdu",
|
||||
"uz": "uzbek",
|
||||
"vi": "vietnamese",
|
||||
"xh": "xhosa",
|
||||
"zh": "jiantizhongwen",
|
||||
"zh_Hant": "fantizhengwen",
|
||||
"zu": "zulu"
|
||||
},
|
||||
"regions": {
|
||||
"ar-EG": "ar_EG",
|
||||
"bg-BG": "bg_BG",
|
||||
"ca-ES": "ca_ES",
|
||||
"cs-CZ": "cs_CZ",
|
||||
"da-DK": "da_DK",
|
||||
"de-AT": "de_AT",
|
||||
"de-CH": "de_CH",
|
||||
"de-DE": "de_DE",
|
||||
"el-GR": "el_GR",
|
||||
"en-AU": "en_AU",
|
||||
"en-CA": "en_CA",
|
||||
"en-GB": "en-GB_GB",
|
||||
"en-ID": "en_ID",
|
||||
"en-IE": "en_IE",
|
||||
"en-IN": "en_IN",
|
||||
"en-MY": "en_MY",
|
||||
"en-NZ": "en_NZ",
|
||||
"en-PH": "en_PH",
|
||||
"en-SG": "en_SG",
|
||||
"en-US": "en_US",
|
||||
"en-ZA": "en_ZA",
|
||||
"es-AR": "es_AR",
|
||||
"es-CL": "es_CL",
|
||||
"es-CO": "es_CO",
|
||||
"es-ES": "es_ES",
|
||||
"es-MX": "es_MX",
|
||||
"es-PE": "es_PE",
|
||||
"es-US": "es_US",
|
||||
"es-UY": "es_UY",
|
||||
"es-VE": "es_VE",
|
||||
"et-EE": "et_EE",
|
||||
"fi-FI": "fi_FI",
|
||||
"fil-PH": "fil_PH",
|
||||
"fr-BE": "fr_BE",
|
||||
"fr-CA": "fr_CA",
|
||||
"fr-CH": "fr_CH",
|
||||
"fr-FR": "fr_FR",
|
||||
"hi-IN": "hi_IN",
|
||||
"hu-HU": "hu_HU",
|
||||
"id-ID": "id_ID",
|
||||
"it-CH": "it_CH",
|
||||
"it-IT": "it_IT",
|
||||
"ja-JP": "ja_JP",
|
||||
"ko-KR": "ko_KR",
|
||||
"ms-MY": "ms_MY",
|
||||
"ms-SG": "ms_SG",
|
||||
"nb-NO": "no_NO",
|
||||
"nl-BE": "nl_BE",
|
||||
"nl-NL": "nl_NL",
|
||||
"pl-PL": "pl_PL",
|
||||
"pt-BR": "pt-BR_BR",
|
||||
"pt-PT": "pt_PT",
|
||||
"ro-RO": "ro_RO",
|
||||
"ru-BY": "ru_BY",
|
||||
"ru-RU": "ru_RU",
|
||||
"sv-SE": "sv_SE",
|
||||
"tr-TR": "tr_TR",
|
||||
"uk-UA": "uk_UA",
|
||||
"vi-VN": "vi_VN",
|
||||
"zh-CN": "zh-CN_CN",
|
||||
"zh-HK": "zh-TW_HK",
|
||||
"zh-TW": "zh-TW_TW"
|
||||
}
|
||||
"languages": {},
|
||||
"regions": {}
|
||||
},
|
||||
"wikidata": {
|
||||
"all_locale": null,
|
||||
@@ -8394,6 +7960,7 @@
|
||||
"inh",
|
||||
"io",
|
||||
"is",
|
||||
"isv",
|
||||
"it",
|
||||
"iu",
|
||||
"ja",
|
||||
@@ -8442,6 +8009,7 @@
|
||||
"ltg",
|
||||
"lv",
|
||||
"mad",
|
||||
"mag",
|
||||
"mai",
|
||||
"map-bms",
|
||||
"mdf",
|
||||
@@ -9585,4 +9153,4 @@
|
||||
},
|
||||
"regions": {}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -207,6 +207,7 @@ class EngineAbout(msgspec.Struct, kw_only=True):
|
||||
|
||||
use_official_api: bool = False
|
||||
"""SearXNG engine makes use of the official API or not"""
|
||||
|
||||
require_api_key: bool = False
|
||||
"""API requires a key or not."""
|
||||
|
||||
@@ -217,8 +218,7 @@ class EngineAbout(msgspec.Struct, kw_only=True):
|
||||
"""Brief description of the engine and where it gets its data from.
|
||||
|
||||
This value should only be set as long as no description of the data source
|
||||
is available via a :py:obj:`EngineAbout.wikidata_id`.
|
||||
"""
|
||||
is available via a :py:obj:`EngineAbout.wikidata_id`."""
|
||||
|
||||
language: str = ""
|
||||
"""Deprecated! Migrate your setting from `engine.about.language` to
|
||||
|
||||
@@ -1,210 +0,0 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""AOL supports WEB, image, and video search. Internally, it uses the Bing
|
||||
index.
|
||||
|
||||
AOL doesn't seem to support setting the language via request parameters, instead
|
||||
the results are based on the URL. For example, there is
|
||||
|
||||
- `search.aol.com <https://search.aol.com>`_ for English results
|
||||
- `suche.aol.de <https://suche.aol.de>`_ for German results
|
||||
|
||||
However, AOL offers its services only in a few regions:
|
||||
|
||||
- en-US: search.aol.com
|
||||
- de-DE: suche.aol.de
|
||||
- fr-FR: recherche.aol.fr
|
||||
- en-GB: search.aol.co.uk
|
||||
- en-CA: search.aol.ca
|
||||
|
||||
In order to still offer sufficient support for language and region, the `search
|
||||
keywords`_ known from Bing, ``language`` and ``loc`` (region), are added to the
|
||||
search term (AOL is basically just a proxy for Bing).
|
||||
|
||||
.. _search keywords:
|
||||
https://support.microsoft.com/en-us/topic/advanced-search-keywords-ea595928-5d63-4a0b-9c6b-0b769865e78a
|
||||
|
||||
"""
|
||||
|
||||
from urllib.parse import urlencode, unquote_plus
|
||||
import typing as t
|
||||
|
||||
from lxml import html
|
||||
from dateutil import parser
|
||||
|
||||
from searx.result_types import EngineResults
|
||||
from searx.utils import eval_xpath_list, eval_xpath, extract_text
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.extended_types import SXNG_Response
|
||||
from searx.search.processors import OnlineParams
|
||||
|
||||
about = {
|
||||
"website": "https://www.aol.com",
|
||||
"wikidata_id": "Q27585",
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "HTML",
|
||||
}
|
||||
|
||||
categories = ["general"]
|
||||
search_type = "search" # supported: search, image, video
|
||||
|
||||
paging = True
|
||||
safesearch = True
|
||||
time_range_support = True
|
||||
results_per_page = 10
|
||||
|
||||
|
||||
base_url = "https://search.aol.com"
|
||||
time_range_map = {"day": "1d", "week": "1w", "month": "1m", "year": "1y"}
|
||||
safesearch_map = {0: "p", 1: "r", 2: "i"}
|
||||
|
||||
enable_http2 = False
|
||||
|
||||
|
||||
def init(_):
|
||||
if search_type not in ("search", "image", "video"):
|
||||
raise ValueError(f"unsupported search type {search_type}")
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams") -> None:
|
||||
|
||||
language, region = (params["searxng_locale"].split("-") + [None])[:2]
|
||||
if language and language != "all":
|
||||
query = f"{query} language:{language}"
|
||||
if region:
|
||||
query = f"{query} loc:{region}"
|
||||
|
||||
args: dict[str, str | int | None] = {
|
||||
"q": query,
|
||||
"b": params["pageno"] * results_per_page + 1, # page is 1-indexed
|
||||
"pz": results_per_page,
|
||||
}
|
||||
|
||||
if params["time_range"]:
|
||||
args["fr2"] = "time"
|
||||
args["age"] = params["time_range"]
|
||||
else:
|
||||
args["fr2"] = "sb-top-search"
|
||||
|
||||
params["cookies"]["sB"] = f"vm={safesearch_map[params['safesearch']]}"
|
||||
params["url"] = f"{base_url}/aol/{search_type}?{urlencode(args)}"
|
||||
logger.debug(params)
|
||||
|
||||
|
||||
def _deobfuscate_url(obfuscated_url: str) -> str | None:
|
||||
# URL looks like "https://search.aol.com/click/_ylt=AwjFSDjd;_ylu=JfsdjDFd/RV=2/RE=1774058166/RO=10/RU=https%3a%2f%2fen.wikipedia.org%2fwiki%2fTree/RK=0/RS=BP2CqeMLjscg4n8cTmuddlEQA2I-" # pylint: disable=line-too-long
|
||||
if not obfuscated_url:
|
||||
return None
|
||||
|
||||
for part in obfuscated_url.split("/"):
|
||||
if part.startswith("RU="):
|
||||
return unquote_plus(part[3:])
|
||||
# pattern for de-obfuscating URL not found, fall back to Yahoo's tracking link
|
||||
return obfuscated_url
|
||||
|
||||
|
||||
def _general_results(doc: html.HtmlElement) -> EngineResults:
|
||||
res = EngineResults()
|
||||
|
||||
for result in eval_xpath_list(doc, "//div[@id='web']//ol/li[not(contains(@class, 'first'))]"):
|
||||
obfuscated_url = extract_text(eval_xpath(result, ".//h3/a/@href"))
|
||||
if not obfuscated_url:
|
||||
continue
|
||||
|
||||
url = _deobfuscate_url(obfuscated_url)
|
||||
if not url:
|
||||
continue
|
||||
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=url,
|
||||
title=extract_text(eval_xpath(result, ".//h3/a")) or "",
|
||||
content=extract_text(eval_xpath(result, ".//div[contains(@class, 'compText')]")) or "",
|
||||
thumbnail=extract_text(eval_xpath(result, ".//a[contains(@class, 'thm')]/img/@data-src")) or "",
|
||||
)
|
||||
)
|
||||
return res
|
||||
|
||||
|
||||
def _video_results(doc: html.HtmlElement) -> EngineResults:
|
||||
res = EngineResults()
|
||||
|
||||
for result in eval_xpath_list(doc, "//div[contains(@class, 'results')]//ol/li"):
|
||||
obfuscated_url = extract_text(eval_xpath(result, ".//a/@href"))
|
||||
if not obfuscated_url:
|
||||
continue
|
||||
|
||||
url = _deobfuscate_url(obfuscated_url)
|
||||
if not url:
|
||||
continue
|
||||
|
||||
published_date_raw = extract_text(eval_xpath(result, ".//div[contains(@class, 'v-age')]"))
|
||||
try:
|
||||
published_date = parser.parse(published_date_raw or "")
|
||||
except parser.ParserError:
|
||||
published_date = None
|
||||
|
||||
res.add(
|
||||
res.types.LegacyResult(
|
||||
{
|
||||
"template": "videos.html",
|
||||
"url": url,
|
||||
"title": extract_text(eval_xpath(result, ".//h3")),
|
||||
"content": extract_text(eval_xpath(result, ".//div[contains(@class, 'compText')]")),
|
||||
"thumbnail": extract_text(eval_xpath(result, ".//img[contains(@class, 'thm')]/@src")),
|
||||
"length": extract_text(eval_xpath(result, ".//span[contains(@class, 'v-time')]")),
|
||||
"publishedDate": published_date,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
|
||||
|
||||
def _image_results(doc: html.HtmlElement) -> EngineResults:
|
||||
res = EngineResults()
|
||||
|
||||
for result in eval_xpath_list(doc, "//section[@id='results']//ul/li"):
|
||||
obfuscated_url = extract_text(eval_xpath(result, "./a/@href"))
|
||||
if not obfuscated_url:
|
||||
continue
|
||||
|
||||
url = _deobfuscate_url(obfuscated_url)
|
||||
if not url:
|
||||
continue
|
||||
|
||||
res.add(
|
||||
res.types.LegacyResult(
|
||||
{
|
||||
"template": "images.html",
|
||||
# results don't have an extra URL, only the image source
|
||||
"url": url,
|
||||
"title": extract_text(eval_xpath(result, ".//a/@aria-label")),
|
||||
"thumbnail_src": extract_text(eval_xpath(result, ".//img/@src")),
|
||||
"img_src": url,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response") -> EngineResults:
|
||||
doc = html.fromstring(resp.text)
|
||||
|
||||
match search_type:
|
||||
case "search":
|
||||
results = _general_results(doc)
|
||||
case "image":
|
||||
results = _image_results(doc)
|
||||
case "video":
|
||||
results = _video_results(doc)
|
||||
case _:
|
||||
raise ValueError("unsupported search type")
|
||||
|
||||
for suggestion in eval_xpath_list(doc, ".//ol[contains(@class, 'searchRightBottom')]//table//a"):
|
||||
results.add(results.types.LegacyResult({"suggestion": extract_text(suggestion)}))
|
||||
|
||||
return results
|
||||
@@ -14,7 +14,6 @@ from searx.extended_types import SXNG_Response
|
||||
from searx.network import get, post
|
||||
from searx.result_types import EngineResults
|
||||
from searx.utils import html_to_text
|
||||
from searx.enginelib import EngineCache
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.search.processors import OnlineParams
|
||||
@@ -42,21 +41,7 @@ search_index = "cw22"
|
||||
<https://www.chatnoir.eu/docs/api-general>`_ for a full list."""
|
||||
|
||||
|
||||
CACHE: EngineCache
|
||||
"""Cache to store session info (i.e. api key, csrf token, session id)."""
|
||||
|
||||
|
||||
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
||||
global CACHE # pylint: disable=global-statement
|
||||
CACHE = EngineCache(engine_settings["name"])
|
||||
return True
|
||||
|
||||
|
||||
def _obtain_api_key() -> tuple[str, str, str]:
|
||||
cached_session = CACHE.get("session")
|
||||
if cached_session:
|
||||
return tuple(cached_session.split("|"))
|
||||
|
||||
home_resp = get(base_url)
|
||||
if not home_resp.ok:
|
||||
raise SearxEngineAPIException("failed to obtain api key")
|
||||
@@ -76,10 +61,6 @@ def _obtain_api_key() -> tuple[str, str, str]:
|
||||
session_id = token_resp.cookies["sessionid"]
|
||||
scraped_api_key = token_resp.json()["token"]["token"]
|
||||
|
||||
# session keys seem to become rate-limited very fast, so only remembering
|
||||
# for 1 minute here
|
||||
CACHE.set("session", f"{csrf_token}|{session_id}|{scraped_api_key}", expire=60)
|
||||
|
||||
return csrf_token, session_id, scraped_api_key
|
||||
|
||||
|
||||
|
||||
@@ -19,11 +19,9 @@ Configuration
|
||||
|
||||
The engine has the following additional settings:
|
||||
|
||||
- :py:obj:`chinaso_category` (:py:obj:`ChinasoCategoryType`)
|
||||
- :py:obj:`chinaso_news_source` (:py:obj:`ChinasoNewsSourceType`)
|
||||
|
||||
In the example below, all three ChinaSO engines are using the :ref:`network
|
||||
<engine network>` from the ``chinaso news`` engine.
|
||||
In the example below, ChinaSO is configured for news search.
|
||||
|
||||
.. code:: yaml
|
||||
|
||||
@@ -31,23 +29,8 @@ In the example below, all three ChinaSO engines are using the :ref:`network
|
||||
engine: chinaso
|
||||
shortcut: chinaso
|
||||
categories: [news]
|
||||
chinaso_category: news
|
||||
chinaso_news_source: all
|
||||
|
||||
- name: chinaso images
|
||||
engine: chinaso
|
||||
network: chinaso news
|
||||
shortcut: chinasoi
|
||||
categories: [images]
|
||||
chinaso_category: images
|
||||
|
||||
- name: chinaso videos
|
||||
engine: chinaso
|
||||
network: chinaso news
|
||||
shortcut: chinasov
|
||||
categories: [videos]
|
||||
chinaso_category: videos
|
||||
|
||||
|
||||
Implementations
|
||||
===============
|
||||
@@ -78,19 +61,6 @@ results_per_page = 10
|
||||
categories = []
|
||||
language = "zh"
|
||||
|
||||
ChinasoCategoryType = t.Literal['news', 'videos', 'images']
|
||||
"""ChinaSo supports news, videos, images search.
|
||||
|
||||
- ``news``: search for news
|
||||
- ``videos``: search for videos
|
||||
- ``images``: search for images
|
||||
|
||||
In the category ``news`` you can additionally filter by option
|
||||
:py:obj:`chinaso_news_source`.
|
||||
"""
|
||||
chinaso_category = 'news'
|
||||
"""Configure ChinaSo category (:py:obj:`ChinasoCategoryType`)."""
|
||||
|
||||
ChinasoNewsSourceType = t.Literal['CENTRAL', 'LOCAL', 'BUSINESS', 'EPAPER', 'all']
|
||||
"""Filtering ChinaSo-News results by source:
|
||||
|
||||
@@ -109,39 +79,24 @@ base_url = "https://www.chinaso.com"
|
||||
|
||||
|
||||
def init(_):
|
||||
if chinaso_category not in ('news', 'videos', 'images'):
|
||||
raise ValueError(f"Unsupported category: {chinaso_category}")
|
||||
if chinaso_category == 'news' and chinaso_news_source not in t.get_args(ChinasoNewsSourceType):
|
||||
if chinaso_news_source not in t.get_args(ChinasoNewsSourceType):
|
||||
raise ValueError(f"Unsupported news source: {chinaso_news_source}")
|
||||
|
||||
|
||||
def request(query, params):
|
||||
query_params = {"q": query}
|
||||
query_params = {'q': query, 'pn': params["pageno"], 'ps': results_per_page}
|
||||
|
||||
if time_range_dict.get(params['time_range']):
|
||||
query_params["stime"] = time_range_dict[params['time_range']]
|
||||
query_params["etime"] = 'now'
|
||||
|
||||
category_config = {
|
||||
'news': {'endpoint': '/v5/general/v1/web/search', 'params': {'pn': params["pageno"], 'ps': results_per_page}},
|
||||
'images': {
|
||||
'endpoint': '/v5/general/v1/search/image',
|
||||
'params': {'start_index': (params["pageno"] - 1) * results_per_page, 'rn': results_per_page},
|
||||
},
|
||||
'videos': {
|
||||
'endpoint': '/v5/general/v1/search/video',
|
||||
'params': {'start_index': (params["pageno"] - 1) * results_per_page, 'rn': results_per_page},
|
||||
},
|
||||
}
|
||||
if chinaso_news_source != 'all':
|
||||
if chinaso_news_source == 'EPAPER':
|
||||
category_config['news']['params']["type"] = 'EPAPER'
|
||||
query_params["type"] = 'EPAPER'
|
||||
else:
|
||||
category_config['news']['params']["cate"] = chinaso_news_source
|
||||
query_params["cate"] = chinaso_news_source
|
||||
|
||||
query_params.update(category_config[chinaso_category]['params'])
|
||||
|
||||
params["url"] = f"{base_url}{category_config[chinaso_category]['endpoint']}?{urlencode(query_params)}"
|
||||
params["url"] = f"{base_url}/v5/general/v1/web/search?{urlencode(query_params)}"
|
||||
cookie = {
|
||||
"uid": base64.b64encode(secrets.token_bytes(16)).decode("utf-8"),
|
||||
}
|
||||
@@ -163,12 +118,6 @@ def response(resp):
|
||||
if not data["data"]:
|
||||
return []
|
||||
|
||||
parsers = {'news': parse_news, 'images': parse_images, 'videos': parse_videos}
|
||||
|
||||
return parsers[chinaso_category](data)
|
||||
|
||||
|
||||
def parse_news(data):
|
||||
results = []
|
||||
if not data.get("data", {}).get("data"):
|
||||
raise SearxEngineAPIException("Invalid response")
|
||||
@@ -190,47 +139,3 @@ def parse_news(data):
|
||||
}
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
def parse_images(data):
|
||||
results = []
|
||||
if not data.get("data", {}).get("arrRes"):
|
||||
raise SearxEngineAPIException("Invalid response")
|
||||
|
||||
for entry in data["data"]["arrRes"]:
|
||||
results.append(
|
||||
{
|
||||
'url': entry["web_url"],
|
||||
'title': html_to_text(entry["title"]),
|
||||
'content': html_to_text(entry.get("ImageInfo", "")),
|
||||
'template': 'images.html',
|
||||
'img_src': entry["url"].replace("http://", "https://"),
|
||||
'thumbnail_src': entry["largeimage"].replace("http://", "https://"),
|
||||
}
|
||||
)
|
||||
return results
|
||||
|
||||
|
||||
def parse_videos(data):
|
||||
results = []
|
||||
if not data.get("data", {}).get("arrRes"):
|
||||
raise SearxEngineAPIException("Invalid response")
|
||||
|
||||
for entry in data["data"]["arrRes"]:
|
||||
published_date = None
|
||||
if entry.get("VideoPubDate"):
|
||||
try:
|
||||
published_date = datetime.fromtimestamp(int(entry["VideoPubDate"]))
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
results.append(
|
||||
{
|
||||
'url': entry["url"],
|
||||
'title': html_to_text(entry["raw_title"]),
|
||||
'template': 'videos.html',
|
||||
'publishedDate': published_date,
|
||||
'thumbnail': entry["image_src"].replace("http://", "https://"),
|
||||
}
|
||||
)
|
||||
return results
|
||||
|
||||
118
searx/engines/findfiles.py
Normal file
118
searx/engines/findfiles.py
Normal file
@@ -0,0 +1,118 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""FindFiles.net_ is a Germany-based file search engine.
|
||||
|
||||
FindFiles.net_ is a specialized file search engine designed to help you search
|
||||
files online with precision. Unlike traditional search engines that mainly index
|
||||
web pages, FindFiles focuses on finding real files on the internet - including
|
||||
PDFs, documents, archives, videos, datasets, and more.
|
||||
|
||||
.. _FindFiles.net: https://findfiles.net
|
||||
"""
|
||||
|
||||
from os.path import basename
|
||||
from urllib.parse import urlencode
|
||||
import typing as t
|
||||
|
||||
from lxml import html
|
||||
|
||||
from searx.result_types import EngineResults
|
||||
from searx.utils import extract_text, eval_xpath, eval_xpath_list
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from extended_types import SXNG_Response
|
||||
from search.processors import OnlineParams
|
||||
|
||||
about = {
|
||||
"website": "https://findfiles.net",
|
||||
"wikidata_id": None,
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "HTML",
|
||||
}
|
||||
|
||||
base_url = "https://findfiles.net"
|
||||
categories = ["files"]
|
||||
paging = True
|
||||
safeserach = True
|
||||
|
||||
safesearch_map = {
|
||||
0: "contentguard.off",
|
||||
1: "contentguard.moderate",
|
||||
2: "contentguard.strict",
|
||||
}
|
||||
|
||||
FindFilesCategory = t.Literal[
|
||||
"all",
|
||||
"document",
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"video",
|
||||
]
|
||||
FINDFILES_CATEGORIES = t.get_args(FindFilesCategory)
|
||||
|
||||
findfiles_categ: FindFilesCategory = "all"
|
||||
"""Category to search in."""
|
||||
|
||||
|
||||
def setup(_: dict[str, t.Any]) -> bool:
|
||||
if findfiles_categ not in FINDFILES_CATEGORIES:
|
||||
raise ValueError("invalid category: %s" % findfiles_categ)
|
||||
return True
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams") -> None:
|
||||
args = {
|
||||
"query": query,
|
||||
"contentguard": safesearch_map[params["safesearch"]],
|
||||
"page": params["pageno"],
|
||||
}
|
||||
# the language in the path doesn't change anything about the results, it
|
||||
# only changes the UI
|
||||
params["url"] = f"{base_url}/en/serp/{findfiles_categ}/?{urlencode(args)}"
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response") -> EngineResults:
|
||||
res = EngineResults()
|
||||
|
||||
dom = html.fromstring(resp.text)
|
||||
if findfiles_categ == "image":
|
||||
for result in eval_xpath_list(
|
||||
dom, "//div[contains(@class, 'image-mosaic')]/div[contains(@class, 'image-item')]"
|
||||
):
|
||||
res.add(
|
||||
res.types.Image(
|
||||
url=extract_text(eval_xpath(result, ".//div[contains(@class, 'caption')]/a/@href")) or "",
|
||||
title=extract_text(eval_xpath(result, ".//div[contains(@class, 'caption')]/a")) or "",
|
||||
thumbnail_src=extract_text(eval_xpath(result, ".//img/@src")) or "",
|
||||
)
|
||||
)
|
||||
elif findfiles_categ == "video":
|
||||
for result in eval_xpath_list(
|
||||
dom, "//div[contains(@class, 'video-mosaic')]/div[contains(@class, 'video-item')]"
|
||||
):
|
||||
video_src = extract_text(eval_xpath(result, ".//video/@src")) or ""
|
||||
res.add(
|
||||
res.types.LegacyResult(
|
||||
template="videos.html",
|
||||
url=video_src,
|
||||
title=extract_text(eval_xpath(result, ".//div[contains(@class, 'caption')]/span")) or "",
|
||||
iframe_src=video_src or "",
|
||||
)
|
||||
)
|
||||
else:
|
||||
for result in eval_xpath_list(dom, "//ol/li[contains(@class, 'result-item')]/article"):
|
||||
filename = basename(extract_text(eval_xpath(result, ".//h3")) or "")
|
||||
res.add(
|
||||
res.types.File(
|
||||
url=extract_text(eval_xpath(result, ".//h3/a/@href")) or "",
|
||||
title=filename,
|
||||
content=" ".join(extract_text(el) or "" for el in eval_xpath_list(result, "./div/span")),
|
||||
filename=filename,
|
||||
size=extract_text(eval_xpath(result, "(.//span[@id])[1]")) or "",
|
||||
embedded=extract_text(eval_xpath(result, ".//audio/@src")) or "",
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
127
searx/engines/giphy.py
Normal file
127
searx/engines/giphy.py
Normal file
@@ -0,0 +1,127 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""Giphy (images)"""
|
||||
|
||||
import random
|
||||
from urllib.parse import urlencode
|
||||
import re
|
||||
|
||||
import typing as t
|
||||
|
||||
from lxml import html
|
||||
|
||||
from searx.enginelib import EngineCache
|
||||
from searx.exceptions import SearxEngineAPIException
|
||||
from searx.network import get
|
||||
from searx.result_types import EngineResults
|
||||
from searx.result_types.image import ImageRef
|
||||
from searx.utils import eval_xpath_list, humanize_bytes
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.extended_types import SXNG_Response
|
||||
from searx.search.processors import OnlineParams
|
||||
|
||||
|
||||
about = {
|
||||
"website": "https://giphy.com",
|
||||
"wikidata_id": "Q17054335",
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "JSON",
|
||||
}
|
||||
|
||||
base_url = "https://giphy.com"
|
||||
api_url = "https://api.giphy.com"
|
||||
|
||||
categories = ["images"]
|
||||
paging = True
|
||||
page_size = 15
|
||||
|
||||
GiphyCategs = t.Literal["gifs", "stickers", "clips"]
|
||||
giphy_categ: GiphyCategs = "gifs"
|
||||
"""Giphy category to search in."""
|
||||
|
||||
CACHE: EngineCache
|
||||
"""Cache for storing the extracted api key."""
|
||||
|
||||
|
||||
_GIPHY_API_KEY_RE = re.compile(r"[Aa]piKey\s*:\s*\"(\w+)\"")
|
||||
|
||||
|
||||
def setup(engine_settings: dict[str, str]) -> bool:
|
||||
if giphy_categ not in t.get_args(GiphyCategs):
|
||||
raise ValueError("invalid category: %s" % giphy_categ)
|
||||
|
||||
global CACHE # pylint: disable=global-statement
|
||||
CACHE = EngineCache(engine_settings["name"])
|
||||
|
||||
return True
|
||||
|
||||
|
||||
def _get_api_key() -> str:
|
||||
"""
|
||||
Extract the Giphy API key from the JavaScript code. There are different API keys
|
||||
(e.g. for mobile, desktop, ...), so we just pick a random one of these.
|
||||
"""
|
||||
cached = CACHE.get("api_key")
|
||||
if cached:
|
||||
return cached
|
||||
|
||||
homepage_resp = get(base_url)
|
||||
homepage_doc = html.fromstring(homepage_resp.text)
|
||||
|
||||
for script_src in eval_xpath_list(homepage_doc, "//script[contains(@src, 'layout')]/@src"):
|
||||
script_resp = get(base_url + script_src)
|
||||
api_keys = _GIPHY_API_KEY_RE.findall(script_resp.text)
|
||||
if api_keys:
|
||||
api_key = random.choice(api_keys)
|
||||
CACHE.set("api_key", api_key, expire=60 * 60 * 6) # 6 hours
|
||||
return api_key
|
||||
|
||||
raise SearxEngineAPIException("failed to extract api keys")
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams") -> None:
|
||||
args = {
|
||||
"q": query,
|
||||
"api_key": _get_api_key(),
|
||||
"limit": page_size,
|
||||
"offset": (params["pageno"] - 1) * page_size,
|
||||
"type": giphy_categ,
|
||||
}
|
||||
params["url"] = f"{api_url}/v1/{giphy_categ}/search?{urlencode(args)}"
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response"):
|
||||
res = EngineResults()
|
||||
|
||||
result: dict[str, t.Any]
|
||||
for result in resp.json()["data"]:
|
||||
img = result['images']['original']
|
||||
formats = [
|
||||
ImageRef(url=img["mp4"], subtype="mp4"), # type: ignore
|
||||
ImageRef(url=img["webp"], subtype="webp"), # type: ignore
|
||||
]
|
||||
thumb = (
|
||||
result["images"].get("downsized")
|
||||
or result["images"].get("downsized_medium")
|
||||
or result["images"].get("downsized_small")
|
||||
or result["images"].get("downsized_large")
|
||||
)
|
||||
res.add(
|
||||
res.types.Image(
|
||||
title=result["title"],
|
||||
content=", ".join(result.get("tags", [])),
|
||||
url=result["url"],
|
||||
thumbnail_src=thumb.get("url") or img["url"],
|
||||
img_src=img["url"],
|
||||
resolution=f"{img['width']}x{img['height']}",
|
||||
img_format="GIF",
|
||||
formats=formats,
|
||||
author=result["username"],
|
||||
filesize=humanize_bytes(int(img["size"])),
|
||||
source=result.get("source_tld") or "",
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
88
searx/engines/iseek.py
Normal file
88
searx/engines/iseek.py
Normal file
@@ -0,0 +1,88 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""iseek_ is a search engine by the AI company Vantage Labs LLC,
|
||||
that focuses on medical and educational applicances.
|
||||
Although it's an AI company, it doesn't include any AI stuff in its results.
|
||||
|
||||
.. _iseek : https://www.iseek.ai/
|
||||
"""
|
||||
|
||||
import base64
|
||||
from hashlib import sha256
|
||||
import typing as t
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from searx.result_types import EngineResults
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.search.processors import OnlineParams
|
||||
from searx.extended_types import SXNG_Response
|
||||
|
||||
|
||||
about = {
|
||||
"website": 'https://www.iseek.com',
|
||||
"wikidata_id": None,
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "JSON",
|
||||
}
|
||||
categories = ["general"]
|
||||
paging = True
|
||||
|
||||
base_url = "https://api.iseek.com"
|
||||
page_size = 10
|
||||
|
||||
|
||||
def _get_new_token(query: str, pageno: int) -> str:
|
||||
"""Create a new ``qToken``. This reduced the time for fetching subsequent pages
|
||||
from 4 seconds to 200ms when testing."""
|
||||
# The website uses a random value as qToken for the first page. For our use case,
|
||||
# it's easier if the qToken can be deterministically re-calculated based on the search query,
|
||||
# so that we can the same result when calling _get_new_token for the second, third, ... page
|
||||
#
|
||||
# var qToken = Math.ceil(Math.random() * parseInt("ZZZZ", 36)).toString(36);
|
||||
# while (qToken.length < 4) qToken = '0' + qToken;
|
||||
# qToken = qToken + "_" + pageno
|
||||
query_hash = sha256(query.encode()).digest()
|
||||
hash_start = base64.b64encode(query_hash).decode()[0:4]
|
||||
return f"{hash_start}_{pageno}"
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams"):
|
||||
offset = (params["pageno"] - 1) * page_size
|
||||
|
||||
# always seems to find 20 results max
|
||||
if offset >= 20:
|
||||
params["url"] = None
|
||||
return
|
||||
|
||||
args = {
|
||||
"q": query,
|
||||
"key": "core-web",
|
||||
"num": str(page_size),
|
||||
"off": offset,
|
||||
"rSort": "__metasearch_score_d:desc",
|
||||
# it supports many more fields, but none of them are really relevant
|
||||
"names": "title_t,content_txt,url_s",
|
||||
"qNames": "title_t",
|
||||
"qToken": _get_new_token(query, params["pageno"]),
|
||||
}
|
||||
params["url"] = f"{base_url}/search?{urlencode(args)}"
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response") -> EngineResults:
|
||||
res = EngineResults()
|
||||
|
||||
for group in resp.json()["data"]:
|
||||
group: dict[str, t.Any]
|
||||
for result in group["doclist"]["docs"]:
|
||||
result: dict[str, str]
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=result["url_s"],
|
||||
title=result["title_t"],
|
||||
content="".join(result["content_txt"]),
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
@@ -44,7 +44,7 @@ about = {
|
||||
|
||||
base_url = "https://api2.marginalia-search.com"
|
||||
safesearch = True
|
||||
categories = ["general"]
|
||||
categories = ["general", "blogs"]
|
||||
paging = True
|
||||
results_per_page = 20
|
||||
api_key = None
|
||||
|
||||
@@ -101,7 +101,7 @@ def _image_results(doc: "ElementBase") -> EngineResults:
|
||||
res.types.Image(
|
||||
url=extract_text(eval_xpath(result, "./@href")) or "",
|
||||
title=extract_text(eval_xpath(result, "./img/@alt")) or "",
|
||||
thumbnail_src=extract_text(eval_xpath(result, "./img/@src")) or "",
|
||||
img_src=extract_text(eval_xpath(result, "./img/@src")) or "",
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
107
searx/engines/startpagina.py
Normal file
107
searx/engines/startpagina.py
Normal file
@@ -0,0 +1,107 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""Startpagina is a Netherlands search engine by `Kompas`_. It takes all its
|
||||
results from Google.
|
||||
|
||||
.. _Kompas: https://www.kompaspublishing.nl/
|
||||
"""
|
||||
|
||||
import typing as t
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from dateutil import parser
|
||||
|
||||
from searx.utils import format_duration
|
||||
from searx.result_types import EngineResults
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.extended_types import SXNG_Response
|
||||
from searx.search.processors import OnlineParams
|
||||
|
||||
about = {
|
||||
"website": "https://startpagina.nl",
|
||||
"wikidata_id": None,
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "JSON",
|
||||
}
|
||||
|
||||
language = "ne"
|
||||
paging = True
|
||||
safesearch = True
|
||||
|
||||
categories = ["general"]
|
||||
startpagina_categ = "web"
|
||||
"""Category to search in. Can be either "web", "images", "videos" or "news"."""
|
||||
page_size = 10
|
||||
|
||||
|
||||
api_url = "https://search.kompas.services"
|
||||
|
||||
|
||||
def init(_):
|
||||
if startpagina_categ not in ("web", "images", "videos", "news"):
|
||||
raise ValueError("invalid search type: %s" % startpagina_categ)
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams") -> None:
|
||||
args = {"q": query, "page_size": page_size, "page": params["pageno"]}
|
||||
params["url"] = f"{api_url}/api/v2/search/{startpagina_categ}/?{urlencode(args)}"
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response"):
|
||||
res = EngineResults()
|
||||
|
||||
json_resp = resp.json()
|
||||
|
||||
for result in json_resp["results"]:
|
||||
if startpagina_categ == "web":
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=result["original_url"],
|
||||
title=result["title"],
|
||||
content=result["description"],
|
||||
)
|
||||
)
|
||||
elif startpagina_categ == "news":
|
||||
publishedDate = None
|
||||
try:
|
||||
publishedDate = parser.parse(result["date"])
|
||||
except parser.ParserError:
|
||||
pass
|
||||
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=result["original_url"],
|
||||
title=result["title"],
|
||||
content=result["description"],
|
||||
thumbnail=result["image"]["thumbnail_url"],
|
||||
publishedDate=publishedDate,
|
||||
)
|
||||
)
|
||||
elif startpagina_categ == "videos":
|
||||
res.add(
|
||||
res.types.LegacyResult(
|
||||
template="videos.html",
|
||||
url=result["original_url"],
|
||||
title=result["title"],
|
||||
content=result["description"],
|
||||
thumbnail=result["video"]["thumbnail_url"],
|
||||
length=format_duration(result["video"]["duration"]),
|
||||
)
|
||||
)
|
||||
elif startpagina_categ == "images":
|
||||
res.add(
|
||||
res.types.Image(
|
||||
url=result["original_url"],
|
||||
title=result["title"],
|
||||
content=result["description"],
|
||||
thumbnail_src=result["image"]["thumbnail_url"],
|
||||
resolution=f"{result['image']['width']}x{result['image']['height']}",
|
||||
)
|
||||
)
|
||||
|
||||
for related in json_resp["related_searches"]:
|
||||
res.add(res.types.LegacyResult(suggestion=related["query"]))
|
||||
|
||||
return res
|
||||
@@ -73,7 +73,7 @@ def _obtain_session_code() -> str:
|
||||
if cached_session:
|
||||
return cached_session
|
||||
|
||||
results_page = get(f"{base_url}/_internCode.aspx")
|
||||
results_page = get(f"{base_url}/checkCode.aspx")
|
||||
doc = html.fromstring(results_page.text)
|
||||
|
||||
extra_data: dict[str, str] = {}
|
||||
@@ -107,7 +107,7 @@ def _obtain_session_code() -> str:
|
||||
}
|
||||
|
||||
challenge_response = post(
|
||||
f"{base_url}/_internCode.aspx",
|
||||
f"{base_url}/checkCode.aspx",
|
||||
cookies=results_page.cookies,
|
||||
data=data,
|
||||
)
|
||||
@@ -125,7 +125,9 @@ def request(query: str, params: "OnlineParams"):
|
||||
code = _obtain_session_code()
|
||||
args = {"w": query, "page": params["pageno"]}
|
||||
params["url"] = f"{base_url}/{tiger_category}?{urlencode(args)}"
|
||||
params["cookies"]["Tiger.ch"] = f"Code={code}"
|
||||
# Setting Checked=1 shows related search terms / suggestions
|
||||
# Language and country could be set with Lng= and Land= in the future
|
||||
params["cookies"]["Tiger.ch"] = f"Tiger.ch=&Code={code}&Checked=1"
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response") -> EngineResults:
|
||||
@@ -144,6 +146,9 @@ def response(resp: "SXNG_Response") -> EngineResults:
|
||||
content=extract_text(eval_xpath(result, ".//*[contains(@class, 'webbodynopic')]")) or "",
|
||||
)
|
||||
)
|
||||
|
||||
for suggestion in eval_xpath_list(doc, "//a[contains(@class, 'linkAnders')]"):
|
||||
res.add(res.types.LegacyResult(suggestion=extract_text(suggestion)))
|
||||
elif tiger_category == "News":
|
||||
for result in eval_xpath_list(doc, "//div[@id='panNews']/div"):
|
||||
publishedDate = None
|
||||
|
||||
162
searx/engines/tusksearch.py
Normal file
162
searx/engines/tusksearch.py
Normal file
@@ -0,0 +1,162 @@
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
"""Tusksearch_ is an American search engine that claims to fight censorship.
|
||||
Its search results are (at least partially) from Brave.
|
||||
|
||||
.. _Tusksearch: https://tusksearch.com/about
|
||||
"""
|
||||
|
||||
from json import loads
|
||||
import random
|
||||
import typing as t
|
||||
from urllib.parse import urlencode
|
||||
from dateutil import parser
|
||||
|
||||
from searx.exceptions import SearxEngineAPIException
|
||||
from searx.network import get
|
||||
from searx.utils import html_to_text
|
||||
from searx.result_types import EngineResults
|
||||
|
||||
if t.TYPE_CHECKING:
|
||||
from searx.extended_types import SXNG_Response
|
||||
from searx.search.processors import OnlineParams
|
||||
|
||||
about = {
|
||||
"website": "https://tusksearch.com",
|
||||
"wikidata_id": None,
|
||||
"official_api_documentation": None,
|
||||
"use_official_api": False,
|
||||
"require_api_key": False,
|
||||
"results": "JSON",
|
||||
}
|
||||
|
||||
paging = True
|
||||
|
||||
categories = ["general"]
|
||||
tusk_categ = "web"
|
||||
"""Category to search in. Can be either "web", "images", "videos" or "news"."""
|
||||
|
||||
|
||||
api_url = "https://api.tusksearch.com"
|
||||
|
||||
|
||||
def init(_):
|
||||
if tusk_categ not in ("web", "images", "videos", "news"):
|
||||
raise ValueError("invalid search type: %s" % tusk_categ)
|
||||
|
||||
|
||||
def _obtain_x_sid() -> tuple[str, str]:
|
||||
"""
|
||||
The session ID ("sid") is encoded as a byte array in ``embed.js``.
|
||||
It is only valid for exactly one request, so we can't cache it.
|
||||
|
||||
The header key is usually called `x-sid-{UUIDv4}`, and the value is
|
||||
usually a plain UUIDv4 (but a different one than in the header key).
|
||||
"""
|
||||
resp = get(f"{api_url}/revcontent/embed.js")
|
||||
if not resp.ok:
|
||||
raise SearxEngineAPIException("failed to obtain request x-sid token")
|
||||
|
||||
# data is prefixed by 'var x='
|
||||
data_array = loads(resp.text[6:])
|
||||
|
||||
def _byte_array_to_ascii(text: list[int]) -> str:
|
||||
"""
|
||||
Converts a byte array (e.g. [81, 101, 97, 114, 88, 78, 71]) to the ASCII
|
||||
string representation (e.g. "SearXNG").
|
||||
"""
|
||||
return "".join([chr(x) for x in text])
|
||||
|
||||
x_sid_header = _byte_array_to_ascii(data_array[3])
|
||||
x_sid_value = _byte_array_to_ascii(data_array[4])
|
||||
return x_sid_header, x_sid_value
|
||||
|
||||
|
||||
def request(query: str, params: "OnlineParams") -> None:
|
||||
# images don't support pagination, news and videos only support two pages
|
||||
if tusk_categ == "images" and params["pageno"] > 1 or tusk_categ in ("news", "videos") and params["pageno"] > 2:
|
||||
params["url"] = None
|
||||
return
|
||||
|
||||
args = {
|
||||
"q": query,
|
||||
"p": params["pageno"],
|
||||
"l": "center", # political direction: "left", "center" or "right"
|
||||
}
|
||||
if tusk_categ == "images":
|
||||
params["url"] = f"{api_url}/Search/Image?{urlencode(args)}"
|
||||
else:
|
||||
# web response also contains news and videos
|
||||
params["url"] = f"{api_url}/Search/Web?{urlencode(args)}"
|
||||
|
||||
x_sid_header, x_sid_value = _obtain_x_sid()
|
||||
params["headers"] = {
|
||||
x_sid_header: x_sid_value,
|
||||
# required - we send a random longitude and latitude instead of the actual user location
|
||||
'x-lon': str(random.random() * 90),
|
||||
'x-lat': str(random.random() * 90),
|
||||
}
|
||||
|
||||
|
||||
def response(resp: "SXNG_Response"):
|
||||
res = EngineResults()
|
||||
|
||||
json_resp = resp.json()["results"]
|
||||
|
||||
if tusk_categ == "web":
|
||||
for result in (json_resp.get("web") or {}).get("results", []):
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=result["url"],
|
||||
title=html_to_text(result["title"]),
|
||||
content=html_to_text(result["description"]),
|
||||
thumbnail=(result["thumbnail"] or {}).get("src") or "",
|
||||
)
|
||||
)
|
||||
elif tusk_categ == "news":
|
||||
for result in (json_resp.get("news") or {}).get("results", []):
|
||||
publishedDate = None
|
||||
try:
|
||||
publishedDate = parser.parse(result["age"])
|
||||
except parser.ParserError:
|
||||
pass
|
||||
|
||||
res.add(
|
||||
res.types.MainResult(
|
||||
url=result["url"],
|
||||
title=html_to_text(result["title"]),
|
||||
content=html_to_text(result["description"]),
|
||||
thumbnail=result["thumbnail"]["src"],
|
||||
publishedDate=publishedDate,
|
||||
)
|
||||
)
|
||||
elif tusk_categ == "videos":
|
||||
for result in (json_resp.get("videos") or {}).get("results", []):
|
||||
publishedDate = None
|
||||
try:
|
||||
publishedDate = parser.parse(result["age"])
|
||||
except parser.ParserError:
|
||||
pass
|
||||
|
||||
res.add(
|
||||
res.types.LegacyResult(
|
||||
template="videos.html",
|
||||
url=result["url"],
|
||||
title=html_to_text(result["title"]),
|
||||
content=html_to_text(result["description"]),
|
||||
thumbnail=result["thumbnail"]["src"],
|
||||
publishedDate=publishedDate,
|
||||
length=result["video"].get("duration"),
|
||||
)
|
||||
)
|
||||
elif tusk_categ == "images":
|
||||
for result in json_resp:
|
||||
res.add(
|
||||
res.types.Image(
|
||||
url=result["url"],
|
||||
title=html_to_text(result["title"]),
|
||||
img_src=result["properties"]["url"],
|
||||
thumbnail_src=result["thumbnail"]["src"],
|
||||
)
|
||||
)
|
||||
|
||||
return res
|
||||
@@ -14,6 +14,7 @@ template.
|
||||
# pylint: disable=too-few-public-methods
|
||||
__all__ = ["Image", "ImageRef"]
|
||||
|
||||
import mimetypes
|
||||
import types
|
||||
import typing as t
|
||||
from collections.abc import Callable
|
||||
@@ -71,7 +72,7 @@ class Image(MainResult, kw_only=True):
|
||||
"""The resolution of the image (e.g. ``1920 x 1080`` pixel)"""
|
||||
|
||||
img_format: str = ""
|
||||
"""The format of the image (e.g. ``png``)."""
|
||||
"""The format of the image :py:obj:`.MainResult.img_src` (e.g. ``png``)."""
|
||||
|
||||
source: str = ""
|
||||
"""Source of the image."""
|
||||
@@ -83,6 +84,19 @@ class Image(MainResult, kw_only=True):
|
||||
formats: list[ImageRef] = []
|
||||
"""List of links to alternative image formats."""
|
||||
|
||||
def __post_init__(self):
|
||||
super().__post_init__()
|
||||
|
||||
if not self.img_format:
|
||||
# automatically guess the image format based on the path of the image
|
||||
mimetype = mimetypes.guess_type(self.img_src)[0]
|
||||
if mimetype:
|
||||
subtype = mimetype.split("/")[-1]
|
||||
if subtype in MIMESUB:
|
||||
self.img_format = MIMESUB[subtype]
|
||||
else:
|
||||
self.img_format = subtype.upper()
|
||||
|
||||
def filter_urls(self, filter_func: "Callable[[Result | LegacyResult, str, str], str | bool ]"):
|
||||
|
||||
for _ref in self.formats[:]:
|
||||
|
||||
@@ -443,27 +443,12 @@ engines:
|
||||
timeout: 6.0
|
||||
shortcut: conda
|
||||
disabled: true
|
||||
|
||||
- name: aol
|
||||
engine: aol
|
||||
search_type: search
|
||||
categories: [general]
|
||||
shortcut: aol
|
||||
disabled: true
|
||||
|
||||
- name: aol images
|
||||
engine: aol
|
||||
search_type: image
|
||||
categories: [images]
|
||||
shortcut: aoli
|
||||
disabled: true
|
||||
|
||||
- name: aol videos
|
||||
engine: aol
|
||||
search_type: video
|
||||
categories: [videos]
|
||||
shortcut: aolv
|
||||
disabled: true
|
||||
about:
|
||||
website: https://anaconda.org/search
|
||||
wikidata_id: Q18209310
|
||||
official_api_documentation: https://api.anaconda.org/search
|
||||
use_official_api: false
|
||||
results: HTML
|
||||
|
||||
- name: arch linux wiki
|
||||
engine: archlinux
|
||||
@@ -665,28 +650,29 @@ engines:
|
||||
engine: chinaso
|
||||
shortcut: chinaso
|
||||
categories: [news]
|
||||
chinaso_category: news
|
||||
chinaso_news_source: all
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: chinaso images
|
||||
engine: chinaso
|
||||
network: chinaso news
|
||||
shortcut: chinasoi
|
||||
categories: [images]
|
||||
chinaso_category: images
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: chinaso videos
|
||||
engine: chinaso
|
||||
network: chinaso news
|
||||
shortcut: chinasov
|
||||
categories: [videos]
|
||||
chinaso_category: videos
|
||||
- name: cl0q
|
||||
engine: json_engine
|
||||
shortcut: cl
|
||||
categories: general
|
||||
paging: true
|
||||
first_page_num: 0
|
||||
page_size: 20
|
||||
search_url: https://cl0q.com/search?q={query}&limit=20&offset={pageno}
|
||||
results_query: results
|
||||
url_query: domain
|
||||
url_prefix: https://
|
||||
title_query: title
|
||||
content_query: description
|
||||
disabled: true
|
||||
inactive: true
|
||||
about:
|
||||
website: https://cl0q.com
|
||||
description: "Open source network for searching domains"
|
||||
results: JSON
|
||||
|
||||
- name: cloudflareai
|
||||
engine: cloudflareai
|
||||
@@ -978,6 +964,34 @@ engines:
|
||||
shortcut: fd
|
||||
disabled: true
|
||||
|
||||
- name: findfiles
|
||||
engine: findfiles
|
||||
findfiles_categ: all
|
||||
categories: files
|
||||
shortcut: fif
|
||||
disabled: true
|
||||
|
||||
- name: findfiles images
|
||||
engine: findfiles
|
||||
findfiles_categ: image
|
||||
categories: images
|
||||
shortcut: fifi
|
||||
disabled: true
|
||||
|
||||
- name: findfiles videos
|
||||
engine: findfiles
|
||||
findfiles_categ: video
|
||||
categories: videos
|
||||
shortcut: fifv
|
||||
disabled: true
|
||||
|
||||
- name: findfiles music
|
||||
engine: findfiles
|
||||
findfiles_categ: audio
|
||||
categories: music
|
||||
shortcut: fifm
|
||||
disabled: true
|
||||
|
||||
- name: findthatmeme
|
||||
engine: findthatmeme
|
||||
shortcut: ftm
|
||||
@@ -1114,6 +1128,11 @@ engines:
|
||||
search_type: text
|
||||
timeout: 10
|
||||
|
||||
- name: giphy
|
||||
engine: giphy
|
||||
shortcut: gif
|
||||
disabled: true
|
||||
|
||||
- name: gitlab
|
||||
engine: gitlab
|
||||
base_url: https://gitlab.com
|
||||
@@ -1295,6 +1314,13 @@ engines:
|
||||
require_api_key: false
|
||||
results: JSON
|
||||
|
||||
- name: iseek
|
||||
engine: iseek
|
||||
shortcut: isk
|
||||
timeout: 4
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: il post
|
||||
engine: il_post
|
||||
shortcut: pst
|
||||
@@ -1383,6 +1409,21 @@ engines:
|
||||
# api_key: "" # required
|
||||
# kagi_categ: videos
|
||||
|
||||
- name: kozmonavt
|
||||
engine: xpath
|
||||
search_url: https://kozmonavt.su/s?q={query}
|
||||
shortcut: koz
|
||||
disabled: true
|
||||
inactive: true
|
||||
results_xpath: //div[contains(@class, 'list')]/section
|
||||
url_xpath: concat('https://', substring-after(.//a/@href, '?q='))
|
||||
title_xpath: .//a
|
||||
content_xpath: .//div[contains(@class, 'snip')]
|
||||
about:
|
||||
website: https://kozmonavt.su
|
||||
description: "Web site directory with its own index"
|
||||
results: HTML
|
||||
|
||||
- name: jisho
|
||||
engine: jisho
|
||||
shortcut: js
|
||||
@@ -1400,6 +1441,22 @@ engines:
|
||||
shortcut: kc
|
||||
timeout: 4.0
|
||||
|
||||
- name: kukei
|
||||
engine: xpath
|
||||
categories: [general, blogs]
|
||||
search_url: https://kukei.eu/?q={query}
|
||||
shortcut: kuk
|
||||
disabled: true
|
||||
inactive: true
|
||||
results_xpath: //ul/li[contains(@class, "result-item-first-level")]
|
||||
url_xpath: .//a/@href
|
||||
title_xpath: .//h3
|
||||
content_xpath: .//p
|
||||
about:
|
||||
website: https://kukei.eu
|
||||
description: "Curated search for the small web"
|
||||
results: HTML
|
||||
|
||||
- name: lemmy communities
|
||||
engine: lemmy
|
||||
lemmy_type: Communities
|
||||
@@ -2033,7 +2090,7 @@ engines:
|
||||
- name: rawweb
|
||||
engine: json_engine
|
||||
shortcut: rw
|
||||
categories: general
|
||||
categories: [general, blogs]
|
||||
paging: true
|
||||
search_url: 'https://api.rawweb.org/api/search?keyword={query}&page={pageno}&lang=*'
|
||||
results_query: data
|
||||
@@ -2088,7 +2145,7 @@ engines:
|
||||
- name: searchmysite
|
||||
engine: xpath
|
||||
shortcut: sms
|
||||
categories: general
|
||||
categories: [general, blogs]
|
||||
paging: true
|
||||
search_url: https://searchmysite.net/search/?q={query}&page={pageno}
|
||||
results_xpath: //div[contains(@class,'search-result')]
|
||||
@@ -2404,6 +2461,35 @@ engines:
|
||||
- 5000
|
||||
inactive: true
|
||||
|
||||
- name: tusksearch
|
||||
engine: tusksearch
|
||||
shortcut: tu
|
||||
tusk_categ: web
|
||||
categories: general
|
||||
disabled: true
|
||||
|
||||
- name: tusksearch images
|
||||
engine: tusksearch
|
||||
shortcut: tui
|
||||
paging: false
|
||||
tusk_categ: images
|
||||
categories: images
|
||||
disabled: true
|
||||
|
||||
- name: tusksearch videos
|
||||
engine: tusksearch
|
||||
shortcut: tuv
|
||||
tusk_categ: videos
|
||||
categories: videos
|
||||
disabled: true
|
||||
|
||||
- name: tusksearch news
|
||||
engine: tusksearch
|
||||
shortcut: tun
|
||||
tusk_categ: news
|
||||
categories: news
|
||||
disabled: true
|
||||
|
||||
# tmp suspended - too slow, too many errors
|
||||
# - name: urbandictionary
|
||||
# engine : xpath
|
||||
@@ -2417,6 +2503,24 @@ engines:
|
||||
engine: unsplash
|
||||
shortcut: us
|
||||
|
||||
- name: unobtanium
|
||||
engine: xpath
|
||||
shortcut: uno
|
||||
categories: [general, blogs]
|
||||
paging: true
|
||||
first_page_num: 0
|
||||
search_url: https://unobtanium.rocks/search?search={query}&page={pageno}
|
||||
results_xpath: //ul[contains(@class, 'search-results')]/li
|
||||
url_xpath: .//a/@href
|
||||
title_xpath: .//a/b
|
||||
content_xpath: .//p[2]
|
||||
disabled: true
|
||||
inactive: true
|
||||
about:
|
||||
website: https://unobtanium.rocks
|
||||
description: "Personal websites focused search engine."
|
||||
results: HTML
|
||||
|
||||
- name: yandex
|
||||
engine: yandex
|
||||
categories: general
|
||||
@@ -2476,7 +2580,7 @@ engines:
|
||||
url_query: URL
|
||||
title_query: Title
|
||||
content_query: Snippet
|
||||
categories: [general, web]
|
||||
categories: [general, blogs]
|
||||
shortcut: wib
|
||||
disabled: true
|
||||
about:
|
||||
@@ -2901,6 +3005,38 @@ engines:
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: startpagina
|
||||
engine: startpagina
|
||||
shortcut: spnl
|
||||
startpagina_categ: web
|
||||
categories: general
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: startpagina images
|
||||
engine: startpagina
|
||||
shortcut: spnli
|
||||
startpagina_categ: images
|
||||
categories: images
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: startpagina videos
|
||||
engine: startpagina
|
||||
shortcut: spnlv
|
||||
startpagina_categ: videos
|
||||
categories: videos
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: startpagina news
|
||||
engine: startpagina
|
||||
shortcut: spnln
|
||||
startpagina_categ: news
|
||||
categories: news
|
||||
disabled: true
|
||||
inactive: true
|
||||
|
||||
- name: swisscows
|
||||
engine: swisscows
|
||||
categories: general
|
||||
@@ -3023,6 +3159,22 @@ engines:
|
||||
shortcut: wttr
|
||||
timeout: 9.0
|
||||
|
||||
- name: xonaly
|
||||
engine: xpath
|
||||
search_url: https://xonaly.com/?query={query}
|
||||
categories: general
|
||||
shortcut: xo
|
||||
disabled: true
|
||||
inactive: true
|
||||
results_xpath: //div[contains(@class, 'results-block')]/ul/li
|
||||
url_xpath: ./div[contains(@class, 'result-title')]/a/@href
|
||||
title_xpath: ./div[contains(@class, 'result-title')]/a
|
||||
content_xpath: ./p[contains(@class, 'excerpt')]
|
||||
about:
|
||||
website: https://xonaly.com
|
||||
description: "Canadian search engine with its own index"
|
||||
results: HTML
|
||||
|
||||
- name: zapmeta
|
||||
engine: xpath
|
||||
shortcut: zpm
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
'''
|
||||
|
||||
sxng_locales = (
|
||||
('af', 'Afrikaans', '', 'Afrikaans', '\U0001f310'),
|
||||
('ar', 'العربية', '', 'Arabic', '\U0001f310'),
|
||||
('ar-SA', 'العربية', 'المملكة العربية السعودية', 'Arabic', '\U0001f1f8\U0001f1e6'),
|
||||
('bg', 'Български', '', 'Bulgarian', '\U0001f310'),
|
||||
@@ -46,7 +45,6 @@ sxng_locales = (
|
||||
('es-ES', 'Español', 'España', 'Spanish', '\U0001f1ea\U0001f1f8'),
|
||||
('es-MX', 'Español', 'México', 'Spanish', '\U0001f1f2\U0001f1fd'),
|
||||
('es-PE', 'Español', 'Perú', 'Spanish', '\U0001f1f5\U0001f1ea'),
|
||||
('es-VE', 'Español', 'Venezuela', 'Spanish', '\U0001f1fb\U0001f1ea'),
|
||||
('et', 'Eesti', '', 'Estonian', '\U0001f310'),
|
||||
('et-EE', 'Eesti', 'Eesti', 'Estonian', '\U0001f1ea\U0001f1ea'),
|
||||
('fi', 'Suomi', '', 'Finnish', '\U0001f310'),
|
||||
@@ -58,7 +56,6 @@ sxng_locales = (
|
||||
('fr-CA', 'Français', 'Canada', 'French', '\U0001f1e8\U0001f1e6'),
|
||||
('fr-CH', 'Français', 'Suisse', 'French', '\U0001f1e8\U0001f1ed'),
|
||||
('fr-FR', 'Français', 'France', 'French', '\U0001f1eb\U0001f1f7'),
|
||||
('gl', 'Galego', '', 'Galician', '\U0001f310'),
|
||||
('hi', 'हिन्दी', '', 'Hindi', '\U0001f310'),
|
||||
('hi-IN', 'हिन्दी', 'भारत', 'Hindi', '\U0001f1ee\U0001f1f3'),
|
||||
('hr', 'Hrvatski', '', 'Croatian', '\U0001f310'),
|
||||
@@ -74,6 +71,8 @@ sxng_locales = (
|
||||
('ja-JP', '日本語', '日本', 'Japanese', '\U0001f1ef\U0001f1f5'),
|
||||
('ko', '한국어', '', 'Korean', '\U0001f310'),
|
||||
('ko-KR', '한국어', '대한민국', 'Korean', '\U0001f1f0\U0001f1f7'),
|
||||
('mi', 'Māori', '', 'Māori', '\U0001f310'),
|
||||
('mi-NZ', 'Māori', 'Aotearoa', 'Māori', '\U0001f1f3\U0001f1ff'),
|
||||
('nb', 'Norsk Bokmål', '', 'Norwegian Bokmål', '\U0001f310'),
|
||||
('nb-NO', 'Norsk Bokmål', 'Norge', 'Norwegian Bokmål', '\U0001f1f3\U0001f1f4'),
|
||||
('nl', 'Nederlands', '', 'Dutch', '\U0001f310'),
|
||||
@@ -90,8 +89,6 @@ sxng_locales = (
|
||||
('ro-RO', 'Română', 'România', 'Romanian', '\U0001f1f7\U0001f1f4'),
|
||||
('ru', 'Русский', '', 'Russian', '\U0001f310'),
|
||||
('ru-RU', 'Русский', 'Россия', 'Russian', '\U0001f1f7\U0001f1fa'),
|
||||
('sk', 'Slovenčina', '', 'Slovak', '\U0001f310'),
|
||||
('sq', 'Shqip', '', 'Albanian', '\U0001f310'),
|
||||
('sv', 'Svenska', '', 'Swedish', '\U0001f310'),
|
||||
('sv-FI', 'Svenska', 'Finland', 'Swedish', '\U0001f1eb\U0001f1ee'),
|
||||
('sv-SE', 'Svenska', 'Sverige', 'Swedish', '\U0001f1f8\U0001f1ea'),
|
||||
@@ -99,7 +96,6 @@ sxng_locales = (
|
||||
('th-TH', 'ไทย', 'ไทย', 'Thai', '\U0001f1f9\U0001f1ed'),
|
||||
('tr', 'Türkçe', '', 'Turkish', '\U0001f310'),
|
||||
('tr-TR', 'Türkçe', 'Türkiye', 'Turkish', '\U0001f1f9\U0001f1f7'),
|
||||
('uk', 'Українська', '', 'Ukrainian', '\U0001f310'),
|
||||
('vi', 'Tiếng Việt', '', 'Vietnamese', '\U0001f310'),
|
||||
('vi-VN', 'Tiếng Việt', 'Việt Nam', 'Vietnamese', '\U0001f1fb\U0001f1f3'),
|
||||
('zh', '中文', '', 'Chinese', '\U0001f310'),
|
||||
|
||||
Binary file not shown.
@@ -33,22 +33,24 @@
|
||||
# IcewindX <icewindx@noreply.codeberg.org>, 2025.
|
||||
# 0ko <0ko@noreply.codeberg.org>, 2025.
|
||||
# greatdng <greatdng@noreply.codeberg.org>, 2026.
|
||||
# code_gremlin <code_gremlin@noreply.codeberg.org>, 2026.
|
||||
msgid ""
|
||||
msgstr ""
|
||||
"Project-Id-Version: searx\n"
|
||||
"Project-Id-Version: searx\n"
|
||||
"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n"
|
||||
"POT-Creation-Date: 2026-06-13 11:31+0000\n"
|
||||
"PO-Revision-Date: 2026-05-19 12:08+0000\n"
|
||||
"Last-Translator: return42 <return42@noreply.codeberg.org>\n"
|
||||
"PO-Revision-Date: 2026-06-24 10:07+0000\n"
|
||||
"Last-Translator: code_gremlin <code_gremlin@noreply.codeberg.org>\n"
|
||||
"Language-Team: Russian <https://translate.codeberg.org/projects/searxng/"
|
||||
"searxng/ru/>\n"
|
||||
"Language: ru\n"
|
||||
"Language-Team: Russian "
|
||||
"<https://translate.codeberg.org/projects/searxng/searxng/ru/>\n"
|
||||
"Plural-Forms: nplurals=4; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && "
|
||||
"n%10<=4 && (n%100<12 || n%100>14) ? 1 : n%10==0 || (n%10>=5 && n%10<=9) "
|
||||
"|| (n%100>=11 && n%100<=14)? 2 : 3);\n"
|
||||
"MIME-Version: 1.0\n"
|
||||
"Content-Type: text/plain; charset=utf-8\n"
|
||||
"Content-Transfer-Encoding: 8bit\n"
|
||||
"Plural-Forms: nplurals=4; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && "
|
||||
"n%10<=4 && (n%100<12 || n%100>14) ? 1 : n%10==0 || (n%10>=5 && n%10<=9) || "
|
||||
"(n%100>=11 && n%100<=14)? 2 : 3);\n"
|
||||
"X-Generator: Weblate 2026.6.1\n"
|
||||
"Generated-By: Babel 2.18.0\n"
|
||||
|
||||
#. CONSTANT_NAMES['NO_SUBGROUPING']
|
||||
@@ -1667,11 +1669,11 @@ msgstr "Разрешение"
|
||||
|
||||
#: searx/templates/simple/result_templates/images.html:55
|
||||
msgid "Image formats"
|
||||
msgstr ""
|
||||
msgstr "Форматы изображений"
|
||||
|
||||
#: searx/templates/simple/result_templates/images.html:56
|
||||
msgid "original format"
|
||||
msgstr ""
|
||||
msgstr "формат оригинала"
|
||||
|
||||
#: searx/templates/simple/result_templates/images.html:63
|
||||
msgid "Engine"
|
||||
@@ -2441,4 +2443,3 @@ msgstr "скрыть видео"
|
||||
|
||||
#~ msgid "Format"
|
||||
#~ msgstr "Формат"
|
||||
|
||||
|
||||
@@ -15,25 +15,26 @@ from os.path import join
|
||||
|
||||
from lxml.html import fromstring
|
||||
|
||||
import searx.engines
|
||||
from searx.engines import wikidata, set_loggers
|
||||
from searx.utils import extract_text, searxng_useragent
|
||||
from searx.utils import extract_text
|
||||
from searx.locales import LOCALE_NAMES, locales_initialize, match_locale
|
||||
from searx import searx_dir
|
||||
from searx.utils import gen_useragent
|
||||
import searx.search
|
||||
import searx.network
|
||||
from searx.data import data_dir
|
||||
from searx.data import data_dir, ENGINE_DESCRIPTIONS
|
||||
|
||||
DATA_FILE = data_dir / 'engine_descriptions.json'
|
||||
DATA_FILE = data_dir / "engine_descriptions.json"
|
||||
|
||||
set_loggers(wikidata, 'wikidata')
|
||||
set_loggers(wikidata, "wikidata")
|
||||
locales_initialize()
|
||||
|
||||
# you can run the query in https://query.wikidata.org
|
||||
# replace %IDS% by Wikidata entities separated by spaces with the prefix wd:
|
||||
# for example wd:Q182496 wd:Q1540899
|
||||
# replace %LANGUAGES_SPARQL% by languages
|
||||
SPARQL_WIKIPEDIA_ARTICLE = """
|
||||
SPARQL_WIKIPEDIA_ARTICLE: str = """
|
||||
SELECT DISTINCT ?item ?name ?article ?lang
|
||||
WHERE {
|
||||
hint:Query hint:optimizer "None".
|
||||
@@ -48,7 +49,7 @@ WHERE {
|
||||
ORDER BY ?item ?lang
|
||||
"""
|
||||
|
||||
SPARQL_DESCRIPTION = """
|
||||
SPARQL_DESCRIPTION: str = """
|
||||
SELECT DISTINCT ?item ?itemDescription
|
||||
WHERE {
|
||||
VALUES ?item { %IDS% }
|
||||
@@ -58,43 +59,43 @@ WHERE {
|
||||
ORDER BY ?itemLang
|
||||
"""
|
||||
|
||||
NOT_A_DESCRIPTION = [
|
||||
'web site',
|
||||
'site web',
|
||||
'komputa serĉilo',
|
||||
'interreta serĉilo',
|
||||
'bilaketa motor',
|
||||
'web search engine',
|
||||
'wikimedia täpsustuslehekülg',
|
||||
NOT_A_DESCRIPTION: list[str] = [
|
||||
"web site",
|
||||
"site web",
|
||||
"komputa serĉilo",
|
||||
"interreta serĉilo",
|
||||
"bilaketa motor",
|
||||
"web search engine",
|
||||
"wikimedia täpsustuslehekülg",
|
||||
]
|
||||
|
||||
SKIP_ENGINE_SOURCE = [
|
||||
SKIP_ENGINE_SOURCE: list[tuple[str, str]] = [
|
||||
# fmt: off
|
||||
('gitlab', 'wikidata')
|
||||
("gitlab", "wikidata")
|
||||
# descriptions are about wikipedia disambiguation pages
|
||||
# fmt: on
|
||||
]
|
||||
|
||||
WIKIPEDIA_LANGUAGES = {}
|
||||
LANGUAGES_SPARQL = ''
|
||||
IDS = None
|
||||
WIKIPEDIA_LANGUAGE_VARIANTS = {'zh_Hant': 'zh-tw'}
|
||||
WIKIPEDIA_LANGUAGES: dict[str, str] = {}
|
||||
LANGUAGES_SPARQL: str = ""
|
||||
IDS: str = ""
|
||||
WIKIPEDIA_LANGUAGE_VARIANTS: dict[str, str] = {"zh_Hant": "zh-tw"}
|
||||
|
||||
# descriptions[engine][language] = [description, source]
|
||||
descriptions: dict[str, dict[str, list[str]]] = {}
|
||||
wd_to_engine_name: dict[str, set[str]] = {}
|
||||
|
||||
|
||||
descriptions = {}
|
||||
wd_to_engine_name = {}
|
||||
|
||||
|
||||
def normalize_description(description):
|
||||
def normalize_description(description: str):
|
||||
for c in [chr(c) for c in range(0, 31)]:
|
||||
description = description.replace(c, ' ')
|
||||
description = ' '.join(description.strip().split())
|
||||
description = description.replace(c, " ")
|
||||
description = " ".join(description.strip().split())
|
||||
return description
|
||||
|
||||
|
||||
def update_description(engine_name, lang, description, source, replace=True):
|
||||
def update_description(engine_name: str, lang: str, description: str, source: str, replace: bool = True) -> None:
|
||||
if not isinstance(description, str):
|
||||
return
|
||||
return # pyright: ignore[reportUnreachable]
|
||||
description = normalize_description(description)
|
||||
if description.lower() == engine_name.lower():
|
||||
return
|
||||
@@ -102,54 +103,71 @@ def update_description(engine_name, lang, description, source, replace=True):
|
||||
return
|
||||
if (engine_name, source) in SKIP_ENGINE_SOURCE:
|
||||
return
|
||||
if ' ' not in description:
|
||||
if " " not in description:
|
||||
# skip unique word description (like "website")
|
||||
return
|
||||
if replace or lang not in descriptions[engine_name]:
|
||||
descriptions[engine_name][lang] = [description, source]
|
||||
|
||||
|
||||
def get_wikipedia_summary(wikipedia_url, searxng_locale):
|
||||
def get_wikipedia_summary(wikipedia_url: str, searxng_locale: str):
|
||||
# get the REST API URL from the HTML URL
|
||||
|
||||
# Headers
|
||||
headers = {'User-Agent': searxng_useragent()}
|
||||
headers = {
|
||||
"User-Agent": gen_useragent(),
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Encoding": "gzip, deflate",
|
||||
"DNT": "1",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Sec-GPC": "1",
|
||||
"Cache-Control": "max-age=0",
|
||||
"Sec-Fetch-Dest": "document",
|
||||
"Sec-Fetch-Mode": "navigate",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"Sec-Fetch-User": "?1",
|
||||
}
|
||||
|
||||
if searxng_locale in WIKIPEDIA_LANGUAGE_VARIANTS:
|
||||
headers['Accept-Language'] = WIKIPEDIA_LANGUAGE_VARIANTS.get(searxng_locale)
|
||||
headers["Accept-Language"] = WIKIPEDIA_LANGUAGE_VARIANTS[searxng_locale]
|
||||
|
||||
# URL path : from HTML URL to REST API URL
|
||||
parsed_url = urlparse(wikipedia_url)
|
||||
# remove the /wiki/ prefix
|
||||
article_name = parsed_url.path.split('/wiki/')[1]
|
||||
article_name = parsed_url.path.split("/wiki/")[1]
|
||||
# article_name is already encoded but not the / which is required for the REST API call
|
||||
encoded_article_name = article_name.replace('/', '%2F')
|
||||
path = '/api/rest_v1/page/summary/' + encoded_article_name
|
||||
encoded_article_name = article_name.replace("/", "%2F")
|
||||
path = "/api/rest_v1/page/summary/" + encoded_article_name
|
||||
wikipedia_rest_url = parsed_url._replace(path=path).geturl()
|
||||
try:
|
||||
response = searx.network.get(wikipedia_rest_url, headers=headers, timeout=10)
|
||||
response.raise_for_status()
|
||||
except Exception as e: # pylint: disable=broad-except
|
||||
print(" ", wikipedia_url, e)
|
||||
print(" get_wikipedia_summary: ", wikipedia_rest_url, e)
|
||||
return None
|
||||
api_result = json.loads(response.text)
|
||||
return api_result.get('extract')
|
||||
return api_result.get("extract")
|
||||
|
||||
|
||||
def get_website_description(url, lang1, lang2=None):
|
||||
def get_website_description(url: str, lang1: str | None, lang2: str | None = None):
|
||||
headers = {
|
||||
'User-Agent': gen_useragent(),
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
|
||||
'DNT': '1',
|
||||
'Upgrade-Insecure-Requests': '1',
|
||||
'Sec-GPC': '1',
|
||||
'Cache-Control': 'max-age=0',
|
||||
"User-Agent": gen_useragent(),
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Encoding": "gzip, deflate",
|
||||
"DNT": "1",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Sec-GPC": "1",
|
||||
"Cache-Control": "max-age=0",
|
||||
"Sec-Fetch-Dest": "document",
|
||||
"Sec-Fetch-Mode": "navigate",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"Sec-Fetch-User": "?1",
|
||||
}
|
||||
if lang1 is not None:
|
||||
lang_list = [lang1]
|
||||
if lang2 is not None:
|
||||
lang_list.append(lang2)
|
||||
headers['Accept-Language'] = f'{",".join(lang_list)};q=0.8'
|
||||
headers["Accept-Language"] = f'{",".join(lang_list)};q=0.8'
|
||||
try:
|
||||
response = searx.network.get(url, headers=headers, timeout=10)
|
||||
response.raise_for_status()
|
||||
@@ -167,7 +185,7 @@ def get_website_description(url, lang1, lang2=None):
|
||||
if not description:
|
||||
description = extract_text(html.xpath('/html/head/title'))
|
||||
lang = extract_text(html.xpath('/html/@lang'))
|
||||
if lang is None and len(lang1) > 0:
|
||||
if lang is None and lang1 and len(lang1) > 0:
|
||||
lang = lang1
|
||||
lang = lang or 'en'
|
||||
lang = lang.split('_')[0]
|
||||
@@ -178,15 +196,15 @@ def get_website_description(url, lang1, lang2=None):
|
||||
def initialize():
|
||||
global IDS, LANGUAGES_SPARQL
|
||||
searx.search.initialize()
|
||||
wikipedia_engine = searx.engines.engines['wikipedia']
|
||||
wikipedia_engine = searx.engines.engines["wikipedia"]
|
||||
|
||||
locale2lang = {'nl-BE': 'nl'}
|
||||
locale2lang = {"nl-BE": "nl"}
|
||||
for sxng_ui_lang in LOCALE_NAMES:
|
||||
|
||||
sxng_ui_alias = locale2lang.get(sxng_ui_lang, sxng_ui_lang)
|
||||
wiki_lang = None
|
||||
|
||||
if sxng_ui_alias in wikipedia_engine.traits.custom['WIKIPEDIA_LANGUAGES']:
|
||||
if sxng_ui_alias in wikipedia_engine.traits.custom["WIKIPEDIA_LANGUAGES"]:
|
||||
wiki_lang = sxng_ui_alias
|
||||
if not wiki_lang:
|
||||
wiki_lang = wikipedia_engine.traits.get_language(sxng_ui_alias)
|
||||
@@ -195,70 +213,84 @@ def initialize():
|
||||
continue
|
||||
WIKIPEDIA_LANGUAGES[sxng_ui_lang] = wiki_lang
|
||||
|
||||
LANGUAGES_SPARQL = ', '.join(f"'{l}'" for l in set(WIKIPEDIA_LANGUAGES.values()))
|
||||
LANGUAGES_SPARQL = ", ".join(f"'{l}'" for l in set(WIKIPEDIA_LANGUAGES.values()))
|
||||
for engine_name, engine in searx.engines.engines.items():
|
||||
descriptions[engine_name] = {}
|
||||
wikidata_id = getattr(engine, "about", {}).get('wikidata_id')
|
||||
if wikidata_id is not None:
|
||||
wd_to_engine_name.setdefault(wikidata_id, set()).add(engine_name)
|
||||
if engine.about.wikidata_id:
|
||||
wd_to_engine_name.setdefault(engine.about.wikidata_id, set()).add(engine_name)
|
||||
|
||||
IDS = ' '.join(list(map(lambda wd_id: 'wd:' + wd_id, wd_to_engine_name.keys())))
|
||||
IDS = " ".join(list(map(lambda wd_id: "wd:" + wd_id, wd_to_engine_name.keys())))
|
||||
|
||||
|
||||
def fetch_wikidata_descriptions():
|
||||
print('Fetching wikidata descriptions')
|
||||
print("Fetching wikidata descriptions")
|
||||
searx.network.set_timeout_for_thread(60)
|
||||
result = wikidata.send_wikidata_query(
|
||||
SPARQL_DESCRIPTION.replace('%IDS%', IDS).replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL)
|
||||
SPARQL_DESCRIPTION.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL)
|
||||
)
|
||||
if result is not None:
|
||||
for binding in result['results']['bindings']:
|
||||
wikidata_id = binding['item']['value'].replace('http://www.wikidata.org/entity/', '')
|
||||
wikidata_lang = binding['itemDescription']['xml:lang']
|
||||
desc = binding['itemDescription']['value']
|
||||
for engine_name in wd_to_engine_name[wikidata_id]:
|
||||
for searxng_locale in LOCALE_NAMES:
|
||||
if WIKIPEDIA_LANGUAGES[searxng_locale] != wikidata_lang:
|
||||
continue
|
||||
print(
|
||||
f" engine: {engine_name:20} / wikidata_lang: {wikidata_lang:5}",
|
||||
f"/ len(wikidata_desc): {len(desc)}",
|
||||
)
|
||||
update_description(engine_name, searxng_locale, desc, 'wikidata')
|
||||
if not result:
|
||||
print("ERROR: fetching wikiDATA descriptions - SPARQL_DESCRIPTION query without results.")
|
||||
return
|
||||
|
||||
for binding in result["results"]["bindings"]:
|
||||
wikidata_id = binding["item"]["value"].replace("http://www.wikidata.org/entity/", "")
|
||||
wikidata_lang = binding["itemDescription"]["xml:lang"]
|
||||
desc = binding["itemDescription"]["value"]
|
||||
for engine_name in wd_to_engine_name[wikidata_id]:
|
||||
for searxng_locale in LOCALE_NAMES:
|
||||
if WIKIPEDIA_LANGUAGES[searxng_locale] != wikidata_lang:
|
||||
continue
|
||||
print(
|
||||
f" engine: {engine_name:20} / wikidata_lang: {wikidata_lang:5}",
|
||||
f"/ len(wikidata_desc): {len(desc)}",
|
||||
)
|
||||
update_description(engine_name, searxng_locale, desc, "wikidata")
|
||||
|
||||
|
||||
def fetch_wikipedia_descriptions():
|
||||
print('Fetching wikipedia descriptions')
|
||||
print("Fetching wikipedia descriptions")
|
||||
result = wikidata.send_wikidata_query(
|
||||
SPARQL_WIKIPEDIA_ARTICLE.replace('%IDS%', IDS).replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL)
|
||||
SPARQL_WIKIPEDIA_ARTICLE.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL)
|
||||
)
|
||||
if result is not None:
|
||||
for binding in result['results']['bindings']:
|
||||
wikidata_id = binding['item']['value'].replace('http://www.wikidata.org/entity/', '')
|
||||
wikidata_lang = binding['name']['xml:lang']
|
||||
wikipedia_url = binding['article']['value'] # for example the URL https://de.wikipedia.org/wiki/PubMed
|
||||
for engine_name in wd_to_engine_name[wikidata_id]:
|
||||
for searxng_locale in LOCALE_NAMES:
|
||||
if WIKIPEDIA_LANGUAGES[searxng_locale] != wikidata_lang:
|
||||
continue
|
||||
desc = get_wikipedia_summary(wikipedia_url, searxng_locale)
|
||||
if not desc:
|
||||
continue
|
||||
print(
|
||||
f" engine: {engine_name:20} / wikidata_lang: {wikidata_lang:5}",
|
||||
f"/ len(wikipedia_desc): {len(desc)}",
|
||||
)
|
||||
update_description(engine_name, searxng_locale, desc, 'wikipedia')
|
||||
if not result:
|
||||
print("ERROR: fetching wikiPEDIA descriptions - SPARQL_WIKIPEDIA_ARTICLE query without results.")
|
||||
return
|
||||
|
||||
# pylint: disable=too-many-nested-blocks
|
||||
for binding in result["results"]["bindings"]:
|
||||
wikidata_id = binding["item"]["value"].replace("http://www.wikidata.org/entity/", "")
|
||||
wikidata_lang = binding["name"]["xml:lang"]
|
||||
wikipedia_url = binding["article"]["value"] # for example the URL https://de.wikipedia.org/wiki/PubMed
|
||||
for engine_name in wd_to_engine_name[wikidata_id]:
|
||||
for searxng_locale in LOCALE_NAMES:
|
||||
if WIKIPEDIA_LANGUAGES[searxng_locale] != wikidata_lang:
|
||||
continue
|
||||
desc = get_wikipedia_summary(wikipedia_url, searxng_locale)
|
||||
if not desc:
|
||||
if descriptions.get(searxng_locale, {}).get(engine_name) is None:
|
||||
_descr = ENGINE_DESCRIPTIONS.get(searxng_locale, {}).get(engine_name)
|
||||
if _descr is not None:
|
||||
if len(_descr) == 2 and _descr[1] == 'ref':
|
||||
ref_engine, ref_lang = _descr[0].split(':')
|
||||
_descr = ENGINE_DESCRIPTIONS[ref_lang][ref_engine]
|
||||
update_description(engine_name, searxng_locale, _descr[0], _descr[1])
|
||||
|
||||
continue
|
||||
print(
|
||||
f" engine: {engine_name:20} / wikidata_lang: {wikidata_lang:5}",
|
||||
f"/ len(wikipedia_desc): {len(desc)}",
|
||||
)
|
||||
update_description(engine_name, searxng_locale, desc, "wikipedia")
|
||||
|
||||
|
||||
def normalize_url(url):
|
||||
url = url.replace('{language}', 'en')
|
||||
url = urlparse(url)._replace(path='/', params='', query='', fragment='').geturl()
|
||||
url = url.replace('https://api.', 'https://')
|
||||
def normalize_url(url: str):
|
||||
url = url.replace("{language}", "en")
|
||||
url = urlparse(url)._replace(path="/", params="", query="", fragment="").geturl()
|
||||
url = url.replace("https://api.", "https://")
|
||||
return url
|
||||
|
||||
|
||||
def fetch_website_description(engine_name, website):
|
||||
def fetch_website_description(engine_name: str, website: str):
|
||||
print(f"- fetch website descr: {engine_name} / {website}")
|
||||
default_lang, default_description = get_website_description(website, None, None)
|
||||
|
||||
@@ -268,11 +300,11 @@ def fetch_website_description(engine_name, website):
|
||||
|
||||
# to specify an order in where the most common languages are in front of the
|
||||
# language list ..
|
||||
languages = ['en', 'es', 'pt', 'ru', 'tr', 'fr']
|
||||
languages = ["en", "es", "pt", "ru", "tr", "fr"]
|
||||
languages = languages + [l for l in LOCALE_NAMES if l not in languages]
|
||||
|
||||
previous_matched_lang = None
|
||||
previous_count = 0
|
||||
previous_matched_lang: str | None = None
|
||||
previous_count: int = 0
|
||||
|
||||
for lang in languages:
|
||||
|
||||
@@ -307,19 +339,15 @@ def fetch_website_description(engine_name, website):
|
||||
f" / fetched lang: {fetched_lang:7} / len(desc): {len(desc)}"
|
||||
)
|
||||
|
||||
matched_lang = match_locale(fetched_lang, LOCALE_NAMES.keys(), fallback=lang)
|
||||
matched_lang = match_locale(fetched_lang, list(LOCALE_NAMES.keys())) or lang
|
||||
update_description(engine_name, matched_lang, desc, website, replace=False)
|
||||
|
||||
|
||||
def fetch_website_descriptions():
|
||||
print('Fetching website descriptions')
|
||||
print("Fetching website descriptions")
|
||||
for engine_name, engine in searx.engines.engines.items():
|
||||
website = getattr(engine, "about", {}).get('website')
|
||||
if website is None and hasattr(engine, "search_url"):
|
||||
website = normalize_url(getattr(engine, "search_url"))
|
||||
if website is None and hasattr(engine, "base_url"):
|
||||
website = normalize_url(getattr(engine, "base_url"))
|
||||
if website is not None:
|
||||
website = engine.about.website or getattr(engine, "search_url", "") or getattr(engine, "base_url", "")
|
||||
if website:
|
||||
fetch_website_description(engine_name, website)
|
||||
|
||||
|
||||
@@ -328,30 +356,32 @@ def get_engine_descriptions_filename():
|
||||
|
||||
|
||||
def get_output():
|
||||
"""Summary of the results, once known descriptions are not duplicated,
|
||||
instead a reference is provided.
|
||||
|
||||
- from: ``descriptions[engine][language] = [description, source]``
|
||||
- to: ``output[language][engine] = description_and_source``
|
||||
|
||||
``description_and_source`` can be:
|
||||
|
||||
- ``[description, source]``
|
||||
- ``description`` (if source = "wikipedia")
|
||||
- ``[f"engine:lang", "ref"]`` reference to another existing description
|
||||
|
||||
"""
|
||||
From descriptions[engine][language] = [description, source]
|
||||
To
|
||||
output: dict[str, dict[str, list[str] | str]] = {locale: {} for locale in LOCALE_NAMES}
|
||||
seen_descriptions: dict[str, tuple[str, str]] = {}
|
||||
|
||||
* output[language][engine] = description_and_source
|
||||
* description_and_source can be:
|
||||
* [description, source]
|
||||
* description (if source = "wikipedia")
|
||||
* [f"engine:lang", "ref"] (reference to another existing description)
|
||||
"""
|
||||
output = {locale: {} for locale in LOCALE_NAMES}
|
||||
|
||||
seen_descriptions = {}
|
||||
|
||||
for engine_name, lang_descriptions in descriptions.items():
|
||||
for language, description in lang_descriptions.items():
|
||||
if description[0] in seen_descriptions:
|
||||
ref = seen_descriptions[description[0]]
|
||||
description = [f'{ref[0]}:{ref[1]}', 'ref']
|
||||
for engine_name, lang_descriptions in sorted(descriptions.items()):
|
||||
for language, descr in sorted(lang_descriptions.items()):
|
||||
if descr[0] in seen_descriptions:
|
||||
ref = seen_descriptions[descr[0]]
|
||||
descr = [f"{ref[0]}:{ref[1]}", "ref"]
|
||||
else:
|
||||
seen_descriptions[description[0]] = (engine_name, language)
|
||||
if description[1] == 'wikipedia':
|
||||
description = description[0]
|
||||
output.setdefault(language, {}).setdefault(engine_name, description)
|
||||
seen_descriptions[descr[0]] = (engine_name, language)
|
||||
if descr[1] == "wikipedia":
|
||||
descr = descr[0]
|
||||
output.setdefault(language, {}).setdefault(engine_name, descr)
|
||||
|
||||
return output
|
||||
|
||||
@@ -363,8 +393,8 @@ def main():
|
||||
fetch_website_descriptions()
|
||||
|
||||
output = get_output()
|
||||
with DATA_FILE.open('w', encoding='utf8') as f:
|
||||
f.write(json.dumps(output, indent=1, separators=(',', ':'), sort_keys=True, ensure_ascii=False))
|
||||
with DATA_FILE.open("w", encoding="utf8") as f:
|
||||
f.write(json.dumps(output, indent=1, separators=(",", ":"), sort_keys=True, ensure_ascii=False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user