mirror of
https://github.com/searxng/searxng.git
synced 2026-09-13 01:36:04 +00:00
Compare commits
34 Commits
a1144dda3e
...
dependabot
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a1d4e26880 | ||
|
|
a0ab3b4f11 | ||
|
|
56b1f64541 | ||
|
|
923a307454 | ||
|
|
87bf8c86ed | ||
|
|
61d660276f | ||
|
|
6a27c21008 | ||
|
|
ffe96f8a6f | ||
|
|
931fd9787b | ||
|
|
42e1d61296 | ||
|
|
765a9999df | ||
|
|
ba055b3e09 | ||
|
|
3fdc6d753a | ||
|
|
3e454637fb | ||
|
|
c7f3080aac | ||
|
|
072311b5e0 | ||
|
|
c06e9f0889 | ||
|
|
4781754dc4 | ||
|
|
28b61729c7 | ||
|
|
14a9f84c6c | ||
|
|
8b01679e8f | ||
|
|
eaf1fcb349 | ||
|
|
e20e370353 | ||
|
|
3605a2d58b | ||
|
|
aef258321c | ||
|
|
a303e9c0ca | ||
|
|
ccffbfc164 | ||
|
|
242dc6e398 | ||
|
|
22056605a6 | ||
|
|
23e7e4da00 | ||
|
|
15a91992e4 | ||
|
|
03c439a5b9 | ||
|
|
be836e614a | ||
|
|
15b0c8ef3a |
9
.github/workflows/container.yml
vendored
9
.github/workflows/container.yml
vendored
@@ -62,7 +62,7 @@ jobs:
|
|||||||
python-version: "${{ env.PYTHON_VERSION }}"
|
python-version: "${{ env.PYTHON_VERSION }}"
|
||||||
|
|
||||||
- name: Setup QEMU
|
- name: Setup QEMU
|
||||||
uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0
|
uses: docker/setup-qemu-action@1f40c72289eff860ee54a304f1438e3cff362e0a # v4.3.0
|
||||||
|
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
@@ -105,8 +105,9 @@ jobs:
|
|||||||
arch: amd64
|
arch: amd64
|
||||||
- runner: ubuntu-26.04-arm
|
- runner: ubuntu-26.04-arm
|
||||||
arch: arm64
|
arch: arm64
|
||||||
- runner: ubuntu-26.04-arm
|
# FIXME: https://github.com/searxng/searxng/pull/6655#issuecomment-5550293085
|
||||||
arch: armv7
|
# - runner: ubuntu-26.04-arm
|
||||||
|
# arch: armv7
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Login to GHCR
|
- name: Login to GHCR
|
||||||
@@ -117,7 +118,7 @@ jobs:
|
|||||||
password: "${{ secrets.GITHUB_TOKEN }}"
|
password: "${{ secrets.GITHUB_TOKEN }}"
|
||||||
|
|
||||||
- name: Setup QEMU
|
- name: Setup QEMU
|
||||||
uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0
|
uses: docker/setup-qemu-action@1f40c72289eff860ee54a304f1438e3cff362e0a # v4.3.0
|
||||||
|
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
|
|||||||
855
client/simple/package-lock.json
generated
855
client/simple/package-lock.json
generated
File diff suppressed because it is too large
Load Diff
@@ -29,14 +29,14 @@
|
|||||||
"swiped-events": "1.2.0"
|
"swiped-events": "1.2.0"
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
"@biomejs/biome": "2.5.10",
|
"@biomejs/biome": "2.5.12",
|
||||||
"@types/node": "^26.3.0",
|
"@types/node": "^26.5.0",
|
||||||
"browserslist": "^4.28.8",
|
"browserslist": "^4.28.8",
|
||||||
"browserslist-to-esbuild": "^2.1.1",
|
"browserslist-to-esbuild": "^2.1.1",
|
||||||
"edge.js": "^6.5.1",
|
"edge.js": "^6.5.1",
|
||||||
"less": "^4.9.0",
|
"less": "^4.9.0",
|
||||||
"mathjs": "^15.2.0",
|
"mathjs": "^15.2.0",
|
||||||
"sharp": "~0.35.3",
|
"sharp": "~0.35.4",
|
||||||
"sort-package-json": "^4.0.0",
|
"sort-package-json": "^4.0.0",
|
||||||
"stylelint": "^17.14.1",
|
"stylelint": "^17.14.1",
|
||||||
"stylelint-config-standard-less": "^4.1.0",
|
"stylelint-config-standard-less": "^4.1.0",
|
||||||
|
|||||||
@@ -58,10 +58,9 @@ engine is shown. Most of the options have a default value or even are optional.
|
|||||||
|
|
||||||
# overwrite values from section 'outgoing:'
|
# overwrite values from section 'outgoing:'
|
||||||
enable_http2: false
|
enable_http2: false
|
||||||
|
enable_http3: false
|
||||||
retries: 1
|
retries: 1
|
||||||
max_connections: 100
|
max_connections: 100
|
||||||
max_keepalive_connections: 10
|
|
||||||
keepalive_expiry: 5.0
|
|
||||||
using_tor_proxy: false
|
using_tor_proxy: false
|
||||||
proxies:
|
proxies:
|
||||||
http:
|
http:
|
||||||
@@ -163,6 +162,16 @@ engine is shown. Most of the options have a default value or even are optional.
|
|||||||
``enable_http`` : optional
|
``enable_http`` : optional
|
||||||
Enable HTTP for this engine (by default only HTTPS is enabled).
|
Enable HTTP for this engine (by default only HTTPS is enabled).
|
||||||
|
|
||||||
|
``enable_http3`` : optional
|
||||||
|
Use HTTP/3 (falls back to HTTP/2). Default ``false``.
|
||||||
|
Ignored when a proxy is set.
|
||||||
|
|
||||||
|
.. hint::
|
||||||
|
|
||||||
|
HTTP/3 places demands on the IP infrastructure that are not met in every
|
||||||
|
environment. Enable this option only if you are aware of these requirements
|
||||||
|
and the extent to which they are met.
|
||||||
|
|
||||||
``retry_on_http_error`` : optional
|
``retry_on_http_error`` : optional
|
||||||
Retry request on some HTTP status code.
|
Retry request on some HTTP status code.
|
||||||
|
|
||||||
@@ -179,20 +188,12 @@ engine is shown. Most of the options have a default value or even are optional.
|
|||||||
Using tor proxy (``true``) or not (``false``) for this engine. The default is
|
Using tor proxy (``true``) or not (``false``) for this engine. The default is
|
||||||
taken from ``using_tor_proxy`` of the :ref:`settings outgoing`.
|
taken from ``using_tor_proxy`` of the :ref:`settings outgoing`.
|
||||||
|
|
||||||
.. _Pool limit configuration: https://www.python-httpx.org/advanced/#pool-limit-configuration
|
.. _Pool limit configuration: https://curl-cffi.readthedocs.io/en/latest/api.html#sessions
|
||||||
|
|
||||||
``max_keepalive_connection#s`` :
|
|
||||||
`Pool limit configuration`_, overwrites value ``pool_maxsize`` from
|
|
||||||
:ref:`settings outgoing` for this engine.
|
|
||||||
|
|
||||||
``max_connections`` :
|
``max_connections`` :
|
||||||
`Pool limit configuration`_, overwrites value ``pool_connections`` from
|
`Pool limit configuration`_, overwrites value ``pool_connections`` from
|
||||||
:ref:`settings outgoing` for this engine.
|
:ref:`settings outgoing` for this engine.
|
||||||
|
|
||||||
``keepalive_expiry`` :
|
|
||||||
`Pool limit configuration`_, overwrites value ``keepalive_expiry`` from
|
|
||||||
:ref:`settings outgoing` for this engine.
|
|
||||||
|
|
||||||
|
|
||||||
.. _private engines:
|
.. _private engines:
|
||||||
|
|
||||||
|
|||||||
@@ -12,20 +12,12 @@ Communication with search engines.
|
|||||||
request_timeout: 2.0 # default timeout in seconds, can be override by engine
|
request_timeout: 2.0 # default timeout in seconds, can be override by engine
|
||||||
max_request_timeout: 10.0 # the maximum timeout in seconds
|
max_request_timeout: 10.0 # the maximum timeout in seconds
|
||||||
useragent_suffix: "" # information like an email address to the administrator
|
useragent_suffix: "" # information like an email address to the administrator
|
||||||
pool_connections: 100 # Maximum number of allowable connections, or null
|
pool_connections: 100 # Maximum number of concurrent connections (default: 100)
|
||||||
# for no limits. The default is 100.
|
enable_http2: true # Enables the use of HTTP2
|
||||||
pool_maxsize: 10 # Number of allowable keep-alive connections, or null
|
|
||||||
# to always allow. The default is 10.
|
|
||||||
enable_http2: true # See https://www.python-httpx.org/http2/
|
|
||||||
# uncomment below section if you want to use a custom server certificate
|
# uncomment below section if you want to use a custom server certificate
|
||||||
# see https://www.python-httpx.org/advanced/#changing-the-verification-defaults
|
|
||||||
# and https://www.python-httpx.org/compatibility/#ssl-configuration
|
|
||||||
# verify: ~/.mitmproxy/mitmproxy-ca-cert.cer
|
# verify: ~/.mitmproxy/mitmproxy-ca-cert.cer
|
||||||
#
|
#
|
||||||
# uncomment below section if you want to use a proxyq see: SOCKS proxies
|
# uncomment below section if you want to use a proxy
|
||||||
# https://2.python-requests.org/en/latest/user/advanced/#proxies
|
|
||||||
# are also supported: see
|
|
||||||
# https://2.python-requests.org/en/latest/user/advanced/#socks
|
|
||||||
#
|
#
|
||||||
# proxies:
|
# proxies:
|
||||||
# all://:
|
# all://:
|
||||||
@@ -46,30 +38,26 @@ Communication with search engines.
|
|||||||
timeout to load). Can be override by ``timeout`` in the :ref:`settings engines`.
|
timeout to load). Can be override by ``timeout`` in the :ref:`settings engines`.
|
||||||
|
|
||||||
``useragent_suffix`` :
|
``useragent_suffix`` :
|
||||||
Suffix to the user-agent SearXNG uses to send requests to others engines. If an
|
Suffix to add when an engine's User-Agent is set via searxng_useragent().
|
||||||
engine wish to block you, a contact info here may be useful to avoid that.
|
Contact info here may be useful to avoid an engine blocking you.
|
||||||
|
|
||||||
.. _Pool limit configuration: https://www.python-httpx.org/advanced/#pool-limit-configuration
|
.. _Pool limit configuration: https://curl-cffi.readthedocs.io/en/latest/api.html#sessions
|
||||||
|
|
||||||
``pool_maxsize``:
|
|
||||||
Number of allowable keep-alive connections, or ``null`` to always allow. The
|
|
||||||
default is 10. See ``max_keepalive_connections`` `Pool limit configuration`_.
|
|
||||||
|
|
||||||
``pool_connections`` :
|
``pool_connections`` :
|
||||||
Maximum number of allowable connections, or ``null`` # for no limits. The
|
Maximum number of concurrent connections. The default is 100.
|
||||||
default is 100. See ``max_connections`` `Pool limit configuration`_.
|
See ``max_clients`` `Pool limit configuration`_.
|
||||||
|
|
||||||
``keepalive_expiry`` :
|
.. _curl_cffi proxies: https://curl-cffi.readthedocs.io/en/latest/quick_start.html
|
||||||
Number of seconds to keep a connection in the pool. By default 5.0 seconds.
|
|
||||||
See ``keepalive_expiry`` `Pool limit configuration`_.
|
|
||||||
|
|
||||||
.. _httpx proxies: https://www.python-httpx.org/advanced/#http-proxying
|
|
||||||
|
|
||||||
``proxies`` :
|
``proxies`` :
|
||||||
Define one or more proxies you wish to use, see `httpx proxies`_.
|
Define one or more proxies you wish to use, see `curl_cffi proxies`_.
|
||||||
If there are more than one proxy for one protocol (http, https),
|
If there are more than one proxy for one protocol (http, https),
|
||||||
requests to the engines are distributed in a round-robin fashion.
|
requests to the engines are distributed in a round-robin fashion.
|
||||||
|
|
||||||
|
HTTP, HTTPS, SOCKS4, SOCKS5 and SOCKS5h proxies are supported
|
||||||
|
(``http://``, ``https://``, ``socks4://``, ``socks5://``, ``socks5h://``). You should
|
||||||
|
use ``socks5h://`` when using Tor so hostnames are resolved by the proxy.
|
||||||
|
|
||||||
``source_ips`` :
|
``source_ips`` :
|
||||||
If you use multiple network interfaces, define from which IP the requests must
|
If you use multiple network interfaces, define from which IP the requests must
|
||||||
be made. Example:
|
be made. Example:
|
||||||
@@ -87,18 +75,15 @@ Communication with search engines.
|
|||||||
different proxy and source ip.
|
different proxy and source ip.
|
||||||
|
|
||||||
``enable_http2`` :
|
``enable_http2`` :
|
||||||
Enable by default. Set to ``false`` to disable HTTP/2.
|
Enable by default (HTTP/2). Set to ``false`` to force HTTP/1.1.
|
||||||
|
HTTP/3 is opt-in per engine (``enable_http3``).
|
||||||
.. _httpx verification defaults: https://www.python-httpx.org/advanced/#changing-the-verification-defaults
|
|
||||||
.. _httpx ssl configuration: https://www.python-httpx.org/compatibility/#ssl-configuration
|
|
||||||
|
|
||||||
``verify``: : ``$SSL_CERT_FILE``, ``$SSL_CERT_DIR``
|
``verify``: : ``$SSL_CERT_FILE``, ``$SSL_CERT_DIR``
|
||||||
Allow to specify a path to certificate.
|
HTTPS verification uses the OS's trust store by default.
|
||||||
see `httpx verification defaults`_.
|
Set a path to use a custom CA file.
|
||||||
|
|
||||||
In addition to ``verify``, SearXNG supports the ``$SSL_CERT_FILE`` (for a file) and
|
In addition to ``verify``, SearXNG supports the ``$SSL_CERT_FILE`` (for a file) and
|
||||||
``$SSL_CERT_DIR`` (for a directory) OpenSSL variables.
|
``$SSL_CERT_DIR`` (for a directory) OpenSSL variables.
|
||||||
see `httpx ssl configuration`_.
|
|
||||||
|
|
||||||
``max_redirects`` :
|
``max_redirects`` :
|
||||||
30 by default. Maximum redirect before it is an error.
|
30 by default. Maximum redirect before it is an error.
|
||||||
|
|||||||
@@ -143,7 +143,7 @@ parameters with default value can be redefined for special purposes.
|
|||||||
data dict ``{}``
|
data dict ``{}``
|
||||||
cookies dict ``{}``
|
cookies dict ``{}``
|
||||||
verify bool ``True``
|
verify bool ``True``
|
||||||
headers.User-Agent str a random User-Agent
|
headers.User-Agent str ``''``
|
||||||
category str current category, like ``'general'``
|
category str current category, like ``'general'``
|
||||||
safesearch int ``0``, between ``0`` and ``2`` (normal, moderate, strict)
|
safesearch int ``0``, between ``0`` and ``2`` (normal, moderate, strict)
|
||||||
time_range Optional[str] ``None``, can be ``day``, ``week``, ``month``, ``year``
|
time_range Optional[str] ``None``, can be ``day``, ``week``, ``month``, ``year``
|
||||||
@@ -229,6 +229,8 @@ following parameters can be used to specify a search request:
|
|||||||
max_redirects int maximum redirects, hard limit
|
max_redirects int maximum redirects, hard limit
|
||||||
soft_max_redirects int maximum redirects, soft limit. Record an error but don't stop the engine
|
soft_max_redirects int maximum redirects, soft limit. Record an error but don't stop the engine
|
||||||
raise_for_httperror bool True by default: raise an exception if the HTTP code of response is >= 300
|
raise_for_httperror bool True by default: raise an exception if the HTTP code of response is >= 300
|
||||||
|
impersonate str curl_cffi impersonate target (default: chrome, none to disable)
|
||||||
|
curl_options dict Any extra libcurl options for the request
|
||||||
=================== =========== ==========================================================================
|
=================== =========== ==========================================================================
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,8 +0,0 @@
|
|||||||
.. _cara engine:
|
|
||||||
|
|
||||||
===========
|
|
||||||
Cara Images
|
|
||||||
===========
|
|
||||||
|
|
||||||
.. automodule:: searx.engines.cara
|
|
||||||
:members:
|
|
||||||
8
docs/dev/engines/online/europepmc.rst
Normal file
8
docs/dev/engines/online/europepmc.rst
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
.. _europepmc engine:
|
||||||
|
|
||||||
|
==========
|
||||||
|
Europe PMC
|
||||||
|
==========
|
||||||
|
|
||||||
|
.. automodule:: searx.engines.europepmc
|
||||||
|
:members:
|
||||||
@@ -286,7 +286,7 @@ content becomes smart.
|
|||||||
files & folders origin :origin:`docs/dev/reST.rst` ``:origin:`docs/dev/reST.rst```
|
files & folders origin :origin:`docs/dev/reST.rst` ``:origin:`docs/dev/reST.rst```
|
||||||
pull request :pull:`4` ``:pull:`4```
|
pull request :pull:`4` ``:pull:`4```
|
||||||
patch :patch:`af2cae6` ``:patch:`af2cae6```
|
patch :patch:`af2cae6` ``:patch:`af2cae6```
|
||||||
PyPi package :pypi:`httpx` ``:pypi:`httpx```
|
PyPi package :pypi:`curl_cffi` ``:pypi:`curl_cffi```
|
||||||
manual page man :man:`bash` ``:man:`bash```
|
manual page man :man:`bash` ``:man:`bash```
|
||||||
intersphinx_
|
intersphinx_
|
||||||
--------------------------------------------------------------------------------------------------
|
--------------------------------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -1,10 +1,10 @@
|
|||||||
mock==5.2.0
|
mock==5.2.0
|
||||||
nose2[coverage_plugin]==0.16.0
|
nose2[coverage_plugin]==0.16.0
|
||||||
cov-core==1.15.0
|
cov-core==1.15.0
|
||||||
black==25.9.0
|
black==26.5.1
|
||||||
pylint==4.0.7
|
pylint==4.0.8
|
||||||
splinter==0.21.0
|
splinter==0.21.0
|
||||||
selenium==4.47.0
|
selenium==4.48.0
|
||||||
Sphinx==8.2.3;python_version <= "3.11"
|
Sphinx==8.2.3;python_version <= "3.11"
|
||||||
Sphinx==9.1.0; python_version > "3.11"
|
Sphinx==9.1.0; python_version > "3.11"
|
||||||
sphinx-issues==6.0.0
|
sphinx-issues==6.0.0
|
||||||
|
|||||||
@@ -7,13 +7,11 @@ lxml==6.1.2
|
|||||||
pygments==2.21.0
|
pygments==2.21.0
|
||||||
python-dateutil==2.9.0.post0
|
python-dateutil==2.9.0.post0
|
||||||
pyyaml==6.0.3
|
pyyaml==6.0.3
|
||||||
httpx[http2]==0.28.1
|
curl_cffi==0.16.1
|
||||||
httpx-socks[asyncio]==0.13.1
|
|
||||||
sniffio==1.3.1
|
|
||||||
valkey==6.1.1
|
valkey==6.1.1
|
||||||
markdown-it-py==4.2.0
|
markdown-it-py==4.2.0
|
||||||
msgspec==0.21.1
|
msgspec==0.21.1
|
||||||
typer==0.27.1
|
typer==0.27.2
|
||||||
isodate==0.7.2
|
isodate==0.7.2
|
||||||
whitenoise==6.12.0
|
whitenoise==6.12.0
|
||||||
typing-extensions==4.16.0
|
typing-extensions==4.16.0
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Implementation of the :py:obj:`preference <searx.preference>` settings."""
|
"""Implementation of the :py:obj:`preference <searx.preference>` settings."""
|
||||||
|
|
||||||
# pylint: disable = too-few-public-methods
|
# pylint: disable = too-few-public-methods
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
|||||||
@@ -38,7 +38,6 @@ area:
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
__all__ = ["AnswererInfo", "Answerer", "AnswerStorage"]
|
__all__ = ["AnswererInfo", "Answerer", "AnswerStorage"]
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -13,7 +13,6 @@ from dataclasses import dataclass
|
|||||||
from searx.utils import load_module
|
from searx.utils import load_module
|
||||||
from searx.result_types.answer import BaseAnswer
|
from searx.result_types.answer import BaseAnswer
|
||||||
|
|
||||||
|
|
||||||
_default = pathlib.Path(__file__).parent
|
_default = pathlib.Path(__file__).parent
|
||||||
log: logging.Logger = logging.getLogger("searx.answerers")
|
log: logging.Logger = logging.getLogger("searx.answerers")
|
||||||
|
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from urllib.parse import urlencode
|
|||||||
|
|
||||||
import lxml.etree
|
import lxml.etree
|
||||||
import lxml.html
|
import lxml.html
|
||||||
from httpx import HTTPError
|
from curl_cffi.requests.exceptions import RequestException
|
||||||
|
|
||||||
from searx import settings
|
from searx import settings
|
||||||
from searx.engines import (
|
from searx.engines import (
|
||||||
@@ -63,7 +63,7 @@ def bing(query: str, _sxng_locale: str) -> list[str]:
|
|||||||
base_url = "https://www.bing.com/AS/Suggestions?"
|
base_url = "https://www.bing.com/AS/Suggestions?"
|
||||||
# cvid has to be a 32 character long string consisting of numbers and uppsercase characters
|
# cvid has to be a 32 character long string consisting of numbers and uppsercase characters
|
||||||
cvid = ''.join(random.choices(string.ascii_uppercase + string.digits, k=32))
|
cvid = ''.join(random.choices(string.ascii_uppercase + string.digits, k=32))
|
||||||
response = get(base_url + urlencode({'qry': query, 'csr': 1, 'cvid': cvid}))
|
response = get(base_url + urlencode({'qry': query, 'csr': 1, 'cvid': cvid}), enable_http3=True)
|
||||||
results: list[str] = []
|
results: list[str] = []
|
||||||
|
|
||||||
if response.ok:
|
if response.ok:
|
||||||
@@ -83,7 +83,7 @@ def brave(query: str, _sxng_locale: str) -> list[str]:
|
|||||||
url = 'https://search.brave.com/api/suggest?'
|
url = 'https://search.brave.com/api/suggest?'
|
||||||
url += urlencode({'q': query})
|
url += urlencode({'q': query})
|
||||||
country = 'all'
|
country = 'all'
|
||||||
kwargs = {'cookies': {'country': country}}
|
kwargs = {'cookies': {'country': country}, 'enable_http3': True}
|
||||||
resp = get(url, **kwargs)
|
resp = get(url, **kwargs)
|
||||||
results: list[str] = []
|
results: list[str] = []
|
||||||
|
|
||||||
@@ -147,7 +147,7 @@ def google_complete(query: str, sxng_locale: str) -> list[str]:
|
|||||||
)
|
)
|
||||||
results: list[str] = []
|
results: list[str] = []
|
||||||
|
|
||||||
resp = get('https://www.google.com/complete/search?' + args)
|
resp = get('https://www.google.com/complete/search?' + args, enable_http3=True)
|
||||||
if resp and resp.ok:
|
if resp and resp.ok:
|
||||||
json_txt = resp.text[resp.text.find('[') : resp.text.find(']', -3) + 1]
|
json_txt = resp.text[resp.text.find('[') : resp.text.find(']', -3) + 1]
|
||||||
data = json.loads(json_txt)
|
data = json.loads(json_txt)
|
||||||
@@ -418,5 +418,5 @@ def search_autocomplete(backend_name: str, query: str, sxng_locale: str) -> list
|
|||||||
return []
|
return []
|
||||||
try:
|
try:
|
||||||
return backend(query, sxng_locale)
|
return backend(query, sxng_locale)
|
||||||
except (HTTPError, SearxEngineResponseException):
|
except (RequestException, SearxEngineResponseException):
|
||||||
return []
|
return []
|
||||||
|
|||||||
@@ -5,7 +5,6 @@ Implementations used for bot detection.
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
__all__ = ["init", "dump_request", "get_network", "too_many_requests", "ProxyFix"]
|
__all__ = ["init", "dump_request", "get_network", "too_many_requests", "ProxyFix"]
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -182,7 +182,7 @@ class Config:
|
|||||||
if default is UNSET:
|
if default is UNSET:
|
||||||
raise KeyError(name)
|
raise KeyError(name)
|
||||||
return default
|
return default
|
||||||
(modulename, name) = str(fqn).rsplit('.', 1)
|
modulename, name = str(fqn).rsplit('.', 1)
|
||||||
m = __import__(modulename, {}, {}, [name], 0)
|
m = __import__(modulename, {}, {}, [name], 0)
|
||||||
return getattr(m, name)
|
return getattr(m, name)
|
||||||
|
|
||||||
|
|||||||
@@ -13,7 +13,6 @@ Accept_ header ..
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from ipaddress import (
|
from ipaddress import (
|
||||||
IPv4Network,
|
IPv4Network,
|
||||||
IPv6Network,
|
IPv6Network,
|
||||||
|
|||||||
@@ -14,7 +14,6 @@ bot if the Accept-Encoding_ header ..
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from ipaddress import (
|
from ipaddress import (
|
||||||
IPv4Network,
|
IPv4Network,
|
||||||
IPv6Network,
|
IPv6Network,
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ if the Accept-Language_ header is unset.
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from ipaddress import (
|
from ipaddress import (
|
||||||
IPv4Network,
|
IPv4Network,
|
||||||
IPv6Network,
|
IPv6Network,
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ the Connection_ header is set to ``close``.
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from ipaddress import (
|
from ipaddress import (
|
||||||
IPv4Network,
|
IPv4Network,
|
||||||
IPv6Network,
|
IPv6Network,
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ Metadata`_. A request is filtered out in case of:
|
|||||||
|
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=unused-argument
|
# pylint: disable=unused-argument
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -12,7 +12,6 @@ the User-Agent_ header is unset or matches the regular expression
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from ipaddress import (
|
from ipaddress import (
|
||||||
IPv4Network,
|
IPv4Network,
|
||||||
@@ -25,7 +24,6 @@ import flask
|
|||||||
from . import config
|
from . import config
|
||||||
from ._helpers import too_many_requests
|
from ._helpers import too_many_requests
|
||||||
|
|
||||||
|
|
||||||
USER_AGENT = (
|
USER_AGENT = (
|
||||||
r'('
|
r'('
|
||||||
+ r'unknown'
|
+ r'unknown'
|
||||||
|
|||||||
@@ -55,7 +55,6 @@ from ._helpers import (
|
|||||||
logger,
|
logger,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
logger = logger.getChild('ip_limit')
|
logger = logger.getChild('ip_limit')
|
||||||
|
|
||||||
BURST_WINDOW = 20
|
BURST_WINDOW = 20
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ The ``ip_lists`` method implements :py:obj:`block-list <block_ip>` and
|
|||||||
]
|
]
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=unused-argument
|
# pylint: disable=unused-argument
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Implementation of a middleware to determine the real IP of an HTTP request
|
"""Implementation of a middleware to determine the real IP of an HTTP request
|
||||||
(:py:obj:`flask.request.remote_addr`) behind a proxy chain."""
|
(:py:obj:`flask.request.remote_addr`) behind a proxy chain."""
|
||||||
|
|
||||||
# pylint: disable=too-many-branches
|
# pylint: disable=too-many-branches
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Providing a Valkey database for the botdetection methods."""
|
"""Providing a Valkey database for the botdetection methods."""
|
||||||
|
|
||||||
|
|
||||||
import valkey
|
import valkey
|
||||||
|
|
||||||
__all__ = ["set_valkey_client", "get_valkey_client"]
|
__all__ = ["set_valkey_client", "get_valkey_client"]
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Implementations needed for a branding of SearXNG."""
|
"""Implementations needed for a branding of SearXNG."""
|
||||||
|
|
||||||
# pylint: disable=too-few-public-methods
|
# pylint: disable=too-few-public-methods
|
||||||
|
|
||||||
# Struct fields aren't discovered in Python 3.14
|
# Struct fields aren't discovered in Python 3.14
|
||||||
|
|||||||
@@ -465,7 +465,7 @@ class ExpireCacheSQLite(sqlitedb.SQLiteAppl, ExpireCache):
|
|||||||
|
|
||||||
# Check if value is expired. It's possible that it's expired but has not
|
# Check if value is expired. It's possible that it's expired but has not
|
||||||
# yet been automatically deleted by the periodic maintenance
|
# yet been automatically deleted by the periodic maintenance
|
||||||
(value, expire) = row
|
value, expire = row
|
||||||
now = time.time()
|
now = time.time()
|
||||||
if expire < now:
|
if expire < now:
|
||||||
# The record is deleted during the maintenance interval. Deleting
|
# The record is deleted during the maintenance interval. Deleting
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
import warnings
|
import warnings
|
||||||
|
|
||||||
|
|
||||||
# limiter backward compatibility
|
# limiter backward compatibility
|
||||||
# ------------------------------
|
# ------------------------------
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
make data.all
|
make data.all
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=invalid-name
|
# pylint: disable=invalid-name
|
||||||
|
|
||||||
__all__ = ["ahmia_blacklist_loader", "data_dir", "get_cache"]
|
__all__ = ["ahmia_blacklist_loader", "data_dir", "get_cache"]
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Simple implementation to store TrackerPatterns data in a SQL database."""
|
"""Simple implementation to store TrackerPatterns data in a SQL database."""
|
||||||
|
|
||||||
# pylint: disable=too-many-branches
|
# pylint: disable=too-many-branches
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
@@ -10,7 +11,7 @@ import re
|
|||||||
from collections.abc import Iterator
|
from collections.abc import Iterator
|
||||||
from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode
|
from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode
|
||||||
|
|
||||||
from httpx import HTTPError
|
from curl_cffi.requests.exceptions import RequestException
|
||||||
|
|
||||||
from searx.data.core import get_cache, log
|
from searx.data.core import get_cache, log
|
||||||
from searx.network import get as http_get
|
from searx.network import get as http_get
|
||||||
@@ -87,8 +88,8 @@ class TrackerPatternsDB:
|
|||||||
try:
|
try:
|
||||||
resp = http_get(url, timeout=3)
|
resp = http_get(url, timeout=3)
|
||||||
|
|
||||||
except HTTPError as exc:
|
except RequestException as exc:
|
||||||
log.warning("TRACKER_PATTERNS: HTTPError (%s) occured while fetching %s", url, exc)
|
log.warning("TRACKER_PATTERNS: RequestException while fetching %s: %s", url, exc)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if resp.status_code != 200:
|
if resp.status_code != 200:
|
||||||
|
|||||||
@@ -305,7 +305,7 @@ class Engine(abc.ABC): # pylint: disable=too-few-public-methods
|
|||||||
|
|
||||||
region: str = ""
|
region: str = ""
|
||||||
"""For an engine, when there is ``region: ...`` in the YAML settings the engine
|
"""For an engine, when there is ``region: ...`` in the YAML settings the engine
|
||||||
does support only this one region::
|
does support only this one region:
|
||||||
|
|
||||||
.. code:: yaml
|
.. code:: yaml
|
||||||
|
|
||||||
@@ -317,6 +317,9 @@ class Engine(abc.ABC): # pylint: disable=too-few-public-methods
|
|||||||
enable_http: bool
|
enable_http: bool
|
||||||
"""Enable HTTP (by default only HTTPS is enabled)."""
|
"""Enable HTTP (by default only HTTPS is enabled)."""
|
||||||
|
|
||||||
|
enable_http3: bool = False
|
||||||
|
"""Enables the use of HTTP/3 if available"""
|
||||||
|
|
||||||
shortcut: str
|
shortcut: str
|
||||||
"""Code used to execute bang requests (``!foo``)"""
|
"""Code used to execute bang requests (``!foo``)"""
|
||||||
|
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ categories: list[str]
|
|||||||
disabled: bool
|
disabled: bool
|
||||||
display_error_messages: bool
|
display_error_messages: bool
|
||||||
enable_http: bool
|
enable_http: bool
|
||||||
|
enable_http3: bool
|
||||||
engine_type: str
|
engine_type: str
|
||||||
inactive: bool
|
inactive: bool
|
||||||
max_page: int
|
max_page: int
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ Implementation
|
|||||||
==============
|
==============
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ To use this engine, add an entry similar to the following to your engine list in
|
|||||||
https://learn.microsoft.com/en-us/entra/identity-platform/quickstart-register-app
|
https://learn.microsoft.com/en-us/entra/identity-platform/quickstart-register-app
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
|
||||||
from searx.enginelib import EngineCache
|
from searx.enginelib import EngineCache
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""BASE (Scholar publications)"""
|
"""BASE (Scholar publications)"""
|
||||||
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
import re
|
import re
|
||||||
|
|
||||||
|
|||||||
@@ -40,6 +40,7 @@ about: dict[str, t.Any] = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ["general", "web"]
|
categories = ["general", "web"]
|
||||||
safesearch = True
|
safesearch = True
|
||||||
|
enable_http3 = True
|
||||||
_safesearch_map: dict[int, str] = {
|
_safesearch_map: dict[int, str] = {
|
||||||
0: "off",
|
0: "off",
|
||||||
1: "moderate",
|
1: "moderate",
|
||||||
@@ -61,8 +62,8 @@ def get_locale_params(engine_region: str | None) -> dict[str, str] | None:
|
|||||||
|
|
||||||
The ``mkt`` parameter takes a full ``<language>-<country>`` code.
|
The ``mkt`` parameter takes a full ``<language>-<country>`` code.
|
||||||
|
|
||||||
This function is shared with :py:mod:`searx.engines.bing_images`,
|
This function is shared with :py:mod:`searx.engines.bing_news`, and
|
||||||
:py:mod:`searx.engines.bing_news`, and :py:mod:`searx.engines.bing_videos`.
|
:py:mod:`searx.engines.bing_videos`.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
if not engine_region or engine_region == "clear":
|
if not engine_region or engine_region == "clear":
|
||||||
@@ -71,43 +72,21 @@ def get_locale_params(engine_region: str | None) -> dict[str, str] | None:
|
|||||||
return {"mkt": engine_region}
|
return {"mkt": engine_region}
|
||||||
|
|
||||||
|
|
||||||
def override_accept_language(params: "OnlineParams", engine_region: str | None) -> None:
|
|
||||||
"""Override the ``Accept-Language`` header.
|
|
||||||
|
|
||||||
The default header built by :py:class:`~searx.search.processors.online.OnlineProcessor`
|
|
||||||
appends ``en;q=0.3`` as a fallback language::
|
|
||||||
|
|
||||||
Accept-Language: de,de-DE;q=0.7,en;q=0.3
|
|
||||||
|
|
||||||
Bing seems to better select the results locale based on the
|
|
||||||
``Accept-Language`` value header.
|
|
||||||
|
|
||||||
This function is shared with :py:mod:`searx.engines.bing_images`,
|
|
||||||
:py:mod:`searx.engines.bing_news`, and :py:mod:`searx.engines.bing_videos`.
|
|
||||||
"""
|
|
||||||
|
|
||||||
if not engine_region or engine_region == "clear":
|
|
||||||
return
|
|
||||||
|
|
||||||
lang = engine_region.split("-")[0]
|
|
||||||
params["headers"]["Accept-Language"] = f"{engine_region},{lang};q=0.9"
|
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams"):
|
def request(query: str, params: "OnlineParams"):
|
||||||
"""Assemble a Bing-Web request."""
|
"""Assemble a Bing-Web request."""
|
||||||
|
|
||||||
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
||||||
|
|
||||||
override_accept_language(params, engine_region)
|
|
||||||
|
|
||||||
query_params: dict[str, str | int] = {
|
query_params: dict[str, str | int] = {
|
||||||
"q": query,
|
"q": query,
|
||||||
"adlt": _safesearch_map.get(params.get("safesearch", 0), "off"),
|
"adlt": _safesearch_map.get(params.get("safesearch", 0), "off"),
|
||||||
}
|
}
|
||||||
|
|
||||||
locale_params = get_locale_params(engine_region)
|
if engine_region and engine_region != "clear":
|
||||||
if locale_params:
|
lang, _, cc = engine_region.partition("-")
|
||||||
query_params.update(locale_params)
|
query_params["setlang"] = lang
|
||||||
|
if cc and cc not in ("us", "cn", "ru"): # bing just sends junk for these
|
||||||
|
query_params["cc"] = cc
|
||||||
|
|
||||||
params["url"] = f"{base_url}/search?{urlencode(query_params)}"
|
params["url"] = f"{base_url}/search?{urlencode(query_params)}"
|
||||||
|
|
||||||
|
|||||||
@@ -1,18 +1,20 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Bing-Images: description see :py:obj:`searx.engines.bing`."""
|
"""Bing-Images: description see :py:obj:`searx.engines.bing`."""
|
||||||
|
|
||||||
|
import typing as t
|
||||||
import json
|
import json
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
from searx.engines.bing import ( # pylint: disable=unused-import
|
from searx.engines.bing import fetch_traits # pylint: disable=unused-import
|
||||||
fetch_traits,
|
from searx.result_types import EngineResults
|
||||||
get_locale_params,
|
|
||||||
override_accept_language,
|
if t.TYPE_CHECKING:
|
||||||
)
|
from searx.extended_types import SXNG_Response
|
||||||
|
from searx.search.processors import OnlineParams
|
||||||
|
|
||||||
|
|
||||||
# about
|
|
||||||
about = {
|
about = {
|
||||||
"website": "https://www.bing.com/images",
|
"website": "https://www.bing.com/images",
|
||||||
"wikidata_id": "Q182496",
|
"wikidata_id": "Q182496",
|
||||||
@@ -22,9 +24,9 @@ about = {
|
|||||||
"results": "HTML",
|
"results": "HTML",
|
||||||
}
|
}
|
||||||
|
|
||||||
# engine dependent config
|
|
||||||
categories = ["images", "web"]
|
categories = ["images", "web"]
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
safesearch = True
|
safesearch = True
|
||||||
time_range_support = True
|
time_range_support = True
|
||||||
time_map = {
|
time_map = {
|
||||||
@@ -38,26 +40,27 @@ base_url = "https://www.bing.com"
|
|||||||
"""Bing-Image search URL"""
|
"""Bing-Image search URL"""
|
||||||
|
|
||||||
|
|
||||||
def request(query, params):
|
def request(query: str, params: "OnlineParams"):
|
||||||
"""Assemble a Bing-Image request."""
|
"""Assemble a Bing-Image request."""
|
||||||
|
|
||||||
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
||||||
|
|
||||||
override_accept_language(params, engine_region)
|
# build URL query / example:
|
||||||
|
# https://www.bing.com/images/async?q=foo&mmasync=1&first=1&count=35
|
||||||
|
|
||||||
# build URL query
|
|
||||||
# - example: https://www.bing.com/images/async?q=foo&async=1&first=1&count=35
|
|
||||||
query_params = {
|
query_params = {
|
||||||
"q": query,
|
"q": query,
|
||||||
"async": "1",
|
"mmasync": "1",
|
||||||
# to simplify the page count lets use the default of 35 images per page
|
# to simplify the page count lets use the default of 35 images per page
|
||||||
"first": (int(params.get("pageno", 1)) - 1) * 35 + 1,
|
"first": (int(params.get("pageno", 1)) - 1) * 35 + 1,
|
||||||
"count": 35,
|
"count": 35,
|
||||||
}
|
}
|
||||||
|
|
||||||
locale_params = get_locale_params(engine_region)
|
if engine_region and engine_region != "clear":
|
||||||
if locale_params:
|
lang, _, cc = engine_region.partition("-")
|
||||||
query_params.update(locale_params)
|
query_params["setlang"] = lang
|
||||||
|
if cc:
|
||||||
|
query_params["cc"] = cc
|
||||||
|
|
||||||
# time range
|
# time range
|
||||||
# - example: one year (525600 minutes) 'qft=filterui:age-lt525600'
|
# - example: one year (525600 minutes) 'qft=filterui:age-lt525600'
|
||||||
@@ -67,10 +70,10 @@ def request(query, params):
|
|||||||
params["url"] = base_url + "/images/async?" + urlencode(query_params)
|
params["url"] = base_url + "/images/async?" + urlencode(query_params)
|
||||||
|
|
||||||
|
|
||||||
def response(resp):
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
"""Get response from Bing-Image"""
|
"""Get response from Bing-Image"""
|
||||||
|
|
||||||
results = []
|
res = EngineResults()
|
||||||
|
|
||||||
dom = html.fromstring(resp.text)
|
dom = html.fromstring(resp.text)
|
||||||
|
|
||||||
@@ -81,19 +84,22 @@ def response(resp):
|
|||||||
|
|
||||||
metadata = json.loads(result.xpath('.//a[@class="iusc"]/@m')[0])
|
metadata = json.loads(result.xpath('.//a[@class="iusc"]/@m')[0])
|
||||||
title = " ".join(result.xpath('.//div[@class="infnmpt"]//a/text()')).strip()
|
title = " ".join(result.xpath('.//div[@class="infnmpt"]//a/text()')).strip()
|
||||||
|
if not title:
|
||||||
|
title = result.xpath('.//div[@class="infnmpt"]//a/@title')[0]
|
||||||
|
|
||||||
img_format = " ".join(result.xpath('.//div[@class="imgpt"]/div/span/text()')).strip().split(" · ")
|
img_format = " ".join(result.xpath('.//div[@class="imgpt"]/div/span/text()')).strip().split(" · ")
|
||||||
source = " ".join(result.xpath('.//div[@class="imgpt"]//div[@class="lnkw"]//a/text()')).strip()
|
source = " ".join(result.xpath('.//div[@class="imgpt"]//div[@class="lnkw"]//a/text()')).strip()
|
||||||
results.append(
|
|
||||||
{
|
res.add(
|
||||||
"template": "images.html",
|
res.types.Image(
|
||||||
"url": metadata["purl"],
|
title=title,
|
||||||
"thumbnail_src": metadata["turl"],
|
url=metadata["purl"],
|
||||||
"img_src": metadata["murl"],
|
thumbnail_src=metadata["turl"],
|
||||||
"content": metadata.get("desc"),
|
img_src=metadata["murl"],
|
||||||
"title": title,
|
content=metadata.get("desc"),
|
||||||
"source": source,
|
source=source,
|
||||||
"resolution": img_format[0],
|
resolution=img_format[0],
|
||||||
"img_format": img_format[1] if len(img_format) >= 2 else None,
|
img_format=img_format[1] if len(img_format) >= 2 else "",
|
||||||
}
|
|
||||||
)
|
)
|
||||||
return results
|
)
|
||||||
|
return res
|
||||||
|
|||||||
@@ -12,10 +12,7 @@ from urllib.parse import urlencode
|
|||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
from searx.enginelib.traits import EngineTraits
|
from searx.enginelib.traits import EngineTraits
|
||||||
from searx.engines.bing import (
|
from searx.engines.bing import get_locale_params
|
||||||
get_locale_params,
|
|
||||||
override_accept_language,
|
|
||||||
)
|
|
||||||
from searx.utils import eval_xpath, eval_xpath_getindex, eval_xpath_list, extract_text
|
from searx.utils import eval_xpath, eval_xpath_getindex, eval_xpath_list, extract_text
|
||||||
|
|
||||||
# about
|
# about
|
||||||
@@ -33,6 +30,7 @@ categories = ["news"]
|
|||||||
paging = True
|
paging = True
|
||||||
"""If go through the pages and there are actually no new results for another
|
"""If go through the pages and there are actually no new results for another
|
||||||
page, then bing returns the results from the last page again."""
|
page, then bing returns the results from the last page again."""
|
||||||
|
enable_http3 = True
|
||||||
|
|
||||||
time_range_support = True
|
time_range_support = True
|
||||||
time_map = {
|
time_map = {
|
||||||
@@ -53,8 +51,6 @@ def request(query, params):
|
|||||||
|
|
||||||
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
||||||
|
|
||||||
override_accept_language(params, engine_region)
|
|
||||||
|
|
||||||
# build URL query
|
# build URL query
|
||||||
# - example: https://www.bing.com/news/infinitescrollajax?q=london&first=1
|
# - example: https://www.bing.com/news/infinitescrollajax?q=london&first=1
|
||||||
page = int(params.get("pageno", 1)) - 1
|
page = int(params.get("pageno", 1)) - 1
|
||||||
|
|||||||
@@ -9,7 +9,6 @@ from lxml import html
|
|||||||
from searx.engines.bing import ( # pylint: disable=unused-import
|
from searx.engines.bing import ( # pylint: disable=unused-import
|
||||||
fetch_traits,
|
fetch_traits,
|
||||||
get_locale_params,
|
get_locale_params,
|
||||||
override_accept_language,
|
|
||||||
)
|
)
|
||||||
from searx.engines.bing_images import time_map
|
from searx.engines.bing_images import time_map
|
||||||
from searx.utils import eval_xpath, eval_xpath_getindex
|
from searx.utils import eval_xpath, eval_xpath_getindex
|
||||||
@@ -26,6 +25,7 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ["videos", "web"]
|
categories = ["videos", "web"]
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
safesearch = True
|
safesearch = True
|
||||||
time_range_support = True
|
time_range_support = True
|
||||||
|
|
||||||
@@ -38,8 +38,6 @@ def request(query, params):
|
|||||||
|
|
||||||
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
engine_region = traits.get_region(params["searxng_locale"], traits.all_locale)
|
||||||
|
|
||||||
override_accept_language(params, engine_region)
|
|
||||||
|
|
||||||
# build URL query
|
# build URL query
|
||||||
# - example: https://www.bing.com/videos/asyncv2?q=foo&async=content&first=1&count=35
|
# - example: https://www.bing.com/videos/asyncv2?q=foo&async=content&first=1&count=35
|
||||||
query_params = {
|
query_params = {
|
||||||
|
|||||||
@@ -151,6 +151,7 @@ about = {
|
|||||||
|
|
||||||
base_url = "https://search.brave.com/"
|
base_url = "https://search.brave.com/"
|
||||||
categories = []
|
categories = []
|
||||||
|
enable_http3 = True
|
||||||
brave_category: t.Literal["search", "videos", "images", "news", "goggles"] = "search"
|
brave_category: t.Literal["search", "videos", "images", "news", "goggles"] = "search"
|
||||||
"""Brave supports common web-search, videos, images, news, and goggles search.
|
"""Brave supports common web-search, videos, images, news, and goggles search.
|
||||||
|
|
||||||
@@ -247,13 +248,13 @@ def extract_json_data(text: str) -> dict[str, t.Any]:
|
|||||||
# node_ids: [0, 19],
|
# node_ids: [0, 19],
|
||||||
# data: [{type:"data",data: .... ["q","goggles_id"],route:1,url:1}}]
|
# data: [{type:"data",data: .... ["q","goggles_id"],route:1,url:1}}]
|
||||||
# ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
# ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
text = text[text.index("<script") : text.index("</script")]
|
# form: null,
|
||||||
if not text:
|
# error: null
|
||||||
raise ValueError("can't find JS/JSON data in the given text")
|
# });
|
||||||
start = text.index("data: [{")
|
start = text.index("data: [{")
|
||||||
end = text.rindex("}}]")
|
newline = text.index("\n", start)
|
||||||
js_obj_str = text[start:end]
|
end = text.rindex("}}]", start, newline)
|
||||||
js_obj_str = "{" + js_obj_str + "}}]}"
|
js_obj_str = "{" + text[start:end] + "}}]}"
|
||||||
# js_obj_str = js_obj_str.replace("\xa0", "") # remove ASCII for
|
# js_obj_str = js_obj_str.replace("\xa0", "") # remove ASCII for
|
||||||
# js_obj_str = js_obj_str.replace(r"\u003C", "<").replace(r"\u003c", "<") # fix broken HTML tags in strings
|
# js_obj_str = js_obj_str.replace(r"\u003C", "<").replace(r"\u003c", "<") # fix broken HTML tags in strings
|
||||||
json_str = js_obj_str_to_json_str(js_obj_str)
|
json_str = js_obj_str_to_json_str(js_obj_str)
|
||||||
@@ -353,14 +354,14 @@ def _parse_news(resp: SXNG_Response) -> EngineResults:
|
|||||||
res = EngineResults()
|
res = EngineResults()
|
||||||
dom = html.fromstring(resp.text)
|
dom = html.fromstring(resp.text)
|
||||||
|
|
||||||
for result in eval_xpath_list(dom, "//div[contains(@class, 'results')]//div[@data-type='news']"):
|
for result in eval_xpath_list(dom, "//div[@data-type='news']"):
|
||||||
url = eval_xpath_getindex(result, ".//a[contains(@class, 'result-header')]/@href", 0, default=None)
|
url = eval_xpath_getindex(result, ".//a/@href", 0, default=None)
|
||||||
if url is None:
|
if url is None:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
title = eval_xpath_list(result, ".//span[contains(@class, 'snippet-title')]")
|
title = eval_xpath_list(result, ".//div[contains(@class, 'title')]")
|
||||||
content = eval_xpath_list(result, ".//p[contains(@class, 'desc')]")
|
content = eval_xpath_list(result, ".//div[contains(@class, 'description')]")
|
||||||
thumbnail = eval_xpath_getindex(result, ".//div[contains(@class, 'image-wrapper')]//img/@src", 0, default="")
|
thumbnail = eval_xpath_getindex(result, ".//a[contains(@class, 'thumbnail')]//img/@src", 0, default="")
|
||||||
|
|
||||||
item = res.types.LegacyResult(
|
item = res.types.LegacyResult(
|
||||||
template="default.html",
|
template="default.html",
|
||||||
|
|||||||
@@ -91,6 +91,7 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
|
|
||||||
params["url"] = f"{base_url}?{urlencode(search_args)}"
|
params["url"] = f"{base_url}?{urlencode(search_args)}"
|
||||||
params["headers"]["X-Subscription-Token"] = api_key
|
params["headers"]["X-Subscription-Token"] = api_key
|
||||||
|
params["headers"]["Accept"] = "application/json"
|
||||||
|
|
||||||
|
|
||||||
def _extract_published_date(published_date_raw: str):
|
def _extract_published_date(published_date_raw: str):
|
||||||
|
|||||||
@@ -1,85 +0,0 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
||||||
# pylint: disable=invalid-name
|
|
||||||
"""Cara_ is a social media and portfolio-sharing platform for artists and art
|
|
||||||
enthusiasts.
|
|
||||||
|
|
||||||
With the widespread use of generative AI, Cara_ decided to build a place that
|
|
||||||
filters out gen AI images so that people searching for authentic creatives and
|
|
||||||
images can do so easily.
|
|
||||||
|
|
||||||
.. _Cara: https://cara.app/about
|
|
||||||
"""
|
|
||||||
|
|
||||||
from urllib.parse import urlencode
|
|
||||||
|
|
||||||
import typing as t
|
|
||||||
|
|
||||||
from searx.result_types import EngineResults
|
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
|
||||||
from searx.extended_types import SXNG_Response
|
|
||||||
from searx.search.processors import OnlineParams
|
|
||||||
|
|
||||||
|
|
||||||
about = {
|
|
||||||
"website": "https://cara.app",
|
|
||||||
"official_api_documentation": None,
|
|
||||||
"use_official_api": False,
|
|
||||||
"require_api_key": False,
|
|
||||||
"results": "JSON",
|
|
||||||
}
|
|
||||||
|
|
||||||
base_url = "https://cara.app"
|
|
||||||
images_url = "https://images.cara.app"
|
|
||||||
|
|
||||||
categories = ["images"]
|
|
||||||
paging = True
|
|
||||||
results_per_page = 24
|
|
||||||
|
|
||||||
# if using HTTP2, we get blocked immediately
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams") -> None:
|
|
||||||
args = {
|
|
||||||
"q": query,
|
|
||||||
"sortBy": "Top",
|
|
||||||
"take": results_per_page,
|
|
||||||
"skip": (params["pageno"] - 1) * results_per_page,
|
|
||||||
}
|
|
||||||
params["url"] = f"{base_url}/api/search/portfolio-posts?{urlencode(args)}"
|
|
||||||
|
|
||||||
|
|
||||||
def response(resp: "SXNG_Response"):
|
|
||||||
res = EngineResults()
|
|
||||||
json_data: list[dict[str, t.Any]] = resp.json()
|
|
||||||
|
|
||||||
for result in json_data:
|
|
||||||
thumbnail, img = None, None
|
|
||||||
|
|
||||||
i: dict[str, str]
|
|
||||||
for i in result["images"]:
|
|
||||||
if thumbnail is None or i["isCoverImg"]:
|
|
||||||
thumbnail = i
|
|
||||||
|
|
||||||
if img is None or not i["isCoverImg"]:
|
|
||||||
img = i
|
|
||||||
|
|
||||||
if not thumbnail or not img:
|
|
||||||
continue
|
|
||||||
|
|
||||||
res.add(
|
|
||||||
res.types.LegacyResult(
|
|
||||||
{
|
|
||||||
"template": "images.html",
|
|
||||||
"url": f"{base_url}/post/{result['id']}",
|
|
||||||
"thumbnail_src": f"{images_url}/{thumbnail['src']}?height=256",
|
|
||||||
"img_src": f"{images_url}/{img['src']}",
|
|
||||||
"title": result["title"],
|
|
||||||
"content": result["content"],
|
|
||||||
"author": result["name"],
|
|
||||||
}
|
|
||||||
)
|
|
||||||
)
|
|
||||||
|
|
||||||
return res
|
|
||||||
@@ -84,6 +84,7 @@ def request(query: str, params: "OnlineParams"):
|
|||||||
|
|
||||||
params["url"] = f"{base_url}/api/v1/_search"
|
params["url"] = f"{base_url}/api/v1/_search"
|
||||||
params["method"] = "POST"
|
params["method"] = "POST"
|
||||||
|
params["impersonate"] = "none"
|
||||||
|
|
||||||
json_data = {
|
json_data = {
|
||||||
"query": query,
|
"query": query,
|
||||||
|
|||||||
@@ -84,7 +84,6 @@ from threading import Thread
|
|||||||
from searx import logger
|
from searx import logger
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
|
|
||||||
engine_type = 'offline'
|
engine_type = 'offline'
|
||||||
paging = True
|
paging = True
|
||||||
command = []
|
command = []
|
||||||
|
|||||||
@@ -141,12 +141,13 @@ def response(resp: "SXNG_Response") -> EngineResults:
|
|||||||
if name:
|
if name:
|
||||||
authors.add(name)
|
authors.add(name)
|
||||||
|
|
||||||
|
tag = result.get("fieldOfStudy")
|
||||||
res.add(
|
res.add(
|
||||||
res.types.Paper(
|
res.types.Paper(
|
||||||
title=result.get("title"),
|
title=result.get("title"),
|
||||||
url=url,
|
url=url,
|
||||||
content=result.get("fullText", "") or "",
|
content=result.get("fullText", "") or "",
|
||||||
tags=result.get("fieldOfStudy", []),
|
tags=[tag] if tag else [],
|
||||||
publishedDate=published_date,
|
publishedDate=published_date,
|
||||||
type=result.get("documentType", "") or "",
|
type=result.get("documentType", "") or "",
|
||||||
authors=authors,
|
authors=authors,
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Docker Hub (IT)"""
|
"""Docker Hub (IT)"""
|
||||||
|
|
||||||
# pylint: disable=use-dict-literal
|
# pylint: disable=use-dict-literal
|
||||||
|
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|||||||
@@ -8,6 +8,9 @@ import typing as t
|
|||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
import html
|
import html
|
||||||
|
|
||||||
|
from searx.enginelib import EngineCache
|
||||||
|
from searx.exceptions import SearxEngineAPIException
|
||||||
|
from searx.network import post
|
||||||
from searx.utils import format_duration, html_to_text, humanize_number
|
from searx.utils import format_duration, html_to_text, humanize_number
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
@@ -35,15 +38,35 @@ dogpile_categ = "search"
|
|||||||
base_url = "https://www.dogpile.com"
|
base_url = "https://www.dogpile.com"
|
||||||
safe_search_map = {0: "none", 1: "moderate", 2: "heavy"}
|
safe_search_map = {0: "none", 1: "moderate", 2: "heavy"}
|
||||||
|
|
||||||
|
CACHE: EngineCache
|
||||||
|
"""Cache for the API token from dogpile"""
|
||||||
|
|
||||||
|
|
||||||
def setup(_: dict[str, t.Any]) -> bool | None:
|
def setup(_: dict[str, t.Any]) -> bool | None:
|
||||||
if dogpile_categ not in ("search", "images", "videos", "news"):
|
if dogpile_categ not in ("search", "images", "videos", "news"):
|
||||||
raise ValueError("invalid search type: %s" % dogpile_categ)
|
raise ValueError("invalid search type: %s" % dogpile_categ)
|
||||||
|
global CACHE # pylint: disable=global-statement
|
||||||
|
CACHE = EngineCache("dogpile") # one token for images/videos/news
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def _obtain_token() -> str:
|
||||||
|
token = CACHE.get("token")
|
||||||
|
if token:
|
||||||
|
return token
|
||||||
|
resp = post(f"{base_url}/api/token/refresh", headers={"Origin": base_url}, cookies={"dp_api_token": "1"})
|
||||||
|
if not resp.ok:
|
||||||
|
raise SearxEngineAPIException("failed to obtain dogpile token")
|
||||||
|
token = resp.json()["token"]
|
||||||
|
CACHE.set("token", token, expire=240) # 300s ttl
|
||||||
|
return token
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams"):
|
def request(query: str, params: "OnlineParams"):
|
||||||
params["url"] = f"{base_url}/api/{dogpile_categ}"
|
params["url"] = f"{base_url}/api/{dogpile_categ}"
|
||||||
params["headers"]["Origin"] = base_url
|
params["headers"]["Origin"] = base_url
|
||||||
|
params["cookies"]["dp_api_token"] = "1"
|
||||||
|
params["headers"]["x-dogpile-token"] = _obtain_token()
|
||||||
|
|
||||||
params["method"] = "POST"
|
params["method"] = "POST"
|
||||||
params["json"] = {"q": query, "qadf": safe_search_map[params["safesearch"]], "page": params["pageno"]}
|
params["json"] = {"q": query, "qadf": safe_search_map[params["safesearch"]], "page": params["pageno"]}
|
||||||
|
|||||||
@@ -164,6 +164,7 @@ Terms / phrases that you keep coming across:
|
|||||||
https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Headers/Accept-Language
|
https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Headers/Accept-Language
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=global-statement
|
# pylint: disable=global-statement
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ least we could not find out how language support should work. It seems that
|
|||||||
most of the features are based on English terms.
|
most of the features are based on English terms.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
|
||||||
from urllib.parse import urlencode, urlparse, urljoin
|
from urllib.parse import urlencode, urlparse, urljoin
|
||||||
|
|||||||
@@ -98,6 +98,7 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
# The vqd value is generated from the query and the UA header. To be able to
|
# The vqd value is generated from the query and the UA header. To be able to
|
||||||
# reuse the vqd value, the UA header must be static.
|
# reuse the vqd value, the UA header must be static.
|
||||||
headers["User-Agent"] = _HTTP_User_Agent
|
headers["User-Agent"] = _HTTP_User_Agent
|
||||||
|
params["impersonate"] = "none"
|
||||||
vqd = get_vqd(query=query, params=params) or fetch_vqd(query=query, params=params)
|
vqd = get_vqd(query=query, params=params) or fetch_vqd(query=query, params=params)
|
||||||
|
|
||||||
headers["Accept"] = "*/*"
|
headers["Accept"] = "*/*"
|
||||||
|
|||||||
@@ -17,7 +17,6 @@ from searx.result_types import EngineResults
|
|||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
from searx import weather
|
from searx import weather
|
||||||
|
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
"website": 'https://duckduckgo.com/',
|
"website": 'https://duckduckgo.com/',
|
||||||
"wikidata_id": 'Q12805',
|
"wikidata_id": 'Q12805',
|
||||||
|
|||||||
@@ -14,11 +14,12 @@ can't build it ourselves and must scrape it from the HTML pages.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
import re
|
||||||
|
|
||||||
from urllib.parse import quote_plus
|
from urllib.parse import quote_plus, urljoin
|
||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
from searx.utils import html_to_text, gen_useragent, extract_text, eval_xpath
|
from searx.utils import html_to_text, extract_text, eval_xpath
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.enginelib import EngineCache
|
from searx.enginelib import EngineCache
|
||||||
from searx.network import get
|
from searx.network import get
|
||||||
@@ -38,7 +39,6 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ["general"]
|
categories = ["general"]
|
||||||
paging = True
|
paging = True
|
||||||
_HTTP_User_Agent: str = gen_useragent()
|
|
||||||
|
|
||||||
base_url = "https://duckduckgo.com"
|
base_url = "https://duckduckgo.com"
|
||||||
|
|
||||||
@@ -73,6 +73,8 @@ def _fetch_first_page_link(
|
|||||||
resp = get(
|
resp = get(
|
||||||
url=f"{base_url}/?q={quote_plus(query)}&t=h_&ia=web",
|
url=f"{base_url}/?q={quote_plus(query)}&t=h_&ia=web",
|
||||||
headers=headers,
|
headers=headers,
|
||||||
|
impersonate="firefox",
|
||||||
|
default_headers=False,
|
||||||
timeout=2,
|
timeout=2,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -96,6 +98,43 @@ def _cache_key(query: str, pageno: int) -> str:
|
|||||||
return f"nextpage_url|{query}|{pageno}"
|
return f"nextpage_url|{query}|{pageno}"
|
||||||
|
|
||||||
|
|
||||||
|
def _solve_jsa(resp: "SXNG_Response") -> "SXNG_Response":
|
||||||
|
"""Duckduckgo sometimes issues a challenge instead of json."""
|
||||||
|
|
||||||
|
# length that a real browser would report for where the broken snippet is
|
||||||
|
html_len = {
|
||||||
|
"<p><div></p><p></div": 32,
|
||||||
|
"<li><div></li><li></div": 29,
|
||||||
|
"<div><div></div><div></div": 33,
|
||||||
|
"<br><div></br><br></div": 23,
|
||||||
|
}
|
||||||
|
|
||||||
|
js = resp.text or ""
|
||||||
|
jsa_match = re.search(r"let jsa = (\d+);.*?DDG\.deep\.initialize\('([^']+)'", js, re.S)
|
||||||
|
if not jsa_match:
|
||||||
|
return resp
|
||||||
|
|
||||||
|
js_functions = dict(re.findall(r"let (\w+) = function\(num\) \{([^}]*)\};", js))
|
||||||
|
jsa = int(jsa_match.group(1))
|
||||||
|
try:
|
||||||
|
for name in re.findall(r"jsa = (\w+)\(jsa\);", js):
|
||||||
|
body = js_functions[name]
|
||||||
|
mul = re.search(r"num \* (\d+)", body)
|
||||||
|
jsa = jsa * int(mul.group(1)) if mul else jsa + html_len[re.search(r"`([^`]+)`", body).group(1)]
|
||||||
|
except (KeyError, AttributeError):
|
||||||
|
return resp
|
||||||
|
|
||||||
|
params = resp.search_params
|
||||||
|
follow = get(
|
||||||
|
urljoin("https://links.duckduckgo.com", jsa_match.group(2) + str(jsa)),
|
||||||
|
headers=params["headers"],
|
||||||
|
impersonate="firefox",
|
||||||
|
default_headers=False,
|
||||||
|
)
|
||||||
|
follow.search_params = params
|
||||||
|
return follow
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams") -> None:
|
def request(query: str, params: "OnlineParams") -> None:
|
||||||
|
|
||||||
if len(query) >= 500:
|
if len(query) >= 500:
|
||||||
@@ -103,25 +142,15 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
params["url"] = None
|
params["url"] = None
|
||||||
return
|
return
|
||||||
|
|
||||||
headers = params["headers"]
|
# firefox TLS only
|
||||||
|
params["impersonate"] = "firefox"
|
||||||
# The vqd value is generated from the query and the UA header. To be able
|
params["default_headers"] = False
|
||||||
# to reuse the vqd value, the UA header must be static.
|
|
||||||
headers["User-Agent"] = _HTTP_User_Agent
|
|
||||||
headers["Accept"] = "*/*"
|
|
||||||
headers["Referer"] = f"{base_url}/"
|
|
||||||
headers["Host"] = "duckduckgo.com"
|
|
||||||
|
|
||||||
# Sec-Fetch headers are required to not get blocked when sending a Firefox user agent
|
|
||||||
headers["Sec-Fetch-Dest"] = "script"
|
|
||||||
headers["Sec-Fetch-Mode"] = "no-cors"
|
|
||||||
headers["Sec-Fetch-Site"] = "same-site"
|
|
||||||
|
|
||||||
api_url = ""
|
api_url = ""
|
||||||
if params["pageno"] > 1:
|
if params["pageno"] > 1:
|
||||||
api_url = CACHE.get(_cache_key(query, params["pageno"]))
|
api_url = CACHE.get(_cache_key(query, params["pageno"]))
|
||||||
else:
|
else:
|
||||||
api_url = _fetch_first_page_link(query, headers)
|
api_url = _fetch_first_page_link(query, params["headers"])
|
||||||
|
|
||||||
if not api_url:
|
if not api_url:
|
||||||
params["url"] = None
|
params["url"] = None
|
||||||
@@ -129,14 +158,27 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
|
|
||||||
params["url"] = api_url.replace("/d.js?", "/d.js?o=json&")
|
params["url"] = api_url.replace("/d.js?", "/d.js?o=json&")
|
||||||
|
|
||||||
|
# loads as a script
|
||||||
|
headers = params["headers"]
|
||||||
|
headers["Accept"] = "*/*"
|
||||||
|
headers["Sec-Fetch-Dest"] = "script"
|
||||||
|
headers["Sec-Fetch-Mode"] = "no-cors"
|
||||||
|
headers["Sec-Fetch-Site"] = "same-site"
|
||||||
|
headers["Referer"] = f"{base_url}/"
|
||||||
|
|
||||||
# TODO: support safesearch, timerange and engine traits # pylint:disable=fixme
|
# TODO: support safesearch, timerange and engine traits # pylint:disable=fixme
|
||||||
|
|
||||||
|
|
||||||
def response(resp: "SXNG_Response"):
|
def response(resp: "SXNG_Response"):
|
||||||
res = EngineResults()
|
res = EngineResults()
|
||||||
res_json = resp.json()
|
|
||||||
|
|
||||||
for result in res_json["results"]:
|
# check if ddg returns a challenge
|
||||||
|
# e.g. 'site:github.com searxng'
|
||||||
|
if "let jsa =" in (resp.text or ""):
|
||||||
|
resp = _solve_jsa(resp)
|
||||||
|
|
||||||
|
results = resp.json()["results"]
|
||||||
|
for result in results:
|
||||||
if "u" not in result:
|
if "u" not in result:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -144,8 +186,8 @@ def response(resp: "SXNG_Response"):
|
|||||||
res.types.MainResult(url=result["u"], title=html_to_text(result["t"]), content=html_to_text(result["a"]))
|
res.types.MainResult(url=result["u"], title=html_to_text(result["t"]), content=html_to_text(result["a"]))
|
||||||
)
|
)
|
||||||
|
|
||||||
# link to next page
|
if results:
|
||||||
next_page_path = res_json["results"][-1].get("n")
|
next_page_path = results[-1].get("n")
|
||||||
if next_page_path:
|
if next_page_path:
|
||||||
CACHE.set(
|
CACHE.set(
|
||||||
_cache_key(resp.search_params["query"], resp.search_params["pageno"] + 1),
|
_cache_key(resp.search_params["query"], resp.search_params["pageno"] + 1),
|
||||||
|
|||||||
@@ -2,7 +2,6 @@
|
|||||||
# pylint: disable=invalid-name
|
# pylint: disable=invalid-name
|
||||||
"""Dummy Offline"""
|
"""Dummy Offline"""
|
||||||
|
|
||||||
|
|
||||||
# about
|
# about
|
||||||
about = {
|
about = {
|
||||||
"wikidata_id": None,
|
"wikidata_id": None,
|
||||||
|
|||||||
150
searx/engines/europepmc.py
Normal file
150
searx/engines/europepmc.py
Normal file
@@ -0,0 +1,150 @@
|
|||||||
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
"""`Europe PMC`_ provides comprehensive access to life sciences literature from
|
||||||
|
trusted sources. With Europe PMC you can search and read millions of
|
||||||
|
publications, preprints and other documents enriched with links to supporting
|
||||||
|
data, reviews, protocols, and other relevant resources.
|
||||||
|
|
||||||
|
.. _Europe PMC: https://europepmc.org/
|
||||||
|
|
||||||
|
Configuration
|
||||||
|
=============
|
||||||
|
|
||||||
|
.. code:: yaml
|
||||||
|
|
||||||
|
- name: europepmc
|
||||||
|
engine: europepmc
|
||||||
|
shortcut: epmc
|
||||||
|
|
||||||
|
Implementations
|
||||||
|
===============
|
||||||
|
|
||||||
|
"""
|
||||||
|
|
||||||
|
import typing as t
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
|
from dateutil.parser import isoparse
|
||||||
|
|
||||||
|
from searx.enginelib import EngineCache
|
||||||
|
from searx.result_types import EngineResults
|
||||||
|
from searx.utils import html_to_text
|
||||||
|
|
||||||
|
if t.TYPE_CHECKING:
|
||||||
|
from searx.extended_types import SXNG_Response
|
||||||
|
from searx.search.processors import OnlineParams
|
||||||
|
|
||||||
|
|
||||||
|
about = {
|
||||||
|
"website": "https://europepmc.org/",
|
||||||
|
"wikidata_id": "Q5412157",
|
||||||
|
"official_api_documentation": "https://europepmc.org/RestfulWebService",
|
||||||
|
"use_official_api": True,
|
||||||
|
"require_api_key": False,
|
||||||
|
"results": "JSON",
|
||||||
|
}
|
||||||
|
|
||||||
|
categories = ["science", "scientific publications"]
|
||||||
|
paging = True
|
||||||
|
|
||||||
|
# engine dependent config
|
||||||
|
search_url = "https://www.ebi.ac.uk/europepmc/webservices/rest/search"
|
||||||
|
article_url = "https://europepmc.org/article/"
|
||||||
|
|
||||||
|
page_size = 20
|
||||||
|
|
||||||
|
CACHE: EngineCache
|
||||||
|
"""Cache for storing the pagination cursor."""
|
||||||
|
|
||||||
|
|
||||||
|
def setup(engine_settings: dict[str, t.Any]):
|
||||||
|
global CACHE # pylint: disable=global-statement
|
||||||
|
CACHE = EngineCache(engine_settings["name"])
|
||||||
|
|
||||||
|
|
||||||
|
def _cache_key(query: str, page: int) -> str:
|
||||||
|
return f"{query}|{page}"
|
||||||
|
|
||||||
|
|
||||||
|
def request(query: str, params: "OnlineParams") -> None:
|
||||||
|
args = {
|
||||||
|
"query": query,
|
||||||
|
"format": "json",
|
||||||
|
"resultType": "core",
|
||||||
|
"pageSize": page_size,
|
||||||
|
}
|
||||||
|
|
||||||
|
if params["pageno"] > 1:
|
||||||
|
if cursor := CACHE.get(_cache_key(query, params["pageno"])):
|
||||||
|
args["cursorMark"] = cursor
|
||||||
|
else:
|
||||||
|
# no cached cursor for that page
|
||||||
|
params["url"] = None
|
||||||
|
return
|
||||||
|
|
||||||
|
params["url"] = f"{search_url}?{urlencode(args)}"
|
||||||
|
|
||||||
|
|
||||||
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
|
res = EngineResults()
|
||||||
|
|
||||||
|
json_resp = resp.json()
|
||||||
|
|
||||||
|
# store pagination cursor for loading next pages in cache
|
||||||
|
if next_cursor := json_resp.get("nextCursorMark"):
|
||||||
|
next_page = resp.search_params["pageno"] + 1
|
||||||
|
query = resp.search_params["query"]
|
||||||
|
CACHE.set(_cache_key(query, next_page), next_cursor)
|
||||||
|
|
||||||
|
all_results = json_resp.get("resultList", {}).get("result", [])
|
||||||
|
|
||||||
|
for item in all_results:
|
||||||
|
source = item.get("source", "")
|
||||||
|
identifier = item.get("id", "")
|
||||||
|
url = f"{article_url}{source}/{identifier}" if source and identifier else ""
|
||||||
|
|
||||||
|
journal_info: dict[str, t.Any] = item.get("journalInfo", {})
|
||||||
|
journal: dict[str, t.Any] = journal_info.get("journal", {})
|
||||||
|
|
||||||
|
res.add(
|
||||||
|
res.types.Paper(
|
||||||
|
url=url,
|
||||||
|
title=html_to_text(item.get("title", "")),
|
||||||
|
content=html_to_text(item.get("abstractText", "")),
|
||||||
|
journal=journal.get("title", ""),
|
||||||
|
issn=[journal.get("issn", "")],
|
||||||
|
authors=_get_authors(item),
|
||||||
|
doi=item.get("doi", ""),
|
||||||
|
publishedDate=_get_published_date(item),
|
||||||
|
type=", ".join((item.get("pubTypeList", {})).get("pubType", [])),
|
||||||
|
pdf_url=_get_pdf_url(item),
|
||||||
|
html_url=url,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return res
|
||||||
|
|
||||||
|
|
||||||
|
def _get_authors(item: dict[str, t.Any]) -> list:
|
||||||
|
"""Extract the list of authors from the item."""
|
||||||
|
if authors := item.get("authorString", None):
|
||||||
|
authors = [author.strip().rstrip(".") for author in authors.split(",") if author.strip()]
|
||||||
|
else:
|
||||||
|
authors = []
|
||||||
|
return authors
|
||||||
|
|
||||||
|
|
||||||
|
def _get_pdf_url(item: dict[str, t.Any]) -> str:
|
||||||
|
"""Extract the PDF URL in case it is open access."""
|
||||||
|
for url_info in (item.get("fullTextUrlList", {})).get("fullTextUrl", []):
|
||||||
|
if url_info.get("documentStyle") == "pdf" and url_info.get("availabilityCode") == "OA":
|
||||||
|
return url_info.get("url", "")
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def _get_published_date(item: dict[str, t.Any]) -> datetime | None:
|
||||||
|
"""Extract the published date from the item and convert it to a datetime object."""
|
||||||
|
if unformatted_date := item.get("firstPublicationDate"):
|
||||||
|
return isoparse(unformatted_date)
|
||||||
|
return None
|
||||||
@@ -65,7 +65,6 @@ code lines are just relabeled (starting from 1) and appended (a disjoint set of
|
|||||||
code blocks in a single file might be returned from the API).
|
code blocks in a single file might be returned from the API).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
|
|||||||
@@ -327,6 +327,7 @@ def google_request(
|
|||||||
|
|
||||||
params["url"] = f"https://www.google.com/wml/search?{urlencode(args)}"
|
params["url"] = f"https://www.google.com/wml/search?{urlencode(args)}"
|
||||||
params["headers"]["User-Agent"] = random.choice(nokia_useragents)
|
params["headers"]["User-Agent"] = random.choice(nokia_useragents)
|
||||||
|
params["impersonate"] = "chrome99_android"
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams") -> None:
|
def request(query: str, params: "OnlineParams") -> None:
|
||||||
|
|||||||
@@ -30,6 +30,7 @@ about = {
|
|||||||
|
|
||||||
categories = ["general", "web"]
|
categories = ["general", "web"]
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
max_page = 5
|
max_page = 5
|
||||||
page_size = 20
|
page_size = 20
|
||||||
time_range_support = True
|
time_range_support = True
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ["images", "web"]
|
categories = ["images", "web"]
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
max_page = 50
|
max_page = 50
|
||||||
"""Google supports up to 50 pages of results, see the `Google max_page discussion`_.
|
"""Google supports up to 50 pages of results, see the `Google max_page discussion`_.
|
||||||
|
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ about = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
play_categ = None # apps|movies
|
play_categ = None # apps|movies
|
||||||
|
enable_http3 = True
|
||||||
base_url = 'https://play.google.com'
|
base_url = 'https://play.google.com'
|
||||||
search_url = base_url + "/store/search?{query}&c={play_categ}"
|
search_url = base_url + "/store/search?{query}&c={play_categ}"
|
||||||
|
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ import typing as t
|
|||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from lxml import html
|
from lxml import html
|
||||||
import httpx
|
from curl_cffi.requests.exceptions import TooManyRedirects
|
||||||
|
|
||||||
from searx.utils import (
|
from searx.utils import (
|
||||||
eval_xpath,
|
eval_xpath,
|
||||||
@@ -63,6 +63,7 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ["science", "scientific publications"]
|
categories = ["science", "scientific publications"]
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
max_page = 50
|
max_page = 50
|
||||||
"""`Google max 50 pages`_
|
"""`Google max 50 pages`_
|
||||||
|
|
||||||
@@ -102,7 +103,7 @@ def response(resp: "SXNG_Response") -> EngineResults: # pylint: disable=too-man
|
|||||||
raise SearxEngineAccessDeniedException(
|
raise SearxEngineAccessDeniedException(
|
||||||
message="google_scholar: unusual traffic detected",
|
message="google_scholar: unusual traffic detected",
|
||||||
)
|
)
|
||||||
raise httpx.TooManyRedirects(f"location {resp.headers['Location'].split('?')[0]}")
|
raise TooManyRedirects(f"location {resp.headers['Location'].split('?')[0]}")
|
||||||
|
|
||||||
res = EngineResults()
|
res = EngineResults()
|
||||||
dom = html.fromstring(resp.text)
|
dom = html.fromstring(resp.text)
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from dateutil import parser
|
from dateutil import parser
|
||||||
|
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
# pylint: disable=line-too-long
|
# pylint: disable=line-too-long
|
||||||
"website": "https://hex.pm/",
|
"website": "https://hex.pm/",
|
||||||
|
|||||||
@@ -108,14 +108,12 @@ def get_infobox(alt_forms, result_url, definitions):
|
|||||||
infobox_content.append(f'<p><i>Other forms:</i> {", ".join(alt_forms[1:])}</p>')
|
infobox_content.append(f'<p><i>Other forms:</i> {", ".join(alt_forms[1:])}</p>')
|
||||||
|
|
||||||
# definitions
|
# definitions
|
||||||
infobox_content.append(
|
infobox_content.append('''
|
||||||
'''
|
|
||||||
<small><a href="https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project">JMdict</a>
|
<small><a href="https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project">JMdict</a>
|
||||||
and <a href="https://www.edrdg.org/enamdict/enamdict_doc.html">JMnedict</a>
|
and <a href="https://www.edrdg.org/enamdict/enamdict_doc.html">JMnedict</a>
|
||||||
by <a href="https://www.edrdg.org/edrdg/licence.html">EDRDG</a>, CC BY-SA 3.0.</small>
|
by <a href="https://www.edrdg.org/edrdg/licence.html">EDRDG</a>, CC BY-SA 3.0.</small>
|
||||||
<ul>
|
<ul>
|
||||||
'''
|
''')
|
||||||
)
|
|
||||||
for pos, engdef, extra in definitions:
|
for pos, engdef, extra in definitions:
|
||||||
if pos == 'Wikipedia definition':
|
if pos == 'Wikipedia definition':
|
||||||
infobox_content.append('</ul><small>Wikipedia, CC BY-SA 3.0.</small><ul>')
|
infobox_content.append('</ul><small>Wikipedia, CC BY-SA 3.0.</small><ul>')
|
||||||
|
|||||||
@@ -37,17 +37,13 @@ about = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
categories = []
|
categories = []
|
||||||
safeseach = True
|
safesearch = True
|
||||||
|
|
||||||
base_url = "https://luxxle.com"
|
base_url = "https://luxxle.com"
|
||||||
|
|
||||||
luxxle_categ = "search"
|
luxxle_categ = "search"
|
||||||
"""Supported categories: "search", "news", "images", "videos"."""
|
"""Supported categories: "search", "news", "images", "videos"."""
|
||||||
|
|
||||||
# otherwise all requests get blocked (http2-fingerprinted probably)
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
|
|
||||||
safe_search_map = {0: "Off", 1: "Moderate", 2: "Strict"}
|
safe_search_map = {0: "Off", 1: "Moderate", 2: "Strict"}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ Lofgren .
|
|||||||
.. _marginalia filters:
|
.. _marginalia filters:
|
||||||
|
|
||||||
Marginalia Filters
|
Marginalia Filters
|
||||||
=================
|
==================
|
||||||
|
|
||||||
Custom filters enable server-side customization of Marginalia search results.
|
Custom filters enable server-side customization of Marginalia search results.
|
||||||
Filter definitions are written in XML and scoped to an API key. Filters can
|
Filter definitions are written in XML and scoped to an API key. Filters can
|
||||||
@@ -82,7 +82,7 @@ api_key = None
|
|||||||
https://about.marginalia-search.com/article/api/
|
https://about.marginalia-search.com/article/api/
|
||||||
|
|
||||||
"""
|
"""
|
||||||
filter_name: str | None = None
|
filter_name: str = ""
|
||||||
"""The name of the custom filter to apply to each search."""
|
"""The name of the custom filter to apply to each search."""
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -49,7 +49,6 @@ except ImportError:
|
|||||||
|
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
|
|
||||||
engine_type = 'offline'
|
engine_type = 'offline'
|
||||||
|
|
||||||
# mongodb connection variables
|
# mongodb connection variables
|
||||||
|
|||||||
@@ -6,10 +6,14 @@
|
|||||||
|
|
||||||
from json import loads
|
from json import loads
|
||||||
import typing as t
|
import typing as t
|
||||||
from urllib.parse import urlencode
|
|
||||||
|
|
||||||
|
from lxml import html
|
||||||
|
|
||||||
|
from searx.exceptions import SearxEngineAPIException
|
||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
|
from searx.network import get
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
from searx.utils import eval_xpath, extract_text
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
if t.TYPE_CHECKING:
|
||||||
from searx.enginelib.traits import EngineTraits
|
from searx.enginelib.traits import EngineTraits
|
||||||
@@ -25,18 +29,33 @@ about = {
|
|||||||
"results": "JSON",
|
"results": "JSON",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
paging = False
|
||||||
|
enable_http3 = True
|
||||||
|
|
||||||
base_url = "https://neosearch.org"
|
base_url = "https://neosearch.org"
|
||||||
categories = ["general"]
|
categories = ["general"]
|
||||||
|
|
||||||
paging = False
|
|
||||||
|
def _obtain_xsrf_token() -> str:
|
||||||
|
resp = get(base_url)
|
||||||
|
doc = html.fromstring(resp.text)
|
||||||
|
|
||||||
|
xsrf_token = extract_text(eval_xpath(doc, "//meta[@name='xsrf-token']/@content"))
|
||||||
|
if not xsrf_token:
|
||||||
|
raise SearxEngineAPIException("failed to obtain xsrf token")
|
||||||
|
return xsrf_token
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams"):
|
def request(query: str, params: "OnlineParams"):
|
||||||
|
params["url"] = f"{base_url}/search"
|
||||||
|
params["headers"]["X-XSRF-TOKEN"] = _obtain_xsrf_token()
|
||||||
|
params["method"] = "POST"
|
||||||
|
|
||||||
args = {"q": query, "generate": "auto"}
|
args = {"q": query, "generate": "auto"}
|
||||||
countrycode = params["searxng_locale"].split("-")[-1].upper()
|
countrycode = params["searxng_locale"].split("-")[-1].upper()
|
||||||
if countrycode in traits.custom["countrycodes"]:
|
if countrycode in traits.custom["countrycodes"]:
|
||||||
args["loc"] = countrycode
|
args["loc"] = countrycode
|
||||||
params["url"] = f"{base_url}/search?{urlencode(args)}"
|
params["json"] = args
|
||||||
|
|
||||||
|
|
||||||
def response(resp: "SXNG_Response") -> EngineResults:
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
@@ -67,7 +86,6 @@ def response(resp: "SXNG_Response") -> EngineResults:
|
|||||||
|
|
||||||
def fetch_traits(engine_traits: "EngineTraits") -> None:
|
def fetch_traits(engine_traits: "EngineTraits") -> None:
|
||||||
# pylint: disable=import-outside-toplevel
|
# pylint: disable=import-outside-toplevel
|
||||||
from searx.network import get
|
|
||||||
from searx.utils import extr, js_obj_str_to_python
|
from searx.utils import extr, js_obj_str_to_python
|
||||||
from babel.core import get_global
|
from babel.core import get_global
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from dateutil import parser
|
from dateutil import parser
|
||||||
|
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
"website": "https://npms.io/",
|
"website": "https://npms.io/",
|
||||||
"wikidata_id": "Q7067518",
|
"wikidata_id": "Q7067518",
|
||||||
|
|||||||
@@ -9,7 +9,6 @@ from datetime import datetime
|
|||||||
from searx.result_types import EngineResults, WeatherAnswer
|
from searx.result_types import EngineResults, WeatherAnswer
|
||||||
from searx import weather
|
from searx import weather
|
||||||
|
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
"website": "https://open-meteo.com",
|
"website": "https://open-meteo.com",
|
||||||
"wikidata_id": None,
|
"wikidata_id": None,
|
||||||
|
|||||||
@@ -8,7 +8,6 @@ Openverse (formerly known as: Creative Commons search engine) [Images]
|
|||||||
from json import loads
|
from json import loads
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
"website": 'https://openverse.org/',
|
"website": 'https://openverse.org/',
|
||||||
"wikidata_id": None,
|
"wikidata_id": None,
|
||||||
|
|||||||
@@ -8,12 +8,11 @@ from urllib.parse import urlencode
|
|||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.utils import eval_xpath_list, gen_useragent
|
from searx.utils import eval_xpath_list
|
||||||
from searx.enginelib import EngineCache
|
from searx.enginelib import EngineCache
|
||||||
from searx.exceptions import SearxEngineAPIException, SearxEngineAccessDeniedException
|
from searx.exceptions import SearxEngineAPIException, SearxEngineAccessDeniedException
|
||||||
from searx.network import get
|
from searx.network import get
|
||||||
|
|
||||||
|
|
||||||
# about
|
# about
|
||||||
about = {
|
about = {
|
||||||
"website": 'https://www.pexels.com',
|
"website": 'https://www.pexels.com',
|
||||||
@@ -44,8 +43,6 @@ SECRET_KEY_DB_KEY = "secret-key"
|
|||||||
CACHE: EngineCache
|
CACHE: EngineCache
|
||||||
"""Cache to store the secret API key for the engine."""
|
"""Cache to store the secret API key for the engine."""
|
||||||
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
|
|
||||||
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
||||||
global CACHE # pylint: disable=global-statement
|
global CACHE # pylint: disable=global-statement
|
||||||
@@ -56,13 +53,7 @@ def setup(engine_settings: dict[str, t.Any]) -> bool:
|
|||||||
def _get_secret_key():
|
def _get_secret_key():
|
||||||
resp = get(
|
resp = get(
|
||||||
base_url,
|
base_url,
|
||||||
headers={
|
headers={"Referer": base_url},
|
||||||
# circumvents Cloudflare bot protections
|
|
||||||
"User-Agent": gen_useragent(),
|
|
||||||
"Referer": base_url,
|
|
||||||
"Sec-GPC": "1",
|
|
||||||
"Connection": "keep-alive",
|
|
||||||
},
|
|
||||||
)
|
)
|
||||||
|
|
||||||
if resp.status_code != 200:
|
if resp.status_code != 200:
|
||||||
@@ -105,8 +96,6 @@ def request(query, params):
|
|||||||
|
|
||||||
params["headers"]["secret-key"] = secret_key
|
params["headers"]["secret-key"] = secret_key
|
||||||
|
|
||||||
return params
|
|
||||||
|
|
||||||
|
|
||||||
def response(resp):
|
def response(resp):
|
||||||
res = EngineResults()
|
res = EngineResults()
|
||||||
|
|||||||
@@ -2,72 +2,85 @@
|
|||||||
"""Pinterest (images)"""
|
"""Pinterest (images)"""
|
||||||
|
|
||||||
from json import dumps
|
from json import dumps
|
||||||
|
import typing as t
|
||||||
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
|
if t.TYPE_CHECKING:
|
||||||
|
from searx.extended_types import SXNG_Response
|
||||||
|
from searx.search.processors import OnlineParams
|
||||||
|
|
||||||
about = {
|
about = {
|
||||||
"website": 'https://www.pinterest.com/',
|
"website": "https://www.pinterest.com/",
|
||||||
"wikidata_id": 'Q255381',
|
"wikidata_id": "Q255381",
|
||||||
"official_api_documentation": 'https://developers.pinterest.com/docs/api/v5/',
|
"official_api_documentation": "https://developers.pinterest.com/docs/api/v5/",
|
||||||
"use_official_api": False,
|
"use_official_api": False,
|
||||||
"require_api_key": False,
|
"require_api_key": False,
|
||||||
"results": 'JSON',
|
"results": "JSON",
|
||||||
}
|
}
|
||||||
|
|
||||||
categories = ['images']
|
categories = ["images"]
|
||||||
paging = True
|
paging = True
|
||||||
|
|
||||||
base_url = 'https://www.pinterest.com'
|
base_url = "https://www.pinterest.com"
|
||||||
|
|
||||||
|
|
||||||
def request(query, params):
|
def request(query: str, params: "OnlineParams") -> None:
|
||||||
|
|
||||||
args = {
|
args = {
|
||||||
'options': {
|
"options": {
|
||||||
'query': query,
|
"query": query,
|
||||||
'bookmarks': [params['engine_data'].get('bookmark', '')],
|
"bookmarks": [params["engine_data"].get("bookmark", "")],
|
||||||
},
|
},
|
||||||
'context': {},
|
"context": {},
|
||||||
}
|
}
|
||||||
params['url'] = f"{base_url}/resource/BaseSearchResource/get/?data={dumps(args)}"
|
params["url"] = f"{base_url}/resource/BaseSearchResource/get/?data={dumps(args)}"
|
||||||
params['headers'] = {
|
params["headers"] = {
|
||||||
'X-Pinterest-AppState': 'active',
|
"X-Requested-With": "XMLHttpRequest",
|
||||||
'X-Pinterest-Source-Url': '/ideas/',
|
"X-Pinterest-AppState": "active",
|
||||||
'X-Pinterest-PWS-Handler': 'www/ideas.js',
|
"X-Pinterest-Source-Url": "/ideas/",
|
||||||
|
"X-Pinterest-PWS-Handler": "www/ideas.js",
|
||||||
}
|
}
|
||||||
|
|
||||||
return params
|
|
||||||
|
|
||||||
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
|
res = EngineResults()
|
||||||
|
json_resp: dict[str, t.Any] = resp.json() # type: ignore
|
||||||
|
|
||||||
def response(resp):
|
res.add(
|
||||||
results = []
|
|
||||||
|
|
||||||
json_resp = resp.json()
|
|
||||||
|
|
||||||
results.append(
|
|
||||||
{
|
{
|
||||||
'engine_data': json_resp['resource_response']['bookmark'],
|
"engine_data": json_resp["resource_response"]["bookmark"],
|
||||||
# it's called bookmark by pinterest, but it's rather a nextpage
|
# it's called bookmark by pinterest, but it's rather a nextpage
|
||||||
# parameter to get the next results
|
# parameter to get the next results
|
||||||
'key': 'bookmark',
|
"key": "bookmark",
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
for result in json_resp['resource_response']['data']['results']:
|
for result in json_resp["resource_response"]["data"]["results"]:
|
||||||
|
|
||||||
if result['type'] == 'story':
|
if result["type"] == "story":
|
||||||
continue
|
continue
|
||||||
|
|
||||||
main_image = result['images']['orig']
|
main_image = result["images"]["orig"]
|
||||||
results.append(
|
|
||||||
{
|
title = result.get("title") or result.get("grid_title") or ""
|
||||||
'template': 'images.html',
|
if len(title) < 5:
|
||||||
'url': result.get('link') or f"{base_url}/pin/{result['id']}/",
|
visual_annotation = result.get("pin_join", {}).get("visual_annotation")
|
||||||
'title': result.get('title') or result.get('grid_title'),
|
if visual_annotation:
|
||||||
'content': (result.get('rich_summary') or {}).get('display_description') or "",
|
title = visual_annotation[0]
|
||||||
'img_src': main_image['url'],
|
else:
|
||||||
'thumbnail_src': result['images']['236x']['url'],
|
title = result.get("name") or result.get("auto_alt_text") or ""
|
||||||
'source': (result.get('rich_summary') or {}).get('site_name'),
|
|
||||||
'resolution': f"{main_image['width']}x{main_image['height']}",
|
res.add(
|
||||||
'author': f"{result['pinner'].get('full_name')} ({result['pinner']['username']})",
|
res.types.Image(
|
||||||
}
|
url=result.get("link") or f"{base_url}/pin/{result['id']}/",
|
||||||
|
title=title,
|
||||||
|
content=(result.get("rich_summary") or {}).get("display_description") or "",
|
||||||
|
img_src=main_image["url"],
|
||||||
|
thumbnail_src=result["images"]["236x"]["url"],
|
||||||
|
source=(result.get("rich_summary") or {}).get("site_name") or "",
|
||||||
|
resolution=f"{main_image['width']}x{main_image['height']}",
|
||||||
|
author=f"{result['pinner'].get('full_name')} ({result['pinner']['username']})",
|
||||||
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
return results
|
return res
|
||||||
|
|||||||
@@ -48,7 +48,6 @@ Implementations
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
import time
|
import time
|
||||||
import random
|
import random
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
|||||||
@@ -27,9 +27,6 @@ time_range_support = True
|
|||||||
safesearch_map = {0: 'off', 1: '1', 2: '1'}
|
safesearch_map = {0: 'off', 1: '1', 2: '1'}
|
||||||
time_range_map = {'day': '1d', 'week': '1w', 'month': '1m', 'year': '1y'}
|
time_range_map = {'day': '1d', 'week': '1w', 'month': '1m', 'year': '1y'}
|
||||||
|
|
||||||
# using http2 returns forbidden errors
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
|
|
||||||
def request(query, params):
|
def request(query, params):
|
||||||
args = {
|
args = {
|
||||||
@@ -50,8 +47,6 @@ def request(query, params):
|
|||||||
# prevent automatic redirects to first page on pagination
|
# prevent automatic redirects to first page on pagination
|
||||||
params['allow_redirects'] = False
|
params['allow_redirects'] = False
|
||||||
|
|
||||||
return params
|
|
||||||
|
|
||||||
|
|
||||||
def _image_result(result):
|
def _image_result(result):
|
||||||
return {
|
return {
|
||||||
|
|||||||
@@ -18,7 +18,6 @@ from searx.utils import eval_xpath_list, eval_xpath, extract_text, extr
|
|||||||
from searx.locales import region_tag
|
from searx.locales import region_tag
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
if t.TYPE_CHECKING:
|
||||||
from lxml.etree import ElementBase
|
from lxml.etree import ElementBase
|
||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ Implementations
|
|||||||
===============
|
===============
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
|
||||||
from datetime import date, timedelta
|
from datetime import date, timedelta
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ from urllib.parse import urlencode
|
|||||||
from lxml import html
|
from lxml import html
|
||||||
|
|
||||||
from searx import locales
|
from searx import locales
|
||||||
|
from searx.exceptions import SearxEngineResponseException
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.utils import eval_xpath_list, eval_xpath, extract_text
|
from searx.utils import eval_xpath_list, eval_xpath, extract_text
|
||||||
|
|
||||||
@@ -52,6 +53,7 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
"q": query,
|
"q": query,
|
||||||
"search_type": resulthunter_categ,
|
"search_type": resulthunter_categ,
|
||||||
"offset": params["pageno"] - 1,
|
"offset": params["pageno"] - 1,
|
||||||
|
"search_source": "other",
|
||||||
}
|
}
|
||||||
|
|
||||||
# uses Brave's engine traits
|
# uses Brave's engine traits
|
||||||
@@ -111,6 +113,11 @@ def _image_results(doc: "ElementBase") -> EngineResults:
|
|||||||
def response(resp: "SXNG_Response") -> EngineResults:
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
doc = html.fromstring(resp.text)
|
doc = html.fromstring(resp.text)
|
||||||
|
|
||||||
|
# if the request was wrong (e.g. missing params), the site doesn't contain a result container
|
||||||
|
# and instead shows an "Installation required" page to download the resulthunter browser extension
|
||||||
|
if not eval_xpath(doc, "//div[contains(@class, 'organic-results-container')]"):
|
||||||
|
raise SearxEngineResponseException()
|
||||||
|
|
||||||
match resulthunter_categ:
|
match resulthunter_categ:
|
||||||
case "web":
|
case "web":
|
||||||
return _general_results(doc)
|
return _general_results(doc)
|
||||||
|
|||||||
63
searx/engines/s1search_rampjs.py
Normal file
63
searx/engines/s1search_rampjs.py
Normal file
@@ -0,0 +1,63 @@
|
|||||||
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
"""JavaScript-based s1search implementation. See :ref:`s1search engine`.
|
||||||
|
|
||||||
|
Works for all s1search sites that contain the ``__RAMPJS__`` JavaScript variable.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import typing as t
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
|
||||||
|
from searx.result_types import EngineResults
|
||||||
|
from searx.utils import extr, html_to_text
|
||||||
|
|
||||||
|
if t.TYPE_CHECKING:
|
||||||
|
from searx.search.processors import OnlineParams
|
||||||
|
from searx.extended_types import SXNG_Response
|
||||||
|
|
||||||
|
about = {
|
||||||
|
"website": "https://s1search.co",
|
||||||
|
"official_api_documentation": None,
|
||||||
|
"use_official_api": False,
|
||||||
|
"require_api_key": False,
|
||||||
|
"results": "JSON",
|
||||||
|
}
|
||||||
|
|
||||||
|
categories = ["general"]
|
||||||
|
paging = True
|
||||||
|
|
||||||
|
base_url = "https://search.answers.com"
|
||||||
|
# other working base URLs:
|
||||||
|
# - https://search.nation.online
|
||||||
|
# - https://search.activebeat.com
|
||||||
|
# - https://search.legalboulevard.com
|
||||||
|
# - https://search.walletgenius.com
|
||||||
|
# - https://search.legalboulevard.com
|
||||||
|
|
||||||
|
|
||||||
|
def request(query: str, params: "OnlineParams"):
|
||||||
|
args = {"q": query, "page": params["pageno"]}
|
||||||
|
params["url"] = f"{base_url}/?{urlencode(args)}"
|
||||||
|
|
||||||
|
|
||||||
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
|
res = EngineResults()
|
||||||
|
|
||||||
|
data_raw = extr(resp.text, "response: ", " };")
|
||||||
|
data = json.loads(data_raw)
|
||||||
|
|
||||||
|
mainline = [s for s in data["search"]["regions"] if s["name"] == "mainline"][0]
|
||||||
|
for group in mainline["groups"]:
|
||||||
|
for result in group["results"]:
|
||||||
|
if not ("url" in result or "clickUrl" in result):
|
||||||
|
continue
|
||||||
|
|
||||||
|
res.add(
|
||||||
|
res.types.MainResult(
|
||||||
|
url=result.get("url") or result.get("clickUrl"),
|
||||||
|
title=html_to_text(result["title"]),
|
||||||
|
content=html_to_text(result["description"]),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return res
|
||||||
88
searx/engines/searchrockit.py
Normal file
88
searx/engines/searchrockit.py
Normal file
@@ -0,0 +1,88 @@
|
|||||||
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
"""SearchRockit is an American search engine. It allegedly has its own index,
|
||||||
|
but the results seem to come from Google."""
|
||||||
|
|
||||||
|
import typing as t
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
from lxml import html
|
||||||
|
from dateutil import parser
|
||||||
|
|
||||||
|
from searx.result_types import EngineResults
|
||||||
|
from searx.utils import (
|
||||||
|
eval_xpath_list,
|
||||||
|
extract_text,
|
||||||
|
eval_xpath,
|
||||||
|
)
|
||||||
|
|
||||||
|
if t.TYPE_CHECKING:
|
||||||
|
from searx.search.processors import OnlineParams
|
||||||
|
from searx.extended_types import SXNG_Response
|
||||||
|
|
||||||
|
about = {
|
||||||
|
"website": "https://searchrockit.com",
|
||||||
|
"official_api_documentation": None,
|
||||||
|
"use_official_api": False,
|
||||||
|
"require_api_key": False,
|
||||||
|
"results": "HTML",
|
||||||
|
}
|
||||||
|
|
||||||
|
categories = ["general"]
|
||||||
|
paging = True
|
||||||
|
|
||||||
|
SearchrockitCateg = t.Literal["web", "images", "news"]
|
||||||
|
searchrockit_categ: SearchrockitCateg = "web"
|
||||||
|
|
||||||
|
base_url = "https://searchrockit.com"
|
||||||
|
|
||||||
|
|
||||||
|
def setup(_):
|
||||||
|
if searchrockit_categ not in t.get_args(SearchrockitCateg):
|
||||||
|
raise ValueError("invalid search category: %s" % searchrockit_categ)
|
||||||
|
|
||||||
|
|
||||||
|
def request(query: str, params: "OnlineParams") -> None:
|
||||||
|
args = {"q": query, "p": params["pageno"]}
|
||||||
|
params["url"] = f"{base_url}/results/{searchrockit_categ}?{urlencode(args)}"
|
||||||
|
|
||||||
|
|
||||||
|
def response(resp: "SXNG_Response") -> EngineResults:
|
||||||
|
doc = html.fromstring(resp.text)
|
||||||
|
res = EngineResults()
|
||||||
|
|
||||||
|
match searchrockit_categ:
|
||||||
|
case "web" | "news":
|
||||||
|
for result in eval_xpath_list(
|
||||||
|
doc, "//div[contains(@class, 'results-list')]/div[contains(@class, 'result-item')]"
|
||||||
|
):
|
||||||
|
publishedDate = None
|
||||||
|
try:
|
||||||
|
d = extract_text(eval_xpath(result, ".//span[contains(@class, 'result-item--publishedAt')]")) or ""
|
||||||
|
publishedDate = parser.parse(d)
|
||||||
|
except parser.ParserError:
|
||||||
|
pass
|
||||||
|
res.add(
|
||||||
|
res.types.MainResult(
|
||||||
|
url=extract_text(eval_xpath(result, ".//a[contains(@class, 'result-item--title')]/@href")),
|
||||||
|
title=extract_text(eval_xpath(result, ".//a[contains(@class, 'result-item--title')]")) or "",
|
||||||
|
content=extract_text(eval_xpath(result, ".//a[contains(@class, 'result-item--desc')]")) or "",
|
||||||
|
thumbnail=extract_text(
|
||||||
|
eval_xpath(result, ".//a[contains(@class, 'result-item--thumb')]/img/@src")
|
||||||
|
)
|
||||||
|
or "",
|
||||||
|
publishedDate=publishedDate,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
case "images":
|
||||||
|
for result in eval_xpath_list(
|
||||||
|
doc, "//div[contains(@class, 'image-grid')]/a[contains(@class, 'image-card')]"
|
||||||
|
):
|
||||||
|
res.add(
|
||||||
|
res.types.Image(
|
||||||
|
url=extract_text(eval_xpath(result, "./@href")),
|
||||||
|
title=extract_text(eval_xpath(result, "./div[contains(@class, 'image-title')]")) or "",
|
||||||
|
thumbnail_src=extract_text(eval_xpath(result, "./img/@src")) or "",
|
||||||
|
img_src=extract_text(eval_xpath(result, "./@data-full-url")) or "",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return res
|
||||||
@@ -4,13 +4,11 @@ independent search infrastructure."""
|
|||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
|
import uuid
|
||||||
|
|
||||||
from searx.exceptions import SearxEngineAPIException
|
|
||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
from searx.network import get
|
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.utils import extr, html_to_text
|
from searx.utils import html_to_text
|
||||||
from searx.enginelib import EngineCache
|
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
if t.TYPE_CHECKING:
|
||||||
from searx.search.processors import OnlineParams
|
from searx.search.processors import OnlineParams
|
||||||
@@ -34,43 +32,19 @@ SearchzeeCategType = t.Literal["web", "news"]
|
|||||||
searchzee_categ: SearchzeeCategType = None # type: ignore[reportAssignmentType]
|
searchzee_categ: SearchzeeCategType = None # type: ignore[reportAssignmentType]
|
||||||
|
|
||||||
|
|
||||||
CACHE: EngineCache
|
|
||||||
"""Cache for storing the scraped API Token."""
|
|
||||||
|
|
||||||
base_url = "https://searchzee.com"
|
base_url = "https://searchzee.com"
|
||||||
|
|
||||||
# only supports for news
|
# only supports for news
|
||||||
time_range_map = {"day": "pd", "week": "pw", "month": "pm", "year": "py"}
|
time_range_map = {"day": "pd", "week": "pw", "month": "pm", "year": "py"}
|
||||||
|
|
||||||
|
|
||||||
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
def setup(_: dict[str, t.Any]):
|
||||||
if searchzee_categ not in t.get_args(SearchzeeCategType):
|
if searchzee_categ not in t.get_args(SearchzeeCategType):
|
||||||
raise ValueError("invalid category: %s" % searchzee_categ)
|
raise ValueError("invalid category: %s" % searchzee_categ)
|
||||||
|
|
||||||
global CACHE # pylint: disable=global-statement
|
|
||||||
CACHE = EngineCache(engine_settings["name"]) # type: ignore[reportAny]
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def _obtain_api_token() -> str:
|
|
||||||
token: str | None = CACHE.get("token") # type: ignore[reportAny]
|
|
||||||
if token:
|
|
||||||
return token
|
|
||||||
|
|
||||||
token_resp = get(
|
|
||||||
f"{base_url}/app.js",
|
|
||||||
)
|
|
||||||
if not token_resp.ok:
|
|
||||||
raise SearxEngineAPIException("failed to obtain api key")
|
|
||||||
|
|
||||||
token = extr(token_resp.text, "const SEARCHZEE_API_TOKEN = \"", "\";")
|
|
||||||
CACHE.set("token", token, expire=3600)
|
|
||||||
|
|
||||||
return token
|
|
||||||
|
|
||||||
|
|
||||||
def request(query: str, params: "OnlineParams"):
|
def request(query: str, params: "OnlineParams"):
|
||||||
params["headers"]["X-SearchZee-Token"] = _obtain_api_token()
|
params["cookies"]["szs"] = str(uuid.uuid4())
|
||||||
|
|
||||||
args = {"q": query, "type": searchzee_categ, "offset": params["pageno"] - 1}
|
args = {"q": query, "type": searchzee_categ, "offset": params["pageno"] - 1}
|
||||||
if params["time_range"]:
|
if params["time_range"]:
|
||||||
|
|||||||
@@ -1,57 +0,0 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
||||||
"""
|
|
||||||
Searx (all)
|
|
||||||
"""
|
|
||||||
|
|
||||||
from json import loads
|
|
||||||
from searx.engines import categories as searx_categories
|
|
||||||
|
|
||||||
# about
|
|
||||||
about = {
|
|
||||||
"website": 'https://github.com/searxng/searxng',
|
|
||||||
"wikidata_id": 'Q17639196',
|
|
||||||
"official_api_documentation": 'https://docs.searxng.org/dev/search_api.html',
|
|
||||||
"use_official_api": True,
|
|
||||||
"require_api_key": False,
|
|
||||||
"results": 'JSON',
|
|
||||||
}
|
|
||||||
|
|
||||||
categories = searx_categories.keys()
|
|
||||||
|
|
||||||
# search-url
|
|
||||||
instance_urls = []
|
|
||||||
instance_index = 0
|
|
||||||
|
|
||||||
|
|
||||||
# do search-request
|
|
||||||
def request(query, params):
|
|
||||||
global instance_index # pylint: disable=global-statement
|
|
||||||
params['url'] = instance_urls[instance_index % len(instance_urls)]
|
|
||||||
params['method'] = 'POST'
|
|
||||||
|
|
||||||
instance_index += 1
|
|
||||||
|
|
||||||
params['data'] = {
|
|
||||||
'q': query,
|
|
||||||
'pageno': params['pageno'],
|
|
||||||
'language': params['language'],
|
|
||||||
'time_range': params['time_range'],
|
|
||||||
'category': params['category'],
|
|
||||||
'format': 'json',
|
|
||||||
}
|
|
||||||
|
|
||||||
return params
|
|
||||||
|
|
||||||
|
|
||||||
# get response from search-request
|
|
||||||
def response(resp):
|
|
||||||
|
|
||||||
response_json = loads(resp.text)
|
|
||||||
results = response_json['results']
|
|
||||||
|
|
||||||
for i in ('answers', 'infoboxes'):
|
|
||||||
results.extend(response_json[i])
|
|
||||||
|
|
||||||
results.extend({'suggestion': s} for s in response_json['suggestions'])
|
|
||||||
|
|
||||||
return results
|
|
||||||
@@ -34,7 +34,6 @@ from searx.exceptions import SearxEngineAPIException
|
|||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
|
|
||||||
|
|
||||||
base_url = 'http://localhost:8983'
|
base_url = 'http://localhost:8983'
|
||||||
collection = ''
|
collection = ''
|
||||||
rows = 10
|
rows = 10
|
||||||
|
|||||||
@@ -118,7 +118,7 @@ def response(resp):
|
|||||||
|
|
||||||
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
def setup(engine_settings: dict[str, t.Any]) -> bool:
|
||||||
global CACHE # pylint: disable=global-statement
|
global CACHE # pylint: disable=global-statement
|
||||||
CACHE = EngineCache(engine_settings["name"]) # type:ignore
|
CACHE = EngineCache(engine_settings["name"]) # type: ignore
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -44,6 +44,7 @@ Implementations
|
|||||||
===============
|
===============
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
import sqlite3
|
import sqlite3
|
||||||
import contextlib
|
import contextlib
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
"""Startpage's language & region selectors are a mess ..
|
"""Startpage requires solving an Anubis POW captcha (difficulty 4).
|
||||||
|
Solving it requires a lot of CPU, so the engine is set inactive by default.
|
||||||
|
|
||||||
|
Startpage's language & region selectors are a mess ..
|
||||||
|
|
||||||
.. _startpage regions:
|
.. _startpage regions:
|
||||||
|
|
||||||
@@ -82,8 +85,10 @@ Startpage's category (for Web-search, News, Videos, ..) is set by
|
|||||||
Supported categories are ``web``, ``news`` and ``images``.
|
Supported categories are ``web``, ``news`` and ``images``.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=too-many-statements
|
# pylint: disable=too-many-statements
|
||||||
|
|
||||||
|
import hashlib
|
||||||
import re
|
import re
|
||||||
import typing as t
|
import typing as t
|
||||||
from collections import OrderedDict
|
from collections import OrderedDict
|
||||||
@@ -99,7 +104,7 @@ from searx.enginelib import EngineCache
|
|||||||
from searx.enginelib.traits import EngineTraits
|
from searx.enginelib.traits import EngineTraits
|
||||||
from searx.exceptions import SearxEngineCaptchaException
|
from searx.exceptions import SearxEngineCaptchaException
|
||||||
from searx.locales import region_tag
|
from searx.locales import region_tag
|
||||||
from searx.network import get # see https://github.com/searxng/searxng/issues/762
|
from searx.network import get, post # see https://github.com/searxng/searxng/issues/762
|
||||||
from searx.utils import (
|
from searx.utils import (
|
||||||
eval_xpath,
|
eval_xpath,
|
||||||
extr,
|
extr,
|
||||||
@@ -176,6 +181,45 @@ def setup(_: dict[str, t.Any]) -> bool | None:
|
|||||||
sc_code_cache_sec = 3600
|
sc_code_cache_sec = 3600
|
||||||
"""Time in seconds the sc-code is cached in memory :py:obj:`get_sc_code`."""
|
"""Time in seconds the sc-code is cached in memory :py:obj:`get_sc_code`."""
|
||||||
|
|
||||||
|
# startpage's anubis difficulty is set to 4
|
||||||
|
max_difficulty = 4
|
||||||
|
|
||||||
|
|
||||||
|
def _solve_anubis(resp) -> str:
|
||||||
|
"""Anubis POW solver"""
|
||||||
|
payload = loads(extr(resp.text, '<script id="anubis_challenge" type="application/json">', "</script>"))
|
||||||
|
challenge = payload["challenge"]
|
||||||
|
difficulty = int(payload["rules"]["difficulty"])
|
||||||
|
if difficulty > max_difficulty:
|
||||||
|
raise SearxEngineCaptchaException(message="startpage: Anubis difficulty too high")
|
||||||
|
prefix = "0" * difficulty
|
||||||
|
blob = challenge["randomData"].encode()
|
||||||
|
for nonce in range(16**difficulty * 8): # max search is 8x average search, e^-8 = 0.034% will fail
|
||||||
|
digest = hashlib.sha256(blob + str(nonce).encode()).hexdigest()
|
||||||
|
if digest.startswith(prefix):
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
raise SearxEngineCaptchaException(message="startpage: Anubis failed")
|
||||||
|
|
||||||
|
pass_resp = get(
|
||||||
|
f"{base_url}/.within.website/x/cmd/anubis/api/pass-challenge",
|
||||||
|
params={
|
||||||
|
"id": challenge["id"],
|
||||||
|
"response": digest,
|
||||||
|
"nonce": nonce,
|
||||||
|
"redir": str(resp.url),
|
||||||
|
"elapsedTime": "1",
|
||||||
|
},
|
||||||
|
cookies=resp.cookies,
|
||||||
|
allow_redirects=False,
|
||||||
|
)
|
||||||
|
auth = pass_resp.cookies.get("spchal-auth")
|
||||||
|
if not auth:
|
||||||
|
raise SearxEngineCaptchaException(message="startpage: Anubis pass-challenge failed")
|
||||||
|
auth = str(auth)
|
||||||
|
CACHE.set("SPCHAL_AUTH", auth, expire=240)
|
||||||
|
return auth
|
||||||
|
|
||||||
|
|
||||||
def get_sc_code(params):
|
def get_sc_code(params):
|
||||||
"""Get an actual ``sc`` argument from Startpage's search form (HTML page).
|
"""Get an actual ``sc`` argument from Startpage's search form (HTML page).
|
||||||
@@ -201,6 +245,9 @@ def get_sc_code(params):
|
|||||||
logger.debug("get_sc_code: request headers: %s", headers)
|
logger.debug("get_sc_code: request headers: %s", headers)
|
||||||
resp = get(get_sc_url, headers=headers)
|
resp = get(get_sc_url, headers=headers)
|
||||||
|
|
||||||
|
if 'id="anubis_challenge"' in resp.text:
|
||||||
|
resp = get(get_sc_url, headers=headers, cookies={"spchal-auth": _solve_anubis(resp)})
|
||||||
|
|
||||||
# ?? x = network.get('https://www.startpage.com/sp/cdn/images/filter-chevron.svg', headers=headers)
|
# ?? x = network.get('https://www.startpage.com/sp/cdn/images/filter-chevron.svg', headers=headers)
|
||||||
# ?? https://www.startpage.com/sp/cdn/images/filter-chevron.svg
|
# ?? https://www.startpage.com/sp/cdn/images/filter-chevron.svg
|
||||||
# ?? ping-back URL: https://www.startpage.com/sp/pb?sc=TLsB0oITjZ8F21
|
# ?? ping-back URL: https://www.startpage.com/sp/pb?sc=TLsB0oITjZ8F21
|
||||||
@@ -239,8 +286,8 @@ def request(query, params):
|
|||||||
Additionally the arguments form Startpage's search form needs to be set in
|
Additionally the arguments form Startpage's search form needs to be set in
|
||||||
HTML POST data / compare ``<input>`` elements: :py:obj:`search_form_xpath`.
|
HTML POST data / compare ``<input>`` elements: :py:obj:`search_form_xpath`.
|
||||||
"""
|
"""
|
||||||
engine_region = traits.get_region(params["searxng_locale"], "en-US")
|
engine_region = traits.get_region(params["searxng_locale"], "en_US")
|
||||||
engine_language = traits.get_language(params["searxng_locale"], "en")
|
engine_language = traits.get_language(params["searxng_locale"], "english")
|
||||||
|
|
||||||
params["headers"]["Origin"] = base_url
|
params["headers"]["Origin"] = base_url
|
||||||
params["headers"]["Referer"] = base_url + "/"
|
params["headers"]["Referer"] = base_url + "/"
|
||||||
@@ -262,9 +309,9 @@ def request(query, params):
|
|||||||
args["language"] = engine_language
|
args["language"] = engine_language
|
||||||
args["lui"] = engine_language
|
args["lui"] = engine_language
|
||||||
|
|
||||||
|
args["segment"] = "startpage.udog"
|
||||||
if params["pageno"] > 1:
|
if params["pageno"] > 1:
|
||||||
args["page"] = params["pageno"]
|
args["page"] = params["pageno"]
|
||||||
args["segment"] = "startpage.udog"
|
|
||||||
|
|
||||||
# Build cookie
|
# Build cookie
|
||||||
lang_homepage = "en"
|
lang_homepage = "en"
|
||||||
@@ -289,6 +336,8 @@ def request(query, params):
|
|||||||
cookie["search_results_region"] = engine_region
|
cookie["search_results_region"] = engine_region
|
||||||
|
|
||||||
params["cookies"]["preferences"] = "N1N".join(["%sEEE%s" % x for x in cookie.items()])
|
params["cookies"]["preferences"] = "N1N".join(["%sEEE%s" % x for x in cookie.items()])
|
||||||
|
if auth := CACHE.get("SPCHAL_AUTH"):
|
||||||
|
params["cookies"]["spchal-auth"] = auth
|
||||||
logger.debug("cookie preferences: %s", params["cookies"]["preferences"])
|
logger.debug("cookie preferences: %s", params["cookies"]["preferences"])
|
||||||
|
|
||||||
logger.debug("data: %s", args)
|
logger.debug("data: %s", args)
|
||||||
@@ -400,6 +449,18 @@ def _get_image_result(result) -> dict[str, t.Any] | None:
|
|||||||
|
|
||||||
|
|
||||||
def response(resp):
|
def response(resp):
|
||||||
|
if 'id="anubis_challenge"' in resp.text:
|
||||||
|
params = resp.search_params
|
||||||
|
params["cookies"]["spchal-auth"] = _solve_anubis(resp)
|
||||||
|
resp = post(
|
||||||
|
params["url"] or search_url,
|
||||||
|
data=params["data"],
|
||||||
|
headers=params["headers"],
|
||||||
|
cookies=params["cookies"],
|
||||||
|
)
|
||||||
|
if 'id="anubis_challenge"' in resp.text:
|
||||||
|
raise SearxEngineCaptchaException()
|
||||||
|
|
||||||
categ = startpage_categ.capitalize()
|
categ = startpage_categ.capitalize()
|
||||||
results_raw = "{" + extr(resp.text, f"React.createElement(UIStartpage.AppSerp{categ}, {{", "}})") + "}}"
|
results_raw = "{" + extr(resp.text, f"React.createElement(UIStartpage.AppSerp{categ}, {{", "}})") + "}}"
|
||||||
|
|
||||||
|
|||||||
@@ -21,8 +21,6 @@ about = {
|
|||||||
"require_api_key": False,
|
"require_api_key": False,
|
||||||
"results": "JSON",
|
"results": "JSON",
|
||||||
}
|
}
|
||||||
# otherwise all requests get blocked, probably HTTP2 fingerprinting
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
base_url = "https://stocksnap.io"
|
base_url = "https://stocksnap.io"
|
||||||
cdn_url = "https://cdn.stocksnap.io"
|
cdn_url = "https://cdn.stocksnap.io"
|
||||||
|
|||||||
@@ -74,7 +74,6 @@ Implementations
|
|||||||
===============
|
===============
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from dateutil.parser import parse
|
from dateutil.parser import parse
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ from dateutil import parser
|
|||||||
|
|
||||||
from searx.exceptions import SearxEngineAPIException
|
from searx.exceptions import SearxEngineAPIException
|
||||||
from searx.network import get
|
from searx.network import get
|
||||||
from searx.utils import gen_useragent, html_to_text
|
from searx.utils import html_to_text
|
||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
if t.TYPE_CHECKING:
|
||||||
@@ -52,7 +52,7 @@ def _obtain_x_sid() -> tuple[str, str]:
|
|||||||
The header key is usually called `x-sid-{UUIDv4}`, and the value is
|
The header key is usually called `x-sid-{UUIDv4}`, and the value is
|
||||||
usually a plain UUIDv4 (but a different one than in the header key).
|
usually a plain UUIDv4 (but a different one than in the header key).
|
||||||
"""
|
"""
|
||||||
resp = get(f"{api_url}/revcontent/embed.js", headers={"User-Agent": gen_useragent()})
|
resp = get(f"{api_url}/revcontent/embed.js", headers={"Referer": "https://tusksearch.com/"})
|
||||||
if not resp.ok:
|
if not resp.ok:
|
||||||
raise SearxEngineAPIException("failed to obtain request x-sid token")
|
raise SearxEngineAPIException("failed to obtain request x-sid token")
|
||||||
|
|
||||||
@@ -95,6 +95,7 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
# required - we send a random longitude and latitude instead of the actual user location
|
# required - we send a random longitude and latitude instead of the actual user location
|
||||||
"x-lon": str(round(random.random() * 90, 4)),
|
"x-lon": str(round(random.random() * 90, 4)),
|
||||||
"x-lat": str(round(random.random() * 90, 4)),
|
"x-lat": str(round(random.random() * 90, 4)),
|
||||||
|
"Referer": "https://tusksearch.com/",
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -17,12 +17,10 @@ about = {
|
|||||||
categories = ['images', 'icons']
|
categories = ['images', 'icons']
|
||||||
|
|
||||||
base_url = "https://uxwing.com"
|
base_url = "https://uxwing.com"
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
|
|
||||||
def request(query, params):
|
def request(query, params):
|
||||||
params['url'] = f"{base_url}/?s={quote_plus(query)}"
|
params['url'] = f"{base_url}/?s={quote_plus(query)}"
|
||||||
return params
|
|
||||||
|
|
||||||
|
|
||||||
def response(resp):
|
def response(resp):
|
||||||
|
|||||||
@@ -12,7 +12,6 @@ from lxml import html
|
|||||||
from searx.result_types import EngineResults
|
from searx.result_types import EngineResults
|
||||||
from searx.utils import eval_xpath_list, eval_xpath, extract_text
|
from searx.utils import eval_xpath_list, eval_xpath, extract_text
|
||||||
|
|
||||||
|
|
||||||
if t.TYPE_CHECKING:
|
if t.TYPE_CHECKING:
|
||||||
from lxml.etree import ElementBase
|
from lxml.etree import ElementBase
|
||||||
from searx.extended_types import SXNG_Response
|
from searx.extended_types import SXNG_Response
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
|
|
||||||
Some implementations are shared from :ref:`wikipedia engine`.
|
Some implementations are shared from :ref:`wikipedia engine`.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=missing-class-docstring
|
# pylint: disable=missing-class-docstring
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
|
|||||||
@@ -53,7 +53,7 @@ seconds."""
|
|||||||
|
|
||||||
def setup(engine_settings: dict[str, t.Any]) -> bool | None:
|
def setup(engine_settings: dict[str, t.Any]) -> bool | None:
|
||||||
global CACHE # pylint: disable=global-statement
|
global CACHE # pylint: disable=global-statement
|
||||||
CACHE = EngineCache(engine_settings["name"]) # type:ignore
|
CACHE = EngineCache(engine_settings["name"]) # type: ignore
|
||||||
|
|
||||||
|
|
||||||
def obtain_token() -> str:
|
def obtain_token() -> str:
|
||||||
|
|||||||
@@ -50,6 +50,7 @@ the engine).
|
|||||||
Implementations
|
Implementations
|
||||||
===============
|
===============
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=fixme
|
# pylint: disable=fixme
|
||||||
|
|
||||||
import typing as t
|
import typing as t
|
||||||
@@ -58,7 +59,7 @@ from json import loads
|
|||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
from dateutil import parser
|
from dateutil import parser
|
||||||
|
|
||||||
from httpx import DigestAuth
|
from curl_cffi import CurlOpt
|
||||||
|
|
||||||
from searx.utils import html_to_text
|
from searx.utils import html_to_text
|
||||||
|
|
||||||
@@ -141,7 +142,10 @@ def request(query, params):
|
|||||||
params["url"] = f"{_base_url()}/yacysearch.json?{urlencode(args)}"
|
params["url"] = f"{_base_url()}/yacysearch.json?{urlencode(args)}"
|
||||||
|
|
||||||
if http_digest_auth_user and http_digest_auth_pass:
|
if http_digest_auth_user and http_digest_auth_pass:
|
||||||
params['auth'] = DigestAuth(http_digest_auth_user, http_digest_auth_pass)
|
params['curl_options'] = {
|
||||||
|
CurlOpt.HTTPAUTH: 2, # CURLAUTH_DIGEST
|
||||||
|
CurlOpt.USERPWD: f"{http_digest_auth_user}:{http_digest_auth_pass}",
|
||||||
|
}
|
||||||
|
|
||||||
return params
|
return params
|
||||||
|
|
||||||
|
|||||||
@@ -8,7 +8,6 @@ from lxml import html
|
|||||||
from searx.exceptions import SearxEngineCaptchaException
|
from searx.exceptions import SearxEngineCaptchaException
|
||||||
from searx.utils import humanize_bytes, eval_xpath, eval_xpath_list, extract_text, extr
|
from searx.utils import humanize_bytes, eval_xpath, eval_xpath_list, extract_text, extr
|
||||||
|
|
||||||
|
|
||||||
# Engine metadata
|
# Engine metadata
|
||||||
about = {
|
about = {
|
||||||
"website": 'https://yandex.com/',
|
"website": 'https://yandex.com/',
|
||||||
@@ -22,6 +21,7 @@ about = {
|
|||||||
# Engine configuration
|
# Engine configuration
|
||||||
categories = []
|
categories = []
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
search_type = ""
|
search_type = ""
|
||||||
|
|
||||||
# Search URL
|
# Search URL
|
||||||
|
|||||||
@@ -77,10 +77,10 @@ notifications, but only as a fallback -- a request whose own locale matches
|
|||||||
``kk``, ``uk``, ``tr`` or ``en``."""
|
``kk``, ``uk``, ``tr`` or ``en``."""
|
||||||
|
|
||||||
region: str = ""
|
region: str = ""
|
||||||
"""Optional Yandex `region id`.
|
"""Optional Yandex `region id`_.
|
||||||
Only meaningful together with ``SEARCH_TYPE_RU``.
|
Only meaningful together with ``SEARCH_TYPE_RU``.
|
||||||
|
|
||||||
__ https://aistudio.yandex.ru/docs/en/search-api/reference/regions.html
|
.. _region id: https://aistudio.yandex.ru/docs/en/search-api/reference/regions.html
|
||||||
"""
|
"""
|
||||||
|
|
||||||
page_size: int = 10
|
page_size: int = 10
|
||||||
|
|||||||
@@ -30,8 +30,6 @@ web_base_url = "https://yep.com"
|
|||||||
safesearch = True
|
safesearch = True
|
||||||
safesearch_map = {0: "off", 1: "moderate", 2: "strict"}
|
safesearch_map = {0: "off", 1: "moderate", 2: "strict"}
|
||||||
|
|
||||||
enable_http2 = False
|
|
||||||
|
|
||||||
results_per_page = 20
|
results_per_page = 20
|
||||||
|
|
||||||
_IMPORT_RE = re.compile(r"import\"(.*?)\";")
|
_IMPORT_RE = re.compile(r"import\"(.*?)\";")
|
||||||
@@ -50,9 +48,6 @@ def request(query: str, params: "OnlineParams") -> None:
|
|||||||
{
|
{
|
||||||
"Referer": f"{web_base_url}/",
|
"Referer": f"{web_base_url}/",
|
||||||
"Origin": web_base_url,
|
"Origin": web_base_url,
|
||||||
"Sec-Fetch-Dest": "empty",
|
|
||||||
"Sec-Fetch-Mode": "cors",
|
|
||||||
"Sec-Fetch-Site": "same-site",
|
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ['videos', 'music']
|
categories = ['videos', 'music']
|
||||||
paging = False
|
paging = False
|
||||||
|
enable_http3 = True
|
||||||
api_key = None
|
api_key = None
|
||||||
|
|
||||||
# search-url
|
# search-url
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ about = {
|
|||||||
# engine dependent config
|
# engine dependent config
|
||||||
categories = ['videos', 'music']
|
categories = ['videos', 'music']
|
||||||
paging = True
|
paging = True
|
||||||
|
enable_http3 = True
|
||||||
language_support = False
|
language_support = False
|
||||||
time_range_support = True
|
time_range_support = True
|
||||||
|
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
|
|
||||||
- :py:obj:`flask.request` is replaced by :py:obj:`sxng_request`
|
- :py:obj:`flask.request` is replaced by :py:obj:`sxng_request`
|
||||||
- :py:obj:`flask.Request` is replaced by :py:obj:`SXNG_Request`
|
- :py:obj:`flask.Request` is replaced by :py:obj:`SXNG_Request`
|
||||||
- :py:obj:`httpx.response` is replaced by :py:obj:`SXNG_Response`
|
- :py:obj:`curl_cffi.requests.Response` is replaced by :py:obj:`SXNG_Response`
|
||||||
|
|
||||||
----
|
----
|
||||||
|
|
||||||
@@ -19,13 +19,16 @@
|
|||||||
:members:
|
:members:
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# pylint: disable=invalid-name
|
# pylint: disable=invalid-name
|
||||||
|
|
||||||
__all__ = ["SXNG_Request", "sxng_request", "SXNG_Response"]
|
__all__ = ["SXNG_Request", "sxng_request", "SXNG_Response"]
|
||||||
|
|
||||||
import typing
|
import typing
|
||||||
|
from urllib.parse import urlsplit
|
||||||
|
|
||||||
import flask
|
import flask
|
||||||
import httpx
|
from curl_cffi.requests import Response as CurlResponse
|
||||||
|
|
||||||
if typing.TYPE_CHECKING:
|
if typing.TYPE_CHECKING:
|
||||||
import searx.preferences
|
import searx.preferences
|
||||||
@@ -69,18 +72,37 @@ class SXNG_Request(flask.Request):
|
|||||||
sxng_request = typing.cast(SXNG_Request, flask.request)
|
sxng_request = typing.cast(SXNG_Request, flask.request)
|
||||||
|
|
||||||
|
|
||||||
class SXNG_Response(httpx.Response):
|
class SXNG_URL(str):
|
||||||
"""SearXNG extends the class :py:obj:`httpx.Response` with properties from
|
"""String URL"""
|
||||||
*this* class (type cast of :py:obj:`httpx.Response`).
|
|
||||||
|
@property
|
||||||
|
def host(self) -> str | None:
|
||||||
|
return urlsplit(self).hostname
|
||||||
|
|
||||||
|
@property
|
||||||
|
def path(self) -> str:
|
||||||
|
return urlsplit(self).path
|
||||||
|
|
||||||
|
|
||||||
|
class SXNG_Response(CurlResponse):
|
||||||
|
"""SearXNG extends :py:obj:`curl_cffi.requests.Response` with properties from
|
||||||
|
*this* class (type cast of the curl_cffi response).
|
||||||
|
|
||||||
.. code:: python
|
.. code:: python
|
||||||
|
|
||||||
response = httpx.get("https://example.org")
|
|
||||||
response = typing.cast(SXNG_Response, response)
|
response = typing.cast(SXNG_Response, response)
|
||||||
if response.ok:
|
if response.ok:
|
||||||
...
|
...
|
||||||
query_was = search_params["query"]
|
query_was = search_params["query"]
|
||||||
"""
|
"""
|
||||||
|
|
||||||
ok: bool
|
|
||||||
search_params: "OnlineParamTypes | OnlineDictParams | OnlineCurrenciesParams"
|
search_params: "OnlineParamTypes | OnlineDictParams | OnlineCurrenciesParams"
|
||||||
|
_url: str = ""
|
||||||
|
|
||||||
|
@property
|
||||||
|
def url(self) -> SXNG_URL: # type: ignore[override]
|
||||||
|
return SXNG_URL(self._url)
|
||||||
|
|
||||||
|
@url.setter
|
||||||
|
def url(self, value: str) -> None:
|
||||||
|
self._url = str(value or "")
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user