[refactor] wikidata: cache wikidata properties in searxng data

This commit is contained in:
Bnyro
2026-08-14 23:01:19 +02:00
parent d226b78bc4
commit 2b1c88c54c
15 changed files with 695 additions and 671 deletions

View File

@@ -31,7 +31,7 @@ jobs:
- update_external_bangs.py - update_external_bangs.py
- update_firefox_version.py - update_firefox_version.py
- update_engine_traits.py - update_engine_traits.py
- update_wikidata_units.py - update_wikidata.py
- update_engine_descriptions.py - update_engine_descriptions.py
permissions: permissions:

View File

@@ -80,8 +80,8 @@ same environment, here are a few examples::
# to test one of the update scripts # to test one of the update scripts
(dev.env)$ searxng_extra/update/update_engine_traits.py --help (dev.env)$ searxng_extra/update/update_engine_traits.py --help
# to test the update of the wikidata units # to test the update of the wikidata units and property names
(dev.env)$ searxng_extra/update/update_wikidata_units.py (dev.env)$ searxng_extra/update/update_wikidata.py
.. sidebar:: further read .. sidebar:: further read

View File

@@ -90,10 +90,10 @@ Scripts to update static data in :origin:`searx/data/`
:members: :members:
``update_wikidata_units.py`` ``update_wikidata.py``
============================ ============================
:origin:`[source] <searxng_extra/update/update_wikidata_units.py>` :origin:`[source] <searxng_extra/update/update_wikidata.py>`
.. automodule:: searxng_extra.update.update_wikidata_units .. automodule:: searxng_extra.update.update_wikidata
:members: :members:

View File

@@ -32,6 +32,13 @@ class WikiDataUnitType(t.TypedDict):
to_si_factor: float to_si_factor: float
WikiDataPropertyNameType = str | dict[str, str]
"""Name of a Wikidata property. Can be either the plain name or a dictionary of
language code to property name, e.g. ``{"en": "Date of birth"}``."""
WikiDataPropertiesType = dict[str, WikiDataPropertyNameType]
"""Dictionary from wikidata property ID to property name."""
class LocalesType(t.TypedDict): class LocalesType(t.TypedDict):
"""Data structure of an item in ``locales.json``""" """Data structure of an item in ``locales.json``"""
@@ -41,6 +48,7 @@ class LocalesType(t.TypedDict):
USER_AGENTS: UserAgentType USER_AGENTS: UserAgentType
WIKIDATA_UNITS: dict[str, WikiDataUnitType] WIKIDATA_UNITS: dict[str, WikiDataUnitType]
WIKIDATA_PROPERTIES: WikiDataPropertiesType
TRACKER_PATTERNS: TrackerPatternsDB TRACKER_PATTERNS: TrackerPatternsDB
LOCALES: LocalesType LOCALES: LocalesType
CURRENCIES: CurrenciesDB CURRENCIES: CurrenciesDB
@@ -52,11 +60,12 @@ ENGINE_DESCRIPTIONS: dict[str, dict[str, t.Any]]
ENGINE_TRAITS: dict[str, dict[str, t.Any]] ENGINE_TRAITS: dict[str, dict[str, t.Any]]
lazy_globals = { lazy_globals: dict[str, t.Any] = {
"CURRENCIES": CurrenciesDB(), "CURRENCIES": CurrenciesDB(),
"USER_AGENTS": None, "USER_AGENTS": None,
"EXTERNAL_URLS": None, "EXTERNAL_URLS": None,
"WIKIDATA_UNITS": None, "WIKIDATA_UNITS": None,
"WIKIDATA_PROPERTIES": None,
"EXTERNAL_BANGS": None, "EXTERNAL_BANGS": None,
"OSM_KEYS_TAGS": None, "OSM_KEYS_TAGS": None,
"ENGINE_DESCRIPTIONS": None, "ENGINE_DESCRIPTIONS": None,
@@ -69,6 +78,7 @@ data_json_files = {
"USER_AGENTS": "useragents.json", "USER_AGENTS": "useragents.json",
"EXTERNAL_URLS": "external_urls.json", "EXTERNAL_URLS": "external_urls.json",
"WIKIDATA_UNITS": "wikidata_units.json", "WIKIDATA_UNITS": "wikidata_units.json",
"WIKIDATA_PROPERTIES": "wikidata_properties.json",
"EXTERNAL_BANGS": "external_bangs.json", "EXTERNAL_BANGS": "external_bangs.json",
"OSM_KEYS_TAGS": "osm_keys_tags.json", "OSM_KEYS_TAGS": "osm_keys_tags.json",
"ENGINE_DESCRIPTIONS": "engine_descriptions.json", "ENGINE_DESCRIPTIONS": "engine_descriptions.json",

View File

@@ -10,7 +10,8 @@ from flask_babel import gettext
from searx.data import OSM_KEYS_TAGS, CURRENCIES from searx.data import OSM_KEYS_TAGS, CURRENCIES
from searx.external_urls import get_external_url from searx.external_urls import get_external_url
from searx.engines.wikidata import send_wikidata_query, sparql_string_escape, get_thumbnail from searx.wikidata import send_wikidata_query
from searx.engines.wikidata import sparql_string_escape, get_thumbnail
from searx.result_types import EngineResults from searx.result_types import EngineResults
# about # about
@@ -290,7 +291,8 @@ def get_title_address(result):
'house_number': address_raw.get('house_number'), 'house_number': address_raw.get('house_number'),
'road': address_raw.get('road'), 'road': address_raw.get('road'),
'locality': address_raw.get( 'locality': address_raw.get(
'city', address_raw.get('town', address_raw.get('village')) # noqa 'city',
address_raw.get('town', address_raw.get('village')), # noqa
), # noqa ), # noqa
'postcode': address_raw.get('postcode'), 'postcode': address_raw.get('postcode'),
'country': address_raw.get('country'), 'country': address_raw.get('country'),

View File

@@ -7,24 +7,29 @@ Some implementations are shared from :ref:`wikipedia engine`.
import typing as t import typing as t
import os
from hashlib import md5 from hashlib import md5
from urllib.parse import urlencode, unquote from urllib.parse import urlencode, unquote
from json import loads from json import loads
from dateutil.parser import isoparse
from babel.dates import format_datetime, format_date, format_time, get_datetime_format
from searx.enginelib import EngineCache
from searx.data import WIKIDATA_UNITS
from searx.network import post, get from searx.network import post, get
from searx.utils import searxng_useragent, get_string_replaces_function from searx.utils import get_string_replaces_function
from searx.external_urls import get_external_url, get_earth_coordinates_url, area_to_osm_zoom from searx.external_urls import area_to_osm_zoom
from searx.engines.wikipedia import ( from searx.engines.wikipedia import (
fetch_wikimedia_traits, fetch_wikimedia_traits,
get_wiki_params, get_wiki_params,
) )
from searx.enginelib.traits import EngineTraits from searx.enginelib.traits import EngineTraits
from searx.wikidata_properties import (
QUERY_TEMPLATE,
WDArticle,
WDAttrList,
WDGeoAttribute,
WDImageAttribute,
WDURLAttribute,
get_attributes,
)
from searx.wikidata import SPARQL_ENDPOINT_URL, SPARQL_EXPLAIN_URL, get_wikidata_headers
if t.TYPE_CHECKING: if t.TYPE_CHECKING:
from searx.extended_types import SXNG_Response from searx.extended_types import SXNG_Response
@@ -47,78 +52,6 @@ display_type = ["infobox"]
one will add a hit to the result list. The first one will show a hit in the one will add a hit to the result list. The first one will show a hit in the
info box. Both values can be set, or one of the two can be set.""" info box. Both values can be set, or one of the two can be set."""
CACHE: EngineCache
"""Persistent (SQLite) key/value cache that deletes its values after ``expire``
seconds."""
# SPARQL
SPARQL_ENDPOINT_URL = "https://query.wikidata.org/sparql"
SPARQL_EXPLAIN_URL = "https://query.wikidata.org/bigdata/namespace/wdq/sparql?explain"
WDPType = dict[str | tuple[str, str], str]
WIKIDATA_PROPERTIES: WDPType = {
"P434": "MusicBrainz",
"P435": "MusicBrainz",
"P436": "MusicBrainz",
"P966": "MusicBrainz",
"P345": "IMDb",
"P2397": "YouTube",
"P1651": "YouTube",
"P2002": "Twitter",
"P2013": "Facebook",
"P2003": "Instagram",
"P4033": "Mastodon",
"P11947": "Lemmy",
"P12622": "PeerTube",
}
# SERVICE wikibase:mwapi : https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual/MWAPI
# SERVICE wikibase:label: https://en.wikibooks.org/wiki/SPARQL/SERVICE_-_Label#Manual_Label_SERVICE
# https://en.wikibooks.org/wiki/SPARQL/WIKIDATA_Precision,_Units_and_Coordinates
# https://www.mediawiki.org/wiki/Wikibase/Indexing/RDF_Dump_Format#Data_model
# optimization:
# * https://www.wikidata.org/wiki/Wikidata:SPARQL_query_service/query_optimization
# * https://github.com/blazegraph/database/wiki/QueryHints
QUERY_TEMPLATE = """
SELECT ?item ?itemLabel ?itemDescription ?lat ?long %SELECT%
WHERE
{
SERVICE wikibase:mwapi {
bd:serviceParam wikibase:endpoint "www.wikidata.org";
wikibase:api "EntitySearch";
wikibase:limit 1;
mwapi:search "%QUERY%";
mwapi:language "%LANGUAGE%".
?item wikibase:apiOutputItem mwapi:item.
}
hint:Prior hint:runFirst "true".
%WHERE%
SERVICE wikibase:label {
bd:serviceParam wikibase:language "%LANGUAGE%,en".
?item rdfs:label ?itemLabel .
?item schema:description ?itemDescription .
%WIKIBASE_LABELS%
}
}
GROUP BY ?item ?itemLabel ?itemDescription ?lat ?long %GROUP_BY%
"""
# Get the calendar names and the property names
QUERY_PROPERTY_NAMES = """
SELECT ?item ?name
WHERE {
{
SELECT ?item
WHERE { ?item wdt:P279* wd:Q12132 }
} UNION {
VALUES ?item { %ATTRIBUTES% }
}
OPTIONAL { ?item rdfs:label ?name. }
}
"""
# see the property "dummy value" of https://www.wikidata.org/wiki/Q2013 (Wikidata) # see the property "dummy value" of https://www.wikidata.org/wiki/Q2013 (Wikidata)
# hard coded here to avoid to an additional SPARQL request when the server starts # hard coded here to avoid to an additional SPARQL request when the server starts
DUMMY_ENTITY_URLS = set( DUMMY_ENTITY_URLS = set(
@@ -130,357 +63,13 @@ DUMMY_ENTITY_URLS = set(
# https://lists.w3.org/Archives/Public/public-rdf-dawg/2011OctDec/0175.html # https://lists.w3.org/Archives/Public/public-rdf-dawg/2011OctDec/0175.html
sparql_string_escape = get_string_replaces_function( sparql_string_escape = get_string_replaces_function(
# fmt: off # fmt: off
{ {"\t": "\\\t", "\n": "\\\n", "\r": "\\\r", "\b": "\\\b", "\f": "\\\f", "\"": "\\\"", "'": "\\'", "\\": "\\\\"}
"\t": "\\\t",
"\n": "\\\n",
"\r": "\\\r",
"\b": "\\\b",
"\f": "\\\f",
"\"": "\\\"",
"\'": "\\\'",
"\\": "\\\\"
}
# fmt: on # fmt: on
) )
replace_http_by_https = get_string_replaces_function({"http:": "https:"}) replace_http_by_https = get_string_replaces_function({"http:": "https:"})
class WDAttribute:
def __init__(self, name: str):
self.name: str = name
def get_select(self):
return "(group_concat(distinct ?{name};separator=', ') as ?{name}s)".replace("{name}", self.name)
def get_label(self, language: str):
return get_label_for_entity(self.name, language)
def get_where(self):
return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name)
def get_wikibase_label(self) -> str:
return ""
def get_group_by(self) -> str:
return ""
def get_str(self, result: dict[str, t.Any], language: str) -> str | None: # pylint: disable=unused-argument
return result.get(self.name + "s")
def __repr__(self):
return "<" + str(type(self).__name__) + ":" + self.name + ">"
class WDAmountAttribute(WDAttribute):
def get_select(self) -> str:
return "?{name} ?{name}Unit".replace("{name}", self.name)
def get_where(self):
return """ OPTIONAL { ?item p:{name} ?{name}Node .
?{name}Node rdf:type wikibase:BestRank ; ps:{name} ?{name} .
OPTIONAL { ?{name}Node psv:{name}/wikibase:quantityUnit ?{name}Unit. } }""".replace(
'{name}', self.name
)
def get_group_by(self) -> str:
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name)
unit: str | None = result.get(self.name + "Unit")
if unit is not None:
unit = unit.replace("http://www.wikidata.org/entity/", "")
return str(value) + " " + get_label_for_entity(unit, language)
return value
class WDArticle(WDAttribute):
def __init__(self, language: str, kwargs: dict[str, t.Any] | None = None):
super().__init__("wikipedia")
self.language: str = language
self.kwargs: dict[str, t.Any] = kwargs or {}
def get_label(self, language: str):
# language parameter is ignored
return "Wikipedia ({language})".replace("{language}", self.language)
def get_select(self):
return "?article{language} ?articleName{language}".replace("{language}", self.language)
def get_where(self):
return """OPTIONAL { ?article{language} schema:about ?item ;
schema:inLanguage "{language}" ;
schema:isPartOf <https://{language}.wikipedia.org/> ;
schema:name ?articleName{language} . }""".replace(
'{language}', self.language
)
def get_group_by(self):
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
key = "article{language}".replace("{language}", self.language)
return result.get(key)
class WDLabelAttribute(WDAttribute):
def get_select(self):
return "(group_concat(distinct ?{name}Label;separator=', ') as ?{name}Labels)".replace("{name}", self.name)
def get_where(self):
return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name)
def get_wikibase_label(self) -> str:
return "?{name} rdfs:label ?{name}Label .".replace("{name}", self.name)
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
return result.get(self.name + "Labels")
class WDURLAttribute(WDAttribute):
HTTP_WIKIMEDIA_IMAGE: str = "http://commons.wikimedia.org/wiki/Special:FilePath/"
def __init__(
self,
name: str,
url_id: str | None = None,
url_path_prefix: str | None = None,
kwargs: dict[str, t.Any] | None = None,
):
"""
:param url_id: ID matching one key in ``external_urls.json`` for
converting IDs to full URLs.
:param url_path_prefix: Path prefix if the values are of format
``account@domain``. If provided, value are rewritten to
``https://<domain><url_path_prefix><account>``. For example::
WDURLAttribute('P4033', url_path_prefix='/@')
Adds Property `P4033 <https://www.wikidata.org/wiki/Property:P4033>`_
to the wikidata query. This field might return for example
``libreoffice@fosstodon.org`` and the URL built from this is then:
- account: ``libreoffice``
- domain: ``fosstodon.org``
- result url: https://fosstodon.org/@libreoffice
"""
super().__init__(name)
self.url_id: str | None = url_id
self.url_path_prefix: str | None = url_path_prefix
self.kwargs: dict[str, t.Any] = kwargs or {}
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name + "s")
if not value:
return None
value = value.split(",")[0]
if self.url_id:
url_id = self.url_id
if value.startswith(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE):
value = value[len(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE) :]
url_id = "wikimedia_image"
return get_external_url(url_id, value)
if self.url_path_prefix:
[account, domain] = [x.strip("@ ") for x in value.rsplit("@", 1)]
return f"https://{domain}{self.url_path_prefix}{account}"
return value
class WDGeoAttribute(WDAttribute):
def get_label(self, language: str):
return "OpenStreetMap"
def get_select(self):
return "?{name}Lat ?{name}Long".replace("{name}", self.name)
def get_where(self):
return """OPTIONAL { ?item p:{name}/psv:{name} [
wikibase:geoLatitude ?{name}Lat ;
wikibase:geoLongitude ?{name}Long ] }""".replace(
'{name}', self.name
)
def get_group_by(self):
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
latitude: str | None = result.get(self.name + "Lat")
longitude: str | None = result.get(self.name + "Long")
if latitude and longitude:
return latitude + " " + longitude
return None
def get_geo_url(self, result: dict[str, t.Any], osm_zoom: int = 19) -> str | None:
latitude: str | None = result.get(self.name + "Lat")
longitude: str | None = result.get(self.name + "Long")
if latitude and longitude:
return get_earth_coordinates_url(latitude, longitude, osm_zoom)
return None
class WDImageAttribute(WDURLAttribute):
def __init__(self, name: str, url_id: str | None = None, priority: int = 100):
super().__init__(name, url_id)
self.priority: int = priority
class WDDateAttribute(WDAttribute):
def get_select(self):
return "?{name} ?{name}timePrecision ?{name}timeZone ?{name}timeCalendar".replace("{name}", self.name)
def get_where(self):
# To remove duplicate, add
# FILTER NOT EXISTS { ?item p:{name}/psv:{name}/wikibase:timeValue ?{name}bis FILTER (?{name}bis < ?{name}) }
# this filter is too slow, so the response function ignore duplicate results
# (see the seen_entities variable)
return """OPTIONAL { ?item p:{name}/psv:{name} [
wikibase:timeValue ?{name} ;
wikibase:timePrecision ?{name}timePrecision ;
wikibase:timeTimezone ?{name}timeZone ;
wikibase:timeCalendarModel ?{name}timeCalendar ] . }
hint:Prior hint:rangeSafe true;""".replace(
'{name}', self.name
)
def get_group_by(self):
return self.get_select()
def format_8(self, value: str, locale: str) -> str: # pylint: disable=unused-argument
# precision: less than a year
return value
def format_9(self, value: str, locale: str) -> str:
year = int(value)
# precision: year
if year < 1584:
if year < 0:
return str(year - 1)
return str(year)
timestamp = isoparse(value)
return format_date(timestamp, format="yyyy", locale=locale)
def format_10(self, value: str, locale: str) -> str:
# precision: month
timestamp = isoparse(value)
return format_date(timestamp, format="MMMM y", locale=locale)
def format_11(self, value: str, locale: str) -> str:
# precision: day
timestamp = isoparse(value)
return format_date(timestamp, format="full", locale=locale)
def format_13(self, value: str, locale: str) -> str:
timestamp = isoparse(value)
# precision: minute
return (
get_datetime_format(format, locale=locale)
.replace("'", "")
.replace("{0}", format_time(timestamp, "full", tzinfo=None, locale=locale))
.replace("{1}", format_date(timestamp, "short", locale=locale))
)
def format_14(self, value: str, locale: str) -> str:
# precision: second.
return format_datetime(isoparse(value), format="full", locale=locale)
DATE_FORMAT: dict[str, tuple[str, int]] = {
"0": ("format_8", 1000000000),
"1": ("format_8", 100000000),
"2": ("format_8", 10000000),
"3": ("format_8", 1000000),
"4": ("format_8", 100000),
"5": ("format_8", 10000),
"6": ("format_8", 1000),
"7": ("format_8", 100),
"8": ("format_8", 10),
"9": ("format_9", 1), # year
"10": ("format_10", 1), # month
"11": ("format_11", 0), # day
"12": ("format_13", 0), # hour (not supported by babel, display minute)
"13": ("format_13", 0), # minute
"14": ("format_14", 0), # second
}
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name)
if value == "" or value is None:
return None
_p: str = result.get(self.name + "timePrecision") or "1"
date_format = WDDateAttribute.DATE_FORMAT.get(_p)
if date_format is not None:
format_method = getattr(self, date_format[0])
precision: int = date_format[1]
try:
if precision >= 1:
_t = value.split("-")
if value.startswith("-"):
value = "-" + _t[1]
else:
value = _t[0]
return format_method(value, language)
except Exception: # pylint: disable=broad-except
return value
return value
WDAttrType = (
WDAttribute
| WDAmountAttribute
| WDArticle
| WDLabelAttribute
| WDURLAttribute
| WDGeoAttribute
| WDImageAttribute
| WDDateAttribute
)
WDAttrList = list[WDAttrType]
def get_headers() -> dict[str, str]:
# user agent: https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual#Query_limits
return {
"Accept": "application/sparql-results+json",
"User-Agent": f"wikidata engine - {searxng_useragent()}",
}
def get_label_for_entity(entity_id: str, language: str) -> str:
name = WIKIDATA_PROPERTIES.get(entity_id)
if name is None:
name = WIKIDATA_PROPERTIES.get((entity_id, language))
if name is None:
name = WIKIDATA_PROPERTIES.get((entity_id, language.split("-")[0]))
if name is None:
name = WIKIDATA_PROPERTIES.get((entity_id, "en"))
if name is None:
name = entity_id
return name
def send_wikidata_query(query: str, method: str = "GET", **kwargs: dict[str, t.Any]) -> dict[str, t.Any]:
if method == "GET":
# query will be cached by wikidata
http_response = get(SPARQL_ENDPOINT_URL + "?" + urlencode({"query": query}), headers=get_headers(), **kwargs)
else:
# query won't be cached by wikidata
http_response = post(SPARQL_ENDPOINT_URL, data={"query": query}, headers=get_headers(), **kwargs)
if http_response.status_code != 200:
logger.debug("SPARQL endpoint error %s", http_response.content.decode())
logger.debug("request time %s", str(http_response.elapsed))
http_response.raise_for_status()
return loads(http_response.content.decode())
def request(query: str, params: "OnlineParams") -> None: def request(query: str, params: "OnlineParams") -> None:
attributes: WDAttrList attributes: WDAttrList
@@ -491,7 +80,7 @@ def request(query: str, params: "OnlineParams") -> None:
params["method"] = "POST" params["method"] = "POST"
params["url"] = SPARQL_ENDPOINT_URL params["url"] = SPARQL_ENDPOINT_URL
params["data"] = {"query": query} params["data"] = {"query": query}
params["headers"] = get_headers() params["headers"] = get_wikidata_headers()
# additional parameters (not a part of OnlineParams) # additional parameters (not a part of OnlineParams)
params["language"] = eng_tag # type: ignore params["language"] = eng_tag # type: ignore
@@ -584,7 +173,6 @@ def get_results(
for attribute in attributes: for attribute in attributes:
value: str | None = attribute.get_str(attribute_result, language) value: str | None = attribute.get_str(attribute_result, language)
if value is not None and value != "": if value is not None and value != "":
if isinstance(attribute, (WDURLAttribute, WDArticle)): if isinstance(attribute, (WDURLAttribute, WDArticle)):
# get_select() method : there is group_concat(distinct ...;separator=", ") # get_select() method : there is group_concat(distinct ...;separator=", ")
# split the value here # split the value here
@@ -670,212 +258,15 @@ def get_query(query: str, language: str) -> tuple[str, WDAttrList]:
return query, attributes return query, attributes
def get_attributes(language: str):
# pylint: disable=too-many-statements
attributes: WDAttrList = []
def add_value(name: str):
attributes.append(WDAttribute(name))
def add_amount(name: str):
attributes.append(WDAmountAttribute(name))
def add_label(name: str):
attributes.append(WDLabelAttribute(name))
def add_url(name: str, url_id: str | None = None, url_path_prefix: str | None = None, **kwargs: dict[str, t.Any]):
attributes.append(WDURLAttribute(name, url_id, url_path_prefix, kwargs))
def add_image(name: str, url_id: str | None = None, priority: int = 1):
attributes.append(WDImageAttribute(name, url_id, priority))
def add_date(name: str):
attributes.append(WDDateAttribute(name))
# Dates
for p in [
"P571", # inception date
"P576", # dissolution date
"P580", # start date
"P582", # end date
"P569", # date of birth
"P570", # date of death
"P619", # date of spacecraft launch
"P620",
]: # date of spacecraft landing
add_date(p)
for p in [
"P27", # country of citizenship
"P495", # country of origin
"P17", # country
"P159",
]: # headquarters location
add_label(p)
# Places
for p in [
"P36", # capital
"P35", # head of state
"P6", # head of government
"P122", # basic form of government
"P37",
]: # official language
add_label(p)
add_value("P1082") # population
add_amount("P2046") # area
add_amount("P281") # postal code
add_label("P38") # currency
add_amount("P2048") # height (building)
# Media
for p in [
"P400", # platform (videogames, computing)
"P50", # author
"P170", # creator
"P57", # director
"P175", # performer
"P178", # developer
"P162", # producer
"P176", # manufacturer
"P58", # screenwriter
"P272", # production company
"P264", # record label
"P123", # publisher
"P449", # original network
"P750", # distributed by
"P86",
]: # composer
add_label(p)
add_date("P577") # publication date
add_label("P136") # genre (music, film, artistic...)
add_label("P364") # original language
add_value("P212") # ISBN-13
add_value("P957") # ISBN-10
add_label("P275") # copyright license
add_label("P277") # programming language
add_value("P348") # version
add_label("P840") # narrative location
# Languages
add_value("P1098") # number of speakers
add_label("P282") # writing system
add_label("P1018") # language regulatory body
add_value("P218") # language code (ISO 639-1)
# Other
add_label("P169") # ceo
add_label("P112") # founded by
add_label("P1454") # legal form (company, organization)
add_label("P137") # operator (service, facility, ...)
add_label("P1029") # crew members (tripulation)
add_label("P225") # taxon name
add_value("P274") # chemical formula
add_label("P1346") # winner (sports, contests, ...)
add_value("P1120") # number of deaths
add_value("P498") # currency code (ISO 4217)
# URL
kwargs: dict[str, t.Any] = {"official": True}
add_url("P856", **kwargs) # official website
attributes.append(WDArticle(language)) # wikipedia (user language)
if not language.startswith("en"):
attributes.append(WDArticle("en")) # wikipedia (english)
add_url("P1324") # source code repository
add_url("P1581") # blog
add_url("P434", url_id="musicbrainz_artist")
add_url("P435", url_id="musicbrainz_work")
add_url("P436", url_id="musicbrainz_release_group")
add_url("P966", url_id="musicbrainz_label")
add_url("P345", url_id="imdb_id")
add_url("P2397", url_id="youtube_channel")
add_url("P1651", url_id="youtube_video")
add_url("P2002", url_id="twitter_profile")
add_url("P2013", url_id="facebook_profile")
add_url("P2003", url_id="instagram_profile")
# Fediverse
add_url("P4033", url_path_prefix="/@") # Mastodon user
add_url("P11947", url_path_prefix="/c/") # Lemmy community
add_url("P12622", url_path_prefix="/c/") # PeerTube channel
# Map
attributes.append(WDGeoAttribute("P625"))
# Image
add_image("P15", priority=1, url_id="wikimedia_image") # route map
add_image("P242", priority=2, url_id="wikimedia_image") # locator map
add_image("P154", priority=3, url_id="wikimedia_image") # logo
add_image("P18", priority=4, url_id="wikimedia_image") # image
add_image("P41", priority=5, url_id="wikimedia_image") # flag
add_image("P2716", priority=6, url_id="wikimedia_image") # collage
add_image("P2910", priority=7, url_id="wikimedia_image") # icon
return attributes
def debug_explain_wikidata_query(query: str, method: str = "GET"): def debug_explain_wikidata_query(query: str, method: str = "GET"):
if method == "GET": if method == "GET":
http_response = get(SPARQL_EXPLAIN_URL + "&" + urlencode({"query": query}), headers=get_headers()) http_response = get(SPARQL_EXPLAIN_URL + "&" + urlencode({"query": query}), headers=get_wikidata_headers())
else: else:
http_response = post(SPARQL_EXPLAIN_URL, data={"query": query}, headers=get_headers()) http_response = post(SPARQL_EXPLAIN_URL, data={"query": query}, headers=get_wikidata_headers())
http_response.raise_for_status() http_response.raise_for_status()
return http_response.content return http_response.content
def init(_):
global CACHE # pylint: disable=global-statement
CACHE = EngineCache("wikidata")
# In an environment with competing processes, the initial loading of the
# cache is required only once.
eng_state: str | None = CACHE.get("eng_state")
if not eng_state or not eng_state.startswith("STATE:"):
CACHE.set("eng_state", f"STATE: being initialized by PID {os.getpid()}")
try:
init_wikidata_properties()
except Exception:
CACHE.set("eng_state", f"ERROR: initialization by PID {os.getpid()} failed.")
raise
else:
logger.debug(eng_state)
def init_wikidata_properties():
global WIKIDATA_PROPERTIES # pylint: disable=global-statement
p: WDPType = CACHE.get(key="WIKIDATA_PROPERTIES")
if p:
WIKIDATA_PROPERTIES = p
return
# WIKIDATA_PROPERTIES : add unit symbols
for k, v in WIKIDATA_UNITS.items():
WIKIDATA_PROPERTIES[k] = v["symbol"]
# WIKIDATA_PROPERTIES : add property labels
wikidata_property_names: list[str] = []
for attribute in get_attributes("en"):
if type(attribute) in (WDAttribute, WDAmountAttribute, WDURLAttribute, WDDateAttribute, WDLabelAttribute):
if attribute.name not in WIKIDATA_PROPERTIES:
wikidata_property_names.append("wd:" + attribute.name)
query = QUERY_PROPERTY_NAMES.replace("%ATTRIBUTES%", " ".join(wikidata_property_names))
kwargs: dict[str, t.Any] = {"timeout": 20}
jsonresponse = send_wikidata_query(query, **kwargs)
for result in jsonresponse.get("results", {}).get("bindings", {}):
name_field = result.get("name")
if not name_field:
continue
name = name_field["value"]
lang = name_field["xml:lang"]
entity_id = result["item"]["value"].replace("http://www.wikidata.org/entity/", "")
WIKIDATA_PROPERTIES[(entity_id, lang)] = name.capitalize()
CACHE.set(key="WIKIDATA_PROPERTIES", value=WIKIDATA_PROPERTIES)
def fetch_traits(engine_traits: EngineTraits): def fetch_traits(engine_traits: EngineTraits):
"""Uses languages evaluated from :py:obj:`wikipedia.fetch_wikimedia_traits """Uses languages evaluated from :py:obj:`wikipedia.fetch_wikimedia_traits
<searx.engines.wikipedia.fetch_wikimedia_traits>` and removes <searx.engines.wikipedia.fetch_wikimedia_traits>` and removes

35
searx/wikidata.py Normal file
View File

@@ -0,0 +1,35 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
"""Shared methods for accessing Wikidata."""
import typing as t
from urllib.parse import urlencode
from searx.network import get, post
from searx.utils import gen_useragent
# SPARQL
SPARQL_ENDPOINT_URL = "https://query.wikidata.org/sparql"
SPARQL_EXPLAIN_URL = "https://query.wikidata.org/bigdata/namespace/wdq/sparql?explain"
def send_wikidata_query(query: str, method: str = "GET", **kwargs: dict[str, t.Any]) -> dict[str, t.Any]:
if method == "GET":
# query will be cached by wikidata
http_response = get(
SPARQL_ENDPOINT_URL + "?" + urlencode({"query": query}), headers=get_wikidata_headers(), **kwargs
)
else:
# query won't be cached by wikidata
http_response = post(SPARQL_ENDPOINT_URL, data={"query": query}, headers=get_wikidata_headers(), **kwargs)
http_response.raise_for_status()
return http_response.json()
def get_wikidata_headers() -> dict[str, str]:
# user agent: https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual#Query_limits
return {
"Accept": "application/sparql-results+json",
"User-Agent": gen_useragent(),
}

View File

@@ -0,0 +1,580 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
# pylint: disable=missing-class-docstring
"""Fetch property names from :origin:`searx/engines/wikidata.py` engine."""
import typing as t
from dateutil.parser import isoparse
from babel.dates import format_datetime, format_date, format_time, get_datetime_format
from searx.external_urls import get_earth_coordinates_url, get_external_url
from searx.data import WikiDataPropertiesType, WikiDataUnitType
from searx.wikidata import send_wikidata_query
# SERVICE wikibase:mwapi : https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual/MWAPI
# SERVICE wikibase:label: https://en.wikibooks.org/wiki/SPARQL/SERVICE_-_Label#Manual_Label_SERVICE
# https://en.wikibooks.org/wiki/SPARQL/WIKIDATA_Precision,_Units_and_Coordinates
# https://www.mediawiki.org/wiki/Wikibase/Indexing/RDF_Dump_Format#Data_model
# optimization:
# * https://www.wikidata.org/wiki/Wikidata:SPARQL_query_service/query_optimization
# * https://github.com/blazegraph/database/wiki/QueryHints
QUERY_TEMPLATE = """
SELECT ?item ?itemLabel ?itemDescription ?lat ?long %SELECT%
WHERE
{
SERVICE wikibase:mwapi {
bd:serviceParam wikibase:endpoint "www.wikidata.org";
wikibase:api "EntitySearch";
wikibase:limit 1;
mwapi:search "%QUERY%";
mwapi:language "%LANGUAGE%".
?item wikibase:apiOutputItem mwapi:item.
}
hint:Prior hint:runFirst "true".
%WHERE%
SERVICE wikibase:label {
bd:serviceParam wikibase:language "%LANGUAGE%,en".
?item rdfs:label ?itemLabel .
?item schema:description ?itemDescription .
%WIKIBASE_LABELS%
}
}
GROUP BY ?item ?itemLabel ?itemDescription ?lat ?long %GROUP_BY%
"""
# Get the calendar names and the property names
QUERY_PROPERTY_NAMES = """
SELECT ?item ?name
WHERE {
{
SELECT ?item
WHERE { ?item wdt:P279* wd:Q12132 }
} UNION {
VALUES ?item { %ATTRIBUTES% }
}
OPTIONAL { ?item rdfs:label ?name. }
}
"""
class WDAttribute:
def __init__(self, name: str):
self.name: str = name
def get_select(self):
return "(group_concat(distinct ?{name};separator=', ') as ?{name}s)".replace("{name}", self.name)
def get_label(self, language: str):
return get_label_for_entity(self.name, language)
def get_where(self):
return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name)
def get_wikibase_label(self) -> str:
return ""
def get_group_by(self) -> str:
return ""
def get_str(self, result: dict[str, t.Any], language: str) -> str | None: # pylint: disable=unused-argument
return result.get(self.name + "s")
def __repr__(self):
return "<" + str(type(self).__name__) + ":" + self.name + ">"
class WDAmountAttribute(WDAttribute):
def get_select(self) -> str:
return "?{name} ?{name}Unit".replace("{name}", self.name)
def get_where(self):
return """ OPTIONAL { ?item p:{name} ?{name}Node .
?{name}Node rdf:type wikibase:BestRank ; ps:{name} ?{name} .
OPTIONAL { ?{name}Node psv:{name}/wikibase:quantityUnit ?{name}Unit. } }""".replace(
'{name}', self.name
)
def get_group_by(self) -> str:
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name)
unit: str | None = result.get(self.name + "Unit")
if unit is not None:
unit = unit.replace("http://www.wikidata.org/entity/", "")
return str(value) + " " + get_label_for_entity(unit, language)
return value
class WDArticle(WDAttribute):
def __init__(self, language: str, kwargs: dict[str, t.Any] | None = None):
super().__init__("wikipedia")
self.language: str = language
self.kwargs: dict[str, t.Any] = kwargs or {}
def get_label(self, language: str):
# language parameter is ignored
return "Wikipedia ({language})".replace("{language}", self.language)
def get_select(self):
return "?article{language} ?articleName{language}".replace("{language}", self.language)
def get_where(self):
return """OPTIONAL { ?article{language} schema:about ?item ;
schema:inLanguage "{language}" ;
schema:isPartOf <https://{language}.wikipedia.org/> ;
schema:name ?articleName{language} . }""".replace(
'{language}', self.language
)
def get_group_by(self):
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
key = "article{language}".replace("{language}", self.language)
return result.get(key)
class WDLabelAttribute(WDAttribute):
def get_select(self):
return "(group_concat(distinct ?{name}Label;separator=', ') as ?{name}Labels)".replace("{name}", self.name)
def get_where(self):
return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name)
def get_wikibase_label(self) -> str:
return "?{name} rdfs:label ?{name}Label .".replace("{name}", self.name)
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
return result.get(self.name + "Labels")
class WDURLAttribute(WDAttribute):
HTTP_WIKIMEDIA_IMAGE: str = "http://commons.wikimedia.org/wiki/Special:FilePath/"
def __init__(
self,
name: str,
url_id: str | None = None,
url_path_prefix: str | None = None,
kwargs: dict[str, t.Any] | None = None,
):
"""
:param url_id: ID matching one key in ``external_urls.json`` for
converting IDs to full URLs.
:param url_path_prefix: Path prefix if the values are of format
``account@domain``. If provided, value are rewritten to
``https://<domain><url_path_prefix><account>``. For example::
WDURLAttribute('P4033', url_path_prefix='/@')
Adds Property `P4033 <https://www.wikidata.org/wiki/Property:P4033>`_
to the wikidata query. This field might return for example
``libreoffice@fosstodon.org`` and the URL built from this is then:
- account: ``libreoffice``
- domain: ``fosstodon.org``
- result url: https://fosstodon.org/@libreoffice
"""
super().__init__(name)
self.url_id: str | None = url_id
self.url_path_prefix: str | None = url_path_prefix
self.kwargs: dict[str, t.Any] = kwargs or {}
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name + "s")
if not value:
return None
value = value.split(",")[0]
if self.url_id:
url_id = self.url_id
if value.startswith(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE):
value = value[len(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE) :]
url_id = "wikimedia_image"
return get_external_url(url_id, value)
if self.url_path_prefix:
[account, domain] = [x.strip("@ ") for x in value.rsplit("@", 1)]
return f"https://{domain}{self.url_path_prefix}{account}"
return value
class WDGeoAttribute(WDAttribute):
def get_label(self, language: str):
return "OpenStreetMap"
def get_select(self):
return "?{name}Lat ?{name}Long".replace("{name}", self.name)
def get_where(self):
return """OPTIONAL { ?item p:{name}/psv:{name} [
wikibase:geoLatitude ?{name}Lat ;
wikibase:geoLongitude ?{name}Long ] }""".replace(
'{name}', self.name
)
def get_group_by(self):
return self.get_select()
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
latitude: str | None = result.get(self.name + "Lat")
longitude: str | None = result.get(self.name + "Long")
if latitude and longitude:
return latitude + " " + longitude
return None
def get_geo_url(self, result: dict[str, t.Any], osm_zoom: int = 19) -> str | None:
latitude: str | None = result.get(self.name + "Lat")
longitude: str | None = result.get(self.name + "Long")
if latitude and longitude:
return get_earth_coordinates_url(latitude, longitude, osm_zoom)
return None
class WDImageAttribute(WDURLAttribute):
def __init__(self, name: str, url_id: str | None = None, priority: int = 100):
super().__init__(name, url_id)
self.priority: int = priority
class WDDateAttribute(WDAttribute):
def get_select(self):
return "?{name} ?{name}timePrecision ?{name}timeZone ?{name}timeCalendar".replace("{name}", self.name)
def get_where(self):
# To remove duplicate, add
# FILTER NOT EXISTS { ?item p:{name}/psv:{name}/wikibase:timeValue ?{name}bis FILTER (?{name}bis < ?{name}) }
# this filter is too slow, so the response function ignore duplicate results
# (see the seen_entities variable)
return """OPTIONAL { ?item p:{name}/psv:{name} [
wikibase:timeValue ?{name} ;
wikibase:timePrecision ?{name}timePrecision ;
wikibase:timeTimezone ?{name}timeZone ;
wikibase:timeCalendarModel ?{name}timeCalendar ] . }
hint:Prior hint:rangeSafe true;""".replace(
'{name}', self.name
)
def get_group_by(self):
return self.get_select()
def format_8(self, value: str, locale: str) -> str: # pylint: disable=unused-argument
# precision: less than a year
return value
def format_9(self, value: str, locale: str) -> str:
year = int(value)
# precision: year
if year < 1584:
if year < 0:
return str(year - 1)
return str(year)
timestamp = isoparse(value)
return format_date(timestamp, format="yyyy", locale=locale)
def format_10(self, value: str, locale: str) -> str:
# precision: month
timestamp = isoparse(value)
return format_date(timestamp, format="MMMM y", locale=locale)
def format_11(self, value: str, locale: str) -> str:
# precision: day
timestamp = isoparse(value)
return format_date(timestamp, format="full", locale=locale)
def format_13(self, value: str, locale: str) -> str:
timestamp = isoparse(value)
# precision: minute
return (
get_datetime_format("medium", locale=locale)
.replace("'", "")
.replace("{0}", format_time(timestamp, "full", tzinfo=None, locale=locale))
.replace("{1}", format_date(timestamp, "short", locale=locale))
)
def format_14(self, value: str, locale: str) -> str:
# precision: second.
return format_datetime(isoparse(value), format="full", locale=locale)
DATE_FORMAT: dict[str, tuple[str, int]] = {
"0": ("format_8", 1000000000),
"1": ("format_8", 100000000),
"2": ("format_8", 10000000),
"3": ("format_8", 1000000),
"4": ("format_8", 100000),
"5": ("format_8", 10000),
"6": ("format_8", 1000),
"7": ("format_8", 100),
"8": ("format_8", 10),
"9": ("format_9", 1), # year
"10": ("format_10", 1), # month
"11": ("format_11", 0), # day
"12": ("format_13", 0), # hour (not supported by babel, display minute)
"13": ("format_13", 0), # minute
"14": ("format_14", 0), # second
}
def get_str(self, result: dict[str, t.Any], language: str) -> str | None:
value: str | None = result.get(self.name)
if value == "" or value is None:
return None
_p: str = result.get(self.name + "timePrecision") or "1"
date_format = WDDateAttribute.DATE_FORMAT.get(_p)
if date_format is not None:
format_method = getattr(self, date_format[0])
precision: int = date_format[1]
try:
if precision >= 1:
_t = value.split("-")
if value.startswith("-"):
value = "-" + _t[1]
else:
value = _t[0]
return format_method(value, language)
except Exception: # pylint: disable=broad-except
return value
return value
WDAttrType = (
WDAttribute
| WDAmountAttribute
| WDArticle
| WDLabelAttribute
| WDURLAttribute
| WDGeoAttribute
| WDImageAttribute
| WDDateAttribute
)
WDAttrList = list[WDAttrType]
_WIKIDATA_PROPERTIES_OVERRIDE: WikiDataPropertiesType = {
"P434": "MusicBrainz",
"P435": "MusicBrainz",
"P436": "MusicBrainz",
"P966": "MusicBrainz",
"P345": "IMDb",
"P2397": "YouTube",
"P1651": "YouTube",
"P2002": "Twitter",
"P2013": "Facebook",
"P2003": "Instagram",
"P4033": "Mastodon",
"P11947": "Lemmy",
"P12622": "PeerTube",
}
"""Custom hardcoded property names for some Wikidata IDs. Only used if the
name Wikidata assigned isn't user-friendly."""
def fetch_properties(units: dict[str, WikiDataUnitType]) -> WikiDataPropertiesType:
properties = _WIKIDATA_PROPERTIES_OVERRIDE
# WIKIDATA_PROPERTIES : add unit symbols
for k, v in units.items():
properties[k] = v["symbol"]
# WIKIDATA_PROPERTIES : add property labels
wikidata_property_names: list[str] = []
for attribute in get_attributes("en"):
if type(attribute) in (WDAttribute, WDAmountAttribute, WDURLAttribute, WDDateAttribute, WDLabelAttribute):
if attribute.name not in properties:
wikidata_property_names.append("wd:" + attribute.name)
query = QUERY_PROPERTY_NAMES.replace("%ATTRIBUTES%", " ".join(wikidata_property_names))
kwargs: dict[str, t.Any] = {"timeout": 60}
json_response = send_wikidata_query(query, **kwargs)
for result in json_response.get("results", {}).get("bindings", {}):
name_field = result.get("name")
if not name_field:
continue
name = name_field["value"]
lang = name_field["xml:lang"]
entity_id = result["item"]["value"].replace("http://www.wikidata.org/entity/", "")
if name:
prop = properties.get(entity_id) or {}
prop[lang] = name.capitalize() # pyright: ignore[reportIndexIssue]
properties[entity_id] = prop
else:
properties[entity_id] = name.capitalize()
return properties
def get_attributes(language: str):
# pylint: disable=too-many-statements
attributes: WDAttrList = []
def add_value(name: str):
attributes.append(WDAttribute(name))
def add_amount(name: str):
attributes.append(WDAmountAttribute(name))
def add_label(name: str):
attributes.append(WDLabelAttribute(name))
def add_url(name: str, url_id: str | None = None, url_path_prefix: str | None = None, **kwargs: dict[str, t.Any]):
attributes.append(WDURLAttribute(name, url_id, url_path_prefix, kwargs))
def add_image(name: str, url_id: str | None = None, priority: int = 1):
attributes.append(WDImageAttribute(name, url_id, priority))
def add_date(name: str):
attributes.append(WDDateAttribute(name))
# Dates
for p in [
"P571", # inception date
"P576", # dissolution date
"P580", # start date
"P582", # end date
"P569", # date of birth
"P570", # date of death
"P619", # date of spacecraft launch
"P620",
]: # date of spacecraft landing
add_date(p)
for p in [
"P27", # country of citizenship
"P495", # country of origin
"P17", # country
"P159",
]: # headquarters location
add_label(p)
# Places
for p in [
"P36", # capital
"P35", # head of state
"P6", # head of government
"P122", # basic form of government
"P37",
]: # official language
add_label(p)
add_value("P1082") # population
add_amount("P2046") # area
add_amount("P281") # postal code
add_label("P38") # currency
add_amount("P2048") # height (building)
# Media
for p in [
"P400", # platform (videogames, computing)
"P50", # author
"P170", # creator
"P57", # director
"P175", # performer
"P178", # developer
"P162", # producer
"P176", # manufacturer
"P58", # screenwriter
"P272", # production company
"P264", # record label
"P123", # publisher
"P449", # original network
"P750", # distributed by
"P86",
]: # composer
add_label(p)
add_date("P577") # publication date
add_label("P136") # genre (music, film, artistic...)
add_label("P364") # original language
add_value("P212") # ISBN-13
add_value("P957") # ISBN-10
add_label("P275") # copyright license
add_label("P277") # programming language
add_value("P348") # version
add_label("P840") # narrative location
# Languages
add_value("P1098") # number of speakers
add_label("P282") # writing system
add_label("P1018") # language regulatory body
add_value("P218") # language code (ISO 639-1)
# Other
add_label("P169") # ceo
add_label("P112") # founded by
add_label("P1454") # legal form (company, organization)
add_label("P137") # operator (service, facility, ...)
add_label("P1029") # crew members (tripulation)
add_label("P225") # taxon name
add_value("P274") # chemical formula
add_label("P1346") # winner (sports, contests, ...)
add_value("P1120") # number of deaths
add_value("P498") # currency code (ISO 4217)
# URL
kwargs: dict[str, t.Any] = {"official": True}
add_url("P856", **kwargs) # official website
attributes.append(WDArticle(language)) # wikipedia (user language)
if not language.startswith("en"):
attributes.append(WDArticle("en")) # wikipedia (english)
add_url("P1324") # source code repository
add_url("P1581") # blog
add_url("P434", url_id="musicbrainz_artist")
add_url("P435", url_id="musicbrainz_work")
add_url("P436", url_id="musicbrainz_release_group")
add_url("P966", url_id="musicbrainz_label")
add_url("P345", url_id="imdb_id")
add_url("P2397", url_id="youtube_channel")
add_url("P1651", url_id="youtube_video")
add_url("P2002", url_id="twitter_profile")
add_url("P2013", url_id="facebook_profile")
add_url("P2003", url_id="instagram_profile")
# Fediverse
add_url("P4033", url_path_prefix="/@") # Mastodon user
add_url("P11947", url_path_prefix="/c/") # Lemmy community
add_url("P12622", url_path_prefix="/c/") # PeerTube channel
# Map
attributes.append(WDGeoAttribute("P625"))
# Image
add_image("P15", priority=1, url_id="wikimedia_image") # route map
add_image("P242", priority=2, url_id="wikimedia_image") # locator map
add_image("P154", priority=3, url_id="wikimedia_image") # logo
add_image("P18", priority=4, url_id="wikimedia_image") # image
add_image("P41", priority=5, url_id="wikimedia_image") # flag
add_image("P2716", priority=6, url_id="wikimedia_image") # collage
add_image("P2910", priority=7, url_id="wikimedia_image") # icon
return attributes
def get_label_for_entity(entity_id: str, language: str) -> str:
# only import properties locally to prevent cyclic import when initializing WIKIDATA_PROPERTIES
from searx.data import WIKIDATA_PROPERTIES # pylint: disable=import-outside-toplevel
property_name = WIKIDATA_PROPERTIES.get(entity_id)
if property_name is None:
return entity_id
if isinstance(property_name, str):
return property_name
if name := property_name.get(language):
return name
if name := property_name.get(language.split("-")[0]):
return name
if name := property_name.get("en"):
return name
return entity_id

View File

@@ -11,7 +11,7 @@ __all__ = ["convert_from_si", "convert_to_si", "symbol_to_si"]
import collections import collections
from searx import data from searx import data
from searx.engines import wikidata from searx.wikidata import send_wikidata_query
class Beaufort: class Beaufort:
@@ -142,7 +142,6 @@ def units_by_si_name(si_name):
# build the catalog .. # build the catalog ..
for item in symbol_to_si(): for item in symbol_to_si():
item_si_name = item[pos_si_name] item_si_name = item[pos_si_name]
item_symbol = item[pos_symbol] item_symbol = item[pos_symbol]
@@ -266,14 +265,13 @@ ORDER BY ?item DESC(?rank) ?symbol
""" """
def fetch_units(): def fetch_units() -> dict[str, data.WikiDataUnitType]:
"""Fetch units from Wikidata. Function is used to update persistence of """Fetch units from Wikidata. Function is used to update persistence of
:py:obj:`searx.data.WIKIDATA_UNITS`.""" :py:obj:`searx.data.WIKIDATA_UNITS`."""
results = collections.OrderedDict() results = collections.OrderedDict()
response = wikidata.send_wikidata_query(SARQL_REQUEST) response = send_wikidata_query(SARQL_REQUEST)
for unit in response['results']['bindings']: for unit in response['results']['bindings']:
symbol = unit['symbol']['value'] symbol = unit['symbol']['value']
name = unit['item']['value'].rsplit('/', 1)[1] name = unit['item']['value'].rsplit('/', 1)[1]
si_name = unit.get('tosiUnit', {}).get('value', '') si_name = unit.get('tosiUnit', {}).get('value', '')

View File

@@ -16,6 +16,7 @@ import json
from searx.locales import LOCALE_NAMES, locales_initialize from searx.locales import LOCALE_NAMES, locales_initialize
from searx.engines import wikidata, set_loggers from searx.engines import wikidata, set_loggers
from searx.data.currencies import CurrenciesDB from searx.data.currencies import CurrenciesDB
from searx.wikidata import send_wikidata_query
set_loggers(wikidata, 'wikidata') set_loggers(wikidata, 'wikidata')
locales_initialize() locales_initialize()
@@ -90,7 +91,7 @@ def add_currency_label(db, label, iso4217, language):
def wikidata_request_result_iterator(request): def wikidata_request_result_iterator(request):
result = wikidata.send_wikidata_query(request.replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL), timeout=20) result = send_wikidata_query(request.replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL), timeout=20)
if result is not None: if result is not None:
yield from result['results']['bindings'] yield from result['results']['bindings']

View File

@@ -1,7 +1,7 @@
#!/usr/bin/env python #!/usr/bin/env python
# SPDX-License-Identifier: AGPL-3.0-or-later # SPDX-License-Identifier: AGPL-3.0-or-later
"""Fetch website description from websites and from """Fetch website description from websites and from
:origin:`searx/engines/wikidata.py` engine. :origin:`searx/wikidata.py`.
Output file: :origin:`searx/data/engine_descriptions.json`. Output file: :origin:`searx/data/engine_descriptions.json`.
@@ -17,6 +17,7 @@ from lxml.html import fromstring
import searx.engines import searx.engines
from searx.engines import wikidata, set_loggers from searx.engines import wikidata, set_loggers
from searx.wikidata import send_wikidata_query
from searx.utils import extract_text from searx.utils import extract_text
from searx.locales import LOCALE_NAMES, locales_initialize, match_locale from searx.locales import LOCALE_NAMES, locales_initialize, match_locale
from searx import searx_dir from searx import searx_dir
@@ -200,7 +201,6 @@ def initialize():
locale2lang = {"nl-BE": "nl"} locale2lang = {"nl-BE": "nl"}
for sxng_ui_lang in LOCALE_NAMES: for sxng_ui_lang in LOCALE_NAMES:
sxng_ui_alias = locale2lang.get(sxng_ui_lang, sxng_ui_lang) sxng_ui_alias = locale2lang.get(sxng_ui_lang, sxng_ui_lang)
wiki_lang = None wiki_lang = None
@@ -225,7 +225,7 @@ def initialize():
def fetch_wikidata_descriptions(): def fetch_wikidata_descriptions():
print("Fetching wikidata descriptions") print("Fetching wikidata descriptions")
searx.network.set_timeout_for_thread(60) searx.network.set_timeout_for_thread(60)
result = wikidata.send_wikidata_query( result = send_wikidata_query(
SPARQL_DESCRIPTION.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL) SPARQL_DESCRIPTION.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL)
) )
if not result: if not result:
@@ -249,7 +249,7 @@ def fetch_wikidata_descriptions():
def fetch_wikipedia_descriptions(): def fetch_wikipedia_descriptions():
print("Fetching wikipedia descriptions") print("Fetching wikipedia descriptions")
result = wikidata.send_wikidata_query( result = send_wikidata_query(
SPARQL_WIKIPEDIA_ARTICLE.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL) SPARQL_WIKIPEDIA_ARTICLE.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL)
) )
if not result: if not result:
@@ -307,7 +307,6 @@ def fetch_website_description(engine_name: str, website: str):
previous_count: int = 0 previous_count: int = 0
for lang in languages: for lang in languages:
if lang in descriptions[engine_name]: if lang in descriptions[engine_name]:
continue continue

View File

@@ -50,6 +50,7 @@ from searx.engines import wikidata, set_loggers
from searx.sxng_locales import sxng_locales from searx.sxng_locales import sxng_locales
from searx.engines.openstreetmap import get_key_rank, VALUE_TO_LINK from searx.engines.openstreetmap import get_key_rank, VALUE_TO_LINK
from searx.data import data_dir from searx.data import data_dir
from searx.wikidata import send_wikidata_query
DATA_FILE = data_dir / 'osm_keys_tags.json' DATA_FILE = data_dir / 'osm_keys_tags.json'
@@ -102,7 +103,7 @@ def get_preset_keys():
def get_keys(): def get_keys():
results = get_preset_keys() results = get_preset_keys()
response = wikidata.send_wikidata_query(SPARQL_KEYS_REQUEST) response = send_wikidata_query(SPARQL_KEYS_REQUEST)
for key in response['results']['bindings']: for key in response['results']['bindings']:
keys = key['key']['value'].split(':')[1:] keys = key['key']['value'].split(':')[1:]
@@ -148,7 +149,7 @@ def get_keys():
def get_tags(): def get_tags():
results = collections.OrderedDict() results = collections.OrderedDict()
response = wikidata.send_wikidata_query(SPARQL_TAGS_REQUEST) response = send_wikidata_query(SPARQL_TAGS_REQUEST)
for tag in response['results']['bindings']: for tag in response['results']['bindings']:
tag_names = tag['tag']['value'].split(':')[1].split('=') tag_names = tag['tag']['value'].split(':')[1].split('=')
if len(tag_names) == 2: if len(tag_names) == 2:
@@ -204,7 +205,6 @@ def optimize_keys(data):
if __name__ == '__main__': if __name__ == '__main__':
set_timeout_for_thread(60) set_timeout_for_thread(60)
result = { result = {
'keys': optimize_keys(get_keys()), 'keys': optimize_keys(get_keys()),

View File

@@ -0,0 +1,29 @@
#!/usr/bin/env python
# SPDX-License-Identifier: AGPL-3.0-or-later
"""Fetch units and property names from :origin:`searx/engines/wikidata.py` engine.
Output files: (:origin:`CI Update data <.github/workflows/data-update.yml>`).
- :origin:`searx/data/wikidata_units.json`
- :origin:`searx/data/wikidata_properties.json`
"""
import json
from searx.engines import wikidata, set_loggers
from searx.data import data_dir
from searx.wikidata_properties import fetch_properties
from searx.wikidata_units import fetch_units
UNITS_DATA_FILE = data_dir / 'wikidata_units.json'
PROPERTIES_DATA_FILE = data_dir / 'wikidata_properties.json'
set_loggers(wikidata, 'wikidata')
if __name__ == '__main__':
units = fetch_units()
with UNITS_DATA_FILE.open('w', encoding="utf8") as f:
json.dump(units, f, indent=4, sort_keys=True, ensure_ascii=False)
properties = fetch_properties(units)
with PROPERTIES_DATA_FILE.open('w', encoding="utf8") as f:
json.dump(properties, f, indent=4, sort_keys=True, ensure_ascii=False)

View File

@@ -1,22 +0,0 @@
#!/usr/bin/env python
# SPDX-License-Identifier: AGPL-3.0-or-later
"""Fetch units from :origin:`searx/engines/wikidata.py` engine.
Output file: :origin:`searx/data/wikidata_units.json` (:origin:`CI Update data
... <.github/workflows/data-update.yml>`).
"""
import json
from searx.engines import wikidata, set_loggers
from searx.data import data_dir
from searx.wikidata_units import fetch_units
DATA_FILE = data_dir / 'wikidata_units.json'
set_loggers(wikidata, 'wikidata')
if __name__ == '__main__':
with DATA_FILE.open('w', encoding="utf8") as f:
json.dump(fetch_units(), f, indent=4, sort_keys=True, ensure_ascii=False)

View File

@@ -26,7 +26,8 @@ data.all() {
build_msg DATA "update searx/data/ahmia_blacklist.txt" build_msg DATA "update searx/data/ahmia_blacklist.txt"
python searxng_extra/update/update_ahmia_blacklist.py python searxng_extra/update/update_ahmia_blacklist.py
build_msg DATA "update searx/data/wikidata_units.json" build_msg DATA "update searx/data/wikidata_units.json"
python searxng_extra/update/update_wikidata_units.py build_msg DATA "update searx/data/wikidata_properties.json"
python searxng_extra/update/update_wikidata.py
build_msg DATA "update searx/data/currencies.json" build_msg DATA "update searx/data/currencies.json"
python searxng_extra/update/update_currencies.py python searxng_extra/update/update_currencies.py
build_msg DATA "update searx/data/external_bangs.json" build_msg DATA "update searx/data/external_bangs.json"