From 2b1c88c54c6900a3684a6733b9749e91ecf7de4b Mon Sep 17 00:00:00 2001 From: Bnyro Date: Fri, 14 Aug 2026 23:01:19 +0200 Subject: [PATCH] [refactor] wikidata: cache wikidata properties in searxng data --- .github/workflows/data-update.yml | 2 +- docs/dev/quickstart.rst | 4 +- docs/dev/searxng_extra/update.rst | 6 +- searx/data/__init__.py | 12 +- searx/engines/openstreetmap.py | 6 +- searx/engines/wikidata.py | 641 +----------------- searx/wikidata.py | 35 + searx/wikidata_properties.py | 580 ++++++++++++++++ searx/wikidata_units.py | 8 +- searxng_extra/update/update_currencies.py | 3 +- .../update/update_engine_descriptions.py | 9 +- searxng_extra/update/update_osm_keys_tags.py | 6 +- searxng_extra/update/update_wikidata.py | 29 + searxng_extra/update/update_wikidata_units.py | 22 - utils/lib_sxng_data.sh | 3 +- 15 files changed, 695 insertions(+), 671 deletions(-) create mode 100644 searx/wikidata.py create mode 100644 searx/wikidata_properties.py create mode 100755 searxng_extra/update/update_wikidata.py delete mode 100755 searxng_extra/update/update_wikidata_units.py diff --git a/.github/workflows/data-update.yml b/.github/workflows/data-update.yml index 58e807e7a..e3e04ee95 100644 --- a/.github/workflows/data-update.yml +++ b/.github/workflows/data-update.yml @@ -31,7 +31,7 @@ jobs: - update_external_bangs.py - update_firefox_version.py - update_engine_traits.py - - update_wikidata_units.py + - update_wikidata.py - update_engine_descriptions.py permissions: diff --git a/docs/dev/quickstart.rst b/docs/dev/quickstart.rst index 3cf956c06..9ed41cace 100644 --- a/docs/dev/quickstart.rst +++ b/docs/dev/quickstart.rst @@ -80,8 +80,8 @@ same environment, here are a few examples:: # to test one of the update scripts (dev.env)$ searxng_extra/update/update_engine_traits.py --help - # to test the update of the wikidata units - (dev.env)$ searxng_extra/update/update_wikidata_units.py + # to test the update of the wikidata units and property names + (dev.env)$ searxng_extra/update/update_wikidata.py .. sidebar:: further read diff --git a/docs/dev/searxng_extra/update.rst b/docs/dev/searxng_extra/update.rst index 000085970..cb808273c 100644 --- a/docs/dev/searxng_extra/update.rst +++ b/docs/dev/searxng_extra/update.rst @@ -90,10 +90,10 @@ Scripts to update static data in :origin:`searx/data/` :members: -``update_wikidata_units.py`` +``update_wikidata.py`` ============================ -:origin:`[source] ` +:origin:`[source] ` -.. automodule:: searxng_extra.update.update_wikidata_units +.. automodule:: searxng_extra.update.update_wikidata :members: diff --git a/searx/data/__init__.py b/searx/data/__init__.py index fc3031680..ff423622a 100644 --- a/searx/data/__init__.py +++ b/searx/data/__init__.py @@ -32,6 +32,13 @@ class WikiDataUnitType(t.TypedDict): to_si_factor: float +WikiDataPropertyNameType = str | dict[str, str] +"""Name of a Wikidata property. Can be either the plain name or a dictionary of +language code to property name, e.g. ``{"en": "Date of birth"}``.""" +WikiDataPropertiesType = dict[str, WikiDataPropertyNameType] +"""Dictionary from wikidata property ID to property name.""" + + class LocalesType(t.TypedDict): """Data structure of an item in ``locales.json``""" @@ -41,6 +48,7 @@ class LocalesType(t.TypedDict): USER_AGENTS: UserAgentType WIKIDATA_UNITS: dict[str, WikiDataUnitType] +WIKIDATA_PROPERTIES: WikiDataPropertiesType TRACKER_PATTERNS: TrackerPatternsDB LOCALES: LocalesType CURRENCIES: CurrenciesDB @@ -52,11 +60,12 @@ ENGINE_DESCRIPTIONS: dict[str, dict[str, t.Any]] ENGINE_TRAITS: dict[str, dict[str, t.Any]] -lazy_globals = { +lazy_globals: dict[str, t.Any] = { "CURRENCIES": CurrenciesDB(), "USER_AGENTS": None, "EXTERNAL_URLS": None, "WIKIDATA_UNITS": None, + "WIKIDATA_PROPERTIES": None, "EXTERNAL_BANGS": None, "OSM_KEYS_TAGS": None, "ENGINE_DESCRIPTIONS": None, @@ -69,6 +78,7 @@ data_json_files = { "USER_AGENTS": "useragents.json", "EXTERNAL_URLS": "external_urls.json", "WIKIDATA_UNITS": "wikidata_units.json", + "WIKIDATA_PROPERTIES": "wikidata_properties.json", "EXTERNAL_BANGS": "external_bangs.json", "OSM_KEYS_TAGS": "osm_keys_tags.json", "ENGINE_DESCRIPTIONS": "engine_descriptions.json", diff --git a/searx/engines/openstreetmap.py b/searx/engines/openstreetmap.py index e62bd205a..c76f05aa3 100644 --- a/searx/engines/openstreetmap.py +++ b/searx/engines/openstreetmap.py @@ -10,7 +10,8 @@ from flask_babel import gettext from searx.data import OSM_KEYS_TAGS, CURRENCIES from searx.external_urls import get_external_url -from searx.engines.wikidata import send_wikidata_query, sparql_string_escape, get_thumbnail +from searx.wikidata import send_wikidata_query +from searx.engines.wikidata import sparql_string_escape, get_thumbnail from searx.result_types import EngineResults # about @@ -290,7 +291,8 @@ def get_title_address(result): 'house_number': address_raw.get('house_number'), 'road': address_raw.get('road'), 'locality': address_raw.get( - 'city', address_raw.get('town', address_raw.get('village')) # noqa + 'city', + address_raw.get('town', address_raw.get('village')), # noqa ), # noqa 'postcode': address_raw.get('postcode'), 'country': address_raw.get('country'), diff --git a/searx/engines/wikidata.py b/searx/engines/wikidata.py index ffb5dc136..6b8181eb4 100644 --- a/searx/engines/wikidata.py +++ b/searx/engines/wikidata.py @@ -7,24 +7,29 @@ Some implementations are shared from :ref:`wikipedia engine`. import typing as t -import os from hashlib import md5 from urllib.parse import urlencode, unquote from json import loads -from dateutil.parser import isoparse -from babel.dates import format_datetime, format_date, format_time, get_datetime_format -from searx.enginelib import EngineCache -from searx.data import WIKIDATA_UNITS from searx.network import post, get -from searx.utils import searxng_useragent, get_string_replaces_function -from searx.external_urls import get_external_url, get_earth_coordinates_url, area_to_osm_zoom +from searx.utils import get_string_replaces_function +from searx.external_urls import area_to_osm_zoom from searx.engines.wikipedia import ( fetch_wikimedia_traits, get_wiki_params, ) from searx.enginelib.traits import EngineTraits +from searx.wikidata_properties import ( + QUERY_TEMPLATE, + WDArticle, + WDAttrList, + WDGeoAttribute, + WDImageAttribute, + WDURLAttribute, + get_attributes, +) +from searx.wikidata import SPARQL_ENDPOINT_URL, SPARQL_EXPLAIN_URL, get_wikidata_headers if t.TYPE_CHECKING: from searx.extended_types import SXNG_Response @@ -47,78 +52,6 @@ display_type = ["infobox"] one will add a hit to the result list. The first one will show a hit in the info box. Both values can be set, or one of the two can be set.""" -CACHE: EngineCache -"""Persistent (SQLite) key/value cache that deletes its values after ``expire`` -seconds.""" - -# SPARQL -SPARQL_ENDPOINT_URL = "https://query.wikidata.org/sparql" -SPARQL_EXPLAIN_URL = "https://query.wikidata.org/bigdata/namespace/wdq/sparql?explain" -WDPType = dict[str | tuple[str, str], str] -WIKIDATA_PROPERTIES: WDPType = { - "P434": "MusicBrainz", - "P435": "MusicBrainz", - "P436": "MusicBrainz", - "P966": "MusicBrainz", - "P345": "IMDb", - "P2397": "YouTube", - "P1651": "YouTube", - "P2002": "Twitter", - "P2013": "Facebook", - "P2003": "Instagram", - "P4033": "Mastodon", - "P11947": "Lemmy", - "P12622": "PeerTube", -} - -# SERVICE wikibase:mwapi : https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual/MWAPI -# SERVICE wikibase:label: https://en.wikibooks.org/wiki/SPARQL/SERVICE_-_Label#Manual_Label_SERVICE -# https://en.wikibooks.org/wiki/SPARQL/WIKIDATA_Precision,_Units_and_Coordinates -# https://www.mediawiki.org/wiki/Wikibase/Indexing/RDF_Dump_Format#Data_model -# optimization: -# * https://www.wikidata.org/wiki/Wikidata:SPARQL_query_service/query_optimization -# * https://github.com/blazegraph/database/wiki/QueryHints -QUERY_TEMPLATE = """ -SELECT ?item ?itemLabel ?itemDescription ?lat ?long %SELECT% -WHERE -{ - SERVICE wikibase:mwapi { - bd:serviceParam wikibase:endpoint "www.wikidata.org"; - wikibase:api "EntitySearch"; - wikibase:limit 1; - mwapi:search "%QUERY%"; - mwapi:language "%LANGUAGE%". - ?item wikibase:apiOutputItem mwapi:item. - } - hint:Prior hint:runFirst "true". - - %WHERE% - - SERVICE wikibase:label { - bd:serviceParam wikibase:language "%LANGUAGE%,en". - ?item rdfs:label ?itemLabel . - ?item schema:description ?itemDescription . - %WIKIBASE_LABELS% - } - -} -GROUP BY ?item ?itemLabel ?itemDescription ?lat ?long %GROUP_BY% -""" - -# Get the calendar names and the property names -QUERY_PROPERTY_NAMES = """ -SELECT ?item ?name -WHERE { - { - SELECT ?item - WHERE { ?item wdt:P279* wd:Q12132 } - } UNION { - VALUES ?item { %ATTRIBUTES% } - } - OPTIONAL { ?item rdfs:label ?name. } -} -""" - # see the property "dummy value" of https://www.wikidata.org/wiki/Q2013 (Wikidata) # hard coded here to avoid to an additional SPARQL request when the server starts DUMMY_ENTITY_URLS = set( @@ -130,357 +63,13 @@ DUMMY_ENTITY_URLS = set( # https://lists.w3.org/Archives/Public/public-rdf-dawg/2011OctDec/0175.html sparql_string_escape = get_string_replaces_function( # fmt: off - { - "\t": "\\\t", - "\n": "\\\n", - "\r": "\\\r", - "\b": "\\\b", - "\f": "\\\f", - "\"": "\\\"", - "\'": "\\\'", - "\\": "\\\\" - } + {"\t": "\\\t", "\n": "\\\n", "\r": "\\\r", "\b": "\\\b", "\f": "\\\f", "\"": "\\\"", "'": "\\'", "\\": "\\\\"} # fmt: on ) replace_http_by_https = get_string_replaces_function({"http:": "https:"}) -class WDAttribute: - - def __init__(self, name: str): - self.name: str = name - - def get_select(self): - return "(group_concat(distinct ?{name};separator=', ') as ?{name}s)".replace("{name}", self.name) - - def get_label(self, language: str): - return get_label_for_entity(self.name, language) - - def get_where(self): - return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name) - - def get_wikibase_label(self) -> str: - return "" - - def get_group_by(self) -> str: - return "" - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: # pylint: disable=unused-argument - return result.get(self.name + "s") - - def __repr__(self): - return "<" + str(type(self).__name__) + ":" + self.name + ">" - - -class WDAmountAttribute(WDAttribute): - def get_select(self) -> str: - return "?{name} ?{name}Unit".replace("{name}", self.name) - - def get_where(self): - return """ OPTIONAL { ?item p:{name} ?{name}Node . - ?{name}Node rdf:type wikibase:BestRank ; ps:{name} ?{name} . - OPTIONAL { ?{name}Node psv:{name}/wikibase:quantityUnit ?{name}Unit. } }""".replace( - '{name}', self.name - ) - - def get_group_by(self) -> str: - return self.get_select() - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - value: str | None = result.get(self.name) - unit: str | None = result.get(self.name + "Unit") - if unit is not None: - unit = unit.replace("http://www.wikidata.org/entity/", "") - return str(value) + " " + get_label_for_entity(unit, language) - return value - - -class WDArticle(WDAttribute): - - def __init__(self, language: str, kwargs: dict[str, t.Any] | None = None): - super().__init__("wikipedia") - self.language: str = language - self.kwargs: dict[str, t.Any] = kwargs or {} - - def get_label(self, language: str): - # language parameter is ignored - return "Wikipedia ({language})".replace("{language}", self.language) - - def get_select(self): - return "?article{language} ?articleName{language}".replace("{language}", self.language) - - def get_where(self): - return """OPTIONAL { ?article{language} schema:about ?item ; - schema:inLanguage "{language}" ; - schema:isPartOf ; - schema:name ?articleName{language} . }""".replace( - '{language}', self.language - ) - - def get_group_by(self): - return self.get_select() - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - key = "article{language}".replace("{language}", self.language) - return result.get(key) - - -class WDLabelAttribute(WDAttribute): - def get_select(self): - return "(group_concat(distinct ?{name}Label;separator=', ') as ?{name}Labels)".replace("{name}", self.name) - - def get_where(self): - return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name) - - def get_wikibase_label(self) -> str: - return "?{name} rdfs:label ?{name}Label .".replace("{name}", self.name) - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - return result.get(self.name + "Labels") - - -class WDURLAttribute(WDAttribute): - - HTTP_WIKIMEDIA_IMAGE: str = "http://commons.wikimedia.org/wiki/Special:FilePath/" - - def __init__( - self, - name: str, - url_id: str | None = None, - url_path_prefix: str | None = None, - kwargs: dict[str, t.Any] | None = None, - ): - """ - :param url_id: ID matching one key in ``external_urls.json`` for - converting IDs to full URLs. - - :param url_path_prefix: Path prefix if the values are of format - ``account@domain``. If provided, value are rewritten to - ``https://``. For example:: - - WDURLAttribute('P4033', url_path_prefix='/@') - - Adds Property `P4033 `_ - to the wikidata query. This field might return for example - ``libreoffice@fosstodon.org`` and the URL built from this is then: - - - account: ``libreoffice`` - - domain: ``fosstodon.org`` - - result url: https://fosstodon.org/@libreoffice - """ - - super().__init__(name) - self.url_id: str | None = url_id - self.url_path_prefix: str | None = url_path_prefix - self.kwargs: dict[str, t.Any] = kwargs or {} - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - value: str | None = result.get(self.name + "s") - if not value: - return None - - value = value.split(",")[0] - if self.url_id: - url_id = self.url_id - if value.startswith(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE): - value = value[len(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE) :] - url_id = "wikimedia_image" - return get_external_url(url_id, value) - - if self.url_path_prefix: - [account, domain] = [x.strip("@ ") for x in value.rsplit("@", 1)] - return f"https://{domain}{self.url_path_prefix}{account}" - - return value - - -class WDGeoAttribute(WDAttribute): - def get_label(self, language: str): - return "OpenStreetMap" - - def get_select(self): - return "?{name}Lat ?{name}Long".replace("{name}", self.name) - - def get_where(self): - return """OPTIONAL { ?item p:{name}/psv:{name} [ - wikibase:geoLatitude ?{name}Lat ; - wikibase:geoLongitude ?{name}Long ] }""".replace( - '{name}', self.name - ) - - def get_group_by(self): - return self.get_select() - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - latitude: str | None = result.get(self.name + "Lat") - longitude: str | None = result.get(self.name + "Long") - if latitude and longitude: - return latitude + " " + longitude - return None - - def get_geo_url(self, result: dict[str, t.Any], osm_zoom: int = 19) -> str | None: - latitude: str | None = result.get(self.name + "Lat") - longitude: str | None = result.get(self.name + "Long") - if latitude and longitude: - return get_earth_coordinates_url(latitude, longitude, osm_zoom) - return None - - -class WDImageAttribute(WDURLAttribute): - - def __init__(self, name: str, url_id: str | None = None, priority: int = 100): - super().__init__(name, url_id) - self.priority: int = priority - - -class WDDateAttribute(WDAttribute): - def get_select(self): - return "?{name} ?{name}timePrecision ?{name}timeZone ?{name}timeCalendar".replace("{name}", self.name) - - def get_where(self): - # To remove duplicate, add - # FILTER NOT EXISTS { ?item p:{name}/psv:{name}/wikibase:timeValue ?{name}bis FILTER (?{name}bis < ?{name}) } - # this filter is too slow, so the response function ignore duplicate results - # (see the seen_entities variable) - return """OPTIONAL { ?item p:{name}/psv:{name} [ - wikibase:timeValue ?{name} ; - wikibase:timePrecision ?{name}timePrecision ; - wikibase:timeTimezone ?{name}timeZone ; - wikibase:timeCalendarModel ?{name}timeCalendar ] . } - hint:Prior hint:rangeSafe true;""".replace( - '{name}', self.name - ) - - def get_group_by(self): - return self.get_select() - - def format_8(self, value: str, locale: str) -> str: # pylint: disable=unused-argument - # precision: less than a year - return value - - def format_9(self, value: str, locale: str) -> str: - year = int(value) - # precision: year - if year < 1584: - if year < 0: - return str(year - 1) - return str(year) - timestamp = isoparse(value) - return format_date(timestamp, format="yyyy", locale=locale) - - def format_10(self, value: str, locale: str) -> str: - # precision: month - timestamp = isoparse(value) - return format_date(timestamp, format="MMMM y", locale=locale) - - def format_11(self, value: str, locale: str) -> str: - # precision: day - timestamp = isoparse(value) - return format_date(timestamp, format="full", locale=locale) - - def format_13(self, value: str, locale: str) -> str: - timestamp = isoparse(value) - # precision: minute - return ( - get_datetime_format(format, locale=locale) - .replace("'", "") - .replace("{0}", format_time(timestamp, "full", tzinfo=None, locale=locale)) - .replace("{1}", format_date(timestamp, "short", locale=locale)) - ) - - def format_14(self, value: str, locale: str) -> str: - # precision: second. - return format_datetime(isoparse(value), format="full", locale=locale) - - DATE_FORMAT: dict[str, tuple[str, int]] = { - "0": ("format_8", 1000000000), - "1": ("format_8", 100000000), - "2": ("format_8", 10000000), - "3": ("format_8", 1000000), - "4": ("format_8", 100000), - "5": ("format_8", 10000), - "6": ("format_8", 1000), - "7": ("format_8", 100), - "8": ("format_8", 10), - "9": ("format_9", 1), # year - "10": ("format_10", 1), # month - "11": ("format_11", 0), # day - "12": ("format_13", 0), # hour (not supported by babel, display minute) - "13": ("format_13", 0), # minute - "14": ("format_14", 0), # second - } - - def get_str(self, result: dict[str, t.Any], language: str) -> str | None: - value: str | None = result.get(self.name) - if value == "" or value is None: - return None - _p: str = result.get(self.name + "timePrecision") or "1" - date_format = WDDateAttribute.DATE_FORMAT.get(_p) - if date_format is not None: - format_method = getattr(self, date_format[0]) - precision: int = date_format[1] - try: - if precision >= 1: - _t = value.split("-") - if value.startswith("-"): - value = "-" + _t[1] - else: - value = _t[0] - return format_method(value, language) - except Exception: # pylint: disable=broad-except - return value - return value - - -WDAttrType = ( - WDAttribute - | WDAmountAttribute - | WDArticle - | WDLabelAttribute - | WDURLAttribute - | WDGeoAttribute - | WDImageAttribute - | WDDateAttribute -) -WDAttrList = list[WDAttrType] - - -def get_headers() -> dict[str, str]: - # user agent: https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual#Query_limits - return { - "Accept": "application/sparql-results+json", - "User-Agent": f"wikidata engine - {searxng_useragent()}", - } - - -def get_label_for_entity(entity_id: str, language: str) -> str: - name = WIKIDATA_PROPERTIES.get(entity_id) - if name is None: - name = WIKIDATA_PROPERTIES.get((entity_id, language)) - if name is None: - name = WIKIDATA_PROPERTIES.get((entity_id, language.split("-")[0])) - if name is None: - name = WIKIDATA_PROPERTIES.get((entity_id, "en")) - if name is None: - name = entity_id - return name - - -def send_wikidata_query(query: str, method: str = "GET", **kwargs: dict[str, t.Any]) -> dict[str, t.Any]: - if method == "GET": - # query will be cached by wikidata - http_response = get(SPARQL_ENDPOINT_URL + "?" + urlencode({"query": query}), headers=get_headers(), **kwargs) - else: - # query won't be cached by wikidata - http_response = post(SPARQL_ENDPOINT_URL, data={"query": query}, headers=get_headers(), **kwargs) - if http_response.status_code != 200: - logger.debug("SPARQL endpoint error %s", http_response.content.decode()) - logger.debug("request time %s", str(http_response.elapsed)) - http_response.raise_for_status() - return loads(http_response.content.decode()) - - def request(query: str, params: "OnlineParams") -> None: attributes: WDAttrList @@ -491,7 +80,7 @@ def request(query: str, params: "OnlineParams") -> None: params["method"] = "POST" params["url"] = SPARQL_ENDPOINT_URL params["data"] = {"query": query} - params["headers"] = get_headers() + params["headers"] = get_wikidata_headers() # additional parameters (not a part of OnlineParams) params["language"] = eng_tag # type: ignore @@ -584,7 +173,6 @@ def get_results( for attribute in attributes: value: str | None = attribute.get_str(attribute_result, language) if value is not None and value != "": - if isinstance(attribute, (WDURLAttribute, WDArticle)): # get_select() method : there is group_concat(distinct ...;separator=", ") # split the value here @@ -670,212 +258,15 @@ def get_query(query: str, language: str) -> tuple[str, WDAttrList]: return query, attributes -def get_attributes(language: str): - # pylint: disable=too-many-statements - attributes: WDAttrList = [] - - def add_value(name: str): - attributes.append(WDAttribute(name)) - - def add_amount(name: str): - attributes.append(WDAmountAttribute(name)) - - def add_label(name: str): - attributes.append(WDLabelAttribute(name)) - - def add_url(name: str, url_id: str | None = None, url_path_prefix: str | None = None, **kwargs: dict[str, t.Any]): - attributes.append(WDURLAttribute(name, url_id, url_path_prefix, kwargs)) - - def add_image(name: str, url_id: str | None = None, priority: int = 1): - attributes.append(WDImageAttribute(name, url_id, priority)) - - def add_date(name: str): - attributes.append(WDDateAttribute(name)) - - # Dates - for p in [ - "P571", # inception date - "P576", # dissolution date - "P580", # start date - "P582", # end date - "P569", # date of birth - "P570", # date of death - "P619", # date of spacecraft launch - "P620", - ]: # date of spacecraft landing - add_date(p) - - for p in [ - "P27", # country of citizenship - "P495", # country of origin - "P17", # country - "P159", - ]: # headquarters location - add_label(p) - - # Places - for p in [ - "P36", # capital - "P35", # head of state - "P6", # head of government - "P122", # basic form of government - "P37", - ]: # official language - add_label(p) - - add_value("P1082") # population - add_amount("P2046") # area - add_amount("P281") # postal code - add_label("P38") # currency - add_amount("P2048") # height (building) - - # Media - for p in [ - "P400", # platform (videogames, computing) - "P50", # author - "P170", # creator - "P57", # director - "P175", # performer - "P178", # developer - "P162", # producer - "P176", # manufacturer - "P58", # screenwriter - "P272", # production company - "P264", # record label - "P123", # publisher - "P449", # original network - "P750", # distributed by - "P86", - ]: # composer - add_label(p) - - add_date("P577") # publication date - add_label("P136") # genre (music, film, artistic...) - add_label("P364") # original language - add_value("P212") # ISBN-13 - add_value("P957") # ISBN-10 - add_label("P275") # copyright license - add_label("P277") # programming language - add_value("P348") # version - add_label("P840") # narrative location - - # Languages - add_value("P1098") # number of speakers - add_label("P282") # writing system - add_label("P1018") # language regulatory body - add_value("P218") # language code (ISO 639-1) - - # Other - add_label("P169") # ceo - add_label("P112") # founded by - add_label("P1454") # legal form (company, organization) - add_label("P137") # operator (service, facility, ...) - add_label("P1029") # crew members (tripulation) - add_label("P225") # taxon name - add_value("P274") # chemical formula - add_label("P1346") # winner (sports, contests, ...) - add_value("P1120") # number of deaths - add_value("P498") # currency code (ISO 4217) - - # URL - kwargs: dict[str, t.Any] = {"official": True} - add_url("P856", **kwargs) # official website - attributes.append(WDArticle(language)) # wikipedia (user language) - if not language.startswith("en"): - attributes.append(WDArticle("en")) # wikipedia (english) - - add_url("P1324") # source code repository - add_url("P1581") # blog - add_url("P434", url_id="musicbrainz_artist") - add_url("P435", url_id="musicbrainz_work") - add_url("P436", url_id="musicbrainz_release_group") - add_url("P966", url_id="musicbrainz_label") - add_url("P345", url_id="imdb_id") - add_url("P2397", url_id="youtube_channel") - add_url("P1651", url_id="youtube_video") - add_url("P2002", url_id="twitter_profile") - add_url("P2013", url_id="facebook_profile") - add_url("P2003", url_id="instagram_profile") - - # Fediverse - add_url("P4033", url_path_prefix="/@") # Mastodon user - add_url("P11947", url_path_prefix="/c/") # Lemmy community - add_url("P12622", url_path_prefix="/c/") # PeerTube channel - - # Map - attributes.append(WDGeoAttribute("P625")) - - # Image - add_image("P15", priority=1, url_id="wikimedia_image") # route map - add_image("P242", priority=2, url_id="wikimedia_image") # locator map - add_image("P154", priority=3, url_id="wikimedia_image") # logo - add_image("P18", priority=4, url_id="wikimedia_image") # image - add_image("P41", priority=5, url_id="wikimedia_image") # flag - add_image("P2716", priority=6, url_id="wikimedia_image") # collage - add_image("P2910", priority=7, url_id="wikimedia_image") # icon - - return attributes - - def debug_explain_wikidata_query(query: str, method: str = "GET"): if method == "GET": - http_response = get(SPARQL_EXPLAIN_URL + "&" + urlencode({"query": query}), headers=get_headers()) + http_response = get(SPARQL_EXPLAIN_URL + "&" + urlencode({"query": query}), headers=get_wikidata_headers()) else: - http_response = post(SPARQL_EXPLAIN_URL, data={"query": query}, headers=get_headers()) + http_response = post(SPARQL_EXPLAIN_URL, data={"query": query}, headers=get_wikidata_headers()) http_response.raise_for_status() return http_response.content -def init(_): - global CACHE # pylint: disable=global-statement - CACHE = EngineCache("wikidata") - - # In an environment with competing processes, the initial loading of the - # cache is required only once. - eng_state: str | None = CACHE.get("eng_state") - if not eng_state or not eng_state.startswith("STATE:"): - CACHE.set("eng_state", f"STATE: being initialized by PID {os.getpid()}") - try: - init_wikidata_properties() - except Exception: - CACHE.set("eng_state", f"ERROR: initialization by PID {os.getpid()} failed.") - raise - else: - logger.debug(eng_state) - - -def init_wikidata_properties(): - global WIKIDATA_PROPERTIES # pylint: disable=global-statement - p: WDPType = CACHE.get(key="WIKIDATA_PROPERTIES") - if p: - WIKIDATA_PROPERTIES = p - return - - # WIKIDATA_PROPERTIES : add unit symbols - for k, v in WIKIDATA_UNITS.items(): - WIKIDATA_PROPERTIES[k] = v["symbol"] - - # WIKIDATA_PROPERTIES : add property labels - wikidata_property_names: list[str] = [] - for attribute in get_attributes("en"): - if type(attribute) in (WDAttribute, WDAmountAttribute, WDURLAttribute, WDDateAttribute, WDLabelAttribute): - if attribute.name not in WIKIDATA_PROPERTIES: - wikidata_property_names.append("wd:" + attribute.name) - query = QUERY_PROPERTY_NAMES.replace("%ATTRIBUTES%", " ".join(wikidata_property_names)) - kwargs: dict[str, t.Any] = {"timeout": 20} - jsonresponse = send_wikidata_query(query, **kwargs) - for result in jsonresponse.get("results", {}).get("bindings", {}): - name_field = result.get("name") - if not name_field: - continue - name = name_field["value"] - lang = name_field["xml:lang"] - entity_id = result["item"]["value"].replace("http://www.wikidata.org/entity/", "") - WIKIDATA_PROPERTIES[(entity_id, lang)] = name.capitalize() - - CACHE.set(key="WIKIDATA_PROPERTIES", value=WIKIDATA_PROPERTIES) - - def fetch_traits(engine_traits: EngineTraits): """Uses languages evaluated from :py:obj:`wikipedia.fetch_wikimedia_traits ` and removes diff --git a/searx/wikidata.py b/searx/wikidata.py new file mode 100644 index 000000000..defb2a0bb --- /dev/null +++ b/searx/wikidata.py @@ -0,0 +1,35 @@ +# SPDX-License-Identifier: AGPL-3.0-or-later +"""Shared methods for accessing Wikidata.""" + +import typing as t +from urllib.parse import urlencode + +from searx.network import get, post +from searx.utils import gen_useragent + + +# SPARQL +SPARQL_ENDPOINT_URL = "https://query.wikidata.org/sparql" +SPARQL_EXPLAIN_URL = "https://query.wikidata.org/bigdata/namespace/wdq/sparql?explain" + + +def send_wikidata_query(query: str, method: str = "GET", **kwargs: dict[str, t.Any]) -> dict[str, t.Any]: + if method == "GET": + # query will be cached by wikidata + http_response = get( + SPARQL_ENDPOINT_URL + "?" + urlencode({"query": query}), headers=get_wikidata_headers(), **kwargs + ) + else: + # query won't be cached by wikidata + http_response = post(SPARQL_ENDPOINT_URL, data={"query": query}, headers=get_wikidata_headers(), **kwargs) + + http_response.raise_for_status() + return http_response.json() + + +def get_wikidata_headers() -> dict[str, str]: + # user agent: https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual#Query_limits + return { + "Accept": "application/sparql-results+json", + "User-Agent": gen_useragent(), + } diff --git a/searx/wikidata_properties.py b/searx/wikidata_properties.py new file mode 100644 index 000000000..81b5df498 --- /dev/null +++ b/searx/wikidata_properties.py @@ -0,0 +1,580 @@ +# SPDX-License-Identifier: AGPL-3.0-or-later +# pylint: disable=missing-class-docstring +"""Fetch property names from :origin:`searx/engines/wikidata.py` engine.""" + +import typing as t + +from dateutil.parser import isoparse +from babel.dates import format_datetime, format_date, format_time, get_datetime_format + +from searx.external_urls import get_earth_coordinates_url, get_external_url +from searx.data import WikiDataPropertiesType, WikiDataUnitType +from searx.wikidata import send_wikidata_query + +# SERVICE wikibase:mwapi : https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual/MWAPI +# SERVICE wikibase:label: https://en.wikibooks.org/wiki/SPARQL/SERVICE_-_Label#Manual_Label_SERVICE +# https://en.wikibooks.org/wiki/SPARQL/WIKIDATA_Precision,_Units_and_Coordinates +# https://www.mediawiki.org/wiki/Wikibase/Indexing/RDF_Dump_Format#Data_model +# optimization: +# * https://www.wikidata.org/wiki/Wikidata:SPARQL_query_service/query_optimization +# * https://github.com/blazegraph/database/wiki/QueryHints +QUERY_TEMPLATE = """ +SELECT ?item ?itemLabel ?itemDescription ?lat ?long %SELECT% +WHERE +{ + SERVICE wikibase:mwapi { + bd:serviceParam wikibase:endpoint "www.wikidata.org"; + wikibase:api "EntitySearch"; + wikibase:limit 1; + mwapi:search "%QUERY%"; + mwapi:language "%LANGUAGE%". + ?item wikibase:apiOutputItem mwapi:item. + } + hint:Prior hint:runFirst "true". + + %WHERE% + + SERVICE wikibase:label { + bd:serviceParam wikibase:language "%LANGUAGE%,en". + ?item rdfs:label ?itemLabel . + ?item schema:description ?itemDescription . + %WIKIBASE_LABELS% + } + +} +GROUP BY ?item ?itemLabel ?itemDescription ?lat ?long %GROUP_BY% +""" + +# Get the calendar names and the property names +QUERY_PROPERTY_NAMES = """ +SELECT ?item ?name +WHERE { + { + SELECT ?item + WHERE { ?item wdt:P279* wd:Q12132 } + } UNION { + VALUES ?item { %ATTRIBUTES% } + } + OPTIONAL { ?item rdfs:label ?name. } +} +""" + + +class WDAttribute: + def __init__(self, name: str): + self.name: str = name + + def get_select(self): + return "(group_concat(distinct ?{name};separator=', ') as ?{name}s)".replace("{name}", self.name) + + def get_label(self, language: str): + return get_label_for_entity(self.name, language) + + def get_where(self): + return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name) + + def get_wikibase_label(self) -> str: + return "" + + def get_group_by(self) -> str: + return "" + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: # pylint: disable=unused-argument + return result.get(self.name + "s") + + def __repr__(self): + return "<" + str(type(self).__name__) + ":" + self.name + ">" + + +class WDAmountAttribute(WDAttribute): + def get_select(self) -> str: + return "?{name} ?{name}Unit".replace("{name}", self.name) + + def get_where(self): + return """ OPTIONAL { ?item p:{name} ?{name}Node . + ?{name}Node rdf:type wikibase:BestRank ; ps:{name} ?{name} . + OPTIONAL { ?{name}Node psv:{name}/wikibase:quantityUnit ?{name}Unit. } }""".replace( + '{name}', self.name + ) + + def get_group_by(self) -> str: + return self.get_select() + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + value: str | None = result.get(self.name) + unit: str | None = result.get(self.name + "Unit") + if unit is not None: + unit = unit.replace("http://www.wikidata.org/entity/", "") + return str(value) + " " + get_label_for_entity(unit, language) + return value + + +class WDArticle(WDAttribute): + def __init__(self, language: str, kwargs: dict[str, t.Any] | None = None): + super().__init__("wikipedia") + self.language: str = language + self.kwargs: dict[str, t.Any] = kwargs or {} + + def get_label(self, language: str): + # language parameter is ignored + return "Wikipedia ({language})".replace("{language}", self.language) + + def get_select(self): + return "?article{language} ?articleName{language}".replace("{language}", self.language) + + def get_where(self): + return """OPTIONAL { ?article{language} schema:about ?item ; + schema:inLanguage "{language}" ; + schema:isPartOf ; + schema:name ?articleName{language} . }""".replace( + '{language}', self.language + ) + + def get_group_by(self): + return self.get_select() + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + key = "article{language}".replace("{language}", self.language) + return result.get(key) + + +class WDLabelAttribute(WDAttribute): + def get_select(self): + return "(group_concat(distinct ?{name}Label;separator=', ') as ?{name}Labels)".replace("{name}", self.name) + + def get_where(self): + return "OPTIONAL { ?item wdt:{name} ?{name} . }".replace("{name}", self.name) + + def get_wikibase_label(self) -> str: + return "?{name} rdfs:label ?{name}Label .".replace("{name}", self.name) + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + return result.get(self.name + "Labels") + + +class WDURLAttribute(WDAttribute): + HTTP_WIKIMEDIA_IMAGE: str = "http://commons.wikimedia.org/wiki/Special:FilePath/" + + def __init__( + self, + name: str, + url_id: str | None = None, + url_path_prefix: str | None = None, + kwargs: dict[str, t.Any] | None = None, + ): + """ + :param url_id: ID matching one key in ``external_urls.json`` for + converting IDs to full URLs. + + :param url_path_prefix: Path prefix if the values are of format + ``account@domain``. If provided, value are rewritten to + ``https://``. For example:: + + WDURLAttribute('P4033', url_path_prefix='/@') + + Adds Property `P4033 `_ + to the wikidata query. This field might return for example + ``libreoffice@fosstodon.org`` and the URL built from this is then: + + - account: ``libreoffice`` + - domain: ``fosstodon.org`` + - result url: https://fosstodon.org/@libreoffice + """ + + super().__init__(name) + self.url_id: str | None = url_id + self.url_path_prefix: str | None = url_path_prefix + self.kwargs: dict[str, t.Any] = kwargs or {} + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + value: str | None = result.get(self.name + "s") + if not value: + return None + + value = value.split(",")[0] + if self.url_id: + url_id = self.url_id + if value.startswith(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE): + value = value[len(WDURLAttribute.HTTP_WIKIMEDIA_IMAGE) :] + url_id = "wikimedia_image" + return get_external_url(url_id, value) + + if self.url_path_prefix: + [account, domain] = [x.strip("@ ") for x in value.rsplit("@", 1)] + return f"https://{domain}{self.url_path_prefix}{account}" + + return value + + +class WDGeoAttribute(WDAttribute): + def get_label(self, language: str): + return "OpenStreetMap" + + def get_select(self): + return "?{name}Lat ?{name}Long".replace("{name}", self.name) + + def get_where(self): + return """OPTIONAL { ?item p:{name}/psv:{name} [ + wikibase:geoLatitude ?{name}Lat ; + wikibase:geoLongitude ?{name}Long ] }""".replace( + '{name}', self.name + ) + + def get_group_by(self): + return self.get_select() + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + latitude: str | None = result.get(self.name + "Lat") + longitude: str | None = result.get(self.name + "Long") + if latitude and longitude: + return latitude + " " + longitude + return None + + def get_geo_url(self, result: dict[str, t.Any], osm_zoom: int = 19) -> str | None: + latitude: str | None = result.get(self.name + "Lat") + longitude: str | None = result.get(self.name + "Long") + if latitude and longitude: + return get_earth_coordinates_url(latitude, longitude, osm_zoom) + return None + + +class WDImageAttribute(WDURLAttribute): + def __init__(self, name: str, url_id: str | None = None, priority: int = 100): + super().__init__(name, url_id) + self.priority: int = priority + + +class WDDateAttribute(WDAttribute): + def get_select(self): + return "?{name} ?{name}timePrecision ?{name}timeZone ?{name}timeCalendar".replace("{name}", self.name) + + def get_where(self): + # To remove duplicate, add + # FILTER NOT EXISTS { ?item p:{name}/psv:{name}/wikibase:timeValue ?{name}bis FILTER (?{name}bis < ?{name}) } + # this filter is too slow, so the response function ignore duplicate results + # (see the seen_entities variable) + return """OPTIONAL { ?item p:{name}/psv:{name} [ + wikibase:timeValue ?{name} ; + wikibase:timePrecision ?{name}timePrecision ; + wikibase:timeTimezone ?{name}timeZone ; + wikibase:timeCalendarModel ?{name}timeCalendar ] . } + hint:Prior hint:rangeSafe true;""".replace( + '{name}', self.name + ) + + def get_group_by(self): + return self.get_select() + + def format_8(self, value: str, locale: str) -> str: # pylint: disable=unused-argument + # precision: less than a year + return value + + def format_9(self, value: str, locale: str) -> str: + year = int(value) + # precision: year + if year < 1584: + if year < 0: + return str(year - 1) + return str(year) + timestamp = isoparse(value) + return format_date(timestamp, format="yyyy", locale=locale) + + def format_10(self, value: str, locale: str) -> str: + # precision: month + timestamp = isoparse(value) + return format_date(timestamp, format="MMMM y", locale=locale) + + def format_11(self, value: str, locale: str) -> str: + # precision: day + timestamp = isoparse(value) + return format_date(timestamp, format="full", locale=locale) + + def format_13(self, value: str, locale: str) -> str: + timestamp = isoparse(value) + # precision: minute + return ( + get_datetime_format("medium", locale=locale) + .replace("'", "") + .replace("{0}", format_time(timestamp, "full", tzinfo=None, locale=locale)) + .replace("{1}", format_date(timestamp, "short", locale=locale)) + ) + + def format_14(self, value: str, locale: str) -> str: + # precision: second. + return format_datetime(isoparse(value), format="full", locale=locale) + + DATE_FORMAT: dict[str, tuple[str, int]] = { + "0": ("format_8", 1000000000), + "1": ("format_8", 100000000), + "2": ("format_8", 10000000), + "3": ("format_8", 1000000), + "4": ("format_8", 100000), + "5": ("format_8", 10000), + "6": ("format_8", 1000), + "7": ("format_8", 100), + "8": ("format_8", 10), + "9": ("format_9", 1), # year + "10": ("format_10", 1), # month + "11": ("format_11", 0), # day + "12": ("format_13", 0), # hour (not supported by babel, display minute) + "13": ("format_13", 0), # minute + "14": ("format_14", 0), # second + } + + def get_str(self, result: dict[str, t.Any], language: str) -> str | None: + value: str | None = result.get(self.name) + if value == "" or value is None: + return None + _p: str = result.get(self.name + "timePrecision") or "1" + date_format = WDDateAttribute.DATE_FORMAT.get(_p) + if date_format is not None: + format_method = getattr(self, date_format[0]) + precision: int = date_format[1] + try: + if precision >= 1: + _t = value.split("-") + if value.startswith("-"): + value = "-" + _t[1] + else: + value = _t[0] + return format_method(value, language) + except Exception: # pylint: disable=broad-except + return value + return value + + +WDAttrType = ( + WDAttribute + | WDAmountAttribute + | WDArticle + | WDLabelAttribute + | WDURLAttribute + | WDGeoAttribute + | WDImageAttribute + | WDDateAttribute +) +WDAttrList = list[WDAttrType] + +_WIKIDATA_PROPERTIES_OVERRIDE: WikiDataPropertiesType = { + "P434": "MusicBrainz", + "P435": "MusicBrainz", + "P436": "MusicBrainz", + "P966": "MusicBrainz", + "P345": "IMDb", + "P2397": "YouTube", + "P1651": "YouTube", + "P2002": "Twitter", + "P2013": "Facebook", + "P2003": "Instagram", + "P4033": "Mastodon", + "P11947": "Lemmy", + "P12622": "PeerTube", +} +"""Custom hardcoded property names for some Wikidata IDs. Only used if the +name Wikidata assigned isn't user-friendly.""" + + +def fetch_properties(units: dict[str, WikiDataUnitType]) -> WikiDataPropertiesType: + properties = _WIKIDATA_PROPERTIES_OVERRIDE + + # WIKIDATA_PROPERTIES : add unit symbols + for k, v in units.items(): + properties[k] = v["symbol"] + + # WIKIDATA_PROPERTIES : add property labels + wikidata_property_names: list[str] = [] + for attribute in get_attributes("en"): + if type(attribute) in (WDAttribute, WDAmountAttribute, WDURLAttribute, WDDateAttribute, WDLabelAttribute): + if attribute.name not in properties: + wikidata_property_names.append("wd:" + attribute.name) + + query = QUERY_PROPERTY_NAMES.replace("%ATTRIBUTES%", " ".join(wikidata_property_names)) + kwargs: dict[str, t.Any] = {"timeout": 60} + json_response = send_wikidata_query(query, **kwargs) + + for result in json_response.get("results", {}).get("bindings", {}): + name_field = result.get("name") + if not name_field: + continue + name = name_field["value"] + lang = name_field["xml:lang"] + entity_id = result["item"]["value"].replace("http://www.wikidata.org/entity/", "") + + if name: + prop = properties.get(entity_id) or {} + prop[lang] = name.capitalize() # pyright: ignore[reportIndexIssue] + properties[entity_id] = prop + else: + properties[entity_id] = name.capitalize() + + return properties + + +def get_attributes(language: str): + # pylint: disable=too-many-statements + attributes: WDAttrList = [] + + def add_value(name: str): + attributes.append(WDAttribute(name)) + + def add_amount(name: str): + attributes.append(WDAmountAttribute(name)) + + def add_label(name: str): + attributes.append(WDLabelAttribute(name)) + + def add_url(name: str, url_id: str | None = None, url_path_prefix: str | None = None, **kwargs: dict[str, t.Any]): + attributes.append(WDURLAttribute(name, url_id, url_path_prefix, kwargs)) + + def add_image(name: str, url_id: str | None = None, priority: int = 1): + attributes.append(WDImageAttribute(name, url_id, priority)) + + def add_date(name: str): + attributes.append(WDDateAttribute(name)) + + # Dates + for p in [ + "P571", # inception date + "P576", # dissolution date + "P580", # start date + "P582", # end date + "P569", # date of birth + "P570", # date of death + "P619", # date of spacecraft launch + "P620", + ]: # date of spacecraft landing + add_date(p) + + for p in [ + "P27", # country of citizenship + "P495", # country of origin + "P17", # country + "P159", + ]: # headquarters location + add_label(p) + + # Places + for p in [ + "P36", # capital + "P35", # head of state + "P6", # head of government + "P122", # basic form of government + "P37", + ]: # official language + add_label(p) + + add_value("P1082") # population + add_amount("P2046") # area + add_amount("P281") # postal code + add_label("P38") # currency + add_amount("P2048") # height (building) + + # Media + for p in [ + "P400", # platform (videogames, computing) + "P50", # author + "P170", # creator + "P57", # director + "P175", # performer + "P178", # developer + "P162", # producer + "P176", # manufacturer + "P58", # screenwriter + "P272", # production company + "P264", # record label + "P123", # publisher + "P449", # original network + "P750", # distributed by + "P86", + ]: # composer + add_label(p) + + add_date("P577") # publication date + add_label("P136") # genre (music, film, artistic...) + add_label("P364") # original language + add_value("P212") # ISBN-13 + add_value("P957") # ISBN-10 + add_label("P275") # copyright license + add_label("P277") # programming language + add_value("P348") # version + add_label("P840") # narrative location + + # Languages + add_value("P1098") # number of speakers + add_label("P282") # writing system + add_label("P1018") # language regulatory body + add_value("P218") # language code (ISO 639-1) + + # Other + add_label("P169") # ceo + add_label("P112") # founded by + add_label("P1454") # legal form (company, organization) + add_label("P137") # operator (service, facility, ...) + add_label("P1029") # crew members (tripulation) + add_label("P225") # taxon name + add_value("P274") # chemical formula + add_label("P1346") # winner (sports, contests, ...) + add_value("P1120") # number of deaths + add_value("P498") # currency code (ISO 4217) + + # URL + kwargs: dict[str, t.Any] = {"official": True} + add_url("P856", **kwargs) # official website + attributes.append(WDArticle(language)) # wikipedia (user language) + if not language.startswith("en"): + attributes.append(WDArticle("en")) # wikipedia (english) + + add_url("P1324") # source code repository + add_url("P1581") # blog + add_url("P434", url_id="musicbrainz_artist") + add_url("P435", url_id="musicbrainz_work") + add_url("P436", url_id="musicbrainz_release_group") + add_url("P966", url_id="musicbrainz_label") + add_url("P345", url_id="imdb_id") + add_url("P2397", url_id="youtube_channel") + add_url("P1651", url_id="youtube_video") + add_url("P2002", url_id="twitter_profile") + add_url("P2013", url_id="facebook_profile") + add_url("P2003", url_id="instagram_profile") + + # Fediverse + add_url("P4033", url_path_prefix="/@") # Mastodon user + add_url("P11947", url_path_prefix="/c/") # Lemmy community + add_url("P12622", url_path_prefix="/c/") # PeerTube channel + + # Map + attributes.append(WDGeoAttribute("P625")) + + # Image + add_image("P15", priority=1, url_id="wikimedia_image") # route map + add_image("P242", priority=2, url_id="wikimedia_image") # locator map + add_image("P154", priority=3, url_id="wikimedia_image") # logo + add_image("P18", priority=4, url_id="wikimedia_image") # image + add_image("P41", priority=5, url_id="wikimedia_image") # flag + add_image("P2716", priority=6, url_id="wikimedia_image") # collage + add_image("P2910", priority=7, url_id="wikimedia_image") # icon + + return attributes + + +def get_label_for_entity(entity_id: str, language: str) -> str: + # only import properties locally to prevent cyclic import when initializing WIKIDATA_PROPERTIES + from searx.data import WIKIDATA_PROPERTIES # pylint: disable=import-outside-toplevel + + property_name = WIKIDATA_PROPERTIES.get(entity_id) + if property_name is None: + return entity_id + + if isinstance(property_name, str): + return property_name + + if name := property_name.get(language): + return name + + if name := property_name.get(language.split("-")[0]): + return name + + if name := property_name.get("en"): + return name + + return entity_id diff --git a/searx/wikidata_units.py b/searx/wikidata_units.py index 821476b80..2169a4858 100644 --- a/searx/wikidata_units.py +++ b/searx/wikidata_units.py @@ -11,7 +11,7 @@ __all__ = ["convert_from_si", "convert_to_si", "symbol_to_si"] import collections from searx import data -from searx.engines import wikidata +from searx.wikidata import send_wikidata_query class Beaufort: @@ -142,7 +142,6 @@ def units_by_si_name(si_name): # build the catalog .. for item in symbol_to_si(): - item_si_name = item[pos_si_name] item_symbol = item[pos_symbol] @@ -266,14 +265,13 @@ ORDER BY ?item DESC(?rank) ?symbol """ -def fetch_units(): +def fetch_units() -> dict[str, data.WikiDataUnitType]: """Fetch units from Wikidata. Function is used to update persistence of :py:obj:`searx.data.WIKIDATA_UNITS`.""" results = collections.OrderedDict() - response = wikidata.send_wikidata_query(SARQL_REQUEST) + response = send_wikidata_query(SARQL_REQUEST) for unit in response['results']['bindings']: - symbol = unit['symbol']['value'] name = unit['item']['value'].rsplit('/', 1)[1] si_name = unit.get('tosiUnit', {}).get('value', '') diff --git a/searxng_extra/update/update_currencies.py b/searxng_extra/update/update_currencies.py index 288f0994e..36e8cfa2e 100755 --- a/searxng_extra/update/update_currencies.py +++ b/searxng_extra/update/update_currencies.py @@ -16,6 +16,7 @@ import json from searx.locales import LOCALE_NAMES, locales_initialize from searx.engines import wikidata, set_loggers from searx.data.currencies import CurrenciesDB +from searx.wikidata import send_wikidata_query set_loggers(wikidata, 'wikidata') locales_initialize() @@ -90,7 +91,7 @@ def add_currency_label(db, label, iso4217, language): def wikidata_request_result_iterator(request): - result = wikidata.send_wikidata_query(request.replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL), timeout=20) + result = send_wikidata_query(request.replace('%LANGUAGES_SPARQL%', LANGUAGES_SPARQL), timeout=20) if result is not None: yield from result['results']['bindings'] diff --git a/searxng_extra/update/update_engine_descriptions.py b/searxng_extra/update/update_engine_descriptions.py index 69e0971f7..f0798aaf7 100755 --- a/searxng_extra/update/update_engine_descriptions.py +++ b/searxng_extra/update/update_engine_descriptions.py @@ -1,7 +1,7 @@ #!/usr/bin/env python # SPDX-License-Identifier: AGPL-3.0-or-later """Fetch website description from websites and from -:origin:`searx/engines/wikidata.py` engine. +:origin:`searx/wikidata.py`. Output file: :origin:`searx/data/engine_descriptions.json`. @@ -17,6 +17,7 @@ from lxml.html import fromstring import searx.engines from searx.engines import wikidata, set_loggers +from searx.wikidata import send_wikidata_query from searx.utils import extract_text from searx.locales import LOCALE_NAMES, locales_initialize, match_locale from searx import searx_dir @@ -200,7 +201,6 @@ def initialize(): locale2lang = {"nl-BE": "nl"} for sxng_ui_lang in LOCALE_NAMES: - sxng_ui_alias = locale2lang.get(sxng_ui_lang, sxng_ui_lang) wiki_lang = None @@ -225,7 +225,7 @@ def initialize(): def fetch_wikidata_descriptions(): print("Fetching wikidata descriptions") searx.network.set_timeout_for_thread(60) - result = wikidata.send_wikidata_query( + result = send_wikidata_query( SPARQL_DESCRIPTION.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL) ) if not result: @@ -249,7 +249,7 @@ def fetch_wikidata_descriptions(): def fetch_wikipedia_descriptions(): print("Fetching wikipedia descriptions") - result = wikidata.send_wikidata_query( + result = send_wikidata_query( SPARQL_WIKIPEDIA_ARTICLE.replace("%IDS%", IDS).replace("%LANGUAGES_SPARQL%", LANGUAGES_SPARQL) ) if not result: @@ -307,7 +307,6 @@ def fetch_website_description(engine_name: str, website: str): previous_count: int = 0 for lang in languages: - if lang in descriptions[engine_name]: continue diff --git a/searxng_extra/update/update_osm_keys_tags.py b/searxng_extra/update/update_osm_keys_tags.py index 2c4f5897f..c4871aedb 100755 --- a/searxng_extra/update/update_osm_keys_tags.py +++ b/searxng_extra/update/update_osm_keys_tags.py @@ -50,6 +50,7 @@ from searx.engines import wikidata, set_loggers from searx.sxng_locales import sxng_locales from searx.engines.openstreetmap import get_key_rank, VALUE_TO_LINK from searx.data import data_dir +from searx.wikidata import send_wikidata_query DATA_FILE = data_dir / 'osm_keys_tags.json' @@ -102,7 +103,7 @@ def get_preset_keys(): def get_keys(): results = get_preset_keys() - response = wikidata.send_wikidata_query(SPARQL_KEYS_REQUEST) + response = send_wikidata_query(SPARQL_KEYS_REQUEST) for key in response['results']['bindings']: keys = key['key']['value'].split(':')[1:] @@ -148,7 +149,7 @@ def get_keys(): def get_tags(): results = collections.OrderedDict() - response = wikidata.send_wikidata_query(SPARQL_TAGS_REQUEST) + response = send_wikidata_query(SPARQL_TAGS_REQUEST) for tag in response['results']['bindings']: tag_names = tag['tag']['value'].split(':')[1].split('=') if len(tag_names) == 2: @@ -204,7 +205,6 @@ def optimize_keys(data): if __name__ == '__main__': - set_timeout_for_thread(60) result = { 'keys': optimize_keys(get_keys()), diff --git a/searxng_extra/update/update_wikidata.py b/searxng_extra/update/update_wikidata.py new file mode 100755 index 000000000..20a0ff100 --- /dev/null +++ b/searxng_extra/update/update_wikidata.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python +# SPDX-License-Identifier: AGPL-3.0-or-later +"""Fetch units and property names from :origin:`searx/engines/wikidata.py` engine. + +Output files: (:origin:`CI Update data <.github/workflows/data-update.yml>`). +- :origin:`searx/data/wikidata_units.json` +- :origin:`searx/data/wikidata_properties.json` +""" + +import json + +from searx.engines import wikidata, set_loggers +from searx.data import data_dir +from searx.wikidata_properties import fetch_properties +from searx.wikidata_units import fetch_units + +UNITS_DATA_FILE = data_dir / 'wikidata_units.json' +PROPERTIES_DATA_FILE = data_dir / 'wikidata_properties.json' +set_loggers(wikidata, 'wikidata') + + +if __name__ == '__main__': + units = fetch_units() + with UNITS_DATA_FILE.open('w', encoding="utf8") as f: + json.dump(units, f, indent=4, sort_keys=True, ensure_ascii=False) + + properties = fetch_properties(units) + with PROPERTIES_DATA_FILE.open('w', encoding="utf8") as f: + json.dump(properties, f, indent=4, sort_keys=True, ensure_ascii=False) diff --git a/searxng_extra/update/update_wikidata_units.py b/searxng_extra/update/update_wikidata_units.py deleted file mode 100755 index d815dc85d..000000000 --- a/searxng_extra/update/update_wikidata_units.py +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env python -# SPDX-License-Identifier: AGPL-3.0-or-later -"""Fetch units from :origin:`searx/engines/wikidata.py` engine. - -Output file: :origin:`searx/data/wikidata_units.json` (:origin:`CI Update data -... <.github/workflows/data-update.yml>`). - -""" - -import json - -from searx.engines import wikidata, set_loggers -from searx.data import data_dir -from searx.wikidata_units import fetch_units - -DATA_FILE = data_dir / 'wikidata_units.json' -set_loggers(wikidata, 'wikidata') - - -if __name__ == '__main__': - with DATA_FILE.open('w', encoding="utf8") as f: - json.dump(fetch_units(), f, indent=4, sort_keys=True, ensure_ascii=False) diff --git a/utils/lib_sxng_data.sh b/utils/lib_sxng_data.sh index 2cfe76876..4964bf1e2 100755 --- a/utils/lib_sxng_data.sh +++ b/utils/lib_sxng_data.sh @@ -26,7 +26,8 @@ data.all() { build_msg DATA "update searx/data/ahmia_blacklist.txt" python searxng_extra/update/update_ahmia_blacklist.py build_msg DATA "update searx/data/wikidata_units.json" - python searxng_extra/update/update_wikidata_units.py + build_msg DATA "update searx/data/wikidata_properties.json" + python searxng_extra/update/update_wikidata.py build_msg DATA "update searx/data/currencies.json" python searxng_extra/update/update_currencies.py build_msg DATA "update searx/data/external_bangs.json"