[mod] brave - modernize response handling (#6742)

This commit is contained in:
WookieeOnCrack
2026-09-19 11:04:37 +00:00
committed by GitHub
parent fe5bdd1bf2
commit c0014f015e

View File

@@ -119,25 +119,20 @@ Implementations
import json import json
import typing as t import typing as t
from urllib.parse import ( from collections.abc import Callable
urlencode, from urllib.parse import urlencode
urlparse,
)
from dateutil import parser from dateutil import parser
from searx import locales from searx import locales, logger
from searx.enginelib.traits import EngineTraits from searx.enginelib.traits import EngineTraits
from searx.exceptions import SearxEngineResponseException
from searx.result_types import EngineResults, MainResult
from searx.result_types.image import Image
from searx.utils import html_to_text, js_obj_str_to_json_str, js_obj_str_to_python
if t.TYPE_CHECKING:
from searx.extended_types import SXNG_Response from searx.extended_types import SXNG_Response
from searx.result_types import EngineResults
from searx.utils import (
eval_xpath_getindex,
eval_xpath_list,
extract_text,
get_embedded_stream_url,
js_obj_str_to_json_str,
js_obj_str_to_python,
)
about = { about = {
"website": "https://search.brave.com/", "website": "https://search.brave.com/",
@@ -261,155 +256,143 @@ def extract_json_data(text: str) -> dict[str, t.Any]:
return data return data
def response(resp: SXNG_Response) -> EngineResults: def response(resp: "SXNG_Response") -> EngineResults:
# delegate the response to the appropriate parser based on search type
if brave_category in ("search", "goggles"): match brave_category:
return _parse_search(resp) case "search" | "goggles":
return _parse_results(_parse_search_result, resp)
case "news":
return _parse_results(_parse_news_result, resp)
case "images":
return _parse_results(_parse_image_result, resp)
case "videos":
return _parse_results(_parse_video_result, resp)
case _:
raise ValueError(f"Unsupported brave category: {brave_category}") # pyright: ignore[reportUnreachable]
if brave_category in ("news"):
return _parse_news(resp)
def _parse_search_result(result: dict[str, t.Any]) -> MainResult:
thumbnail: dict[str, t.Any] = result.get("thumbnail", {})
return MainResult(
template="default.html",
title=result.get("title", ""),
content=html_to_text(result.get("description", "")),
url=result.get("url", ""),
publishedDate=_extract_published_date(result.get("page_age")),
pubdate=result.get("age", ""),
thumbnail=thumbnail.get("src", "") if thumbnail and not thumbnail.get("logo") else "",
)
def _parse_secondary_items(json_data: dict[str, t.Any], results: EngineResults):
# video results utilize same schema as video search -> re-use _parse_video_result
body_resp: dict[str, t.Any] = _get_response_data(json_data)
videos_resp: dict[str, t.Any] = body_resp.get("videos", {})
if videos_resp and "results" in videos_resp:
for result in videos_resp.get("results", []):
results.add(_parse_video_result(result))
# related queries -> suggestion
query: dict[str, t.Any] = body_resp.get("query", {})
if query and "related_queries" in query:
for x in query.get("related_queries", []):
suggestion = " ".join(val[1] for val in x)
results.add(results.types.LegacyResult(suggestion=suggestion))
def _parse_news_result(result: dict[str, t.Any]) -> MainResult:
thumbnail: dict[str, t.Any] = result.get("thumbnail", {})
return MainResult(
title=result.get("title", ""),
content=result.get("description", ""),
url=result.get("url"),
publishedDate=_extract_published_date(result.get("age")),
pubdate=result.get("age", ""),
thumbnail=thumbnail.get("src", "") if thumbnail else "",
)
def _parse_image_result(result: dict[str, t.Any]) -> Image:
properties: dict[str, t.Any] = result.get("properties", {})
thumbnail: dict[str, t.Any] = result.get("thumbnail", {})
width, height = properties.get("width"), properties.get("height")
return Image(
title=result.get("title", ""),
url=result.get("url"),
img_src=properties.get("url", ""),
thumbnail_src=thumbnail.get("src", "") if thumbnail else "",
source=result.get("source", ""),
resolution=f"{width}x{height}" if width and height else "",
)
def _parse_video_result(result: dict[str, t.Any]) -> MainResult:
video: dict[str, t.Any] = result.get("video", {})
thumbnail: dict[str, t.Any] = result.get("thumbnail", {})
return MainResult(
template="videos.html",
title=result.get("title", ""),
url=result.get("url"),
content=result.get("description", ""),
length=video.get("duration"),
publishedDate=_extract_published_date(result.get("age")),
pubdate=result.get("age", ""),
views=video.get("views", ""),
thumbnail=thumbnail.get("src", "") if thumbnail else "",
)
def _get_response_data(json_data: dict[str, t.Any], category: str | None = None) -> dict[str, t.Any]:
"""Navigate the Brave embedded JSON to the category-specific response object."""
# Brave’s structure is mostly consistent but has a couple of quirks:
# - most categories live under data[1].data.body.response.<category>
# - news omits the intermediate "body" key
try:
data: dict[str, t.Any] = json_data["data"][1]["data"]
if data.get("noResults"): # Boolean Value
return {}
if category == "news":
return data["response"]["news"]
body_resp = data["body"]["response"]
if category in ("search", "goggles"):
return body_resp["web"]
# images / videos / secondary items
return body_resp
except (KeyError, IndexError, TypeError) as e:
raise SearxEngineResponseException(f"Unexpected Brave JSON structure for category {category!r}") from e
def _parse_results(parse_func: Callable[..., MainResult | Image], resp: "SXNG_Response") -> EngineResults:
"""Extract json data and loop through result list
The suppled :py.obj:`parse_func` parses individual result items
General search / goggle search relies on :py.obj:`_parse_secondary_items` for mixed result-types"""
# Example script source containing the data: # Example script source containing the data:
# #
# kit.start(app, element, { # kit.start(app, element, {
# node_ids: [0, 19], # node_ids: [0, 19],
# data: [{type:"data",data: .... ["q","goggles_id"],route:1,url:1}}] # data: [{type:"data",data: .... ["q","goggles_id"],route:1,url:1}}]
# ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ # ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
results = EngineResults()
json_data: dict[str, t.Any] = extract_json_data(resp.text) json_data: dict[str, t.Any] = extract_json_data(resp.text)
json_resp: dict[str, t.Any] = json_data["data"][1]["data"]["body"]["response"] json_resp: dict[str, t.Any] = _get_response_data(json_data, brave_category)
if not json_resp:
# if _get_response_data returns {} - indicates it was parsed successfully but had "noResults" = True
return results
if brave_category == "images": json_results: list[dict[str, t.Any]] = json_resp["results"]
return _parse_images(json_resp) for result in json_results:
if brave_category == "videos": results.add(parse_func(result))
return _parse_videos(json_resp)
raise ValueError(f"Unsupported brave category: {brave_category}") # general search / goggle might have secondary items
if brave_category in ("search", "goggles"):
_parse_secondary_items(json_data, results)
return results
def _parse_search(resp: SXNG_Response) -> EngineResults:
res = EngineResults()
dom = resp.html()
for result in eval_xpath_list(dom, "//div[contains(@class, 'snippet ')]"):
url: str | None = eval_xpath_getindex(result, ".//a/@href", 0, default=None)
title_tag = eval_xpath_getindex(result, ".//div[contains(@class, 'title')]", 0, default=None)
if url is None or title_tag is None or not urlparse(url).netloc: # partial url likely means it's an ad
continue
content: str = ""
pub_date = None
# there are other classes like 'site-name-content' we don't want to match,
# however only using contains(@class, 'content') would e.g. also match `site-name-content`
# thus, we explicitly also require the spaces as class separator
_content = eval_xpath_getindex(
result,
".//div[contains(concat(' ', @class, ' '), ' content ')]",
0,
default="",
)
if len(_content):
content = extract_text(_content) # type: ignore
_pub_date = extract_text(
eval_xpath_getindex(_content, ".//span[contains(@class, 't-secondary')]", 0, default="")
)
if _pub_date:
pub_date = _extract_published_date(_pub_date)
content = content.lstrip(_pub_date).strip("- \n\t")
thumbnail: str = eval_xpath_getindex(result, ".//a[contains(@class, 'thumbnail')]//img/@src", 0, default="")
item = res.types.LegacyResult(
template="default.html",
url=url,
title=extract_text(title_tag),
content=content,
publishedDate=pub_date,
thumbnail=thumbnail,
)
res.add(item)
video_tag = eval_xpath_getindex(
result,
".//div[contains(@class, 'video-snippet') and @data-macro='video']",
0,
default=[],
)
if len(video_tag):
# In my tests a video tag in the WEB search was most often not a
# video, except the ones from youtube ..
iframe_src = get_embedded_stream_url(url)
if iframe_src:
item["iframe_src"] = iframe_src
item["template"] = "videos.html"
for suggestion in eval_xpath_list(dom, "//a[contains(@class, 'related-query')]"):
res.append(res.types.LegacyResult({"suggestion": extract_text(suggestion)}))
return res
def _parse_news(resp: SXNG_Response) -> EngineResults:
res = EngineResults()
dom = resp.html()
for result in eval_xpath_list(dom, "//div[@data-type='news']"):
url = eval_xpath_getindex(result, ".//a/@href", 0, default=None)
if url is None:
continue
title = eval_xpath_list(result, ".//div[contains(@class, 'title')]")
content = eval_xpath_list(result, ".//div[contains(@class, 'description')]")
thumbnail = eval_xpath_getindex(result, ".//a[contains(@class, 'thumbnail')]//img/@src", 0, default="")
item = res.types.LegacyResult(
template="default.html",
url=url,
title=extract_text(title),
thumbnail=thumbnail,
content=extract_text(content),
)
res.add(item)
return res
def _parse_images(json_resp: dict[str, t.Any]) -> EngineResults:
res = EngineResults()
for result in json_resp["results"]:
item = res.types.LegacyResult(
template="images.html",
url=result["url"],
title=result["title"],
source=result["source"],
img_src=result["properties"]["url"],
thumbnail_src=result["thumbnail"]["src"],
)
res.add(item)
return res
def _parse_videos(json_resp: dict[str, t.Any]) -> EngineResults:
res = EngineResults()
for result in json_resp["results"]:
item = res.types.LegacyResult(
template="videos.html",
url=result["url"],
title=result["title"],
content=result["description"],
length=result["video"]["duration"],
duration=result["video"]["duration"],
publishedDate=_extract_published_date(result["age"]),
)
if result["thumbnail"] is not None:
item["thumbnail"] = result["thumbnail"]["src"]
res.add(item)
return res
def fetch_traits(engine_traits: EngineTraits): def fetch_traits(engine_traits: EngineTraits):
@@ -447,10 +430,9 @@ def fetch_traits(engine_traits: EngineTraits):
# silently ignore unknown languages # silently ignore unknown languages
continue continue
conflict = engine_traits.custom["ui_lang"].get(sxng_tag) # type: ignore if conflict := engine_traits.custom["ui_lang"].get(sxng_tag):
if conflict:
if conflict != ui_lang: if conflict != ui_lang:
print("CONFLICT: babel %s --> %s, %s" % (sxng_tag, conflict, ui_lang)) print(f"CONFLICT: babel {sxng_tag} --> {conflict}, {ui_lang}")
continue continue
engine_traits.custom["ui_lang"][sxng_tag] = ui_lang engine_traits.custom["ui_lang"][sxng_tag] = ui_lang
@@ -477,15 +459,14 @@ def fetch_traits(engine_traits: EngineTraits):
for lang_tag in babel.languages.get_official_languages(country_tag, de_facto=True): for lang_tag in babel.languages.get_official_languages(country_tag, de_facto=True):
lang_tag = lang_map.get(lang_tag, lang_tag) lang_tag = lang_map.get(lang_tag, lang_tag)
try: try:
sxng_tag = region_tag(babel.Locale.parse("%s_%s" % (lang_tag, country_tag.upper()))) sxng_tag = region_tag(babel.Locale.parse(f"{lang_tag}_{country_tag.upper()}"))
except babel.UnknownLocaleError: except babel.UnknownLocaleError:
# silently ignore unknown languages # silently ignore unknown languages
continue continue
# print("%-20s: %s <-- %s" % (v["label"], country_tag, sxng_tag)) # print("%-20s: %s <-- %s" % (v["label"], country_tag, sxng_tag))
conflict = engine_traits.regions.get(sxng_tag) conflict = engine_traits.regions.get(sxng_tag)
if conflict: if conflict and conflict != country_tag:
if conflict != country_tag: print(f"CONFLICT: babel {sxng_tag} --> {conflict}, {country_tag}")
print("CONFLICT: babel %s --> %s, %s" % (sxng_tag, conflict, country_tag))
continue continue
engine_traits.regions[sxng_tag] = country_tag engine_traits.regions[sxng_tag] = country_tag