searxng/searx/engines/google_scholar.py

# SPDX-License-Identifier: AGPL-3.0-or-later
# lint: pylint
"""Google (Scholar)

For detailed description of the *REST-full* API see: `Query Parameter
Definitions`_.

.. _Query Parameter Definitions:
   https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions
"""

# pylint: disable=invalid-name

from urllib.parse import urlencode
from datetime import datetime
from lxml import html

from searx.utils import (
    eval_xpath,
    eval_xpath_list,
    extract_text,
)

from searx.engines.google import (
    get_lang_info,
    time_range_dict,
    detect_google_sorry,
)

# pylint: disable=unused-import
from searx.engines.google import (
    supported_languages_url,
    _fetch_supported_languages,
)

# pylint: enable=unused-import

# about
about = {
    "website": 'https://scholar.google.com',
    "wikidata_id": 'Q494817',
    "official_api_documentation": 'https://developers.google.com/custom-search',
    "use_official_api": False,
    "require_api_key": False,
    "results": 'HTML',
}

# engine dependent config
categories = ['science']
paging = True
language_support = True
use_locale_domain = True
time_range_support = True
safesearch = False


def time_range_url(params):
    """Returns a URL query component for a google-Scholar time range based on
    ``params['time_range']``.  Google-Scholar does only support ranges in years.
    To have any effect, all the Searx ranges (*day*, *week*, *month*, *year*)
    are mapped to *year*.  If no range is set, an empty string is returned.
    Example::

        &as_ylo=2019
    """
    # as_ylo=2016&as_yhi=2019
    ret_val = ''
    if params['time_range'] in time_range_dict:
        ret_val = urlencode({'as_ylo': datetime.now().year - 1})
    return '&' + ret_val


def request(query, params):
    """Google-Scholar search request"""

    offset = (params['pageno'] - 1) * 10
    lang_info = get_lang_info(params, supported_languages, language_aliases, False)
    logger.debug("HTTP header Accept-Language --> %s", lang_info['headers']['Accept-Language'])

    # subdomain is: scholar.google.xy
    lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")

    query_url = (
        'https://'
        + lang_info['subdomain']
        + '/scholar'
        + "?"
        + urlencode(
            {
                'q': query,
                **lang_info['params'],
                'ie': "utf8",
                'oe': "utf8",
                'start': offset,
            }
        )
    )

    query_url += time_range_url(params)
    params['url'] = query_url

    params['headers'].update(lang_info['headers'])
    params['headers']['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'

    # params['google_subdomain'] = subdomain
    return params


def response(resp):
    """Get response from google's search request"""
    results = []

    detect_google_sorry(resp)

    # which subdomain ?
    # subdomain = resp.search_params.get('google_subdomain')

    # convert the text to dom
    dom = html.fromstring(resp.text)

    # parse results
    for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):

        title = extract_text(eval_xpath(result, './h3[1]//a'))

        if not title:
            # this is a [ZITATION] block
            continue

        url = eval_xpath(result, './h3[1]//a/@href')[0]
        content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''

        pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))
        if pub_info:
            content += "[%s]" % pub_info

        pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))
        if pub_type:
            title = title + " " + pub_type

        results.append(
            {
                'url': url,
                'title': title,
                'content': content,
            }
        )

    # parse suggestion
    for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):
        # append suggestion
        results.append({'suggestion': extract_text(suggestion)})

    for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):
        results.append({'correction': extract_text(correction)})

    return results
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# SPDX-License-Identifier: AGPL-3.0-or-later`
[pylint] tag PYLINT_FILES by comment `# lint: pylint` These py files are linted by `test.pylint`, all other files are linted by `test.pep8`. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-04-26 18:18:20 +00:00			`# lint: pylint`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`"""Google (Scholar)`

			For detailed description of the REST-full API see: `Query Parameter
			Definitions`_.

			`.. _Query Parameter Definitions:`
			`https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions`
			`"""`

[pylint] engines: drop no longer needed 'missing-function-docstring' Suggested-by: @dalf https://github.com/searxng/searxng/issues/102#issuecomment-914168470 Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-09-07 11:26:59 +00:00			`# pylint: disable=invalid-name`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`from urllib.parse import urlencode`
			`from datetime import datetime`
			`from lxml import html`

			`from searx.utils import (`
			`eval_xpath,`
			`eval_xpath_list,`
			`extract_text,`
			`)`

			`from searx.engines.google import (`
			`get_lang_info,`
			`time_range_dict,`
			`detect_google_sorry,`
			`)`

			`# pylint: disable=unused-import`
			`from searx.engines.google import (`
			`supported_languages_url,`
			`_fetch_supported_languages,`
			`)`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# pylint: enable=unused-import`

			`# about`
			`about = {`
			`"website": 'https://scholar.google.com',`
			`"wikidata_id": 'Q494817',`
			`"official_api_documentation": 'https://developers.google.com/custom-search',`
			`"use_official_api": False,`
			`"require_api_key": False,`
			`"results": 'HTML',`
			`}`

			`# engine dependent config`
			`categories = ['science']`
			`paging = True`
			`language_support = True`
			`use_locale_domain = True`
			`time_range_support = True`
			`safesearch = False`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`def time_range_url(params):`
			`"""Returns a URL query component for a google-Scholar time range based on`
			``params['time_range']``. Google-Scholar does only support ranges in years.
			`To have any effect, all the Searx ranges (day, week, month, year)`
			`are mapped to year. If no range is set, an empty string is returned.`
			`Example::`

			`&as_ylo=2019`
			`"""`
			`# as_ylo=2016&as_yhi=2019`
			`ret_val = ''`
			`if params['time_range'] in time_range_dict:`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`ret_val = urlencode({'as_ylo': datetime.now().year - 1})`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`return '&' + ret_val`


			`def request(query, params):`
			`"""Google-Scholar search request"""`

			`offset = (params['pageno'] - 1) * 10`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`lang_info = get_lang_info(params, supported_languages, language_aliases, False)`
			`logger.debug("HTTP header Accept-Language --> %s", lang_info['headers']['Accept-Language'])`
[fix] log messages from: google- images, news, scholar, videos - HTTP header Accept-Language --> lang_info['headers']['Accept-Language'] - remove obsolete query_url log messages which is already logged by httpx._client:HTTP request Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-06-11 14:31:50 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# subdomain is: scholar.google.xy`
			`lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`query_url = (`
			`'https://'`
			`+ lang_info['subdomain']`
			`+ '/scholar'`
			`+ "?"`
			`+ urlencode(`
			`{`
			`'q': query,`
			`**lang_info['params'],`
			`'ie': "utf8",`
			`'oe': "utf8",`
			`'start': offset,`
			`}`
			`)`
			`)`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`query_url += time_range_url(params)`
			`params['url'] = query_url`

[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 06:18:07 +00:00			`params['headers'].update(lang_info['headers'])`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`params['headers']['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,/;q=0.8'`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`# params['google_subdomain'] = subdomain`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`return params`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`def response(resp):`
			`"""Get response from google's search request"""`
			`results = []`

			`detect_google_sorry(resp)`

			`# which subdomain ?`
			`# subdomain = resp.search_params.get('google_subdomain')`

			`# convert the text to dom`
			`dom = html.fromstring(resp.text)`

			`# parse results`
			`for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):`

			`title = extract_text(eval_xpath(result, './h3[1]//a'))`

			`if not title:`
			`# this is a [ZITATION] block`
			`continue`

			`url = eval_xpath(result, './h3[1]//a/@href')[0]`
			`content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''`

			`pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))`
			`if pub_info:`
			`content += "[%s]" % pub_info`

			`pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))`
			`if pub_type:`
			`title = title + " " + pub_type`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`results.append(`
			`{`
			`'url': url,`
			`'title': title,`
			`'content': content,`
			`}`
			`)`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`# parse suggestion`
			`for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):`
			`# append suggestion`
			`results.append({'suggestion': extract_text(suggestion)})`

			`for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):`
			`results.append({'correction': extract_text(correction)})`

			`return results`