searxng/searx/engines/google_scholar.py

# SPDX-License-Identifier: AGPL-3.0-or-later
# lint: pylint
"""Google (Scholar)

For detailed description of the *REST-full* API see: `Query Parameter
Definitions`_.

.. _Query Parameter Definitions:
   https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions
"""

# pylint: disable=invalid-name, missing-function-docstring

from urllib.parse import urlencode
from datetime import datetime
from lxml import html

from searx.utils import (
    eval_xpath,
    eval_xpath_list,
    extract_text,
)

from searx.engines.google import (
    get_lang_info,
    time_range_dict,
    detect_google_sorry,
)

# pylint: disable=unused-import
from searx.engines.google import (
    supported_languages_url,
    _fetch_supported_languages,
)
# pylint: enable=unused-import

# about
about = {
    "website": 'https://scholar.google.com',
    "wikidata_id": 'Q494817',
    "official_api_documentation": 'https://developers.google.com/custom-search',
    "use_official_api": False,
    "require_api_key": False,
    "results": 'HTML',
}

# engine dependent config
categories = ['science']
paging = True
language_support = True
use_locale_domain = True
time_range_support = True
safesearch = False

def time_range_url(params):
    """Returns a URL query component for a google-Scholar time range based on
    ``params['time_range']``.  Google-Scholar does only support ranges in years.
    To have any effect, all the Searx ranges (*day*, *week*, *month*, *year*)
    are mapped to *year*.  If no range is set, an empty string is returned.
    Example::

        &as_ylo=2019
    """
    # as_ylo=2016&as_yhi=2019
    ret_val = ''
    if params['time_range'] in time_range_dict:
        ret_val= urlencode({'as_ylo': datetime.now().year -1 })
    return '&' + ret_val


def request(query, params):
    """Google-Scholar search request"""

    offset = (params['pageno'] - 1) * 10
    lang_info = get_lang_info(
        # pylint: disable=undefined-variable
        params, supported_languages, language_aliases, False
    )
    logger.debug(
        "HTTP header Accept-Language --> %s", lang_info['headers']['Accept-Language'])

    # subdomain is: scholar.google.xy
    lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")

    query_url = 'https://'+ lang_info['subdomain'] + '/scholar' + "?" + urlencode({
        'q':  query,
        **lang_info['params'],
        'ie': "utf8",
        'oe':  "utf8",
        'start' : offset,
    })

    query_url += time_range_url(params)
    params['url'] = query_url

    params['headers'].update(lang_info['headers'])
    params['headers']['Accept'] = (
        'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'
    )

    #params['google_subdomain'] = subdomain
    return params

def response(resp):
    """Get response from google's search request"""
    results = []

    detect_google_sorry(resp)

    # which subdomain ?
    # subdomain = resp.search_params.get('google_subdomain')

    # convert the text to dom
    dom = html.fromstring(resp.text)

    # parse results
    for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):

        title = extract_text(eval_xpath(result, './h3[1]//a'))

        if not title:
            # this is a [ZITATION] block
            continue

        url = eval_xpath(result, './h3[1]//a/@href')[0]
        content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''

        pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))
        if pub_info:
            content += "[%s]" % pub_info

        pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))
        if pub_type:
            title = title + " " + pub_type

        results.append({
            'url':      url,
            'title':    title,
            'content':  content,
        })

    # parse suggestion
    for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):
        # append suggestion
        results.append({'suggestion': extract_text(suggestion)})

    for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):
        results.append({'correction': extract_text(correction)})

    return results
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# SPDX-License-Identifier: AGPL-3.0-or-later`
[pylint] tag PYLINT_FILES by comment `# lint: pylint` These py files are linted by `test.pylint`, all other files are linted by `test.pep8`. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-04-26 18:18:20 +00:00			`# lint: pylint`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`"""Google (Scholar)`

			For detailed description of the REST-full API see: `Query Parameter
			Definitions`_.

			`.. _Query Parameter Definitions:`
			`https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions`
			`"""`

			`# pylint: disable=invalid-name, missing-function-docstring`

			`from urllib.parse import urlencode`
			`from datetime import datetime`
			`from lxml import html`

			`from searx.utils import (`
			`eval_xpath,`
			`eval_xpath_list,`
			`extract_text,`
			`)`

			`from searx.engines.google import (`
			`get_lang_info,`
			`time_range_dict,`
			`detect_google_sorry,`
			`)`

			`# pylint: disable=unused-import`
			`from searx.engines.google import (`
			`supported_languages_url,`
			`_fetch_supported_languages,`
			`)`
			`# pylint: enable=unused-import`

			`# about`
			`about = {`
			`"website": 'https://scholar.google.com',`
			`"wikidata_id": 'Q494817',`
			`"official_api_documentation": 'https://developers.google.com/custom-search',`
			`"use_official_api": False,`
			`"require_api_key": False,`
			`"results": 'HTML',`
			`}`

			`# engine dependent config`
			`categories = ['science']`
			`paging = True`
			`language_support = True`
			`use_locale_domain = True`
			`time_range_support = True`
			`safesearch = False`

			`def time_range_url(params):`
			`"""Returns a URL query component for a google-Scholar time range based on`
			``params['time_range']``. Google-Scholar does only support ranges in years.
			`To have any effect, all the Searx ranges (day, week, month, year)`
			`are mapped to year. If no range is set, an empty string is returned.`
			`Example::`

			`&as_ylo=2019`
			`"""`
			`# as_ylo=2016&as_yhi=2019`
			`ret_val = ''`
			`if params['time_range'] in time_range_dict:`
			`ret_val= urlencode({'as_ylo': datetime.now().year -1 })`
			`return '&' + ret_val`


			`def request(query, params):`
			`"""Google-Scholar search request"""`

			`offset = (params['pageno'] - 1) * 10`
			`lang_info = get_lang_info(`
			`# pylint: disable=undefined-variable`
[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 06:18:07 +00:00			`params, supported_languages, language_aliases, False`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`)`
[fix] log messages from: google- images, news, scholar, videos - HTTP header Accept-Language --> lang_info['headers']['Accept-Language'] - remove obsolete query_url log messages which is already logged by httpx._client:HTTP request Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-06-11 14:31:50 +00:00			`logger.debug(`
			`"HTTP header Accept-Language --> %s", lang_info['headers']['Accept-Language'])`

[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# subdomain is: scholar.google.xy`
			`lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")`

			`query_url = 'https://'+ lang_info['subdomain'] + '/scholar' + "?" + urlencode({`
			`'q': query,`
[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 06:18:07 +00:00			`**lang_info['params'],`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`'ie': "utf8",`
			`'oe': "utf8",`
			`'start' : offset,`
			`})`

			`query_url += time_range_url(params)`
			`params['url'] = query_url`

[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 06:18:07 +00:00			`params['headers'].update(lang_info['headers'])`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`params['headers']['Accept'] = (`
			`'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,/;q=0.8'`
			`)`

			`#params['google_subdomain'] = subdomain`
			`return params`

			`def response(resp):`
			`"""Get response from google's search request"""`
			`results = []`

			`detect_google_sorry(resp)`

			`# which subdomain ?`
			`# subdomain = resp.search_params.get('google_subdomain')`

			`# convert the text to dom`
			`dom = html.fromstring(resp.text)`

			`# parse results`
			`for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):`

			`title = extract_text(eval_xpath(result, './h3[1]//a'))`

			`if not title:`
			`# this is a [ZITATION] block`
			`continue`

			`url = eval_xpath(result, './h3[1]//a/@href')[0]`
			`content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''`

			`pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))`
			`if pub_info:`
			`content += "[%s]" % pub_info`

			`pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))`
			`if pub_type:`
			`title = title + " " + pub_type`

			`results.append({`
			`'url': url,`
			`'title': title,`
			`'content': content,`
			`})`

			`# parse suggestion`
			`for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):`
			`# append suggestion`
			`results.append({'suggestion': extract_text(suggestion)})`

			`for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):`
			`results.append({'correction': extract_text(correction)})`

			`return results`