searx/searx/engines/google_scholar.py

# SPDX-License-Identifier: AGPL-3.0-or-later
"""Google (Scholar)

For detailed description of the *REST-full* API see: `Query Parameter
Definitions`_.

.. _Query Parameter Definitions:
   https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions
"""

# pylint: disable=invalid-name, missing-function-docstring

from urllib.parse import urlencode
from datetime import datetime
from random import random
from lxml import html
from searx import logger

from searx.utils import (
    eval_xpath,
    eval_xpath_list,
    extract_text,
)

from searx.engines.google import (
    get_lang_info,
    time_range_dict,
    detect_google_sorry,
)

# pylint: disable=unused-import
from searx.engines.google import (
    supported_languages_url,
    _fetch_supported_languages,
)
# pylint: enable=unused-import

# about
about = {
    "website": 'https://scholar.google.com',
    "wikidata_id": 'Q494817',
    "official_api_documentation": 'https://developers.google.com/custom-search',
    "use_official_api": False,
    "require_api_key": False,
    "results": 'HTML',
}

# engine dependent config
categories = ['science']
paging = True
language_support = True
use_locale_domain = True
time_range_support = True
safesearch = False

logger = logger.getChild('google scholar')

def time_range_url(params):
    """Returns a URL query component for a google-Scholar time range based on
    ``params['time_range']``.  Google-Scholar does only support ranges in years.
    To have any effect, all the Searx ranges (*day*, *week*, *month*, *year*)
    are mapped to *year*.  If no range is set, an empty string is returned.
    Example::

        &as_ylo=2019
    """
    # as_ylo=2016&as_yhi=2019
    ret_val = ''
    if params['time_range'] in time_range_dict:
        ret_val= urlencode({'as_ylo': datetime.now().year -1 })
    return '&' + ret_val


def request(query, params):
    """Google-Scholar search request"""

    offset = (params['pageno'] - 1) * 10
    lang_info = get_lang_info(
        # pylint: disable=undefined-variable


        # params, {}, language_aliases

        params, supported_languages, language_aliases, False
    )
    # subdomain is: scholar.google.xy
    lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")

    query_url = (
        'https://'
        + lang_info['subdomain']
        + '/scholar'
        + "?"
        + urlencode({'q': query, **lang_info['params'], 'ie': "utf8", 'oe': "utf8", 'start': offset, 'ucbcb': 1})
    )

    query_url += time_range_url(params)

    logger.debug("query_url --> %s", query_url)
    params['url'] = query_url

    logger.debug("HTTP header Accept-Language --> %s", lang_info.get('Accept-Language'))
    params['cookies']['CONSENT'] = "PENDING+" + str(random()*100)
    params['headers'].update(lang_info['headers'])
    params['headers']['Accept'] = (
        'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'
    )

    #params['google_subdomain'] = subdomain
    return params

def response(resp):
    """Get response from google's search request"""
    results = []

    detect_google_sorry(resp)

    # which subdomain ?
    # subdomain = resp.search_params.get('google_subdomain')

    # convert the text to dom
    dom = html.fromstring(resp.text)

    # parse results
    for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):

        title = extract_text(eval_xpath(result, './h3[1]//a'))

        if not title:
            # this is a [ZITATION] block
            continue

        url = eval_xpath(result, './h3[1]//a/@href')[0]
        content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''

        pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))
        if pub_info:
            content += "[%s]" % pub_info

        pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))
        if pub_type:
            title = title + " " + pub_type

        results.append({
            'url':      url,
            'title':    title,
            'content':  content,
        })

    # parse suggestion
    for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):
        # append suggestion
        results.append({'suggestion': extract_text(suggestion)})

    for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):
        results.append({'correction': extract_text(correction)})

    return results
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 15:02:11 +01:00			`# SPDX-License-Identifier: AGPL-3.0-or-later`
			`"""Google (Scholar)`

			For detailed description of the REST-full API see: `Query Parameter
			Definitions`_.

			`.. _Query Parameter Definitions:`
			`https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions`
			`"""`

			`# pylint: disable=invalid-name, missing-function-docstring`

			`from urllib.parse import urlencode`
			`from datetime import datetime`
Do not consent to tracking when using google 2022-08-02 19:19:12 +02:00			`from random import random`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 15:02:11 +01:00			`from lxml import html`
			`from searx import logger`

			`from searx.utils import (`
			`eval_xpath,`
			`eval_xpath_list,`
			`extract_text,`
			`)`

			`from searx.engines.google import (`
			`get_lang_info,`
			`time_range_dict,`
			`detect_google_sorry,`
			`)`

			`# pylint: disable=unused-import`
			`from searx.engines.google import (`
			`supported_languages_url,`
			`_fetch_supported_languages,`
			`)`
			`# pylint: enable=unused-import`

			`# about`
			`about = {`
			`"website": 'https://scholar.google.com',`
			`"wikidata_id": 'Q494817',`
			`"official_api_documentation": 'https://developers.google.com/custom-search',`
			`"use_official_api": False,`
			`"require_api_key": False,`
			`"results": 'HTML',`
			`}`

			`# engine dependent config`
			`categories = ['science']`
			`paging = True`
			`language_support = True`
			`use_locale_domain = True`
			`time_range_support = True`
			`safesearch = False`

			`logger = logger.getChild('google scholar')`

			`def time_range_url(params):`
			`"""Returns a URL query component for a google-Scholar time range based on`
			``params['time_range']``. Google-Scholar does only support ranges in years.
			`To have any effect, all the Searx ranges (day, week, month, year)`
			`are mapped to year. If no range is set, an empty string is returned.`
			`Example::`

			`&as_ylo=2019`
			`"""`
			`# as_ylo=2016&as_yhi=2019`
			`ret_val = ''`
			`if params['time_range'] in time_range_dict:`
			`ret_val= urlencode({'as_ylo': datetime.now().year -1 })`
			`return '&' + ret_val`


			`def request(query, params):`
			`"""Google-Scholar search request"""`

			`offset = (params['pageno'] - 1) * 10`
			`lang_info = get_lang_info(`
			`# pylint: disable=undefined-variable`


			`# params, {}, language_aliases`

[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 08:18:07 +02:00			`params, supported_languages, language_aliases, False`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 15:02:11 +01:00			`)`
			`# subdomain is: scholar.google.xy`
			`lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")`

pick engine fixes (#3306) * [fix] google engine: results XPath * [fix] google & youtube - set EU consent cookie This change the previous bypass method for Google consent using ``ucbcb=1`` (6face215b8) to accept the consent using ``CONSENT=YES+``. The youtube_noapi and google have a similar API, at least for the consent[1]. Get CONSENT cookie from google reguest:: curl -i "https://www.google.com/search?q=time&tbm=isch" \ -A "Mozilla/5.0 (X11; Linux i686; rv:102.0) Gecko/20100101 Firefox/102.0" \ \| grep -i consent ... location: https://consent.google.com/m?continue=https://www.google.com/search?q%3Dtime%26tbm%3Disch&gl=DE&m=0&pc=irp&uxe=eomtm&hl=en-US&src=1 set-cookie: CONSENT=PENDING+936; expires=Wed, 24-Jul-2024 11:26:20 GMT; path=/; domain=.google.com; Secure ... PENDING & YES [2]: Google change the way for consent about YouTube cookies agreement in EU countries. Instead of showing a popup in the website, YouTube redirects the user to a new webpage at consent.youtube.com domain ... Fix for this is to put a cookie CONSENT with YES+ value for every YouTube request [1] https://github.com/iv-org/invidious/pull/2207 [2] https://github.com/TeamNewPipe/NewPipeExtractor/issues/592 Closes: https://github.com/searxng/searxng/issues/1432 * [fix] sjp engine - convert enginename to a latin1 compliance name The engine name is not only a name its also a identifier that is used in logs, HTTP headers and more. Unicode characters in the name of an engine could cause various issues. Closes: https://github.com/searxng/searxng/issues/1544 Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> * [fix] engine tineye: handle 422 response of not supported img format Closes: https://github.com/searxng/searxng/issues/1449 Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> * bypass google consent with ucbcb=1 * [mod] Adds Lingva translate engine Add the lingva engine (which grabs data from google translate). Results from Lingva are added to the infobox results. * openstreetmap engine: return the localized named. For example: display "Tokyo" instead of "東京都" when the language is English. * [fix] engines/openstreetmap.py typo: user_langage --> user_language Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> * Wikidata engine: ignore dummy entities * Wikidata engine: minor change of the SPARQL request The engine can be slow especially when the query won't return any answer. See https://www.mediawiki.org/wiki/Wikidata_Query_Service/User_Manual/MWAPI#Find_articles_in_Wikipedia_speaking_about_cheese_and_see_which_Wikibase_items_they_correspond_to Co-authored-by: Léon Tiekötter <leon@tiekoetter.com> Co-authored-by: Emilien Devos <contact@emiliendevos.be> Co-authored-by: Markus Heiser <markus.heiser@darmarit.de> Co-authored-by: Emilien Devos <github@emiliendevos.be> Co-authored-by: ta <alt3753.7@gmail.com> Co-authored-by: Alexandre Flament <alex@al-f.net> 2022-07-30 21:45:07 +02:00			`query_url = (`
			`'https://'`
			`+ lang_info['subdomain']`
			`+ '/scholar'`
			`+ "?"`
			`+ urlencode({'q': query, **lang_info['params'], 'ie': "utf8", 'oe': "utf8", 'start': offset, 'ucbcb': 1})`
			`)`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 15:02:11 +01:00
			`query_url += time_range_url(params)`

			`logger.debug("query_url --> %s", query_url)`
			`params['url'] = query_url`

[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 08:18:07 +02:00			`logger.debug("HTTP header Accept-Language --> %s", lang_info.get('Accept-Language'))`
Do not consent to tracking when using google 2022-08-02 19:19:12 +02:00			`params['cookies']['CONSENT'] = "PENDING+" + str(random()*100)`
[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 08:18:07 +02:00			`params['headers'].update(lang_info['headers'])`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 15:02:11 +01:00			`params['headers']['Accept'] = (`
			`'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,/;q=0.8'`
			`)`

			`#params['google_subdomain'] = subdomain`
			`return params`

			`def response(resp):`
			`"""Get response from google's search request"""`
			`results = []`

			`detect_google_sorry(resp)`

			`# which subdomain ?`
			`# subdomain = resp.search_params.get('google_subdomain')`

			`# convert the text to dom`
			`dom = html.fromstring(resp.text)`

			`# parse results`
			`for result in eval_xpath_list(dom, '//div[@class="gs_ri"]'):`

			`title = extract_text(eval_xpath(result, './h3[1]//a'))`

			`if not title:`
			`# this is a [ZITATION] block`
			`continue`

			`url = eval_xpath(result, './h3[1]//a/@href')[0]`
			`content = extract_text(eval_xpath(result, './div[@class="gs_rs"]')) or ''`

			`pub_info = extract_text(eval_xpath(result, './div[@class="gs_a"]'))`
			`if pub_info:`
			`content += "[%s]" % pub_info`

			`pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))`
			`if pub_type:`
			`title = title + " " + pub_type`

			`results.append({`
			`'url': url,`
			`'title': title,`
			`'content': content,`
			`})`

			`# parse suggestion`
			`for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):`
			`# append suggestion`
			`results.append({'suggestion': extract_text(suggestion)})`

			`for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):`
			`results.append({'correction': extract_text(correction)})`

			`return results`