searxng/searx/engines/google_scholar.py

# SPDX-License-Identifier: AGPL-3.0-or-later
# lint: pylint
"""Google (Scholar)

For detailed description of the *REST-full* API see: `Query Parameter
Definitions`_.

.. _Query Parameter Definitions:
   https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions
"""

# pylint: disable=invalid-name

from urllib.parse import urlencode
from datetime import datetime
from typing import Optional
from lxml import html

from searx.utils import (
    eval_xpath,
    eval_xpath_getindex,
    eval_xpath_list,
    extract_text,
)

from searx.engines.google import (
    get_lang_info,
    time_range_dict,
    detect_google_sorry,
)

# pylint: disable=unused-import
from searx.engines.google import (
    supported_languages_url,
    _fetch_supported_languages,
)

# pylint: enable=unused-import

# about
about = {
    "website": 'https://scholar.google.com',
    "wikidata_id": 'Q494817',
    "official_api_documentation": 'https://developers.google.com/custom-search',
    "use_official_api": False,
    "require_api_key": False,
    "results": 'HTML',
}

# engine dependent config
categories = ['science', 'scientific publications']
paging = True
language_support = True
use_locale_domain = True
time_range_support = True
safesearch = False
send_accept_language_header = True


def time_range_url(params):
    """Returns a URL query component for a google-Scholar time range based on
    ``params['time_range']``.  Google-Scholar does only support ranges in years.
    To have any effect, all the Searx ranges (*day*, *week*, *month*, *year*)
    are mapped to *year*.  If no range is set, an empty string is returned.
    Example::

        &as_ylo=2019
    """
    # as_ylo=2016&as_yhi=2019
    ret_val = ''
    if params['time_range'] in time_range_dict:
        ret_val = urlencode({'as_ylo': datetime.now().year - 1})
    return '&' + ret_val


def request(query, params):
    """Google-Scholar search request"""

    offset = (params['pageno'] - 1) * 10
    lang_info = get_lang_info(params, supported_languages, language_aliases, False)

    # subdomain is: scholar.google.xy
    lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")

    query_url = (
        'https://'
        + lang_info['subdomain']
        + '/scholar'
        + "?"
        + urlencode({'q': query, **lang_info['params'], 'ie': "utf8", 'oe': "utf8", 'start': offset})
    )

    query_url += time_range_url(params)
    params['url'] = query_url

    params['cookies']['CONSENT'] = "YES+"
    params['headers'].update(lang_info['headers'])
    params['headers']['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'

    # params['google_subdomain'] = subdomain
    return params


def parse_gs_a(text: Optional[str]):
    """Parse the text written in green.

    Possible formats:
    * "{authors} - {journal}, {year} - {publisher}"
    * "{authors} - {year} - {publisher}"
    * "{authors} - {publisher}"
    """
    if text is None or text == "":
        return None, None, None, None

    s_text = text.split(' - ')
    authors = s_text[0].split(', ')
    publisher = s_text[-1]
    if len(s_text) != 3:
        return authors, None, publisher, None

    # the format is "{authors} - {journal}, {year} - {publisher}" or "{authors} - {year} - {publisher}"
    # get journal and year
    journal_year = s_text[1].split(', ')
    # journal is optional and may contains some coma
    if len(journal_year) > 1:
        journal = ', '.join(journal_year[0:-1])
        if journal == '…':
            journal = None
    else:
        journal = None
    # year
    year = journal_year[-1]
    try:
        publishedDate = datetime.strptime(year.strip(), '%Y')
    except ValueError:
        publishedDate = None
    return authors, journal, publisher, publishedDate


def response(resp):  # pylint: disable=too-many-locals
    """Get response from google's search request"""
    results = []

    detect_google_sorry(resp)

    # which subdomain ?
    # subdomain = resp.search_params.get('google_subdomain')

    # convert the text to dom
    dom = html.fromstring(resp.text)

    # parse results
    for result in eval_xpath_list(dom, '//div[@data-cid]'):

        title = extract_text(eval_xpath(result, './/h3[1]//a'))

        if not title:
            # this is a [ZITATION] block
            continue

        pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))
        if pub_type:
            pub_type = pub_type[1:-1].lower()

        url = eval_xpath_getindex(result, './/h3[1]//a/@href', 0)
        content = extract_text(eval_xpath(result, './/div[@class="gs_rs"]'))
        authors, journal, publisher, publishedDate = parse_gs_a(
            extract_text(eval_xpath(result, './/div[@class="gs_a"]'))
        )
        if publisher in url:
            publisher = None

        # cited by
        comments = extract_text(eval_xpath(result, './/div[@class="gs_fl"]/a[starts-with(@href,"/scholar?cites=")]'))

        # link to the html or pdf document
        html_url = None
        pdf_url = None
        doc_url = eval_xpath_getindex(result, './/div[@class="gs_or_ggsm"]/a/@href', 0, default=None)
        doc_type = extract_text(eval_xpath(result, './/span[@class="gs_ctg2"]'))
        if doc_type == "[PDF]":
            pdf_url = doc_url
        else:
            html_url = doc_url

        results.append(
            {
                'template': 'paper.html',
                'type': pub_type,
                'url': url,
                'title': title,
                'authors': authors,
                'publisher': publisher,
                'journal': journal,
                'publishedDate': publishedDate,
                'content': content,
                'comments': comments,
                'html_url': html_url,
                'pdf_url': pdf_url,
            }
        )

    # parse suggestion
    for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):
        # append suggestion
        results.append({'suggestion': extract_text(suggestion)})

    for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):
        results.append({'correction': extract_text(correction)})

    return results
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# SPDX-License-Identifier: AGPL-3.0-or-later`
[pylint] tag PYLINT_FILES by comment `# lint: pylint` These py files are linted by `test.pylint`, all other files are linted by `test.pep8`. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-04-26 18:18:20 +00:00			`# lint: pylint`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`"""Google (Scholar)`

			For detailed description of the REST-full API see: `Query Parameter
			Definitions`_.

			`.. _Query Parameter Definitions:`
			`https://developers.google.com/custom-search/docs/xml_results#WebSearch_Query_Parameter_Definitions`
			`"""`

[pylint] engines: drop no longer needed 'missing-function-docstring' Suggested-by: @dalf https://github.com/searxng/searxng/issues/102#issuecomment-914168470 Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-09-07 11:26:59 +00:00			`# pylint: disable=invalid-name`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`from urllib.parse import urlencode`
			`from datetime import datetime`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`from typing import Optional`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`from lxml import html`

			`from searx.utils import (`
			`eval_xpath,`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`eval_xpath_getindex,`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`eval_xpath_list,`
			`extract_text,`
			`)`

			`from searx.engines.google import (`
			`get_lang_info,`
			`time_range_dict,`
			`detect_google_sorry,`
			`)`

			`# pylint: disable=unused-import`
			`from searx.engines.google import (`
			`supported_languages_url,`
			`_fetch_supported_languages,`
			`)`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# pylint: enable=unused-import`

			`# about`
			`about = {`
			`"website": 'https://scholar.google.com',`
			`"wikidata_id": 'Q494817',`
			`"official_api_documentation": 'https://developers.google.com/custom-search',`
			`"use_official_api": False,`
			`"require_api_key": False,`
			`"results": 'HTML',`
			`}`

			`# engine dependent config`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`categories = ['science', 'scientific publications']`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`paging = True`
			`language_support = True`
			`use_locale_domain = True`
			`time_range_support = True`
			`safesearch = False`
[mod] add 'Accept-Language' HTTP header to online processores Most engines that support languages (and regions) use the Accept-Language from the WEB browser to build a response that fits to the language (and region). - add new engine option: send_accept_language_header Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2022-08-01 15:01:59 +00:00			`send_accept_language_header = True`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`def time_range_url(params):`
			`"""Returns a URL query component for a google-Scholar time range based on`
			``params['time_range']``. Google-Scholar does only support ranges in years.
			`To have any effect, all the Searx ranges (day, week, month, year)`
			`are mapped to year. If no range is set, an empty string is returned.`
			`Example::`

			`&as_ylo=2019`
			`"""`
			`# as_ylo=2016&as_yhi=2019`
			`ret_val = ''`
			`if params['time_range'] in time_range_dict:`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`ret_val = urlencode({'as_ylo': datetime.now().year - 1})`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`return '&' + ret_val`


			`def request(query, params):`
			`"""Google-Scholar search request"""`

			`offset = (params['pageno'] - 1) * 10`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`lang_info = get_lang_info(params, supported_languages, language_aliases, False)`
[fix] log messages from: google- images, news, scholar, videos - HTTP header Accept-Language --> lang_info['headers']['Accept-Language'] - remove obsolete query_url log messages which is already logged by httpx._client:HTTP request Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-06-11 14:31:50 +00:00
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`# subdomain is: scholar.google.xy`
			`lang_info['subdomain'] = lang_info['subdomain'].replace("www.", "scholar.")`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`query_url = (`
			`'https://'`
			`+ lang_info['subdomain']`
			`+ '/scholar'`
			`+ "?"`
[fix] google & youtube - set EU consent cookie This change the previous bypass method for Google consent using ``ucbcb=1`` (6face215b8) to accept the consent using ``CONSENT=YES+``. The youtube_noapi and google have a similar API, at least for the consent[1]. Get CONSENT cookie from google reguest:: curl -i "https://www.google.com/search?q=time&tbm=isch" \ -A "Mozilla/5.0 (X11; Linux i686; rv:102.0) Gecko/20100101 Firefox/102.0" \ \| grep -i consent ... location: https://consent.google.com/m?continue=https://www.google.com/search?q%3Dtime%26tbm%3Disch&gl=DE&m=0&pc=irp&uxe=eomtm&hl=en-US&src=1 set-cookie: CONSENT=PENDING+936; expires=Wed, 24-Jul-2024 11:26:20 GMT; path=/; domain=.google.com; Secure ... PENDING & YES [2]: Google change the way for consent about YouTube cookies agreement in EU countries. Instead of showing a popup in the website, YouTube redirects the user to a new webpage at consent.youtube.com domain ... Fix for this is to put a cookie CONSENT with YES+ value for every YouTube request [1] https://github.com/iv-org/invidious/pull/2207 [2] https://github.com/TeamNewPipe/NewPipeExtractor/issues/592 Closes: https://github.com/searxng/searxng/issues/1432 2022-07-25 10:53:56 +00:00			`+ urlencode({'q': query, **lang_info['params'], 'ie': "utf8", 'oe': "utf8", 'start': offset})`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`)`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`query_url += time_range_url(params)`
			`params['url'] = query_url`

[fix] google & youtube - set EU consent cookie This change the previous bypass method for Google consent using ``ucbcb=1`` (6face215b8) to accept the consent using ``CONSENT=YES+``. The youtube_noapi and google have a similar API, at least for the consent[1]. Get CONSENT cookie from google reguest:: curl -i "https://www.google.com/search?q=time&tbm=isch" \ -A "Mozilla/5.0 (X11; Linux i686; rv:102.0) Gecko/20100101 Firefox/102.0" \ \| grep -i consent ... location: https://consent.google.com/m?continue=https://www.google.com/search?q%3Dtime%26tbm%3Disch&gl=DE&m=0&pc=irp&uxe=eomtm&hl=en-US&src=1 set-cookie: CONSENT=PENDING+936; expires=Wed, 24-Jul-2024 11:26:20 GMT; path=/; domain=.google.com; Secure ... PENDING & YES [2]: Google change the way for consent about YouTube cookies agreement in EU countries. Instead of showing a popup in the website, YouTube redirects the user to a new webpage at consent.youtube.com domain ... Fix for this is to put a cookie CONSENT with YES+ value for every YouTube request [1] https://github.com/iv-org/invidious/pull/2207 [2] https://github.com/TeamNewPipe/NewPipeExtractor/issues/592 Closes: https://github.com/searxng/searxng/issues/1432 2022-07-25 10:53:56 +00:00			`params['cookies']['CONSENT'] = "YES+"`
[enh] google engine: supports "default language" Same behaviour behaviour than Whoogle [1]. Only the google engine with the "Default language" choice "(all)"" is changed by this patch. When searching for a locate place, the result are in the expect language, without missing results [2]: > When a language is not specified, the language interpretation is left up to > Google to decide how the search results should be delivered. The query parameters are copied from Whoogle. With the ``all`` language: - add parameter ``source=lnt`` - don't use parameter ``lr`` - don't add a ``Accept-Language`` HTTP header. The new signature of function ``get_lang_info()`` is: lang_info = get_lang_info(params, lang_list, custom_aliases, supported_any_language) Argument ``supported_any_language`` is True for google.py and False for the other google engines. With this patch the function now returns: - query parameters: ``lang_info['params']`` - HTTP headers: ``lang_info['headers']`` - and as before this patch: - ``lang_info['subdomain']`` - ``lang_info['country']`` - ``lang_info['language']`` [1] https://github.com/benbusby/whoogle-search [2] https://github.com/benbusby/whoogle-search/releases/tag/v0.5.4 2021-06-06 06:18:07 +00:00			`params['headers'].update(lang_info['headers'])`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`params['headers']['Accept'] = 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,/;q=0.8'`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`# params['google_subdomain'] = subdomain`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`return params`

[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`def parse_gs_a(text: Optional[str]):`
			`"""Parse the text written in green.`

			`Possible formats:`
			`* "{authors} - {journal}, {year} - {publisher}"`
			`* "{authors} - {year} - {publisher}"`
			`* "{authors} - {publisher}"`
			`"""`
			`if text is None or text == "":`
			`return None, None, None, None`

			`s_text = text.split(' - ')`
			`authors = s_text[0].split(', ')`
			`publisher = s_text[-1]`
			`if len(s_text) != 3:`
			`return authors, None, publisher, None`

			`# the format is "{authors} - {journal}, {year} - {publisher}" or "{authors} - {year} - {publisher}"`
			`# get journal and year`
			`journal_year = s_text[1].split(', ')`
			`# journal is optional and may contains some coma`
			`if len(journal_year) > 1:`
			`journal = ', '.join(journal_year[0:-1])`
			`if journal == '…':`
			`journal = None`
			`else:`
			`journal = None`
			`# year`
			`year = journal_year[-1]`
			`try:`
			`publishedDate = datetime.strptime(year.strip(), '%Y')`
			`except ValueError:`
			`publishedDate = None`
			`return authors, journal, publisher, publishedDate`


			`def response(resp): # pylint: disable=too-many-locals`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00			`"""Get response from google's search request"""`
			`results = []`

			`detect_google_sorry(resp)`

			`# which subdomain ?`
			`# subdomain = resp.search_params.get('google_subdomain')`

			`# convert the text to dom`
			`dom = html.fromstring(resp.text)`

			`# parse results`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`for result in eval_xpath_list(dom, '//div[@data-cid]'):`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`title = extract_text(eval_xpath(result, './/h3[1]//a'))`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`if not title:`
			`# this is a [ZITATION] block`
			`continue`

			`pub_type = extract_text(eval_xpath(result, './/span[@class="gs_ct1"]'))`
			`if pub_type:`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`pub_type = pub_type[1:-1].lower()`

			`url = eval_xpath_getindex(result, './/h3[1]//a/@href', 0)`
			`content = extract_text(eval_xpath(result, './/div[@class="gs_rs"]'))`
			`authors, journal, publisher, publishedDate = parse_gs_a(`
			`extract_text(eval_xpath(result, './/div[@class="gs_a"]'))`
			`)`
			`if publisher in url:`
			`publisher = None`

			`# cited by`
			`comments = extract_text(eval_xpath(result, './/div[@class="gs_fl"]/a[starts-with(@href,"/scholar?cites=")]'))`

			`# link to the html or pdf document`
			`html_url = None`
			`pdf_url = None`
			`doc_url = eval_xpath_getindex(result, './/div[@class="gs_or_ggsm"]/a/@href', 0, default=None)`
			`doc_type = extract_text(eval_xpath(result, './/span[@class="gs_ctg2"]'))`
			`if doc_type == "[PDF]":`
			`pdf_url = doc_url`
			`else:`
			`html_url = doc_url`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`results.append(`
			`{`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`'template': 'paper.html',`
			`'type': pub_type,`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`'url': url,`
			`'title': title,`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`'authors': authors,`
			`'publisher': publisher,`
			`'journal': journal,`
			`'publishedDate': publishedDate,`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`'content': content,`
Science category: update the engines * use the paper.html template * fetch more data from the engines * add crossref.py 2022-08-26 16:10:12 +00:00			`'comments': comments,`
			`'html_url': html_url,`
			`'pdf_url': pdf_url,`
[format.python] initial formatting of the python code This patch was generated by black [1]:: make format.python [1] https://github.com/psf/black Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-12-27 08:26:22 +00:00			`}`
			`)`
[enh] google scholar - python implementation of the engine The old xpath configuration for google scholar did not work and is replaced by a python implementation. Signed-off-by: Markus Heiser <markus.heiser@darmarit.de> 2021-03-01 14:02:11 +00:00
			`# parse suggestion`
			`for suggestion in eval_xpath(dom, '//div[contains(@class, "gs_qsuggest_wrap")]//li//a'):`
			`# append suggestion`
			`results.append({'suggestion': extract_text(suggestion)})`

			`for correction in eval_xpath(dom, '//div[@class="gs_r gs_pda"]/a'):`
			`results.append({'correction': extract_text(correction)})`

			`return results`