| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149 | """ DuckDuckGo (Web) @website     https://duckduckgo.com/ @provide-api yes (https://duckduckgo.com/api),              but not all results from search-site @using-api   no @results     HTML (using search portal) @stable      no (HTML can change) @parse       url, title, content @todo        rewrite to api"""from lxml.html import fromstringfrom json import loadsfrom searx.engines.xpath import extract_textfrom searx.poolrequests import getfrom searx.url_utils import urlencodefrom searx.utils import match_language, eval_xpath# engine dependent configcategories = ['general']paging = Truelanguage_support = Truesupported_languages_url = 'https://duckduckgo.com/util/u172.js'time_range_support = Truelanguage_aliases = {    'ar-SA': 'ar-XA',    'es-419': 'es-XL',    'ja': 'jp-JP',    'ko': 'kr-KR',    'sl-SI': 'sl-SL',    'zh-TW': 'tzh-TW',    'zh-HK': 'tzh-HK'}# search-urlurl = 'https://duckduckgo.com/html?{query}&s={offset}&dc={dc_param}'time_range_url = '&df={range}'time_range_dict = {'day': 'd',                   'week': 'w',                   'month': 'm'}# specific xpath variablesresult_xpath = '//div[@class="result results_links results_links_deep web-result "]'  # noqaurl_xpath = './/a[@class="result__a"]/@href'title_xpath = './/a[@class="result__a"]'content_xpath = './/a[@class="result__snippet"]'correction_xpath = '//div[@id="did_you_mean"]//a'# match query's language to a region code that duckduckgo will acceptdef get_region_code(lang, lang_list=[]):    if lang == 'all':        return None    lang_code = match_language(lang, lang_list, language_aliases, 'wt-WT')    lang_parts = lang_code.split('-')    # country code goes first    return lang_parts[1].lower() + '-' + lang_parts[0].lower()def request(query, params):    if params['time_range'] not in (None, 'None', '') and params['time_range'] not in time_range_dict:        return params    offset = (params['pageno'] - 1) * 30    region_code = get_region_code(params['language'], supported_languages)    params['url'] = 'https://duckduckgo.com/html/'    if params['pageno'] > 1:        params['method'] = 'POST'        params['data']['q'] = query        params['data']['s'] = offset        params['data']['dc'] = 30        params['data']['nextParams'] = ''        params['data']['v'] = 'l'        params['data']['o'] = 'json'        params['data']['api'] = '/d.js'        if params['time_range'] in time_range_dict:            params['data']['df'] = time_range_dict[params['time_range']]        if region_code:            params['data']['kl'] = region_code    else:        if region_code:            params['url'] = url.format(                query=urlencode({'q': query, 'kl': region_code}), offset=offset, dc_param=offset)        else:            params['url'] = url.format(                query=urlencode({'q': query}), offset=offset, dc_param=offset)        if params['time_range'] in time_range_dict:            params['url'] += time_range_url.format(range=time_range_dict[params['time_range']])    return params# get response from search-requestdef response(resp):    results = []    doc = fromstring(resp.text)    # parse results    for i, r in enumerate(eval_xpath(doc, result_xpath)):        if i >= 30:            break        try:            res_url = eval_xpath(r, url_xpath)[-1]        except:            continue        if not res_url:            continue        title = extract_text(eval_xpath(r, title_xpath))        content = extract_text(eval_xpath(r, content_xpath))        # append result        results.append({'title': title,                        'content': content,                        'url': res_url})    # parse correction    for correction in eval_xpath(doc, correction_xpath):        # append correction        results.append({'correction': extract_text(correction)})    # return results    return results# get supported languages from their sitedef _fetch_supported_languages(resp):    # response is a js file with regions as an embedded object    response_page = resp.text    response_page = response_page[response_page.find('regions:{') + 8:]    response_page = response_page[:response_page.find('}') + 1]    regions_json = loads(response_page)    supported_languages = map((lambda x: x[3:] + '-' + x[:2].upper()), regions_json.keys())    return list(supported_languages)
 |