summaryrefslogtreecommitdiff
path: root/_modules/searx/engines/startpage.html
diff options
context:
space:
mode:
Diffstat (limited to '_modules/searx/engines/startpage.html')
-rw-r--r--_modules/searx/engines/startpage.html611
1 files changed, 611 insertions, 0 deletions
diff --git a/_modules/searx/engines/startpage.html b/_modules/searx/engines/startpage.html
new file mode 100644
index 000000000..53df59b2e
--- /dev/null
+++ b/_modules/searx/engines/startpage.html
@@ -0,0 +1,611 @@
+<!DOCTYPE html>
+
+<html lang="en" data-content_root="../../../">
+ <head>
+ <meta charset="utf-8" />
+ <meta name="viewport" content="width=device-width, initial-scale=1.0" />
+ <meta name="viewport" content="width=device-width, initial-scale=1">
+ <title>searx.engines.startpage &#8212; SearXNG Documentation (2024.5.17+ec41b5358)</title>
+ <link rel="stylesheet" type="text/css" href="../../../_static/pygments.css?v=4f649999" />
+ <link rel="stylesheet" type="text/css" href="../../../_static/searxng.css?v=52e4ff28" />
+ <link rel="stylesheet" type="text/css" href="../../../_static/tabs.css?v=a5c4661c" />
+ <script src="../../../_static/documentation_options.js?v=619ad1c8"></script>
+ <script src="../../../_static/doctools.js?v=9a2dae69"></script>
+ <script src="../../../_static/sphinx_highlight.js?v=dc90522c"></script>
+ <script src="../../../_static/tabs.js?v=3030b3cb"></script>
+ <link rel="index" title="Index" href="../../../genindex.html" />
+ <link rel="search" title="Search" href="../../../search.html" />
+ </head><body>
+ <div class="related" role="navigation" aria-label="related navigation">
+ <h3>Navigation</h3>
+ <ul>
+ <li class="right" style="margin-right: 10px">
+ <a href="../../../genindex.html" title="General Index"
+ accesskey="I">index</a></li>
+ <li class="right" >
+ <a href="../../../py-modindex.html" title="Python Module Index"
+ >modules</a> |</li>
+ <li class="nav-item nav-item-0"><a href="../../../index.html">SearXNG Documentation (2024.5.17+ec41b5358)</a> &#187;</li>
+ <li class="nav-item nav-item-1"><a href="../../index.html" >Module code</a> &#187;</li>
+ <li class="nav-item nav-item-2"><a href="../engines.html" accesskey="U">searx.engines</a> &#187;</li>
+ <li class="nav-item nav-item-this"><a href="">searx.engines.startpage</a></li>
+ </ul>
+ </div>
+
+ <div class="document">
+ <div class="documentwrapper">
+ <div class="bodywrapper">
+ <div class="body" role="main">
+
+ <h1>Source code for searx.engines.startpage</h1><div class="highlight"><pre>
+<span></span><span class="c1"># SPDX-License-Identifier: AGPL-3.0-or-later</span>
+<span class="sd">&quot;&quot;&quot;Startpage&#39;s language &amp; region selectors are a mess ..</span>
+
+<span class="sd">.. _startpage regions:</span>
+
+<span class="sd">Startpage regions</span>
+<span class="sd">=================</span>
+
+<span class="sd">In the list of regions there are tags we need to map to common region tags::</span>
+
+<span class="sd"> pt-BR_BR --&gt; pt_BR</span>
+<span class="sd"> zh-CN_CN --&gt; zh_Hans_CN</span>
+<span class="sd"> zh-TW_TW --&gt; zh_Hant_TW</span>
+<span class="sd"> zh-TW_HK --&gt; zh_Hant_HK</span>
+<span class="sd"> en-GB_GB --&gt; en_GB</span>
+
+<span class="sd">and there is at least one tag with a three letter language tag (ISO 639-2)::</span>
+
+<span class="sd"> fil_PH --&gt; fil_PH</span>
+
+<span class="sd">The locale code ``no_NO`` from Startpage does not exists and is mapped to</span>
+<span class="sd">``nb-NO``::</span>
+
+<span class="sd"> babel.core.UnknownLocaleError: unknown locale &#39;no_NO&#39;</span>
+
+<span class="sd">For reference see languages-subtag at iana; ``no`` is the macrolanguage [1]_ and</span>
+<span class="sd">W3C recommends subtag over macrolanguage [2]_.</span>
+
+<span class="sd">.. [1] `iana: language-subtag-registry</span>
+<span class="sd"> &lt;https://www.iana.org/assignments/language-subtag-registry/language-subtag-registry&gt;`_ ::</span>
+
+<span class="sd"> type: language</span>
+<span class="sd"> Subtag: nb</span>
+<span class="sd"> Description: Norwegian Bokmål</span>
+<span class="sd"> Added: 2005-10-16</span>
+<span class="sd"> Suppress-Script: Latn</span>
+<span class="sd"> Macrolanguage: no</span>
+
+<span class="sd">.. [2]</span>
+<span class="sd"> Use macrolanguages with care. Some language subtags have a Scope field set to</span>
+<span class="sd"> macrolanguage, i.e. this primary language subtag encompasses a number of more</span>
+<span class="sd"> specific primary language subtags in the registry. ... As we recommended for</span>
+<span class="sd"> the collection subtags mentioned above, in most cases you should try to use</span>
+<span class="sd"> the more specific subtags ... `W3: The primary language subtag</span>
+<span class="sd"> &lt;https://www.w3.org/International/questions/qa-choosing-language-tags#langsubtag&gt;`_</span>
+
+<span class="sd">.. _startpage languages:</span>
+
+<span class="sd">Startpage languages</span>
+<span class="sd">===================</span>
+
+<span class="sd">:py:obj:`send_accept_language_header`:</span>
+<span class="sd"> The displayed name in Startpage&#39;s settings page depend on the location of the</span>
+<span class="sd"> IP when ``Accept-Language`` HTTP header is unset. In :py:obj:`fetch_traits`</span>
+<span class="sd"> we use::</span>
+
+<span class="sd"> &#39;Accept-Language&#39;: &quot;en-US,en;q=0.5&quot;,</span>
+<span class="sd"> ..</span>
+
+<span class="sd"> to get uniform names independent from the IP).</span>
+
+<span class="sd">.. _startpage categories:</span>
+
+<span class="sd">Startpage categories</span>
+<span class="sd">====================</span>
+
+<span class="sd">Startpage&#39;s category (for Web-search, News, Videos, ..) is set by</span>
+<span class="sd">:py:obj:`startpage_categ` in settings.yml::</span>
+
+<span class="sd"> - name: startpage</span>
+<span class="sd"> engine: startpage</span>
+<span class="sd"> startpage_categ: web</span>
+<span class="sd"> ...</span>
+
+<span class="sd">.. hint::</span>
+
+<span class="sd"> The default category is ``web`` .. and other categories than ``web`` are not</span>
+<span class="sd"> yet implemented.</span>
+
+<span class="sd">&quot;&quot;&quot;</span>
+
+<span class="kn">from</span> <span class="nn">typing</span> <span class="kn">import</span> <span class="n">TYPE_CHECKING</span>
+<span class="kn">from</span> <span class="nn">collections</span> <span class="kn">import</span> <span class="n">OrderedDict</span>
+<span class="kn">import</span> <span class="nn">re</span>
+<span class="kn">from</span> <span class="nn">unicodedata</span> <span class="kn">import</span> <span class="n">normalize</span><span class="p">,</span> <span class="n">combining</span>
+<span class="kn">from</span> <span class="nn">time</span> <span class="kn">import</span> <span class="n">time</span>
+<span class="kn">from</span> <span class="nn">datetime</span> <span class="kn">import</span> <span class="n">datetime</span><span class="p">,</span> <span class="n">timedelta</span>
+
+<span class="kn">import</span> <span class="nn">dateutil.parser</span>
+<span class="kn">import</span> <span class="nn">lxml.html</span>
+<span class="kn">import</span> <span class="nn">babel</span>
+
+<span class="kn">from</span> <span class="nn">searx.utils</span> <span class="kn">import</span> <span class="n">extract_text</span><span class="p">,</span> <span class="n">eval_xpath</span><span class="p">,</span> <span class="n">gen_useragent</span>
+<span class="kn">from</span> <span class="nn">searx.network</span> <span class="kn">import</span> <span class="n">get</span> <span class="c1"># see https://github.com/searxng/searxng/issues/762</span>
+<span class="kn">from</span> <span class="nn">searx.exceptions</span> <span class="kn">import</span> <span class="n">SearxEngineCaptchaException</span>
+<span class="kn">from</span> <span class="nn">searx.locales</span> <span class="kn">import</span> <span class="n">region_tag</span>
+<span class="kn">from</span> <span class="nn">searx.enginelib.traits</span> <span class="kn">import</span> <span class="n">EngineTraits</span>
+
+<span class="k">if</span> <span class="n">TYPE_CHECKING</span><span class="p">:</span>
+ <span class="kn">import</span> <span class="nn">logging</span>
+
+ <span class="n">logger</span><span class="p">:</span> <span class="n">logging</span><span class="o">.</span><span class="n">Logger</span>
+
+<span class="n">traits</span><span class="p">:</span> <span class="n">EngineTraits</span>
+
+<span class="c1"># about</span>
+<span class="n">about</span> <span class="o">=</span> <span class="p">{</span>
+ <span class="s2">&quot;website&quot;</span><span class="p">:</span> <span class="s1">&#39;https://startpage.com&#39;</span><span class="p">,</span>
+ <span class="s2">&quot;wikidata_id&quot;</span><span class="p">:</span> <span class="s1">&#39;Q2333295&#39;</span><span class="p">,</span>
+ <span class="s2">&quot;official_api_documentation&quot;</span><span class="p">:</span> <span class="kc">None</span><span class="p">,</span>
+ <span class="s2">&quot;use_official_api&quot;</span><span class="p">:</span> <span class="kc">False</span><span class="p">,</span>
+ <span class="s2">&quot;require_api_key&quot;</span><span class="p">:</span> <span class="kc">False</span><span class="p">,</span>
+ <span class="s2">&quot;results&quot;</span><span class="p">:</span> <span class="s1">&#39;HTML&#39;</span><span class="p">,</span>
+<span class="p">}</span>
+
+<span class="n">startpage_categ</span> <span class="o">=</span> <span class="s1">&#39;web&#39;</span>
+<span class="sd">&quot;&quot;&quot;Startpage&#39;s category, visit :ref:`startpage categories`.</span>
+<span class="sd">&quot;&quot;&quot;</span>
+
+<span class="n">send_accept_language_header</span> <span class="o">=</span> <span class="kc">True</span>
+<span class="sd">&quot;&quot;&quot;Startpage tries to guess user&#39;s language and territory from the HTTP</span>
+<span class="sd">``Accept-Language``. Optional the user can select a search-language (can be</span>
+<span class="sd">different to the UI language) and a region filter.</span>
+<span class="sd">&quot;&quot;&quot;</span>
+
+<span class="c1"># engine dependent config</span>
+<span class="n">categories</span> <span class="o">=</span> <span class="p">[</span><span class="s1">&#39;general&#39;</span><span class="p">,</span> <span class="s1">&#39;web&#39;</span><span class="p">]</span>
+<span class="n">paging</span> <span class="o">=</span> <span class="kc">True</span>
+<span class="n">max_page</span> <span class="o">=</span> <span class="mi">18</span>
+<span class="sd">&quot;&quot;&quot;Tested 18 pages maximum (argument ``page``), to be save max is set to 20.&quot;&quot;&quot;</span>
+
+<span class="n">time_range_support</span> <span class="o">=</span> <span class="kc">True</span>
+<span class="n">safesearch</span> <span class="o">=</span> <span class="kc">True</span>
+
+<span class="n">time_range_dict</span> <span class="o">=</span> <span class="p">{</span><span class="s1">&#39;day&#39;</span><span class="p">:</span> <span class="s1">&#39;d&#39;</span><span class="p">,</span> <span class="s1">&#39;week&#39;</span><span class="p">:</span> <span class="s1">&#39;w&#39;</span><span class="p">,</span> <span class="s1">&#39;month&#39;</span><span class="p">:</span> <span class="s1">&#39;m&#39;</span><span class="p">,</span> <span class="s1">&#39;year&#39;</span><span class="p">:</span> <span class="s1">&#39;y&#39;</span><span class="p">}</span>
+<span class="n">safesearch_dict</span> <span class="o">=</span> <span class="p">{</span><span class="mi">0</span><span class="p">:</span> <span class="s1">&#39;0&#39;</span><span class="p">,</span> <span class="mi">1</span><span class="p">:</span> <span class="s1">&#39;1&#39;</span><span class="p">,</span> <span class="mi">2</span><span class="p">:</span> <span class="s1">&#39;1&#39;</span><span class="p">}</span>
+
+<span class="c1"># search-url</span>
+<span class="n">base_url</span> <span class="o">=</span> <span class="s1">&#39;https://www.startpage.com&#39;</span>
+<span class="n">search_url</span> <span class="o">=</span> <span class="n">base_url</span> <span class="o">+</span> <span class="s1">&#39;/sp/search&#39;</span>
+
+<span class="c1"># specific xpath variables</span>
+<span class="c1"># ads xpath //div[@id=&quot;results&quot;]/div[@id=&quot;sponsored&quot;]//div[@class=&quot;result&quot;]</span>
+<span class="c1"># not ads: div[@class=&quot;result&quot;] are the direct childs of div[@id=&quot;results&quot;]</span>
+<span class="n">search_form_xpath</span> <span class="o">=</span> <span class="s1">&#39;//form[@id=&quot;search&quot;]&#39;</span>
+<span class="sd">&quot;&quot;&quot;XPath of Startpage&#39;s origin search form</span>
+
+<span class="sd">.. code: html</span>
+
+<span class="sd"> &lt;form action=&quot;/sp/search&quot; method=&quot;post&quot;&gt;</span>
+<span class="sd"> &lt;input type=&quot;text&quot; name=&quot;query&quot; value=&quot;&quot; ..&gt;</span>
+<span class="sd"> &lt;input type=&quot;hidden&quot; name=&quot;t&quot; value=&quot;device&quot;&gt;</span>
+<span class="sd"> &lt;input type=&quot;hidden&quot; name=&quot;lui&quot; value=&quot;english&quot;&gt;</span>
+<span class="sd"> &lt;input type=&quot;hidden&quot; name=&quot;sc&quot; value=&quot;Q7Mt5TRqowKB00&quot;&gt;</span>
+<span class="sd"> &lt;input type=&quot;hidden&quot; name=&quot;cat&quot; value=&quot;web&quot;&gt;</span>
+<span class="sd"> &lt;input type=&quot;hidden&quot; class=&quot;abp&quot; id=&quot;abp-input&quot; name=&quot;abp&quot; value=&quot;1&quot;&gt;</span>
+<span class="sd"> &lt;/form&gt;</span>
+<span class="sd">&quot;&quot;&quot;</span>
+
+<span class="c1"># timestamp of the last fetch of &#39;sc&#39; code</span>
+<span class="n">sc_code_ts</span> <span class="o">=</span> <span class="mi">0</span>
+<span class="n">sc_code</span> <span class="o">=</span> <span class="s1">&#39;&#39;</span>
+<span class="n">sc_code_cache_sec</span> <span class="o">=</span> <span class="mi">30</span>
+<span class="sd">&quot;&quot;&quot;Time in seconds the sc-code is cached in memory :py:obj:`get_sc_code`.&quot;&quot;&quot;</span>
+
+
+<div class="viewcode-block" id="get_sc_code">
+<a class="viewcode-back" href="../../../dev/engines/online/startpage.html#searx.engines.startpage.get_sc_code">[docs]</a>
+<span class="k">def</span> <span class="nf">get_sc_code</span><span class="p">(</span><span class="n">searxng_locale</span><span class="p">,</span> <span class="n">params</span><span class="p">):</span>
+<span class="w"> </span><span class="sd">&quot;&quot;&quot;Get an actual ``sc`` argument from Startpage&#39;s search form (HTML page).</span>
+
+<span class="sd"> Startpage puts a ``sc`` argument on every HTML :py:obj:`search form</span>
+<span class="sd"> &lt;search_form_xpath&gt;`. Without this argument Startpage considers the request</span>
+<span class="sd"> is from a bot. We do not know what is encoded in the value of the ``sc``</span>
+<span class="sd"> argument, but it seems to be a kind of a *time-stamp*.</span>
+
+<span class="sd"> Startpage&#39;s search form generates a new sc-code on each request. This</span>
+<span class="sd"> function scrap a new sc-code from Startpage&#39;s home page every</span>
+<span class="sd"> :py:obj:`sc_code_cache_sec` seconds.</span>
+
+<span class="sd"> &quot;&quot;&quot;</span>
+
+ <span class="k">global</span> <span class="n">sc_code_ts</span><span class="p">,</span> <span class="n">sc_code</span> <span class="c1"># pylint: disable=global-statement</span>
+
+ <span class="k">if</span> <span class="n">sc_code</span> <span class="ow">and</span> <span class="p">(</span><span class="n">time</span><span class="p">()</span> <span class="o">&lt;</span> <span class="p">(</span><span class="n">sc_code_ts</span> <span class="o">+</span> <span class="n">sc_code_cache_sec</span><span class="p">)):</span>
+ <span class="n">logger</span><span class="o">.</span><span class="n">debug</span><span class="p">(</span><span class="s2">&quot;get_sc_code: reuse &#39;</span><span class="si">%s</span><span class="s2">&#39;&quot;</span><span class="p">,</span> <span class="n">sc_code</span><span class="p">)</span>
+ <span class="k">return</span> <span class="n">sc_code</span>
+
+ <span class="n">headers</span> <span class="o">=</span> <span class="p">{</span><span class="o">**</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;headers&#39;</span><span class="p">]}</span>
+ <span class="n">headers</span><span class="p">[</span><span class="s1">&#39;Origin&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">base_url</span>
+ <span class="n">headers</span><span class="p">[</span><span class="s1">&#39;Referer&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">base_url</span> <span class="o">+</span> <span class="s1">&#39;/&#39;</span>
+ <span class="c1"># headers[&#39;Connection&#39;] = &#39;keep-alive&#39;</span>
+ <span class="c1"># headers[&#39;Accept-Encoding&#39;] = &#39;gzip, deflate, br&#39;</span>
+ <span class="c1"># headers[&#39;Accept&#39;] = &#39;text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8&#39;</span>
+ <span class="c1"># headers[&#39;User-Agent&#39;] = &#39;Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:105.0) Gecko/20100101 Firefox/105.0&#39;</span>
+
+ <span class="c1"># add Accept-Language header</span>
+ <span class="k">if</span> <span class="n">searxng_locale</span> <span class="o">==</span> <span class="s1">&#39;all&#39;</span><span class="p">:</span>
+ <span class="n">searxng_locale</span> <span class="o">=</span> <span class="s1">&#39;en-US&#39;</span>
+ <span class="n">locale</span> <span class="o">=</span> <span class="n">babel</span><span class="o">.</span><span class="n">Locale</span><span class="o">.</span><span class="n">parse</span><span class="p">(</span><span class="n">searxng_locale</span><span class="p">,</span> <span class="n">sep</span><span class="o">=</span><span class="s1">&#39;-&#39;</span><span class="p">)</span>
+
+ <span class="k">if</span> <span class="n">send_accept_language_header</span><span class="p">:</span>
+ <span class="n">ac_lang</span> <span class="o">=</span> <span class="n">locale</span><span class="o">.</span><span class="n">language</span>
+ <span class="k">if</span> <span class="n">locale</span><span class="o">.</span><span class="n">territory</span><span class="p">:</span>
+ <span class="n">ac_lang</span> <span class="o">=</span> <span class="s2">&quot;</span><span class="si">%s</span><span class="s2">-</span><span class="si">%s</span><span class="s2">,</span><span class="si">%s</span><span class="s2">;q=0.9,*;q=0.5&quot;</span> <span class="o">%</span> <span class="p">(</span>
+ <span class="n">locale</span><span class="o">.</span><span class="n">language</span><span class="p">,</span>
+ <span class="n">locale</span><span class="o">.</span><span class="n">territory</span><span class="p">,</span>
+ <span class="n">locale</span><span class="o">.</span><span class="n">language</span><span class="p">,</span>
+ <span class="p">)</span>
+ <span class="n">headers</span><span class="p">[</span><span class="s1">&#39;Accept-Language&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">ac_lang</span>
+
+ <span class="n">get_sc_url</span> <span class="o">=</span> <span class="n">base_url</span> <span class="o">+</span> <span class="s1">&#39;/?sc=</span><span class="si">%s</span><span class="s1">&#39;</span> <span class="o">%</span> <span class="p">(</span><span class="n">sc_code</span><span class="p">)</span>
+ <span class="n">logger</span><span class="o">.</span><span class="n">debug</span><span class="p">(</span><span class="s2">&quot;query new sc time-stamp ... </span><span class="si">%s</span><span class="s2">&quot;</span><span class="p">,</span> <span class="n">get_sc_url</span><span class="p">)</span>
+ <span class="n">logger</span><span class="o">.</span><span class="n">debug</span><span class="p">(</span><span class="s2">&quot;headers: </span><span class="si">%s</span><span class="s2">&quot;</span><span class="p">,</span> <span class="n">headers</span><span class="p">)</span>
+ <span class="n">resp</span> <span class="o">=</span> <span class="n">get</span><span class="p">(</span><span class="n">get_sc_url</span><span class="p">,</span> <span class="n">headers</span><span class="o">=</span><span class="n">headers</span><span class="p">)</span>
+
+ <span class="c1"># ?? x = network.get(&#39;https://www.startpage.com/sp/cdn/images/filter-chevron.svg&#39;, headers=headers)</span>
+ <span class="c1"># ?? https://www.startpage.com/sp/cdn/images/filter-chevron.svg</span>
+ <span class="c1"># ?? ping-back URL: https://www.startpage.com/sp/pb?sc=TLsB0oITjZ8F21</span>
+
+ <span class="k">if</span> <span class="nb">str</span><span class="p">(</span><span class="n">resp</span><span class="o">.</span><span class="n">url</span><span class="p">)</span><span class="o">.</span><span class="n">startswith</span><span class="p">(</span><span class="s1">&#39;https://www.startpage.com/sp/captcha&#39;</span><span class="p">):</span> <span class="c1"># type: ignore</span>
+ <span class="k">raise</span> <span class="n">SearxEngineCaptchaException</span><span class="p">(</span>
+ <span class="n">message</span><span class="o">=</span><span class="s2">&quot;get_sc_code: got redirected to https://www.startpage.com/sp/captcha&quot;</span><span class="p">,</span>
+ <span class="p">)</span>
+
+ <span class="n">dom</span> <span class="o">=</span> <span class="n">lxml</span><span class="o">.</span><span class="n">html</span><span class="o">.</span><span class="n">fromstring</span><span class="p">(</span><span class="n">resp</span><span class="o">.</span><span class="n">text</span><span class="p">)</span> <span class="c1"># type: ignore</span>
+
+ <span class="k">try</span><span class="p">:</span>
+ <span class="n">sc_code</span> <span class="o">=</span> <span class="n">eval_xpath</span><span class="p">(</span><span class="n">dom</span><span class="p">,</span> <span class="n">search_form_xpath</span> <span class="o">+</span> <span class="s1">&#39;//input[@name=&quot;sc&quot;]/@value&#39;</span><span class="p">)[</span><span class="mi">0</span><span class="p">]</span>
+ <span class="k">except</span> <span class="ne">IndexError</span> <span class="k">as</span> <span class="n">exc</span><span class="p">:</span>
+ <span class="n">logger</span><span class="o">.</span><span class="n">debug</span><span class="p">(</span><span class="s2">&quot;suspend startpage API --&gt; https://github.com/searxng/searxng/pull/695&quot;</span><span class="p">)</span>
+ <span class="k">raise</span> <span class="n">SearxEngineCaptchaException</span><span class="p">(</span>
+ <span class="n">message</span><span class="o">=</span><span class="s2">&quot;get_sc_code: [PR-695] query new sc time-stamp failed! (</span><span class="si">%s</span><span class="s2">)&quot;</span> <span class="o">%</span> <span class="n">resp</span><span class="o">.</span><span class="n">url</span><span class="p">,</span> <span class="c1"># type: ignore</span>
+ <span class="p">)</span> <span class="kn">from</span> <span class="nn">exc</span>
+
+ <span class="n">sc_code_ts</span> <span class="o">=</span> <span class="n">time</span><span class="p">()</span>
+ <span class="n">logger</span><span class="o">.</span><span class="n">debug</span><span class="p">(</span><span class="s2">&quot;get_sc_code: new value is: </span><span class="si">%s</span><span class="s2">&quot;</span><span class="p">,</span> <span class="n">sc_code</span><span class="p">)</span>
+ <span class="k">return</span> <span class="n">sc_code</span></div>
+
+
+
+<div class="viewcode-block" id="request">
+<a class="viewcode-back" href="../../../dev/engines/online/startpage.html#searx.engines.startpage.request">[docs]</a>
+<span class="k">def</span> <span class="nf">request</span><span class="p">(</span><span class="n">query</span><span class="p">,</span> <span class="n">params</span><span class="p">):</span>
+<span class="w"> </span><span class="sd">&quot;&quot;&quot;Assemble a Startpage request.</span>
+
+<span class="sd"> To avoid CAPTCHA we need to send a well formed HTTP POST request with a</span>
+<span class="sd"> cookie. We need to form a request that is identical to the request build by</span>
+<span class="sd"> Startpage&#39;s search form:</span>
+
+<span class="sd"> - in the cookie the **region** is selected</span>
+<span class="sd"> - in the HTTP POST data the **language** is selected</span>
+
+<span class="sd"> Additionally the arguments form Startpage&#39;s search form needs to be set in</span>
+<span class="sd"> HTML POST data / compare ``&lt;input&gt;`` elements: :py:obj:`search_form_xpath`.</span>
+<span class="sd"> &quot;&quot;&quot;</span>
+ <span class="k">if</span> <span class="n">startpage_categ</span> <span class="o">==</span> <span class="s1">&#39;web&#39;</span><span class="p">:</span>
+ <span class="k">return</span> <span class="n">_request_cat_web</span><span class="p">(</span><span class="n">query</span><span class="p">,</span> <span class="n">params</span><span class="p">)</span>
+
+ <span class="n">logger</span><span class="o">.</span><span class="n">error</span><span class="p">(</span><span class="s2">&quot;Startpages&#39;s category &#39;%&#39; is not yet implemented.&quot;</span><span class="p">,</span> <span class="n">startpage_categ</span><span class="p">)</span>
+ <span class="k">return</span> <span class="n">params</span></div>
+
+
+
+<span class="k">def</span> <span class="nf">_request_cat_web</span><span class="p">(</span><span class="n">query</span><span class="p">,</span> <span class="n">params</span><span class="p">):</span>
+
+ <span class="n">engine_region</span> <span class="o">=</span> <span class="n">traits</span><span class="o">.</span><span class="n">get_region</span><span class="p">(</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;searxng_locale&#39;</span><span class="p">],</span> <span class="s1">&#39;en-US&#39;</span><span class="p">)</span>
+ <span class="n">engine_language</span> <span class="o">=</span> <span class="n">traits</span><span class="o">.</span><span class="n">get_language</span><span class="p">(</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;searxng_locale&#39;</span><span class="p">],</span> <span class="s1">&#39;en&#39;</span><span class="p">)</span>
+
+ <span class="c1"># build arguments</span>
+ <span class="n">args</span> <span class="o">=</span> <span class="p">{</span>
+ <span class="s1">&#39;query&#39;</span><span class="p">:</span> <span class="n">query</span><span class="p">,</span>
+ <span class="s1">&#39;cat&#39;</span><span class="p">:</span> <span class="s1">&#39;web&#39;</span><span class="p">,</span>
+ <span class="s1">&#39;t&#39;</span><span class="p">:</span> <span class="s1">&#39;device&#39;</span><span class="p">,</span>
+ <span class="s1">&#39;sc&#39;</span><span class="p">:</span> <span class="n">get_sc_code</span><span class="p">(</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;searxng_locale&#39;</span><span class="p">],</span> <span class="n">params</span><span class="p">),</span> <span class="c1"># hint: this func needs HTTP headers,</span>
+ <span class="s1">&#39;with_date&#39;</span><span class="p">:</span> <span class="n">time_range_dict</span><span class="o">.</span><span class="n">get</span><span class="p">(</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;time_range&#39;</span><span class="p">],</span> <span class="s1">&#39;&#39;</span><span class="p">),</span>
+ <span class="p">}</span>
+
+ <span class="k">if</span> <span class="n">engine_language</span><span class="p">:</span>
+ <span class="n">args</span><span class="p">[</span><span class="s1">&#39;language&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">engine_language</span>
+ <span class="n">args</span><span class="p">[</span><span class="s1">&#39;lui&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">engine_language</span>
+
+ <span class="n">args</span><span class="p">[</span><span class="s1">&#39;abp&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;1&#39;</span>
+ <span class="k">if</span> <span class="n">params</span><span class="p">[</span><span class="s1">&#39;pageno&#39;</span><span class="p">]</span> <span class="o">&gt;</span> <span class="mi">1</span><span class="p">:</span>
+ <span class="n">args</span><span class="p">[</span><span class="s1">&#39;page&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">params</span><span class="p">[</span><span class="s1">&#39;pageno&#39;</span><span class="p">]</span>
+
+ <span class="c1"># build cookie</span>
+ <span class="n">lang_homepage</span> <span class="o">=</span> <span class="s1">&#39;en&#39;</span>
+ <span class="n">cookie</span> <span class="o">=</span> <span class="n">OrderedDict</span><span class="p">()</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;date_time&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;world&#39;</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;disable_family_filter&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="n">safesearch_dict</span><span class="p">[</span><span class="n">params</span><span class="p">[</span><span class="s1">&#39;safesearch&#39;</span><span class="p">]]</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;disable_open_in_new_window&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;0&#39;</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;enable_post_method&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;1&#39;</span> <span class="c1"># hint: POST</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;enable_proxy_safety_suggest&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;1&#39;</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;enable_stay_control&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;1&#39;</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;instant_answers&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;1&#39;</span>
+ <span class="n">cookie</span><span class="p">[</span><span class="s1">&#39;lang_homepage&#39;</span><span class="p">]</span> <span class="o">=</span> <span class="s1">&#39;s/device/</span><span cla