summaryrefslogtreecommitdiff
path: root/searx/engines/jina.py
blob: 24ecaa51d127043c5488da60c7177884acbaae9c (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
# SPDX-License-Identifier: AGPL-3.0-or-later
"""Jina is a search AI and part of Elastic, the company behind ElasticSearch.

The engine requires an API key, you can get one from the
`API dashboard <https://jina.ai/api-dashboard/>`_ without signup.

.. code:: yaml

  - name: jina
    engine: jina
    shortcut: ji
    api_key: "jina_..."
    jina_engine: reader
    inactive: false

By default, Jina's own index is used. You can change that by setting a different :py:obj:`jina_engine`.
"""

import typing as t
from urllib.parse import urlencode

from dateutil import parser
from searx.result_types import EngineResults

if t.TYPE_CHECKING:
    from searx.extended_types import SXNG_Response
    from searx.search.processors import OnlineParams


about = {
    "website": "https://jina.ai",
    "wikidata_id": None,
    "official_api_documentation": "https://s.jina.ai/docs",
    "use_official_api": True,
    "require_api_key": True,
    "results": "JSON",
}

categories = ["general"]
paging = True

jina_engine = "reader"
"""Search mode. Currently supported values are 'reader', 'google' and 'bing'."""

base_url = "https://s.jina.ai"
api_key: str | None = None


def setup(_):
    if not api_key:
        raise ValueError("missing api key")


def request(query: str, params: "OnlineParams"):
    # setting 'no-content' pushes the response time down to a third
    args = {"q": query, "page": params["pageno"], "engine": jina_engine, "respondWith": "no-content"}
    params["url"] = f"{base_url}/?{urlencode(args)}"
    params["headers"].update(
        {
            "Accept": "application/json",
            "Authorization": f"Bearer {api_key}",
        }
    )


def response(resp: "SXNG_Response"):
    res = EngineResults()

    json_resp: dict[str, t.Any] = resp.json()

    result: dict[str, str]
    for result in json_resp["data"]:
        published_date = None
        if result.get("date"):
            try:
                published_date = parser.parse(result["date"])
            except parser.ParserError:
                pass

        res.add(
            res.types.MainResult(
                url=result["url"],
                title=result["title"],
                content=result["description"],
                publishedDate=published_date,
            )
        )

    return res