summaryrefslogtreecommitdiff
path: root/searx/engines/tonline.py
blob: 04fb3425059b4a673bd6a0ccb6bbdd9caddac504 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
# SPDX-License-Identifier: AGPL-3.0-or-later
"""T-Online_ is a German news portal, which is powered by Ströer, a German
advertising company, not by Deutsche Telekom (contrary to its name).

It gets its web results from Google, image results from Flickr and videos
results from YouTube.

.. _T-Online: https://www.t-online.de/

"""

import typing as t
from urllib.parse import urlencode

from lxml import html

from searx.utils import eval_xpath_list, eval_xpath, extract_text, get_embeded_stream_url, ElementType
from searx.result_types import EngineResults
from searx.enginelib import EngineAbout

if t.TYPE_CHECKING:
    from searx.extended_types import SXNG_Response
    from searx.search.processors import OnlineParams

about = EngineAbout(
    website="https://www.t-online.de",
    wikidata_id="Q590940",
    results="HTML",
)

paging = True
time_range_support = True

base_url = "https://suche.t-online.de"
tonline_categ = "web"
"""Supported categories are ``web``, ``videos``, ``news`` and ``images``."""

time_range_map = {"day": "d", "week": "w", "month": "m", "year": "y"}

# result provider has to be specified during pagination, pagination can alternatively
# use "tonline" to only search for results from t-online news articles
tonline_channel_map = {"images": "flickr", "videos": "yt"}

language = "de"


def setup(_: dict[str, t.Any]) -> bool | None:
    if tonline_categ not in ("web", "images", "videos", "news"):
        raise ValueError("invalid category: %s" % tonline_categ)


def request(query: str, params: "OnlineParams") -> None:
    # "mandant", "dia" and "ptl" are not needed, but this might reduce changes of captchas
    args = {"q": query, "mandant": "toi", "dia": "suche", "ptl": "std"}
    if params["time_range"]:
        args["age"] = time_range_map[params["time_range"]]

    if params["pageno"] > 1 and tonline_categ in tonline_channel_map:
        ch = tonline_channel_map[tonline_categ]
        args["ch"] = ch
        args[f"{ch}_page"] = str(params["pageno"])
    else:
        args["page"] = str(params["pageno"])

    params["url"] = f"{base_url}/{tonline_categ}?{urlencode(args)}"


def _general_results(doc: ElementType, res: EngineResults):
    result: ElementType
    for result in eval_xpath_list(doc, "//div[@id='google_re']/div[contains(@class, 'doc')]"):
        (
            res.add(
                res.types.MainResult(
                    url=extract_text(eval_xpath(result, "./a/@href") or ""),
                    title=extract_text(eval_xpath(result, ".//span[contains(@class, 'tMMReshl')]") or "") or "",
                    content=extract_text(eval_xpath(result, ".//div[contains(@class, 'tMMRest')]") or "") or "",
                ),
            )
        )
    suggestion: ElementType
    for suggestion in eval_xpath_list(doc, "//div[starts-with(@class, 'rsbl')]/a"):
        res.add(res.types.LegacyResult({"suggestion": extract_text(suggestion)}))


def _image_results(doc: ElementType, res: EngineResults):
    result: ElementType
    for result in eval_xpath_list(doc, "//div[@class='doc']"):
        (
            res.add(
                res.types.Image(
                    url=extract_text(eval_xpath(result, "./a/@href") or ""),
                    title=extract_text(eval_xpath(result, ".//div[contains(@class, 'doc_info')]") or "") or "",
                    thumbnail_src=extract_text(eval_xpath(result, ".//img/@src") or "") or "",
                ),
            )
        )


def _news_results(doc: ElementType, res: EngineResults):
    result: ElementType
    title_parts: list[ElementType]
    for result in eval_xpath_list(doc, "//div[@id='portal_re']/div[contains(@class, 'doc')]"):
        title_parts = eval_xpath(result, ".//a[starts-with(@class, 'tMMReshl')]")
        (
            res.add(
                res.types.MainResult(
                    url=extract_text(eval_xpath(result, "(./a/@href)[1]") or ""),
                    title=" - ".join(extract_text(part) or "" for part in title_parts),
                    content=extract_text(eval_xpath(result, ".//div[contains(@class, 'tMMRest')]") or "") or "",
                    thumbnail=extract_text(eval_xpath(result, ".//img[contains(@class, 'desk')]/@src") or "") or "",
                ),
            )
        )


def _video_results(doc: ElementType, res: EngineResults):
    result: ElementType
    for result in eval_xpath_list(doc, "//div[@class='doc']"):
        url: str | None = extract_text(eval_xpath(result, "./a/@href") or "")
        if url is None:
            continue
        title_parts: list[ElementType] = eval_xpath(result, ".//a[starts-with(@class, 'tMMReshl')]")
        res.add(
            res.types.LegacyResult(
                template="videos.html",
                url=url,
                title=" - ".join(extract_text(part) or "" for part in title_parts),
                thumbnail=extract_text(eval_xpath(result, ".//img/@src") or "") or "",
                iframe_src=get_embeded_stream_url(url) or "",
            )
        )


def response(resp: "SXNG_Response") -> EngineResults:
    doc = html.fromstring(resp.text)
    res = EngineResults()
    match tonline_categ:
        case "web":
            _general_results(doc, res)
        case "news":
            _news_results(doc, res)
        case "images":
            _image_results(doc, res)
        case "videos":
            _video_results(doc, res)
        case _:
            raise ValueError("invalid category: %s" % tonline_categ)
    return res