summaryrefslogtreecommitdiff
path: root/searx/engines/pexels.py
blob: 82d4d702a54a06826f286ecc446a89aa9009c920 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
# SPDX-License-Identifier: AGPL-3.0-or-later
"""Pexels (images)"""

import re
import typing as t

from urllib.parse import urlencode
from lxml import html

from searx.result_types import EngineResults
from searx.utils import eval_xpath_list, gen_useragent
from searx.enginelib import EngineCache
from searx.exceptions import SearxEngineAPIException, SearxEngineAccessDeniedException
from searx.network import get


# about
about = {
    "website": 'https://www.pexels.com',
    "wikidata_id": 'Q101240504',
    "official_api_documentation": 'https://www.pexels.com/api/',
    "use_official_api": False,
    "require_api_key": False,
    "results": 'JSON',
}

base_url = 'https://www.pexels.com'
categories = ['images']

api_key = "H2jk9uKnhRmL6WPwh89zBezWvr"
"""
Fallback API key to use when SearXNG fails to automatically extract one from the website.
"""
results_per_page = 20

paging = True
time_range_support = True
time_range_map = {'day': 'last_24_hours', 'week': 'last_week', 'month': 'last_month', 'year': 'last_year'}

SECRET_KEY_RE = re.compile('"secret-key":\b*"(.*?)"')
SECRET_KEY_DB_KEY = "secret-key"


CACHE: EngineCache
"""Cache to store the secret API key for the engine."""

enable_http2 = False


def setup(engine_settings: dict[str, t.Any]) -> bool:
    global CACHE  # pylint: disable=global-statement
    CACHE = EngineCache(engine_settings["name"])
    return True


def _get_secret_key():
    resp = get(
        base_url,
        headers={
            # circumvents Cloudflare bot protections
            "User-Agent": gen_useragent(),
            "Referer": base_url,
            "Sec-GPC": "1",
            "Connection": "keep-alive",
        },
    )

    if resp.status_code != 200:
        raise SearxEngineAPIException("failed to obtain secret key")

    doc = html.fromstring(resp.text)
    for script_src in eval_xpath_list(doc, "//script/@src"):
        script = get(script_src)
        if script.status_code != 200:
            raise SearxEngineAPIException("failed to obtain secret key")

        match = SECRET_KEY_RE.search(script.text)
        if match:
            return match.groups()[0]

    # all scripts checked, but secret key was not found
    raise SearxEngineAPIException("failed to obtain secret key")


def request(query, params):
    args = {
        'query': query,
        'page': params['pageno'],
        'per_page': results_per_page,
    }
    if params['time_range']:
        args['date_from'] = time_range_map[params['time_range']]

    params["url"] = f"{base_url}/en-us/api/v3/search/photos?{urlencode(args)}"

    # cache api key for future requests
    secret_key = CACHE.get(SECRET_KEY_DB_KEY)
    if not secret_key:
        try:
            secret_key = _get_secret_key()
            CACHE.set(SECRET_KEY_DB_KEY, secret_key)
        except (SearxEngineAPIException, SearxEngineAccessDeniedException) as e:
            logger.debug("failed to extract API key %s" % e)
            secret_key = api_key

    params["headers"]["secret-key"] = secret_key

    return params


def response(resp):
    res = EngineResults()
    json_data = resp.json()

    for result in json_data.get('data', []):
        attrs = result["attributes"]
        res.add(
            res.types.LegacyResult(
                {
                    'template': 'images.html',
                    'url': f"{base_url}/photo/{attrs['slug']}-{attrs['id']}/",
                    'title': attrs["title"],
                    'content': attrs["description"],
                    'thumbnail_src': attrs["image"]["small"],
                    'img_src': attrs["image"]["download_link"],
                    'resolution': f"{attrs['width']}x{attrs['height']}",
                    'author': f"{attrs['user']['username']}",
                }
            )
        )

    return res