summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--searx/engines/google.py10
-rw-r--r--searx/engines/google_cse.py158
-rw-r--r--searx/settings.yml5
-rwxr-xr-xsearxng_extra/update/update_engine_traits.py4
4 files changed, 172 insertions, 5 deletions
diff --git a/searx/engines/google.py b/searx/engines/google.py
index ee31c3835..ee2fbaf36 100644
--- a/searx/engines/google.py
+++ b/searx/engines/google.py
@@ -32,7 +32,7 @@ from searx.utils import (
eval_xpath_getindex,
eval_xpath_list,
extract_text,
- gen_gsa_useragent,
+ # gen_gsa_useragent,
)
if t.TYPE_CHECKING:
@@ -197,7 +197,7 @@ def get_google_info(params: "OnlineParams", eng_traits: EngineTraits) -> dict[st
# https://developers.google.com/custom-search/docs/xml_results_appendices#interfaceLanguages
# https://github.com/searxng/searxng/issues/2515#issuecomment-1607150817
- ret_val["params"]["hl"] = f"{lang_code}-{country}"
+ ret_val["params"]["hl"] = f"{lang_code}"
# lr parameter:
# The lr (language restrict) parameter restricts search results to
@@ -268,7 +268,11 @@ def get_google_info(params: "OnlineParams", eng_traits: EngineTraits) -> dict[st
# HTTP headers
ret_val["headers"]["Accept"] = "*/*"
- ret_val["headers"]["User-Agent"] = gen_gsa_useragent()
+ # GSA for iPhone useragent had been added in
+ # - https://github.com/searxng/searxng/pull/5644
+ # but doe no longer work
+ # - https://github.com/searxng/searxng/issues/6359
+ # ret_val["headers"]["User-Agent"] = gen_gsa_useragent()
# Cookies
diff --git a/searx/engines/google_cse.py b/searx/engines/google_cse.py
new file mode 100644
index 000000000..a0261e324
--- /dev/null
+++ b/searx/engines/google_cse.py
@@ -0,0 +1,158 @@
+# SPDX-License-Identifier: AGPL-3.0-or-later
+"""Google Custom Search Engine"""
+
+import datetime
+import typing as t
+from json import loads
+from urllib.parse import urlencode
+
+from searx.enginelib import EngineCache
+from searx.exceptions import SearxEngineAPIException, SearxEngineTooManyRequestsException
+from searx.network import get
+from searx.result_types import EngineResults
+
+from searx.engines.google import fetch_traits # pylint: disable=unused-import
+from searx.engines.google import filter_mapping, get_google_info
+
+if t.TYPE_CHECKING:
+ from searx.extended_types import SXNG_Response
+ from searx.search.processors import OnlineParams
+
+about = {
+ "website": "https://www.google.com",
+ "wikidata_id": "Q2233943",
+ "official_api_documentation": "https://developers.google.com/custom-search/docs/element",
+ "use_official_api": False,
+ "require_api_key": False,
+ "results": "JSONP",
+ "description": "Platform for creating custom search engines based on Google Search.",
+}
+
+categories = ["general", "web"]
+paging = True
+max_page = 5
+page_size = 20
+time_range_support = True
+language_support = True
+safesearch = True
+
+CX = "partner-pub-8993703457585266:4862972284" # blackle.com
+
+CACHE: EngineCache
+
+
+def setup(engine_settings: dict[str, t.Any]) -> bool:
+ global CACHE # pylint: disable=global-statement
+ CACHE = EngineCache(engine_settings["name"])
+ return True
+
+
+def _cse_token() -> dict[str, str]:
+ token: dict[str, str] = CACHE.get(CX)
+ if token:
+ return token
+
+ resp = get(f"https://www.google.com/cse/cse.js?cx={CX}", timeout=10)
+ if not resp.ok:
+ raise SearxEngineAPIException("failed to obtain cse token")
+
+ end = resp.text.rfind("});")
+ start = resp.text.rfind("({")
+ opts: dict[str, str] = loads(resp.text[start + 1 : end + 1])
+
+ cse_tok = opts.get("cse_token")
+ if not cse_tok:
+ raise SearxEngineAPIException("failed to obtain cse token")
+
+ exp = opts.get("exp")
+ token = {
+ "cse_tok": cse_tok,
+ "cselibv": opts.get("cselibVersion", ""),
+ "exp": ",".join(exp) if exp else "",
+ }
+ CACHE.set(CX, token, expire=3600)
+ return token
+
+
+def _get_start_and_end_date_str(time_range: str) -> tuple[str, str]:
+ time_range_map = {"day": 1, "week": 7, "month": 30, "year": 365}
+
+ end_date = datetime.datetime.now()
+ start_date = end_date - datetime.timedelta(days=time_range_map[time_range])
+
+ return start_date.strftime("%Y%m%d"), end_date.strftime("%Y%m%d")
+
+
+def request(query: str, params: "OnlineParams") -> None:
+ token = _cse_token()
+
+ google_info = get_google_info(params, traits)
+ info: dict[str, str] = google_info["params"]
+
+ hl = info["hl"]
+ if params.get("searxng_locale", "all") == "all" or "ZZ" in hl:
+ hl = "en"
+
+ args = {
+ "rsz": "filtered_cse",
+ "num": str(page_size),
+ "hl": hl,
+ "cselibv": token["cselibv"],
+ "cx": CX,
+ "q": query,
+ "safe": filter_mapping[params["safesearch"]],
+ "cse_tok": token["cse_tok"],
+ "callback": "_",
+ "rurl": "",
+ }
+ if params["time_range"]:
+ start_date, end_date = _get_start_and_end_date_str(params["time_range"])
+ args["sort"] = f"date:r:{start_date}:{end_date}"
+
+ if info.get("lr"):
+ args["lr"] = info["lr"]
+ if info.get("cr"):
+ args["cr"] = info["cr"]
+ if google_info["country"] not in (None, "ZZ"):
+ args["gl"] = google_info["country"]
+ if token["exp"]:
+ args["exp"] = token["exp"]
+
+ start = (params["pageno"] - 1) * page_size
+ if start:
+ args["start"] = str(start)
+
+ params["url"] = "https://cse.google.com/cse/element/v1?" + urlencode(args)
+ params["cookies"] = google_info["cookies"]
+ params["headers"].update(google_info["headers"])
+ params["headers"]["Referer"] = "https://cse.google.com/"
+
+
+def response(resp: "SXNG_Response") -> EngineResults:
+ json_resp = resp.text[resp.text.find("{") : resp.text.rfind("}") + 1]
+ data = loads(json_resp)
+
+ # not the real types, but a sufficient approximation
+ item: dict[str, str]
+ error: dict[str, str | int]
+
+ if error := data.get("error"):
+ message = error.get("message", "unknown error")
+ if error.get("code") == 429:
+ raise SearxEngineTooManyRequestsException(message=f"google cse: {message}")
+ raise SearxEngineAPIException(f"google cse: {message}")
+
+ results = EngineResults()
+ for item in data.get("results", []):
+ url = item.get("unescapedUrl")
+ if not url:
+ continue
+ results.add(
+ results.types.MainResult(
+ url=url,
+ title=item.get("titleNoFormatting", ""),
+ content=item.get("contentNoFormatting", ""),
+ thumbnail=item.get("richSnippet", {}).get("cseThumbnail", {}).get("src", ""), # type: ignore
+ )
+ )
+ return results
diff --git a/searx/settings.yml b/searx/settings.yml
index 6a5a3a3b3..6e3a10c77 100644
--- a/searx/settings.yml
+++ b/searx/settings.yml
@@ -1199,6 +1199,11 @@ engines:
- name: google
engine: google
shortcut: go
+ inactive: true
+
+ - name: google cse
+ engine: google_cse
+ shortcut: goc
- name: google images
engine: google_images
diff --git a/searxng_extra/update/update_engine_traits.py b/searxng_extra/update/update_engine_traits.py
index e955c29c9..cde1b29ec 100755
--- a/searxng_extra/update/update_engine_traits.py
+++ b/searxng_extra/update/update_engine_traits.py
@@ -131,8 +131,8 @@ def fetch_traits_map() -> EngineTraitsMap:
def filter_locales(traits_map: EngineTraitsMap) -> set[str]:
"""Filter language & region tags by a threshold."""
- min_eng_per_region = 18
- min_eng_per_lang = 23
+ min_eng_per_region = 19
+ min_eng_per_lang = 24
_: dict[str, int] = {}
for eng in traits_map.values():