open-webui/backend/open_webui/retrieval/web/google_pse.py

51 lines
1.4 KiB
Python
Raw Normal View History

import logging
2024-08-14 12:46:31 +00:00
from typing import Optional
2024-08-27 22:10:27 +00:00
import requests
2024-12-12 02:05:42 +00:00
from open_webui.retrieval.web.main import SearchResult, get_filtered_results
from open_webui.env import SRC_LOG_LEVELS
log = logging.getLogger(__name__)
log.setLevel(SRC_LOG_LEVELS["RAG"])
def search_google_pse(
2024-06-17 21:32:23 +00:00
api_key: str,
search_engine_id: str,
query: str,
count: int,
2024-08-14 12:46:31 +00:00
filter_list: Optional[list[str]] = None,
) -> list[SearchResult]:
"""Search using Google's Programmable Search Engine API and return the results as a list of SearchResult objects.
Args:
api_key (str): A Programmable Search Engine API key
search_engine_id (str): A Programmable Search Engine ID
query (str): The query to search for
"""
url = "https://www.googleapis.com/customsearch/v1"
headers = {"Content-Type": "application/json"}
params = {
"cx": search_engine_id,
"q": query,
"key": api_key,
2024-06-02 02:57:00 +00:00
"num": count,
}
response = requests.request("GET", url, headers=headers, params=params)
response.raise_for_status()
json_response = response.json()
results = json_response.get("items", [])
if filter_list:
results = get_filtered_results(results, filter_list)
return [
SearchResult(
link=result["link"],
title=result.get("title"),
snippet=result.get("snippet"),
)
for result in results
]