- Panneau peek: les bindings Alpine (x-data absent du conteneur) rendaient loovverture et le redimensionnement inoperants -> cblage direct sur le document. - Helper unique window.fdWirePeekResize (app.js): pointer capture, 300px-90vw, clic=fermer, largeur persiste fd_peek_width partagee entre les 4 peeks. - database-table-container margin:0 (tableau colle a gauche, marge Library). - .lib-container remonte dans app.css (trash etait pleine largeur), .db-index 1100px. - ObsiGate verifie sans code: creation .xlsx OK (openpyxl, #186).
296 lines
11 KiB
Python
296 lines
11 KiB
Python
"""FlowDeck — Web search + GitHub search pour les tools agent (v7.46.0).
|
|
|
|
Alimente les deux tools réseau qui manquaient au registre : la recherche
|
|
**en ligne** (`web_search`) et la recherche **GitHub** (`search_code`).
|
|
|
|
Choix de conception :
|
|
|
|
* **Provider configurable** (``settings.web_search_provider``) : Exa
|
|
(recherche sémantique + snippets, clé requise) en priorité, DuckDuckGo en
|
|
repli sans compte. Le repli parse du HTML — c'est volontairement *dégradé* :
|
|
il n'est jamais utilisé si une clé Exa est configurée, et sert surtout à
|
|
garder le skill utilisable sur une instance sans compte externe.
|
|
* **Aucune dépendance ajoutée** : httpx déjà présent, ``shared_client`` pour le
|
|
pool de connexions (A42), et les garde-fous SSRF déjà éprouvés par
|
|
``app.services.importers.url_fetch`` / ``og_fetcher``.
|
|
* **Toujours des sources** : chaque résultat porte titre + URL + extrait, pour
|
|
que l'IA puisse citer au lieu d'inventer.
|
|
|
|
La couche réseau est injectable (``transport``) pour que les tests n'aient
|
|
jamais besoin du réseau.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import html as htmlmod
|
|
import logging
|
|
import re
|
|
from typing import Any
|
|
from urllib.parse import parse_qs, urlparse
|
|
|
|
from app.config import settings
|
|
from app.services.http_client import shared_client
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
USER_AGENT = "FlowDeck-Agent/1.0 (+web_search)"
|
|
|
|
#: Domaines de repli interdits : réponses « sous conditions » / anti-bot qui
|
|
#: pollueraient les résultats (DuckDuckGo sert parfois une page de challenge).
|
|
_BLOCKED_RESULT_HOSTS = frozenset({
|
|
"duckduckgo.com", "lite.duckduckgo.com", "duckduckgo.com.yahoo.net",
|
|
})
|
|
|
|
EXA_ENDPOINT = "https://api.exa.ai/search"
|
|
DDG_ENDPOINT = "https://lite.duckduckgo.com/lite/"
|
|
GITHUB_API = "https://api.github.com"
|
|
|
|
MAX_RESULTS = 10
|
|
_SNIPPET_CHARS = 300
|
|
|
|
|
|
def _clean(text: str | None) -> str:
|
|
return re.sub(r"\s+", " ", (text or "")).strip()
|
|
|
|
|
|
def _clip(text: str, limit: int = _SNIPPET_CHARS) -> str:
|
|
text = _clean(text)
|
|
return text if len(text) <= limit else text[: limit - 1].rstrip() + "…"
|
|
|
|
|
|
def _is_resultable(url: str) -> bool:
|
|
"""Ignore les redirections vers le moteur lui-même et les URLs non http."""
|
|
parsed = urlparse(url or "")
|
|
if parsed.scheme not in ("http", "https") or not parsed.hostname:
|
|
return False
|
|
return parsed.hostname.lower() not in _BLOCKED_RESULT_HOSTS
|
|
|
|
|
|
# ══════════════════════ Exa ══════════════════════
|
|
|
|
|
|
async def _search_exa(query: str, num_results: int, transport=None) -> list[dict]:
|
|
key = (settings.exa_api_key or "").strip()
|
|
if not key:
|
|
return []
|
|
payload = {
|
|
"query": query,
|
|
"numResults": max(1, min(num_results, MAX_RESULTS)),
|
|
"contents": {"text": {"maxCharacters": 1200}},
|
|
}
|
|
kwargs: dict[str, Any] = {"timeout": 15.0}
|
|
if transport is not None:
|
|
kwargs["transport"] = transport
|
|
async with shared_client(**kwargs) as client:
|
|
resp = await client.post(
|
|
EXA_ENDPOINT,
|
|
json=payload,
|
|
headers={"x-api-key": key, "Content-Type": "application/json",
|
|
"User-Agent": USER_AGENT},
|
|
)
|
|
if resp.status_code != 200:
|
|
raise RuntimeError(f"Exa HTTP {resp.status_code}")
|
|
results = []
|
|
for item in (resp.json() or {}).get("results", []):
|
|
url = item.get("url") or ""
|
|
if not _is_resultable(url):
|
|
continue
|
|
results.append({
|
|
"title": _clip(item.get("title") or url, 160),
|
|
"url": url,
|
|
"snippet": _clip(item.get("text") or item.get("summary") or ""),
|
|
"published": (item.get("publishedDate") or "")[:10],
|
|
"source": "exa",
|
|
})
|
|
return results[:MAX_RESULTS]
|
|
|
|
|
|
# ══════════════════════ DuckDuckGo (repli sans compte) ══════════════════════
|
|
|
|
_DDG_ROW_RE = re.compile(
|
|
r"<a[^>]+class=\"[^\"]*result-link[^\"]*\"[^>]+href=\"(?P<href>[^\"]+)\"[^>]*>(?P<title>.*?)</a>",
|
|
re.I | re.S,
|
|
)
|
|
_DDG_SNIPPET_RE = re.compile(
|
|
r"class=\"[^\"]*result-snippet[^\"]*\"[^>]*>(?P<snippet>.*?)</td>", re.I | re.S
|
|
)
|
|
_TAG_RE = re.compile(r"<[^>]+>")
|
|
|
|
|
|
def _unwrap_ddg(href: str) -> str:
|
|
"""DuckDuckGo emballe les résultats dans ``/l/?uddg=<url encodé>``."""
|
|
if not href:
|
|
return ""
|
|
parsed = urlparse(htmlmod.unescape(href))
|
|
if "uddg" in (parsed.query or ""):
|
|
values = parse_qs(parsed.query).get("uddg")
|
|
if values:
|
|
return values[0]
|
|
return href
|
|
|
|
|
|
async def _search_duckduckgo(query: str, num_results: int, transport=None) -> list[dict]:
|
|
kwargs: dict[str, Any] = {"timeout": 12.0}
|
|
if transport is not None:
|
|
kwargs["transport"] = transport
|
|
async with shared_client(**kwargs) as client:
|
|
resp = await client.post(
|
|
DDG_ENDPOINT,
|
|
data={"q": query, "kl": "wt-wt"},
|
|
headers={"User-Agent": USER_AGENT,
|
|
"Content-Type": "application/x-www-form-urlencoded"},
|
|
)
|
|
if resp.status_code != 200:
|
|
raise RuntimeError(f"DuckDuckGo HTTP {resp.status_code}")
|
|
body = resp.text
|
|
links = list(_DDG_ROW_RE.finditer(body))
|
|
snippets = [_clean(_TAG_RE.sub("", m.group("snippet"))) for m in _DDG_SNIPPET_RE.finditer(body)]
|
|
results = []
|
|
for idx, match in enumerate(links):
|
|
url = _unwrap_ddg(match.group("href"))
|
|
if not _is_resultable(url):
|
|
continue
|
|
results.append({
|
|
"title": _clean(_TAG_RE.sub("", match.group("title"))),
|
|
"url": url,
|
|
"snippet": _clip(snippets[idx]) if idx < len(snippets) else "",
|
|
"published": "",
|
|
"source": "duckduckgo",
|
|
})
|
|
if len(results) >= min(num_results, MAX_RESULTS):
|
|
break
|
|
return results
|
|
|
|
|
|
# ══════════════════════ API publique ══════════════════════
|
|
|
|
|
|
async def search_web(query: str, num_results: int = 5,
|
|
transport=None) -> tuple[list[dict], str]:
|
|
"""Recherche web → ``(results, provider)``.
|
|
|
|
Essaie le provider configuré puis l'autre ; si aucun ne répond, renvoie
|
|
une liste vide — c'est au tool de traduire ça en message actionnable pour
|
|
le modèle (« configurez EXA_API_KEY ») plutôt qu'en erreur réseau opaque.
|
|
"""
|
|
q = (query or "").strip()
|
|
if not q:
|
|
return [], ""
|
|
|
|
provider = (settings.web_search_provider or "exa").strip().lower()
|
|
order = ("exa", "duckduckgo") if provider == "exa" else ("duckduckgo", "exa")
|
|
|
|
errors: list[str] = []
|
|
for name in order:
|
|
try:
|
|
if name == "exa":
|
|
results = await _search_exa(q, num_results, transport)
|
|
else:
|
|
results = await _search_duckduckgo(q, num_results, transport)
|
|
except Exception as exc: # noqa: BLE001 — un provider qui tombe ne doit pas tuer l'autre
|
|
logger.info("web_search provider %s failed: %s", name, exc)
|
|
errors.append(f"{name}: {exc}")
|
|
continue
|
|
if results:
|
|
return results, name
|
|
|
|
if errors:
|
|
logger.warning("web_search sans résultat — %s", "; ".join(errors))
|
|
return [], order[0]
|
|
return [], order[0]
|
|
|
|
|
|
def available_providers() -> dict[str, bool]:
|
|
"""Quel provider est réellement utilisable (affiché dans les settings)."""
|
|
return {
|
|
"exa": bool((settings.exa_api_key or "").strip()),
|
|
"duckduckgo": True,
|
|
"github": True, # sans PAT : 10 req/min suffisent pour un usage agent
|
|
}
|
|
|
|
|
|
# ══════════════════════ GitHub ══════════════════════
|
|
|
|
_GH_KINDS = {
|
|
"repositories": "repos",
|
|
"repos": "repos",
|
|
"code": "code",
|
|
"issues": "issues",
|
|
}
|
|
|
|
|
|
def _gh_headers() -> dict[str, str]:
|
|
headers = {"Accept": "application/vnd.github+json",
|
|
"User-Agent": USER_AGENT,
|
|
"X-GitHub-Api-Version": "2022-11-28"}
|
|
token = (settings.github_token or "").strip()
|
|
if token:
|
|
headers["Authorization"] = f"Bearer {token}"
|
|
return headers
|
|
|
|
|
|
def _gh_item_name(item: dict, kind: str) -> str:
|
|
if kind == "repos":
|
|
return item.get("full_name") or item.get("name") or ""
|
|
if kind == "code":
|
|
return item.get("path") or ""
|
|
return item.get("title") or f"#{item.get('number')}"
|
|
|
|
|
|
def _gh_item_url(item: dict, kind: str) -> str:
|
|
if kind == "code":
|
|
# L'API code search ne renvoie pas d'URL exploitable : on reconstruit.
|
|
repo = (item.get("repository") or {}).get("full_name", "")
|
|
return f"https://github.com/{repo}/blob/HEAD/{item.get('path', '')}" if repo else ""
|
|
return item.get("html_url") or ""
|
|
|
|
|
|
def _gh_item_snippet(item: dict, kind: str) -> str:
|
|
if kind == "code":
|
|
return ""
|
|
if kind == "repos":
|
|
return _clip(item.get("description") or "")
|
|
body = _clean(item.get("body") or "")
|
|
return _clip(body) if body else _clip(item.get("title") or "")
|
|
|
|
|
|
async def search_github(query: str, kind: str = "repositories", limit: int = 5,
|
|
transport=None) -> list[dict]:
|
|
"""Recherche GitHub (repos | code | issues) → liste de résultats sourcés.
|
|
|
|
Fonctionne sans PAT (limite 10 req/min) ; avec ``GITHUB_TOKEN``, 30 req/min.
|
|
"""
|
|
q = (query or "").strip()
|
|
if not q:
|
|
return []
|
|
k = _GH_KINDS.get((kind or "").strip().lower(), "repos")
|
|
per_page = max(1, min(int(limit or 5), MAX_RESULTS))
|
|
|
|
kwargs: dict[str, Any] = {"timeout": 15.0}
|
|
if transport is not None:
|
|
kwargs["transport"] = transport
|
|
async with shared_client(**kwargs) as client:
|
|
resp = await client.get(
|
|
f"{GITHUB_API}/search/{k}",
|
|
params={"q": q, "per_page": per_page},
|
|
headers=_gh_headers(),
|
|
)
|
|
if resp.status_code == 403:
|
|
raise RuntimeError("GitHub: quota dépassé (403) — configurez GITHUB_TOKEN")
|
|
if resp.status_code == 422:
|
|
raise RuntimeError("GitHub: requête de recherche invalide (422)")
|
|
if resp.status_code != 200:
|
|
raise RuntimeError(f"GitHub HTTP {resp.status_code}")
|
|
|
|
items = (resp.json() or {}).get("items", [])
|
|
results = []
|
|
for item in items[:per_page]:
|
|
results.append({
|
|
"title": _gh_item_name(item, k),
|
|
"url": _gh_item_url(item, k),
|
|
"snippet": _gh_item_snippet(item, k),
|
|
"stars": item.get("stargazers_count") if k == "repos" else None,
|
|
"state": item.get("state") if k == "issues" else "",
|
|
"source": f"github/{k}",
|
|
})
|
|
return results
|