diff --git a/CHANGELOG.md b/CHANGELOG.md index 47c7f09..e8ccd9e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,7 +6,7 @@ Format basé sur [Keep a Changelog](https://keepachangelog.com/fr/1.1.0/), et [Semantic Versioning](https://semver.org/spec/v2.0.0.html). > **En cours de développement** : les changements à venir sont listés dans la section -> [Unreleased](#unreleased). La dernière version livrée est **2.51.0**. +> [Unreleased](#unreleased). La dernière version livrée est **2.52.0**. --- @@ -14,6 +14,32 @@ et [Semantic Versioning](https://semver.org/spec/v2.0.0.html). --- +## [2.52.0] — 2026-10-04 + +### Ajouté + +- **#166 — Assistant IA : détection & fusion de doublons** + - Score déterministe (Jaccard 70 % + titre 30 %, frontmatter exclu), + scan borné (500 fichiers, `truncated` exposé). + - Outils `find_duplicates` (READ) / `merge_duplicate_notes` (DANGEROUS, + backup préalable) ; API `GET /api/duplicates` et + `POST /api/duplicates/merge` (`confirm: true` obligatoire, stratégies + `append` / `prefer_target` / `prefer_source`). +- **#168 — Assistant IA : notifications externes Discord / Telegram / SMTP / webhook** + - Canaux avec déclencheurs (`manual`, `schedule_failure`, + `schedule_success`, `duplicate_found`), secrets hors config + (`notify_secrets.json` 0600 ou `OBSIGATE_NOTIFY_SECRET_`), SSRF-safe. + - CRUD admin `/api/notify/channels`, test `/api/notify/test`, + outil `notify_external` (WRITE). +- **#170 — Assistant IA : tâches planifiées type cron** + - Actions `create_file` / `append_to_file` / `notify` (services existants), + planifications `interval_hours` / `daily_time` / `once_at`, tick 60 s + (`OBSIGATE_SCHEDULER=0` pour désactiver), échec enregistré + notifié. + - API `/api/scheduler/tasks` (CRUD + `POST …/run`), outils + `create/list/delete/run_scheduled_task*`. + +--- + ## [2.51.0] — 2026-10-03 ### Ajouté diff --git a/README.fr.md b/README.fr.md index eda32fb..c87de05 100644 --- a/README.fr.md +++ b/README.fr.md @@ -4,7 +4,7 @@ **Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive. -[![Version](https://img.shields.io/badge/Version-2.51.0-blue.svg)]() +[![Version](https://img.shields.io/badge/Version-2.52.0-blue.svg)]() [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://www.docker.com/) [![Python](https://img.shields.io/badge/Python-3.11+-green.svg)](https://www.python.org/) @@ -976,8 +976,8 @@ Ce projet est sous licence **MIT** — voir le fichier [LICENSE](LICENSE) pour l ## 📝 Changelog -Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.51.0). +Consultez le [CHANGELOG.md](./CHANGELOG.md) pour l'historique complet de toutes les versions (v1.0.0 → v2.52.0). --- -*Projet : ObsiGate | Version : 2.51.0 | Dernière mise à jour : Septembre 2026* +*Projet : ObsiGate | Version : 2.52.0 | Dernière mise à jour : Septembre 2026* diff --git a/README.md b/README.md index bd91084..49dbdc8 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ **Ultra-light web gateway for your Obsidian vaults** — Access, browse, and search all your Obsidian notes from any device via a modern, responsive web interface. -[![Version](https://img.shields.io/badge/Version-2.51.0-blue.svg)]() +[![Version](https://img.shields.io/badge/Version-2.52.0-blue.svg)]() [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://www.docker.com/) [![Python](https://img.shields.io/badge/Python-3.11+-green.svg)](https://www.python.org/) @@ -1151,8 +1151,8 @@ This project is licensed under the **MIT License** - see the [LICENSE](LICENSE) ## 📝 Changelog -See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.51.0). +See [CHANGELOG.md](./CHANGELOG.md) for the complete version history (v1.0.0 → v2.52.0). --- -*Project: ObsiGate | Version: 2.51.0 | Last updated: September 2026* +*Project: ObsiGate | Version: 2.52.0 | Last updated: September 2026* diff --git a/VERSION b/VERSION index d66b738..cfa53dc 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -2.51.0 +2.52.0 diff --git a/backend/main.py b/backend/main.py index 9cb3cd2..3dcde45 100644 --- a/backend/main.py +++ b/backend/main.py @@ -368,6 +368,24 @@ async def lifespan(app: FastAPI): asyncio.create_task(_background_startup()) + async def _scheduler_loop(): + """Background tick for scheduled tasks (#170) — every 60 s, best effort.""" + from backend.scheduler import tick + + await asyncio.sleep(60) + while True: + try: + outcomes = await asyncio.to_thread(tick) + if outcomes: + logger.info(f"Scheduler tick: {len(outcomes)} task(s) executed") + except Exception as e: + logger.warning(f"Scheduler tick failed: {e}") + await asyncio.sleep(60) + + if os.environ.get("OBSIGATE_SCHEDULER", "1") != "0": + asyncio.create_task(_scheduler_loop()) + logger.info("Scheduler loop started (#170, 60 s tick).") + logger.info("ObsiGate ready (listening for requests while indexing).") yield @@ -490,12 +508,15 @@ from backend.routers.backups import router as backups_router from backend.routers.config import _load_config from backend.routers.config import router as config_router from backend.routers.conflicts import router as conflicts_router +from backend.routers.duplicates import router as duplicates_router from backend.routers.files_media import router as files_media_router from backend.routers.files_read import router as files_read_router from backend.routers.files_write import router as files_write_router from backend.routers.health import router as health_router from backend.routers.history import router as history_router +from backend.routers.notify import router as notify_router from backend.routers.realtime import router as realtime_router +from backend.routers.scheduler import router as scheduler_router from backend.routers.search import router as search_router from backend.routers.sharing import router as sharing_router from backend.routers.vaults import router as vaults_router @@ -519,6 +540,9 @@ app.include_router(files_write_router) # ROADMAP #85 T6b — Files write app.include_router(webhooks_router) # ROADMAP #85 T2 — Webhooks app.include_router(sharing_router) # ROADMAP #85 T3 — Sharing app.include_router(vaults_router) # ROADMAP #85 T8 — Vaults +app.include_router(duplicates_router) # ROADMAP #166 — Doublons +app.include_router(notify_router) # ROADMAP #168 — Notifications externes +app.include_router(scheduler_router) # ROADMAP #170 — Tâches planifiées # Admin Dashboard endpoints (system stats, audit logs, backups, stream) try: diff --git a/backend/notify.py b/backend/notify.py new file mode 100644 index 0000000..dea7790 --- /dev/null +++ b/backend/notify.py @@ -0,0 +1,350 @@ +"""External notifications — Discord, Telegram, SMTP, generic webhook (#168). + +Configuration is persisted in ``data/notify_channels.json``; secrets live in +``data/notify_secrets.json`` (0600) or in ``OBSIGATE_NOTIFY_SECRET_`` +environment variables — never in the public config file (same pattern as +``backend/webhooks.py``, BUG-026). + +Supported channel types: + +* ``discord`` — Discord webhook URL (``POST {"content": ...}``). +* ``telegram`` — Bot API (``POST https://api.telegram.org/bot/sendMessage``). +* ``smtp`` — Email via stdlib ``smtplib`` (STARTTLS, auth login). +* ``webhook`` — generic JSON ``POST`` (SSRF-safe, same policy as #9). + +Each channel declares ``triggers`` chosen among :data:`VALID_TRIGGERS`. +The scheduler (#170) broadcasts on ``schedule_failure``; file-event fan-out +stays on the historical ``backend/webhooks.py`` path. +""" + +from __future__ import annotations + +import json +import logging +import os +import smtplib +import threading +import uuid +from datetime import datetime, timezone +from email.message import EmailMessage +from pathlib import Path +from typing import Any + +logger = logging.getLogger("obsigate.notify") + +DATA_DIR = Path(os.environ.get("OBSIGATE_DATA_DIR", "data")) +CHANNELS_FILE = DATA_DIR / "notify_channels.json" +SECRETS_FILE = DATA_DIR / "notify_secrets.json" + +CHANNEL_TYPES = ("discord", "telegram", "smtp", "webhook") +VALID_TRIGGERS = ("manual", "schedule_failure", "schedule_success", "duplicate_found") + +_lock = threading.RLock() + + +# ── Store helpers ────────────────────────────────────────────────────────── + + +def _read_channels() -> list[dict[str, Any]]: + if not CHANNELS_FILE.exists(): + return [] + try: + data = json.loads(CHANNELS_FILE.read_text(encoding="utf-8")) + return data if isinstance(data, list) else [] + except (json.JSONDecodeError, OSError): + return [] + + +def _write_channels(channels: list[dict[str, Any]]) -> None: + CHANNELS_FILE.parent.mkdir(parents=True, exist_ok=True) + tmp = CHANNELS_FILE.with_suffix(".tmp") + tmp.write_text(json.dumps(channels, indent=2, default=str), encoding="utf-8") + tmp.replace(CHANNELS_FILE) + + +def _read_secrets() -> dict[str, str]: + if not SECRETS_FILE.exists(): + return {} + try: + data = json.loads(SECRETS_FILE.read_text(encoding="utf-8")) + return data if isinstance(data, dict) else {} + except (json.JSONDecodeError, OSError): + return {} + + +def _write_secrets(secrets: dict[str, str]) -> None: + SECRETS_FILE.parent.mkdir(parents=True, exist_ok=True) + tmp = SECRETS_FILE.with_suffix(".tmp") + tmp.write_text(json.dumps(secrets, indent=2), encoding="utf-8") + tmp.replace(SECRETS_FILE) + try: + SECRETS_FILE.chmod(0o600) + except OSError: + pass # Windows: pas de permissions Unix + + +def _secret_key(channel_id: str) -> str: + return "OBSIGATE_NOTIFY_SECRET_" + channel_id.replace("-", "_").upper() + + +def _get_secret(channel_id: str) -> str | None: + """Resolve a channel secret: env > dedicated store > legacy inline config.""" + env_val = os.environ.get(_secret_key(channel_id)) + if env_val: + return env_val + stored = _read_secrets().get(channel_id) + if stored: + return stored + for ch in _read_channels(): + if ch.get("id") == channel_id: + cfg = ch.get("config", {}) + for key in ("webhook_url", "bot_token", "password"): + if cfg.get(key): + return str(cfg[key]) + return None + + +def _public_view(channel: dict[str, Any]) -> dict[str, Any]: + clean = {k: v for k, v in channel.items() if k != "config"} + cfg = dict(channel.get("config", {})) + for secret_field in ("webhook_url", "bot_token", "password"): + if cfg.get(secret_field): + cfg[secret_field] = "***" + clean["config"] = cfg + clean["has_secret"] = bool(_get_secret(channel["id"])) + return clean + + +# ── CRUD ─────────────────────────────────────────────────────────────────── + + +def _validate_config(channel_type: str, config: dict[str, Any]) -> dict[str, Any]: + """Validate (sans secret) and normalize a channel config. Raises ValueError.""" + config = dict(config or {}) + if channel_type == "discord": + url = str(config.get("webhook_url") or config.get("url") or "").strip() + if not url.startswith(("https://discord.com/api/webhooks/", "https://discordapp.com/api/webhooks/")): + # Laisse passer les URLs de test locales quand le mode privé est ouvert. + from backend.webhooks import validate_webhook_url + + validate_webhook_url(url) + if "discord" not in url and not os.environ.get("OBSIGATE_WEBHOOK_ALLOW_PRIVATE"): + raise ValueError("URL Discord invalide (webhook discord.com attendu)") + config["webhook_url"] = url + elif channel_type == "telegram": + if not str(config.get("chat_id") or "").strip(): + raise ValueError("chat_id Telegram requis") + config["chat_id"] = str(config["chat_id"]).strip() + if config.get("bot_token"): + config["bot_token"] = str(config["bot_token"]).strip() + elif channel_type == "smtp": + for field in ("host", "from_addr", "to_addr"): + if not str(config.get(field) or "").strip(): + raise ValueError(f"Champ SMTP requis : {field}") + config["port"] = int(config.get("port") or 587) + config["use_tls"] = bool(config.get("use_tls", True)) + config["username"] = str(config.get("username") or "").strip() + elif channel_type == "webhook": + from backend.webhooks import validate_webhook_url + + url = str(config.get("url") or "").strip() + validate_webhook_url(url) + config["url"] = url + else: + raise ValueError(f"Type de canal inconnu : {channel_type}") + triggers = [t for t in (config.get("triggers") or ["manual"]) if t in VALID_TRIGGERS] + config["triggers"] = triggers or ["manual"] + return config + + +def list_channels() -> list[dict[str, Any]]: + """Return public views of all notification channels.""" + return [_public_view(ch) for ch in _read_channels()] + + +def create_channel(name: str, channel_type: str, config: dict[str, Any]) -> dict[str, Any]: + """Create a notification channel. Secrets are split into the secret store.""" + if channel_type not in CHANNEL_TYPES: + raise ValueError(f"Type de canal inconnu : {channel_type}") + with _lock: + channels = _read_channels() + channel_id = str(uuid.uuid4()) + normalized = _validate_config(channel_type, config) + secrets = _read_secrets() + for field in ("webhook_url", "bot_token", "password"): + if normalized.get(field) and len(str(normalized[field])) > 8: + secrets[channel_id] = str(normalized[field]) + normalized[field] = "***" # placeholder : le secret vit dans le store dédié + _write_secrets(secrets) + channel = { + "id": channel_id, + "name": (name or channel_type).strip() or channel_type, + "type": channel_type, + "enabled": True, + "config": normalized, + "created_at": datetime.now(timezone.utc).isoformat(), + "last_sent_at": None, + "last_error": None, + } + channels.append(channel) + _write_channels(channels) + logger.info(f"Created notify channel '{name}' ({channel_type})") + return _public_view(channel) + + +def update_channel(channel_id: str, updates: dict[str, Any]) -> dict[str, Any] | None: + """Update a channel (name/enabled/config). Returns None when unknown.""" + with _lock: + channels = _read_channels() + for channel in channels: + if channel.get("id") != channel_id: + continue + if updates.get("name"): + channel["name"] = str(updates["name"]) + if "enabled" in updates: + channel["enabled"] = bool(updates["enabled"]) + if "config" in updates and isinstance(updates["config"], dict): + merged = {**channel.get("config", {}), **updates["config"]} + normalized = _validate_config(channel["type"], merged) + secrets = _read_secrets() + for field in ("webhook_url", "bot_token", "password"): + if updates["config"].get(field): + secrets[channel_id] = str(updates["config"][field]) + normalized[field] = "***" + _write_secrets(secrets) + channel["config"] = normalized + _write_channels(channels) + return _public_view(channel) + return None + + +def delete_channel(channel_id: str) -> bool: + """Delete a channel and its secret. Returns False when unknown.""" + with _lock: + channels = _read_channels() + remaining = [c for c in channels if c.get("id") != channel_id] + if len(remaining) == len(channels): + return False + _write_channels(remaining) + secrets = _read_secrets() + if secrets.pop(channel_id, None) is not None: + _write_secrets(secrets) + return True + + +# ── Dispatch ─────────────────────────────────────────────────────────────── + + +def _send_discord(webhook_url: str, title: str, message: str) -> None: + import httpx + + content = f"**{title}**\n{message}"[:2000] + resp = httpx.post(webhook_url, json={"content": content}, timeout=10.0) + resp.raise_for_status() + + +def _send_telegram(bot_token: str, chat_id: str, title: str, message: str) -> None: + import httpx + + from backend.webhooks import is_safe_target + + url = f"https://api.telegram.org/bot{bot_token}/sendMessage" + if not is_safe_target(url): + raise RuntimeError("Cible Telegram bloquée par la politique SSRF") + text = f"*{title}*\n{message}"[:4000] + resp = httpx.post( + url, + json={"chat_id": chat_id, "text": text, "parse_mode": "Markdown"}, + timeout=10.0, + ) + resp.raise_for_status() + + +def _send_smtp(config: dict[str, Any], password: str | None, title: str, message: str) -> None: + msg = EmailMessage() + msg["Subject"] = f"[ObsiGate] {title}" + msg["From"] = config["from_addr"] + msg["To"] = config["to_addr"] + msg.set_content(message) + with smtplib.SMTP(str(config["host"]), int(config.get("port", 587)), timeout=10) as client: + if config.get("use_tls", True): + client.starttls() + if config.get("username") and password: + client.login(str(config["username"]), password) + client.send_message(msg) + + +def _send_webhook(url: str, title: str, message: str, trigger: str) -> None: + import httpx + + from backend.webhooks import is_safe_target + + if not is_safe_target(url): + raise RuntimeError("Cible webhook bloquée par la politique SSRF") + resp = httpx.post( + url, + json={ + "event": trigger, + "title": title, + "message": message, + "timestamp": datetime.now(timezone.utc).isoformat(), + "source": "obsigate-notify", + }, + timeout=10.0, + ) + resp.raise_for_status() + + +def send_via_channel(channel: dict[str, Any], title: str, message: str, trigger: str = "manual") -> None: + """Send a notification through one raw channel record. Raises on failure.""" + channel_type = channel.get("type") + cfg = dict(channel.get("config", {})) + secret = _get_secret(channel["id"]) + if channel_type == "discord": + url = secret or cfg.get("webhook_url") or "" + if not url or url == "***": + raise RuntimeError("URL webhook Discord manquante") + _send_discord(url, title, message) + elif channel_type == "telegram": + token = secret or cfg.get("bot_token") or os.environ.get("OBSIGATE_TELEGRAM_BOT_TOKEN") or "" + if not token or token == "***": + raise RuntimeError("Token bot Telegram manquant") + _send_telegram(token, str(cfg.get("chat_id", "")), title, message) + elif channel_type == "smtp": + _send_smtp(cfg, secret, title, message) + elif channel_type == "webhook": + url = str(cfg.get("url") or "").strip() + if not url: + raise RuntimeError("URL webhook manquante") + _send_webhook(url, title, message, trigger) + else: + raise RuntimeError(f"Type de canal inconnu : {channel_type}") + + +def broadcast(trigger: str, title: str, message: str) -> list[dict[str, Any]]: + """Send to every enabled channel subscribed to *trigger*. Never raises.""" + results: list[dict[str, Any]] = [] + for channel in _read_channels(): + if not channel.get("enabled", True): + continue + if trigger not in channel.get("config", {}).get("triggers", ["manual"]): + continue + try: + send_via_channel(channel, title, message, trigger) + results.append({"channel_id": channel["id"], "ok": True}) + _mark_sent(channel["id"], None) + except Exception as e: + logger.warning(f"Notify channel '{channel.get('name')}' failed: {e}") + results.append({"channel_id": channel["id"], "ok": False, "error": str(e)}) + _mark_sent(channel["id"], str(e)) + return results + + +def _mark_sent(channel_id: str, error: str | None) -> None: + with _lock: + channels = _read_channels() + for channel in channels: + if channel.get("id") == channel_id: + channel["last_sent_at"] = datetime.now(timezone.utc).isoformat() + channel["last_error"] = error + _write_channels(channels) diff --git a/backend/openapi_docs.py b/backend/openapi_docs.py index ecd702b..22cf9be 100644 --- a/backend/openapi_docs.py +++ b/backend/openapi_docs.py @@ -41,6 +41,9 @@ TAGS_METADATA: list[dict[str, str]] = [ {"name": "Admin", "description": "Admin-only system monitoring: stats, audit log, backup stats and live stream."}, {"name": "Plugins", "description": "Install, enable and manage user plugins."}, {"name": "Push", "description": "Web Push (VAPID) subscription management and test notifications."}, + {"name": "Duplicates", "description": "Duplicate-note detection and confirmed merge (#166)."}, + {"name": "Notify", "description": "External notifications: Discord, Telegram, SMTP and generic webhooks (#168)."}, + {"name": "Scheduler", "description": "Scheduled automatic tasks reusing the vault mutation services (#170)."}, {"name": "Frontend", "description": "Static assets and SPA fallback routes."}, ] @@ -95,6 +98,9 @@ _TAG_RULES: list[tuple[re.Pattern[str], str]] = [ (re.compile(r"^/api/shares"), "Sharing"), (re.compile(r"^/s/"), "Sharing"), (re.compile(r"^/api/webhooks"), "Webhooks"), + (re.compile(r"^/api/duplicates"), "Duplicates"), + (re.compile(r"^/api/notify"), "Notify"), + (re.compile(r"^/api/scheduler"), "Scheduler"), (re.compile(r"^/api/conflicts"), "Conflicts"), (re.compile(r"^/api/backups"), "Backups"), (re.compile(r"^/api/file/[^/]+/(backups|diff|restore)"), "Backups"), @@ -148,6 +154,9 @@ _TAG_ALIASES: dict[str, str] = { "export": "Export", "sharing": "Sharing", "webhooks": "Webhooks", + "duplicates": "Duplicates", + "notify": "Notify", + "scheduler": "Scheduler", "conflicts": "Conflicts", "system": "System", "frontend": "Frontend", diff --git a/backend/routers/duplicates.py b/backend/routers/duplicates.py new file mode 100644 index 0000000..f0c62d4 --- /dev/null +++ b/backend/routers/duplicates.py @@ -0,0 +1,88 @@ +"""Duplicate detection & merge endpoints (#166). + +Read endpoints require vault access; the merge endpoint is destructive +(backup first in the service layer) and additionally requires the +confirmation token pattern used by mutating routes — here enforced by an +explicit ``confirm=true`` body flag, mirroring the agent two-step flow. +""" + +from __future__ import annotations + +from typing import Any + +from fastapi import APIRouter, Body, Depends, HTTPException, Query +from pydantic import BaseModel, ConfigDict, Field + +from backend.auth.middleware import check_vault_access, require_auth +from backend.services import duplicates as _duplicates +from backend.services.errors import ServiceError + +router = APIRouter(prefix="/api/duplicates", tags=["duplicates"]) + + +class DuplicatePair(BaseModel): + """One candidate duplicate pair.""" + + model_config = ConfigDict(extra="allow") + file_a: str = Field(description="First file (vault-relative)") + file_b: str = Field(description="Second file (vault-relative)") + score: float = Field(description="Blended similarity in [0, 1]") + + +class DuplicatesResponse(BaseModel): + """Response for GET /api/duplicates.""" + + model_config = ConfigDict(extra="allow") + vault: str = Field(description="Vault name") + threshold: float = Field(description="Applied threshold") + files_scanned: int = Field(description="Markdown files compared") + truncated: bool = Field(description="True when the scan hit the file cap") + pairs: list[DuplicatePair] = Field(description="Candidate pairs, best score first") + + +class MergeResponse(BaseModel): + """Response for POST /api/duplicates/merge.""" + + model_config = ConfigDict(extra="allow") + strategy: str = Field(description="Applied merge strategy") + target: str = Field(description="Surviving note") + deleted: str = Field(description="Absorbed note (deleted after merge)") + + +@router.get("", response_model=DuplicatesResponse) +async def api_duplicates_list( + vault: str = Query(..., description="Vault name"), + threshold: float = Query(0.75, ge=0.3, le=1.0, description="Minimum similarity"), + limit: int = Query(20, ge=1, le=200, description="Max pairs"), + subdir: str = Query("", description="Directory scope"), + current_user: dict[str, Any] = Depends(require_auth), +): + """List candidate duplicate notes ordered by descending score.""" + if not check_vault_access(vault, current_user): + raise HTTPException(403, f"No access to vault '{vault}'") + try: + return _duplicates.find_duplicate_pairs(vault, threshold=threshold, limit=limit, subdir=subdir) + except ServiceError as e: + raise HTTPException(e.status or 400, e.message) from e + + +@router.post("/merge", response_model=MergeResponse) +async def api_duplicates_merge( + body: dict[str, Any] = Body(...), + current_user: dict[str, Any] = Depends(require_auth), +): + """Merge *source_path* into *target_path* (``confirm: true`` required).""" + vault = str(body.get("vault") or "") + if not check_vault_access(vault, current_user): + raise HTTPException(403, f"No access to vault '{vault}'") + if body.get("confirm") is not True: + raise HTTPException(400, "Fusion destructive : confirmez avec {confirm: true}") + try: + return _duplicates.merge_duplicates( + vault, + str(body.get("source_path") or ""), + str(body.get("target_path") or ""), + strategy=str(body.get("strategy") or "append"), + ) + except ServiceError as e: + raise HTTPException(e.status or 400, e.message) from e diff --git a/backend/routers/notify.py b/backend/routers/notify.py new file mode 100644 index 0000000..3ad7a05 --- /dev/null +++ b/backend/routers/notify.py @@ -0,0 +1,96 @@ +"""External notification channels endpoints (#168). + +Channel CRUD is admin-only (secrets involved); sending a test notification +requires authentication. Responses mask secrets (``***`` + ``has_secret``). +""" + +from __future__ import annotations + +from typing import Any + +from fastapi import APIRouter, Body, Depends, HTTPException +from pydantic import BaseModel, ConfigDict, Field + +from backend import notify as _notify +from backend.auth.middleware import require_admin, require_auth +from backend.schemas import StatusResponse + +router = APIRouter(prefix="/api/notify", tags=["notify"]) + + +class NotifyChannel(BaseModel): + """Public view of a notification channel (secrets masked).""" + + model_config = ConfigDict(extra="allow") + id: str = Field(description="Channel id") + name: str = Field(description="Display name") + type: str = Field(description="discord | telegram | smtp | webhook") + enabled: bool = Field(description="Whether the channel receives broadcasts") + config: dict[str, Any] = Field(description="Channel config (secrets masked)") + has_secret: bool = Field(description="True when a secret is configured") + + +class NotifySendResult(BaseModel): + """Outcome of a test send / broadcast.""" + + model_config = ConfigDict(extra="allow") + ok: bool = Field(description="True when every delivery succeeded") + deliveries: list[dict[str, Any]] = Field(default_factory=list) + + +@router.get("/channels", response_model=list[NotifyChannel]) +async def api_notify_list(current_user=Depends(require_admin)): + """List notification channels (admin).""" + return _notify.list_channels() + + +@router.post("/channels", response_model=NotifyChannel) +async def api_notify_create(body: dict = Body(...), current_user=Depends(require_admin)): + """Create a channel (``{name, type, config}``). Secrets go to the secret store.""" + try: + return _notify.create_channel( + str(body.get("name") or ""), + str(body.get("type") or ""), + dict(body.get("config") or {}), + ) + except ValueError as e: + raise HTTPException(400, str(e)) from e + + +@router.patch("/channels/{channel_id}", response_model=NotifyChannel) +async def api_notify_update(channel_id: str, body: dict = Body(...), current_user=Depends(require_admin)): + """Update a channel (name / enabled / config).""" + try: + result = _notify.update_channel(channel_id, body) + except ValueError as e: + raise HTTPException(400, str(e)) from e + if result is None: + raise HTTPException(404, "Channel not found") + return result + + +@router.delete("/channels/{channel_id}", response_model=StatusResponse) +async def api_notify_delete(channel_id: str, current_user=Depends(require_admin)): + """Delete a channel and its secret.""" + if not _notify.delete_channel(channel_id): + raise HTTPException(404, "Channel not found") + return {"status": "deleted"} + + +@router.post("/test", response_model=NotifySendResult) +async def api_notify_test(body: dict = Body(...), current_user=Depends(require_auth)): + """Send a test notification (broadcast or single ``channel_id``).""" + title = str(body.get("title") or "Test ObsiGate") + message = str(body.get("message") or "Notification de test.") + channel_id = str(body.get("channel_id") or "") + if channel_id: + channel = next((c for c in _notify._read_channels() if c.get("id") == channel_id), None) + if channel is None: + raise HTTPException(404, "Channel not found") + try: + _notify.send_via_channel(channel, title, message, "manual") + except Exception as e: + raise HTTPException(502, f"Envoi échoué : {e}") from e + return {"ok": True, "deliveries": [{"channel_id": channel_id, "ok": True}]} + deliveries = _notify.broadcast("manual", title, message) + return {"ok": all(d.get("ok") for d in deliveries), "deliveries": deliveries} diff --git a/backend/routers/scheduler.py b/backend/routers/scheduler.py new file mode 100644 index 0000000..f81840c --- /dev/null +++ b/backend/routers/scheduler.py @@ -0,0 +1,118 @@ +"""Scheduled tasks endpoints (#170). + +Tasks reuse the existing mutation/notification services — this router only +validates, persists and triggers. File-writing actions check vault access +at creation time; the background tick re-checks nothing (system context) but +records failures and notifies on ``schedule_failure`` (#168). +""" + +from __future__ import annotations + +from typing import Any + +from fastapi import APIRouter, Body, Depends, HTTPException +from pydantic import BaseModel, ConfigDict, Field + +from backend import scheduler as _scheduler +from backend.auth.middleware import check_vault_access, require_auth +from backend.schemas import StatusResponse + +router = APIRouter(prefix="/api/scheduler", tags=["scheduler"]) + + +class ScheduledTask(BaseModel): + """A programmed automatic task.""" + + model_config = ConfigDict(extra="allow") + id: str = Field(description="Task id") + name: str = Field(description="Display name") + action: dict[str, Any] = Field(description="{kind, params}") + schedule: dict[str, Any] = Field(description="{kind, ...}") + enabled: bool = Field(description="Whether the tick executes it") + created_by: str = Field(description="Owner username") + created_at: str = Field(description="ISO-8601 creation time") + last_run_at: str | None = Field(default=None) + last_status: str | None = Field(default=None) + last_error: str | None = Field(default=None) + run_count: int = Field(default=0) + next_run_at: str = Field(description="ISO-8601 next due time") + + +class TaskRunResult(BaseModel): + """Outcome of a manual or due run.""" + + model_config = ConfigDict(extra="allow") + task_id: str = Field(description="Task id") + ok: bool = Field(description="True on success") + result: dict[str, Any] | None = Field(default=None) + error: str | None = Field(default=None) + + +def _check_action_vault(action: dict[str, Any], user: dict[str, Any]) -> None: + from backend.services.errors import ServiceError + from backend.services.vaults import get_vault_root + + kind = (action or {}).get("kind") + params = (action or {}).get("params") or {} + if kind in ("create_file", "append_to_file"): + vault = str(params.get("vault") or "") + if not check_vault_access(vault, user): + raise HTTPException(403, f"No access to vault '{vault}'") + try: + get_vault_root(vault) + except ServiceError as e: + raise HTTPException(404, f"Unknown vault '{vault}'") from e + + +@router.get("/tasks", response_model=list[ScheduledTask]) +async def api_scheduler_list(current_user: dict[str, Any] = Depends(require_auth)): + """List scheduled tasks (newest first).""" + return _scheduler.list_tasks() + + +@router.post("/tasks", response_model=ScheduledTask) +async def api_scheduler_create(body: dict = Body(...), current_user: dict[str, Any] = Depends(require_auth)): + """Create a task (``{name, action, schedule, enabled?}``).""" + action = dict(body.get("action") or {}) + _check_action_vault(action, current_user) + try: + return _scheduler.create_task( + str(body.get("name") or ""), + action, + dict(body.get("schedule") or {}), + created_by=str(current_user.get("username", "api")), + enabled=bool(body.get("enabled", True)), + ) + except ValueError as e: + raise HTTPException(400, str(e)) from e + + +@router.patch("/tasks/{task_id}", response_model=ScheduledTask) +async def api_scheduler_update(task_id: str, body: dict = Body(...), current_user: dict[str, Any] = Depends(require_auth)): + """Update a task (name / enabled / action / schedule).""" + if "action" in body: + _check_action_vault(dict(body["action"] or {}), current_user) + try: + result = _scheduler.update_task(task_id, body) + except ValueError as e: + raise HTTPException(400, str(e)) from e + if result is None: + raise HTTPException(404, "Task not found") + return result + + +@router.delete("/tasks/{task_id}", response_model=StatusResponse) +async def api_scheduler_delete(task_id: str, current_user: dict[str, Any] = Depends(require_auth)): + """Delete a task.""" + if not _scheduler.delete_task(task_id): + raise HTTPException(404, "Task not found") + return {"status": "deleted"} + + +@router.post("/tasks/{task_id}/run", response_model=TaskRunResult) +async def api_scheduler_run(task_id: str, current_user: dict[str, Any] = Depends(require_auth)): + """Execute a task immediately (manual run).""" + try: + return _scheduler.run_task(task_id, manual=True) + except KeyError: + raise HTTPException(404, "Task not found") from None diff --git a/backend/scheduler.py b/backend/scheduler.py new file mode 100644 index 0000000..9aeeae4 --- /dev/null +++ b/backend/scheduler.py @@ -0,0 +1,331 @@ +"""Scheduled tasks — automatic agent actions, type cron (#170). + +Tasks are persisted in ``data/scheduled_tasks.json`` (guarded by an RLock, +same pattern as the other JSON stores). Supported actions reuse the existing +mutation/notification services — no new write path: + +* ``create_file`` → ``backend.services.mutations.create_file``; +* ``append_to_file`` → ``backend.services.mutations.append_to_file``; +* ``notify`` → ``backend.notify.broadcast`` (trigger ``manual``). + +Supported schedules: + +* ``interval_hours`` — every N hours (N >= 0.25); +* ``daily_time`` — once a day at ``HH:MM`` (local server time); +* ``once_at`` — one shot at an ISO-8601 datetime (past = due immediately). + +On failure the task records ``last_error`` and a ``schedule_failure`` +broadcast is emitted to the notification channels (#168) — best effort, +never recursive (a failing ``notify`` action does not rebroadcast). +""" + +from __future__ import annotations + +import json +import logging +import os +import threading +import uuid +from datetime import datetime, timedelta, timezone +from pathlib import Path +from typing import Any + +logger = logging.getLogger("obsigate.scheduler") + +DATA_DIR = Path(os.environ.get("OBSIGATE_DATA_DIR", "data")) +TASKS_FILE = DATA_DIR / "scheduled_tasks.json" + +ACTION_KINDS = ("create_file", "append_to_file", "notify") +SCHEDULE_KINDS = ("interval_hours", "daily_time", "once_at") + +_lock = threading.RLock() + + +# ── Store ────────────────────────────────────────────────────────────────── + + +def _read_tasks() -> list[dict[str, Any]]: + if not TASKS_FILE.exists(): + return [] + try: + data = json.loads(TASKS_FILE.read_text(encoding="utf-8")) + return data if isinstance(data, list) else [] + except (json.JSONDecodeError, OSError): + return [] + + +def _write_tasks(tasks: list[dict[str, Any]]) -> None: + TASKS_FILE.parent.mkdir(parents=True, exist_ok=True) + tmp = TASKS_FILE.with_suffix(".tmp") + tmp.write_text(json.dumps(tasks, indent=2, default=str), encoding="utf-8") + tmp.replace(TASKS_FILE) + + +def list_tasks() -> list[dict[str, Any]]: + """Return all scheduled tasks (newest first).""" + return sorted(_read_tasks(), key=lambda t: t.get("created_at", ""), reverse=True) + + +def get_task(task_id: str) -> dict[str, Any] | None: + """Return one task by id, or None.""" + for task in _read_tasks(): + if task.get("id") == task_id: + return task + return None + + +# ── Validation ───────────────────────────────────────────────────────────── + + +def _validate_action(action: dict[str, Any]) -> dict[str, Any]: + kind = action.get("kind") + if kind not in ACTION_KINDS: + raise ValueError(f"Action inconnue : {kind} (attendu : {', '.join(ACTION_KINDS)})") + params = dict(action.get("params") or {}) + if kind in ("create_file", "append_to_file"): + if not str(params.get("vault") or "").strip(): + raise ValueError("params.vault requis pour create_file/append_to_file") + if not str(params.get("path") or "").strip(): + raise ValueError("params.path requis pour create_file/append_to_file") + if kind == "append_to_file" and not str(params.get("content") or ""): + raise ValueError("params.content requis pour append_to_file") + elif kind == "notify": + if not str(params.get("title") or "").strip(): + raise ValueError("params.title requis pour notify") + if not str(params.get("message") or "").strip(): + raise ValueError("params.message requis pour notify") + return {"kind": kind, "params": params} + + +def _validate_schedule(schedule: dict[str, Any]) -> dict[str, Any]: + kind = schedule.get("kind") + if kind not in SCHEDULE_KINDS: + raise ValueError(f"Planification inconnue : {kind} (attendu : {', '.join(SCHEDULE_KINDS)})") + if kind == "interval_hours": + hours = float(schedule.get("hours") or 0) + if hours < 0.25: + raise ValueError("hours doit être >= 0.25") + return {"kind": kind, "hours": hours} + if kind == "daily_time": + at = str(schedule.get("at") or "").strip() + try: + datetime.strptime(at, "%H:%M") + except ValueError: + raise ValueError("at doit être au format HH:MM (ex. 08:30)") from None + return {"kind": kind, "at": at} + # once_at + at = str(schedule.get("at") or "").strip() + try: + parsed = datetime.fromisoformat(at) + if parsed.tzinfo is None: + parsed = parsed.replace(tzinfo=timezone.utc) + except ValueError: + raise ValueError("at doit être une date ISO-8601 (ex. 2026-10-05T08:30:00)") from None + return {"kind": kind, "at": parsed.isoformat()} + + +# ── CRUD ─────────────────────────────────────────────────────────────────── + + +def create_task( + name: str, + action: dict[str, Any], + schedule: dict[str, Any], + *, + created_by: str = "api", + enabled: bool = True, +) -> dict[str, Any]: + """Create a scheduled task. Raises ValueError on invalid action/schedule.""" + validated_action = _validate_action(action) + validated_schedule = _validate_schedule(schedule) + now = datetime.now(timezone.utc) + with _lock: + tasks = _read_tasks() + task = { + "id": str(uuid.uuid4()), + "name": (name or validated_action["kind"]).strip() or validated_action["kind"], + "action": validated_action, + "schedule": validated_schedule, + "enabled": bool(enabled), + "created_by": created_by, + "created_at": now.isoformat(), + "last_run_at": None, + "last_status": None, + "last_error": None, + "run_count": 0, + "next_run_at": compute_next_run( + {"schedule": validated_schedule, "last_run_at": None}, now + ).isoformat(), + } + tasks.append(task) + _write_tasks(tasks) + logger.info(f"Scheduled task created: '{task['name']}' ({validated_schedule['kind']})") + return task + + +def update_task(task_id: str, updates: dict[str, Any]) -> dict[str, Any] | None: + """Update name/enabled/action/schedule. Returns None when unknown.""" + with _lock: + tasks = _read_tasks() + for task in tasks: + if task.get("id") != task_id: + continue + if updates.get("name"): + task["name"] = str(updates["name"]) + if "enabled" in updates: + task["enabled"] = bool(updates["enabled"]) + if "action" in updates: + task["action"] = _validate_action(updates["action"]) + if "schedule" in updates: + task["schedule"] = _validate_schedule(updates["schedule"]) + task["next_run_at"] = compute_next_run(task).isoformat() + _write_tasks(tasks) + return task + return None + + +def delete_task(task_id: str) -> bool: + """Delete a task. Returns False when unknown.""" + with _lock: + tasks = _read_tasks() + remaining = [t for t in tasks if t.get("id") != task_id] + if len(remaining) == len(tasks): + return False + _write_tasks(remaining) + return True + + +# ── Scheduling ───────────────────────────────────────────────────────────── + + +def compute_next_run(task: dict[str, Any], now: datetime | None = None) -> datetime: + """Compute the next due datetime for *task*.""" + now = now or datetime.now(timezone.utc) + if now.tzinfo is None: + now = now.replace(tzinfo=timezone.utc) + schedule = task.get("schedule", {}) + kind = schedule.get("kind") + last_run_at = task.get("last_run_at") + last = None + if last_run_at: + try: + last = datetime.fromisoformat(str(last_run_at)) + if last.tzinfo is None: + last = last.replace(tzinfo=timezone.utc) + except ValueError: + last = None + if kind == "interval_hours": + hours = float(schedule.get("hours", 24)) + base = last or now + nxt = base + timedelta(hours=hours) + # Première planification : due dès maintenant + intervalle ? Non — + # la tâche démarre au prochain intervalle, sauf retard déjà accumulé. + if last is None: + nxt = now + timedelta(hours=hours) + return max(now, nxt) + if kind == "daily_time": + hour, minute = (str(schedule.get("at", "08:00")) + ":00").split(":")[:2] + candidate = now.replace(hour=int(hour), minute=int(minute), second=0, microsecond=0) + if candidate <= now: + candidate += timedelta(days=1) + return candidate + if kind == "once_at": + try: + at = datetime.fromisoformat(str(schedule.get("at"))) + if at.tzinfo is None: + at = at.replace(tzinfo=timezone.utc) + except ValueError: + return now + if task.get("last_run_at"): + return datetime.max.replace(tzinfo=timezone.utc) # déjà exécutée + return at + return now + timedelta(hours=24) + + +def _execute_action(task: dict[str, Any]) -> dict[str, Any]: + action = task["action"] + kind = action["kind"] + params = action["params"] + if kind == "create_file": + from backend.services.mutations import create_file + + return create_file( + params["vault"], + params["path"], + params.get("content", ""), + overwrite=bool(params.get("overwrite", False)), + ) + if kind == "append_to_file": + from backend.services.mutations import append_to_file + + return append_to_file(params["vault"], params["path"], params.get("content", "")) + if kind == "notify": + from backend.notify import broadcast + + results = broadcast("manual", str(params["title"]), str(params.get("message", ""))) + return {"broadcast": results} + raise ValueError(f"Action inconnue : {kind}") + + +def run_task(task_id: str, *, manual: bool = False) -> dict[str, Any]: + """Execute one task now (manual or due). Records status; notifies on failure.""" + with _lock: + tasks = _read_tasks() + task = next((t for t in tasks if t.get("id") == task_id), None) + if task is None: + raise KeyError(task_id) + if not task.get("enabled", True) and not manual: + return {"task_id": task_id, "skipped": True, "reason": "disabled"} + try: + result = _execute_action(task) + task["last_run_at"] = datetime.now(timezone.utc).isoformat() + task["last_status"] = "ok" + task["last_error"] = None + task["run_count"] = int(task.get("run_count", 0)) + 1 + if task.get("schedule", {}).get("kind") == "once_at": + task["enabled"] = False # one-shot consommé + task["next_run_at"] = compute_next_run(task).isoformat() + _write_tasks(tasks) + if manual: + from backend.notify import broadcast + + broadcast("schedule_success", f"Tâche « {task['name']} » OK", "Exécution manuelle réussie.") + return {"task_id": task_id, "ok": True, "result": result} + except Exception as e: + task["last_run_at"] = datetime.now(timezone.utc).isoformat() + task["last_status"] = "error" + task["last_error"] = str(e) + task["run_count"] = int(task.get("run_count", 0)) + 1 + task["next_run_at"] = compute_next_run(task).isoformat() + _write_tasks(tasks) + logger.warning(f"Scheduled task '{task.get('name')}' failed: {e}") + if task["action"]["kind"] != "notify": + try: + from backend.notify import broadcast + + broadcast( + "schedule_failure", + f"Échec tâche « {task.get('name')} »", + f"{e}", + ) + except Exception: + logger.debug("Failure notification broadcast failed", exc_info=True) + return {"task_id": task_id, "ok": False, "error": str(e)} + + +def tick(now: datetime | None = None) -> list[dict[str, Any]]: + """Run every due task. Returns per-task outcomes (empty when idle).""" + now = now or datetime.now(timezone.utc) + outcomes: list[dict[str, Any]] = [] + for task in _read_tasks(): + if not task.get("enabled", True): + continue + try: + next_run = datetime.fromisoformat(str(task.get("next_run_at") or "")) + if next_run.tzinfo is None: + next_run = next_run.replace(tzinfo=timezone.utc) + except ValueError: + next_run = compute_next_run(task, now) + if next_run <= now: + outcomes.append(run_task(task["id"])) + return outcomes diff --git a/backend/services/duplicates.py b/backend/services/duplicates.py new file mode 100644 index 0000000..9c1c196 --- /dev/null +++ b/backend/services/duplicates.py @@ -0,0 +1,215 @@ +"""Duplicate detection & merge services (#166). + +Single source of truth consumed by the REST routes +(``/api/duplicates``) and the AI tool layer (``find_duplicates``, +``merge_duplicate_notes``). + +Method is deterministic stdlib-only: frontmatter stripped, token-set +Jaccard blended with a title similarity. No embedding dependency — +the semantic index (#70) stays an optional refinement, not a requirement. + +Fusion never runs without an explicit confirmation: the tool layer +registers the merge as ``DANGEROUS`` (two-step propose/apply) and this +service takes an automatic backup before any destructive write. +""" + +from __future__ import annotations + +import logging +import re +from difflib import SequenceMatcher +from pathlib import Path +from typing import Any + +from backend.services.backups import create_backup +from backend.services.errors import ServiceError +from backend.services.paths import resolve_safe_path +from backend.services.vaults import get_vault_root + +logger = logging.getLogger("obsigate.services.duplicates") + +MAX_FILES_SCANNED = 500 +MAX_FILE_BYTES = 200_000 +MAX_CONTENT_CHARS = 50_000 + +_WORD_RE = re.compile(r"[\w]+", re.UNICODE) +_FRONTMATTER_RE = re.compile(r"\A---\s*\n.*?\n---\s*\n", re.DOTALL) + + +def _strip_frontmatter(text: str) -> str: + """Remove a leading YAML frontmatter block, if present.""" + return _FRONTMATTER_RE.sub("", text, count=1) + + +def _tokens(text: str) -> set[str]: + """Lowercase word tokens (keeps accents), stop-words free but tiny tokens dropped.""" + return {t for t in _WORD_RE.findall(text.lower()) if len(t) > 2} + + +def similarity_score(a: str, b: str) -> float: + """Blend Jaccard (0.7) + title/first-line similarity (0.3) in [0, 1]. + + Pure function — unit-tested directly. + """ + ta, tb = _tokens(_strip_frontmatter(a)), _tokens(_strip_frontmatter(b)) + if not ta or not tb: + return 0.0 + jaccard = len(ta & tb) / len(ta | tb) + head_a = (a.strip().splitlines() or [""])[:1][0][:200].lower() + head_b = (b.strip().splitlines() or [""])[:1][0][:200].lower() + title_sim = SequenceMatcher(None, head_a, head_b).ratio() if head_a and head_b else 0.0 + return round(0.7 * jaccard + 0.3 * title_sim, 4) + + +def _iter_markdown_files(root: Path, subdir: str = "") -> list[Path]: + base = resolve_safe_path(root, subdir) if subdir else root.resolve() + if not base.exists() or not base.is_dir(): + raise ServiceError( + f"Directory not found: {subdir or '.'}", + code="not_found", + status=404, + details={"path": subdir}, + ) + files = sorted( + (p for p in base.rglob("*.md") if p.is_file() and not p.is_symlink()), + key=lambda p: str(p), + ) + return files[:MAX_FILES_SCANNED] + + +def _read_capped(path: Path) -> str: + try: + if path.stat().st_size > MAX_FILE_BYTES: + return "" + text = path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + return text[:MAX_CONTENT_CHARS] + + +def find_duplicate_pairs( + vault: str, + threshold: float = 0.75, + limit: int = 50, + subdir: str = "", +) -> dict[str, Any]: + """Return candidate duplicate pairs ordered by descending score. + + Args: + vault: Vault name. + threshold: Minimum blended score in [0.3, 1.0]. + limit: Max pairs returned (1-200). + subdir: Optional vault-relative directory scope. + """ + if not 0.3 <= threshold <= 1.0: + raise ServiceError( + "threshold must be between 0.3 and 1.0", + code="invalid_arguments", + status=400, + ) + limit = max(1, min(limit, 200)) + root = get_vault_root(vault) + files = _iter_markdown_files(root, subdir) + contents: dict[str, str] = {} + token_sets: dict[str, set[str]] = {} + for path in files: + rel = str(path.relative_to(root)).replace("\\", "/") + text = _read_capped(path) + if not text.strip(): + continue + contents[rel] = text + token_sets[rel] = _tokens(_strip_frontmatter(text)) + + rels = sorted(contents) + pairs: list[dict[str, Any]] = [] + for i in range(len(rels)): + for j in range(i + 1, len(rels)): + a, b = rels[i], rels[j] + ta, tb = token_sets[a], token_sets[b] + if not ta or not tb: + continue + # Cheap pre-filter: Jaccard lower bound before the full score. + inter = len(ta & tb) + union = len(ta | tb) + if union == 0 or inter / union < threshold * 0.6: + continue + score = similarity_score(contents[a], contents[b]) + if score >= threshold: + pairs.append({"file_a": a, "file_b": b, "score": score}) + pairs.sort(key=lambda p: p["score"], reverse=True) + return { + "vault": vault, + "threshold": threshold, + "files_scanned": len(contents), + "truncated": len(files) >= MAX_FILES_SCANNED, + "pairs": pairs[:limit], + } + + +def merge_duplicates( + vault: str, + source_path: str, + target_path: str, + strategy: str = "append", +) -> dict[str, Any]: + """Merge *source_path* into *target_path*, then delete the source. + + Strategies: + ``append`` — source content appended after target (separator + origin + marker), source deleted. + ``prefer_target`` — source deleted, target untouched (dedupe only). + ``prefer_source`` — target overwritten with source content, source deleted. + + A backup of both files is taken first; the source deletion also goes + through the backup-aware mutation service. + """ + from backend.services import mutations as _mutations + + if strategy not in ("append", "prefer_target", "prefer_source"): + raise ServiceError( + f"Unknown strategy: {strategy}", + code="invalid_arguments", + status=400, + ) + if source_path == target_path: + raise ServiceError( + "source_path and target_path must differ", + code="invalid_arguments", + status=400, + ) + root = get_vault_root(vault) + src = resolve_safe_path(root, source_path) + dst = resolve_safe_path(root, target_path) + if not src.is_file() or src.suffix.lower() != ".md": + raise ServiceError( + f"Source not found: {source_path}", + code="not_found", + status=404, + details={"path": source_path}, + ) + if not dst.is_file() or dst.suffix.lower() != ".md": + raise ServiceError( + f"Target not found: {target_path}", + code="not_found", + status=404, + details={"path": target_path}, + ) + # Backup préalable (jamais de fusion sans filet — critère #166). + create_backup(src, vault, source_path) + create_backup(dst, vault, target_path) + + if strategy == "prefer_target": + result = _mutations.delete_file(vault, source_path) + return {"strategy": strategy, "target": target_path, "deleted": source_path, "delete": result} + if strategy == "prefer_source": + content = src.read_text(encoding="utf-8", errors="replace") + result = _mutations.edit_file(vault, target_path, content) + deleted = _mutations.delete_file(vault, source_path) + return {"strategy": strategy, "target": target_path, "edit": result, "deleted": source_path, "delete": deleted} + # append + target_text = dst.read_text(encoding="utf-8", errors="replace") + source_text = src.read_text(encoding="utf-8", errors="replace") + merged = target_text.rstrip() + f"\n\n---\n\n_Fusionné depuis `{source_path}` (#166)_\n\n" + source_text.lstrip() + result = _mutations.edit_file(vault, target_path, merged) + deleted = _mutations.delete_file(vault, source_path) + return {"strategy": strategy, "target": target_path, "edit": result, "deleted": source_path, "delete": deleted} diff --git a/backend/tools/api.py b/backend/tools/api.py index 05c66ef..20dc144 100644 --- a/backend/tools/api.py +++ b/backend/tools/api.py @@ -12,6 +12,9 @@ which ``.gitignore`` excludes via ``_*.py``), hence this explicit facade. from backend.tools import connected as _connected # noqa: F401 (registers connected-source tools) from backend.tools import crawler as _crawler # noqa: F401 (registers the site crawler) from backend.tools import documents as _documents # noqa: F401 (registers document tools) +from backend.tools import duplicates as _duplicates # noqa: F401 (registers duplicate tools #166) +from backend.tools import notify as _notify_tools # noqa: F401 (registers notify tool #168) +from backend.tools import scheduled as _scheduled # noqa: F401 (registers scheduler tools #170) from backend.tools import service as _service # noqa: F401 (registers tools) from backend.tools import spreadsheets as _spreadsheets # noqa: F401 (registers existing-workbook tools #153 A6) from backend.tools import web as _web # noqa: F401 (registers web tools) diff --git a/backend/tools/duplicates.py b/backend/tools/duplicates.py new file mode 100644 index 0000000..131f425 --- /dev/null +++ b/backend/tools/duplicates.py @@ -0,0 +1,59 @@ +"""Duplicate detection & merge tools (#166). + +* ``find_duplicates`` — READ, vault-scoped: candidate pairs with scores. +* ``merge_duplicate_notes`` — DANGEROUS: confirmed fusion with automatic + backup (service layer), never without an explicit approval. +""" + +from __future__ import annotations + +from typing import Any + +from backend.services import duplicates as _duplicates +from backend.services.errors import ServiceError +from backend.tools.context import ToolContext, ToolError, ToolRisk +from backend.tools.registry import tool +from backend.tools.schemas import FindDuplicatesInput, MergeDuplicatesInput + + +@tool( + name="find_duplicates", + description="Find candidate duplicate markdown notes in a vault (similarity scores).", + input_model=FindDuplicatesInput, + risk=ToolRisk.READ, + requires_vault=True, +) +def find_duplicates(ctx: ToolContext, params: FindDuplicatesInput) -> dict[str, Any]: + """List duplicate candidates ordered by descending score.""" + try: + return _duplicates.find_duplicate_pairs( + params.vault, + threshold=params.threshold, + limit=params.limit, + subdir=params.subdir, + ) + except ServiceError as e: + raise ToolError(e.message, code=e.code, details=e.details) from e + + +@tool( + name="merge_duplicate_notes", + description=( + "Merge one note into another and delete the source (backup first). " + "Destructive: requires confirmation." + ), + input_model=MergeDuplicatesInput, + risk=ToolRisk.DANGEROUS, + requires_vault=True, +) +def merge_duplicate_notes(ctx: ToolContext, params: MergeDuplicatesInput) -> dict[str, Any]: + """Fuse *source_path* into *target_path* using the chosen strategy.""" + try: + return _duplicates.merge_duplicates( + params.vault, + params.source_path, + params.target_path, + strategy=params.strategy, + ) + except ServiceError as e: + raise ToolError(e.message, code=e.code, details=e.details) from e diff --git a/backend/tools/labels.py b/backend/tools/labels.py index 7ee9a1a..2574ed2 100644 --- a/backend/tools/labels.py +++ b/backend/tools/labels.py @@ -62,6 +62,13 @@ _STEP_LABELS: dict[str, tuple[str, str | None]] = { "create_docx": ("docx_create", "path"), "create_csv": ("csv_create", "path"), "create_pdf": ("pdf_create", "path"), + "find_duplicates": ("duplicates", "vault"), + "merge_duplicate_notes": ("duplicates_merge", "source_path"), + "notify_external": ("notify", "title"), + "create_scheduled_task": ("schedule_create", "name"), + "list_scheduled_tasks": ("schedule_list", None), + "delete_scheduled_task": ("schedule_delete", "task_id"), + "run_scheduled_task_now": ("schedule_run", "task_id"), } GENERIC_KEY = "generic" diff --git a/backend/tools/notify.py b/backend/tools/notify.py new file mode 100644 index 0000000..668380c --- /dev/null +++ b/backend/tools/notify.py @@ -0,0 +1,42 @@ +"""External notification tool (#168) — Discord, Telegram, SMTP, webhook. + +``notify_external`` is WRITE (external side effect → confirmation card in the +UI, propose/apply over MCP). Delivery itself lives in :mod:`backend.notify`. +""" + +from __future__ import annotations + +from typing import Any + +from backend import notify as _notify +from backend.tools.context import ToolContext, ToolError, ToolRisk +from backend.tools.registry import tool +from backend.tools.schemas import NotifyExternalInput + + +@tool( + name="notify_external", + description="Send a notification through external channels (Discord, Telegram, SMTP, webhook).", + input_model=NotifyExternalInput, + risk=ToolRisk.WRITE, +) +def notify_external(ctx: ToolContext, params: NotifyExternalInput) -> dict[str, Any]: + """Broadcast to the trigger scope, or target a single channel id.""" + try: + if params.channel_id: + channel = next( + (c for c in _notify._read_channels() if c.get("id") == params.channel_id), + None, + ) + if channel is None: + raise ToolError(f"Unknown channel: {params.channel_id}", code="not_found") + if not channel.get("enabled", True): + raise ToolError(f"Channel disabled: {params.channel_id}", code="invalid_arguments") + _notify.send_via_channel(channel, params.title, params.message, params.trigger) + return {"ok": True, "channel_id": params.channel_id} + results = _notify.broadcast(params.trigger, params.title, params.message) + return {"ok": True, "deliveries": results} + except ToolError: + raise + except Exception as e: + raise ToolError(f"Notification failed: {e}", code="notify_failed") from e diff --git a/backend/tools/scheduled.py b/backend/tools/scheduled.py new file mode 100644 index 0000000..5cc77b5 --- /dev/null +++ b/backend/tools/scheduled.py @@ -0,0 +1,78 @@ +"""Scheduled-task tools (#170) — the agent programs its own cron. + +* ``create_scheduled_task`` — WRITE (a future write, confirmed once now). +* ``list_scheduled_tasks`` — READ. +* ``delete_scheduled_task`` — WRITE (removes a future side effect). +* ``run_scheduled_task_now`` — WRITE (immediate side effect). +""" + +from __future__ import annotations + +from typing import Any + +from backend import scheduler as _scheduler +from backend.tools.context import ToolContext, ToolError, ToolRisk +from backend.tools.registry import tool +from backend.tools.schemas import ( + CreateScheduledTaskInput, + DeleteScheduledTaskInput, + ListVaultsInput, + RunScheduledTaskInput, +) + + +@tool( + name="create_scheduled_task", + description="Program an automatic task (create_file, append_to_file, notify) on a cron-like schedule.", + input_model=CreateScheduledTaskInput, + risk=ToolRisk.WRITE, +) +def create_scheduled_task(ctx: ToolContext, params: CreateScheduledTaskInput) -> dict[str, Any]: + """Create a task owned by the requesting user.""" + try: + return _scheduler.create_task( + params.name, + params.action, + params.schedule, + created_by=ctx.username, + ) + except ValueError as e: + raise ToolError(str(e), code="invalid_arguments") from e + + +@tool( + name="list_scheduled_tasks", + description="List automatic tasks programmed in ObsiGate.", + input_model=ListVaultsInput, + risk=ToolRisk.READ, +) +def list_scheduled_tasks(ctx: ToolContext, _params: ListVaultsInput) -> list[dict[str, Any]]: + """Return tasks newest first.""" + return _scheduler.list_tasks() + + +@tool( + name="delete_scheduled_task", + description="Delete a programmed automatic task.", + input_model=DeleteScheduledTaskInput, + risk=ToolRisk.WRITE, +) +def delete_scheduled_task(ctx: ToolContext, params: DeleteScheduledTaskInput) -> dict[str, Any]: + """Delete by id; unknown id is a not_found tool error.""" + if not _scheduler.delete_task(params.task_id): + raise ToolError(f"Unknown task: {params.task_id}", code="not_found") + return {"ok": True, "task_id": params.task_id} + + +@tool( + name="run_scheduled_task_now", + description="Execute a programmed task immediately (manual run).", + input_model=RunScheduledTaskInput, + risk=ToolRisk.WRITE, +) +def run_scheduled_task_now(ctx: ToolContext, params: RunScheduledTaskInput) -> dict[str, Any]: + """Run now and return the outcome (failures are recorded + notified).""" + try: + return _scheduler.run_task(params.task_id, manual=True) + except KeyError as e: + raise ToolError(f"Unknown task: {params.task_id}", code="not_found") from e diff --git a/backend/tools/schemas.py b/backend/tools/schemas.py index e8a694a..1a86c0d 100644 --- a/backend/tools/schemas.py +++ b/backend/tools/schemas.py @@ -441,6 +441,53 @@ class PdfInput(BaseModel): overwrite: bool = Field(True, description="Replace an existing file (with backup)") +class FindDuplicatesInput(BaseModel): + """Find candidate duplicate notes in a vault (#166).""" + + vault: str = Field(..., description="Vault name") + threshold: float = Field(0.75, ge=0.3, le=1.0, description="Minimum similarity score") + limit: int = Field(20, ge=1, le=200, description="Maximum number of pairs") + subdir: str = Field("", description="Vault-relative directory scope (empty = whole vault)") + + +class MergeDuplicatesInput(BaseModel): + """Merge one note into another, then delete the source (#166, destructive).""" + + vault: str = Field(..., description="Vault name") + source_path: str = Field(..., description="Vault-relative path of the note to absorb") + target_path: str = Field(..., description="Vault-relative path of the surviving note") + strategy: str = Field("append", description="'append', 'prefer_target' or 'prefer_source'") + + +class NotifyExternalInput(BaseModel): + """Send a notification through external channels (#168).""" + + title: str = Field(..., min_length=1, description="Notification title") + message: str = Field(..., min_length=1, description="Notification body") + trigger: str = Field("manual", description="Trigger scope: manual, schedule_failure, schedule_success") + channel_id: str = Field("", description="Single channel id (empty = broadcast to trigger)") + + +class CreateScheduledTaskInput(BaseModel): + """Create an automatic task executed by the scheduler (#170).""" + + name: str = Field(..., min_length=1, description="Task display name") + action: dict[str, Any] = Field(..., description="{kind, params} (create_file, append_to_file, notify)") + schedule: dict[str, Any] = Field(..., description="{kind, ...} (interval_hours, daily_time, once_at)") + + +class DeleteScheduledTaskInput(BaseModel): + """Delete a scheduled task by id (#170).""" + + task_id: str = Field(..., min_length=1, description="Task id") + + +class RunScheduledTaskInput(BaseModel): + """Execute a scheduled task immediately (#170).""" + + task_id: str = Field(..., min_length=1, description="Task id") + + class ToolResult(BaseModel): """Uniform result returned by :func:`backend.tools.registry.call_tool`.""" diff --git a/desktop/Cargo.lock b/desktop/Cargo.lock index 84db5ce..8f40a3c 100644 --- a/desktop/Cargo.lock +++ b/desktop/Cargo.lock @@ -2626,7 +2626,7 @@ dependencies = [ [[package]] name = "obsigate-desktop" -version = "2.51.0" +version = "2.52.0" dependencies = [ "chrono", "env_logger", diff --git a/desktop/Cargo.toml b/desktop/Cargo.toml index 240575d..904a042 100644 --- a/desktop/Cargo.toml +++ b/desktop/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "obsigate-desktop" -version = "2.51.0" +version = "2.52.0" description = "ObsiGate Desktop — Porte d'entrée native pour vos vaults Obsidian" authors = ["Bruno Charest"] edition = "2021" diff --git a/desktop/tauri.conf.json b/desktop/tauri.conf.json index dee0329..b555347 100644 --- a/desktop/tauri.conf.json +++ b/desktop/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://raw.githubusercontent.com/nicedoc/obsigate/main/desktop/tauri.conf.schema.json", "productName": "ObsiGate", - "version": "2.51.0", + "version": "2.52.0", "identifier": "com.obsigate.desktop", "build": { "frontendDist": "../frontend", diff --git a/docs/GUIDES/API_REST.md b/docs/GUIDES/API_REST.md index 019cad4..758606b 100644 --- a/docs/GUIDES/API_REST.md +++ b/docs/GUIDES/API_REST.md @@ -235,6 +235,9 @@ Gestion : | `/api/conflicts` · `/api/conflicts/resolve` | Conflits Syncthing | GET/POST | | `/api/plugins` | Installer / activer / désactiver | GET/POST/DELETE | | `/api/push/*` | Abonnement Web Push (VAPID) | GET/POST/DELETE | +| `/api/duplicates` · `/api/duplicates/merge` | Doublons : paires candidates, fusion (`confirm: true`) | GET/POST | +| `/api/notify/channels` · `/api/notify/test` | Notifications Discord/Telegram/SMTP/webhook (CRUD admin + test) | GET/POST/PATCH/DELETE | +| `/api/scheduler/tasks` · `/api/scheduler/tasks/{id}/run` | Tâches planifiées (CRUD + exécution manuelle) | GET/POST/PATCH/DELETE | --- diff --git a/docs/GUIDES/ASSISTANT_IA_FORGE.md b/docs/GUIDES/ASSISTANT_IA_FORGE.md index 479be6e..80b56ff 100644 --- a/docs/GUIDES/ASSISTANT_IA_FORGE.md +++ b/docs/GUIDES/ASSISTANT_IA_FORGE.md @@ -182,10 +182,23 @@ partagée par l'assistant in-app et le serveur MCP. | Écriture (propose/apply) | `create_file`, `create_directory`, `edit_file`, `append_to_file`, `restore_backup` | | Destructif (propose/apply) | `rename_file`, `rename_directory`, `move_path`, `replace_in_files`, `delete_file`, `delete_directory` | | Web / sources connectées | `web_search`, `fetch_url`, sources Gitea/GitHub… | +| Doublons (#166) | `find_duplicates` (lecture), `merge_duplicate_notes` (destructif) | +| Notifications (#168) | `notify_external` (Discord, Telegram, SMTP, webhook) | +| Tâches planifiées (#170) | `create/list/delete/run_scheduled_task*` | Les mutations suivent un flux **two-step** : `propose_` renvoie un aperçu et un **jeton signé à usage unique**, puis `apply_` exécute. +Exemples de demandes à l'assistant (mode agent) : + +- « Trouve les notes en double dans ce vault » → `find_duplicates`, puis + « fusionne `brouillon.md` dans `rapport.md` » → `merge_duplicate_notes` + (backup automatique, approbation requise). +- « Préviens-moi sur Discord quand la tâche échoue » → `notify_external` + (canaux configurés via `POST /api/notify/channels`, déclencheurs au choix). +- « Crée une note de veille chaque matin à 8h » → `create_scheduled_task` + (`daily_time 08:00`, action `create_file` ou `append_to_file`). + --- ## 7. Sécurité diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 63d630c..b3e6c1c 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -1,6 +1,6 @@ # ObsiGate — Roadmap -> **Version :** 2.51.0 | **Dernière mise à jour :** 2026-10-03 +> **Version :** 2.52.0 | **Dernière mise à jour :** 2026-10-04 > **Ce fichier ne contient que le travail à venir** (🔵 En cours + ⚪ Backlog) et un index compact > vers les fonctionnalités livrées. > - **Méthode de livraison à appliquer pour toute tâche : [DELIVERY_WORKFLOW.md](./DELIVERY_WORKFLOW.md)** @@ -246,6 +246,12 @@ - **Effort :** 6-8 jours | **Impact :** 🟢 - **Décision 2026-09-26 : reporté (P4)** — axe prioritaire = dette & sécurité (#85/#87) ; #73 hors chemin critique. Si réactivé : partir d'un MVP export/hash/LWW adossé à #59 (PWA offline) + #62 (collab Yjs/CRDT) plutôt qu'un protocole parallèle. +- **Périmètre élargi le 2026-10-04 (liste v1.2 Ph.3) :** la **synchronisation des favoris** + (répertoires compris, cf. #161) et la **synchronisation des préférences utilisateur** (thèmes + #65/#174, langue, historique de recherche et de modifications, recherches sauvegardées #162) sont + **fusionnées dans cet item** plutôt que portées par des items séparés — même API de sync, même + résolution de conflits. Store local actuel : JSON par utilisateur dans `data/` (pas de table + `user_preferences` distincte ; un addendum de schéma suffirait si #73 est réactivé). - **Description :** Synchronisation des vaults entre plusieurs instances d'ObsiGate via un protocole de synchronisation décentralisé ou compatible Obsidian Sync. Alternative self-hosted à Obsidian Sync. - **Sous-tâches :** - [ ] Protocole : évaluation CRDT vs OT vs diff/patch pour fichiers markdown @@ -256,6 +262,9 @@ - [ ] Conflits : UI de résolution manuelle (diff côte à côte entre version locale et distante) - [ ] Chiffrement : optionnel, chiffrement AES-256-GCM avant transmission - [ ] Pairing : échange de clé publique + code QR pour appairage des appareils + - [ ] Périmètre v1.2 : sync des **favoris** (#161) et des **préférences** (thème, langue, + historique, recherches sauvegardées) avec résolution de conflits (priorité à la dernière + modification) + indicateur « préférences synchronisées » --- @@ -279,6 +288,130 @@ --- +## ⚪ Backlog — Améliorations produit & évolutions agent IA (liste v1.2) — P1/P2 + +> **Ouvert le 2026-10-04** à partir de la liste « Roadmap ObsiGate – Développement des +> Fonctionnalités » v1.2 (phases 1-3 + suggestions agent IA numérotées 161-180 dans la liste +> source). Évaluation croisée avec le code réel (2026-10-04) : **7 propositions sont déjà +> livrées** et n'ouvrent donc **pas** d'ID ; seul le périmètre manquant reçoit un ID ObsiGate +> stable `#161` → `#178`. Les numéros de la liste source ne sont **pas** repris tels quels +> (règle du dépôt : un ID n'est jamais attribué à du déjà-fait ni réutilisé). +> Chaque item doit passer par [DELIVERY_WORKFLOW.md](./DELIVERY_WORKFLOW.md) (DoD) à son ouverture. + +### Table de correspondance (liste v1.2 → ObsiGate) + +| Proposition v1.2 | Verdict 2026-10-04 (vérifié dans le code) | ID ObsiGate | +|---|---|---| +| Ph.1 — Sauvegarde des recherches fréquentes | Sauvegarde **déjà livrée** (`backend/saved_searches.py`, store `data/{user}_saved_searches.json`, bouton « Sauver » #158) ; fréquence & rappels manquants | ⚪ #162 | +| Ph.1 — Liens de répertoires aux favoris | Marque-page existant sur **fichiers** uniquement (dashboard) ; répertoires absents | ⚪ #161 | +| Ph.1 — Recherche dans la page | **Déjà livrée** en vue lecture (`FindInPageManager`, Ctrl+F + navigation + regex, tableur via #153-A13) ; reste l'intégration **Forge** | ⚪ #163 | +| Ph.2 — Auto-complétion intelligente | **Déjà livrée** (`frontend/js/autocomplete.js` : Mermaid, code fence, frontmatter, wikilinks, table — partagé Forge/CodeMirror) | ✅ | +| Ph.2 — Suggestions contextuelles (historique & préférences) | Non livré | ⚪ #164 | +| Ph.3 — Synchronisation favoris + préférences multiplateforme | Fusionné dans le périmètre de **#73** (reporté P4, décision 2026-09-26) | ⚪ #73 | +| 161 — Synchronisation Git automatique des vaults | Non livré (les outils `git_*` #92 ciblent Gitea/GitHub pour l'IA, pas la sync de vault) | ⚪ #165 | +| 162 — Génération de résumés automatiques | **Déjà livrée** (`POST /api/ai/summarize`, actions instantanées #106, skills `backend/skills.py`) | ✅ | +| 163 — Détection & fusion de doublons | Non livré (brique embeddings/TF-IDF #70 disponible) | ⚪ #166 | +| 164 — Traduction automatique | **Déjà livrée** (toolbar IA « Traduire », #106) | ✅ | +| 165 — Analyse de ton/style | **Déjà livrée** (toolbar IA « Ton », #106) | ✅ | +| 166 — APIs tierces (Google Drive, Notion) | Partiel : sources connectées **Gitea/GitHub** livrées (#103, `backend/tools/connected.py`) ; Drive/Notion manquants | ⚪ #167 | +| 167 — Notifications externes (Slack, Discord, Email) | Partiel : webhooks #9 + Push API #67 livrés ; canaux Slack/Discord/SMTP et choix des déclencheurs à câbler | ⚪ #168 | +| 168 — Édition collaborative temps réel | **Déjà livrée** (#62 — Yjs/CRDT, WebSocket, awareness, v2.3.0) | ✅ | +| 169 — Scripts personnalisés sandboxés | **Déjà livrée** (#61 — plugins en sandbox Web Worker, hooks, 9 endpoints, v2.2.0) | ✅ | +| 170 — Planification de tâches automatiques (cron) | Non livré | ⚪ #170 | +| 171 — Audit des accès | Partiel : `backend/audit.py` + `/api/admin/audit` + dashboard admin #71 existent ; historique **par fichier**, filtres avancés et export CSV manquants | ⚪ #171 | +| 172 — Chiffrement/déchiffrement de fichiers | Non livré (AES-GCM interne réservé aux tokens) | ⚪ #172 | +| 173 — Vues personnalisées (Kanban, Calendrier) | Non livré | ⚪ #173 | +| 174 — Optimisation mobile | Base **largement livrée** (#69 éditeur mobile, #83 ruban mobile, #114 config responsive 44 px) ; reste un audit du solde | ⚪ #178 | +| 175 — Chat intégré | Non livré (transport WebSocket/rooms de #62 réutilisable) | ⚪ #169 | +| 176 — Thèmes dynamiques (règles heure/date) | Base livrée (#65 thèmes + import/export) ; règles automatiques manquantes | ⚪ #174 | +| 177 — Export formats propriétaires (OneNote, Evernote) | Base livrée (#66 HTML/MD bundle/ePub + PDF #74) ; formats tiers manquants | ⚪ #175 | +| 178 — Webhooks pour événements | **Déjà livrée** (#9 — `backend/webhooks.py`, événements + secrets + validation d'URL SSRF-safe) | ✅ | +| 179 — Gestion fine des permissions | Non livré (ACLs par vault/dossier racine seulement) | ⚪ #176 | +| 180 — Historique des révisions visuel | Partiel : gestion des backups **avec diff** livrée (#40-46, outil IA `diff_backup`) ; UI côte à côte manquante | ⚪ #177 | + +--- + +### 161. Favoris — répertoires et liens de vault (menu contextuel) + +- **Effort :** 0,5-1 jour | **Impact :** 🟡 | **Ouvert :** 2026-10-04 | **Statut :** ⚪ non commencé +- **Description :** le marque-page actuel ne couvre que les fichiers ; ajouter les **répertoires** + aux favoris (proposition « liens de répertoires », liste v1.2 Ph.1). +- **Sous-tâches :** + - [ ] Étendre le modèle de favoris aux répertoires (vérification d'accès backend via + `_resolve_safe_path()`, permissions) + - [ ] Entrée « Ajouter aux favoris » dans le menu contextuel de l'arbre **et** de la page de + navigation (#158-A5) + - [ ] Section « Favoris » dans la sidebar (répertoires + fichiers) + - [ ] i18n FR/EN, tests (unit + E2E si UI), fiche `docs/features/` + +### 162. Recherches fréquentes — fréquence d'utilisation & rappels + +- **Effort :** 1-2 jours | **Impact :** 🟢 | **Ouvert :** 2026-10-04 | **Statut :** ⚪ non commencé +- **Cadrage :** la **sauvegarde** des recherches existe déjà (`backend/saved_searches.py`, bouton + « Sauver » de #158). Cet item n'ouvre que l'incrément manquant de la liste v1.2 Ph.1. +- **Sous-tâches :** + - [ ] Compteur d'utilisation par recherche sauvegardée (horodatage + fréquence) + - [ ] Section dédiée « Recherches fréquentes » (page de résultats + sidebar) et épinglage + - [ ] Rappel optionnel (« Vous avez souvent cherché X ») via notifications #67, interruptible + en Configuration + - [ ] i18n FR/EN + tests + +### 163. Recherche dans la page — éditeur Forge + +- **Effort :** 0,5-1 jour | **Impact :** 🟡 | **Ouvert :** 2026-10-04 | **Statut :** ⚪ non commencé +- **Cadrage :** `FindInPageManager` (Ctrl+F, surbrillance, navigation ↑/↓, casse / mot entier / + regex) est livré pour la vue lecture ; le tableur a sa propre recherche (#153-A13). Reste à + raccorder la recherche à l'éditeur **Forge** et à l'exposer en commande `/`. +- **Sous-tâches :** + - [ ] Raccorder `FindInPageManager` à Forge (textarea **et** CodeMirror) : surbrillance dans le + contenu édité, navigation avec boucle, conservation du curseur + - [ ] Commande `/recherche-dans-page` (mécanisme de commandes #81) + - [ ] Non-régression des raccourcis existants (Ctrl+F vs recherche globale, Échap) ; i18n ; + tests JSDOM + E2E + +### 164. Suggestions contextuelles dans l'éditeur (historique & préférences) + +- **Effort :** 3-5 jours | **Impact :** 🟢 | **Ouvert :** 2026-10-04 | **Statut :** ⚪ non commencé +- **Cadrage :** l'auto-complétion contextuelle (Mermaid, code, Markdown, wikilinks, frontmatter) + est **déjà livrée** (`frontend/js/autocomplete.js`) — la proposition v1.2 Ph.2 + « Auto-complétion intelligente » est couverte. Cet item n'ouvre que les suggestions apprises + du contexte utilisateur. +- **Sous-tâches :** + - [ ] Mesure locale des snippets les plus utilisés par l'utilisateur (fréquence, sans cloud) + - [ ] Suggestions basées sur l'historique de modification (termes récurrents, patterns du type + « souvent inséré après … ») — option locale, IA légère si disponible + - [ ] Bascule on/off en Configuration ; intégration à la complétion existante + - [ ] i18n FR/EN + tests + +--- + +### ⚪ Évolutions agent IA (liste v1.2 « Phase 4 », IDs ObsiGate #165 → #178) + +| ID | Fonctionnalité | Effort | Impact | Dépendances / socle à étendre | +|---|---|---|---|---| +| 165 | Synchronisation automatique des vaults avec Git (add/commit/push/pull, résolution de conflits, auth SSH/token, option planification X heures) | 4-6 j | 🟡 | subprocess/`gitpython` ; UI dans Configuration (#160) ; distinct des outils `git_*` #92 (IA, Gitea/GitHub) | +| 166 | Détection & fusion de doublons (score de similarité, liste des paires potentielles, outil de fusion) — ✅ **livré le 2026-10-04** ([fiche](./features/agent-phase4-166-168-170.md)) | 3-4 j | 🟢 | embeddings/TF-IDF #70 ; **jamais de fusion sans confirmation explicite** (+ backup préalable) | +| 167 | Connexions tierces — Google Drive & Notion (import/export, OAuth2, respect des quotas) | 5-7 j | 🟡 | page « sources connectées » #103 ; rate-limit existant | +| 168 | Notifications externes — canaux Slack/Discord (webhooks) et Email (SMTP) + choix des déclencheurs — ✅ **livré le 2026-10-04** ([fiche](./features/agent-phase4-166-168-170.md) ; périmètre : Discord, Telegram, SMTP, webhook générique — Slack via webhook générique) | 3-4 j | 🟢 | `backend/webhooks.py` #9 ; Push API #67 | +| 169 | Chat intégré par fichier (panneau latéral, historique, notifications de nouveaux messages) | 3-4 j | 🟢 | transport WebSocket/rooms de la collab #62 | +| 170 | Planification de tâches automatiques (type cron ; actions = outils existants `create_file`/`append_to_file` ; notifications d'échec) — ✅ **livré le 2026-10-04** ([fiche](./features/agent-phase4-166-168-170.md) ; tick asyncio 60 s, API + outils, sans UI dédiée) | 3-4 j | 🟢 | scheduler backend (APScheduler ou asyncio) ; UI de création de tâches | +| 171 | Audit des accès — extension : historique lecture/édition par fichier, filtres (utilisateur/fichier/date), export CSV | 2-3 j | 🟡 | `backend/audit.py` + `/api/admin/audit` (#71) | +| 172 | Chiffrement/déchiffrement de fichiers sensibles (AES-256-GCM, gestion de clés, confirmation) | 3-4 j | 🟡 | **backup automatique préalable** ; clés hors dépôt (jamais de secret committé) | +| 173 | Vues personnalisées Kanban & Calendrier (glisser-déposer, colonnes/date par tags & métadonnées) | 5-6 j | 🟢 | facettes/tags #157-#158 ; store JSON par utilisateur (pattern `data/*.json`) | +| 174 | Thèmes dynamiques (règles heure/date, ex. mode sombre 18h-8h, règles personnalisables) | 2-3 j | 🟢 | thèmes #65 ; préférence persistée (candidate à la sync #73) | +| 175 | Export vers formats propriétaires (OneNote, Evernote) avec conservation des métadonnées (tags, liens, images) | 3-4 j | 🟢 | `export.py` #66 (HTML/MD bundle/ePub) + PDF #74 | +| 176 | Permissions fines par fichier/dossier (utilisateur/groupe + actions, héritage, journal des changements) | 4-5 j | 🟡 | ACLs vaults/dossiers existantes ; passerelle obligatoire par `_resolve_safe_path()` | +| 177 | Historique des révisions visuel (diff côte à côte, surlignage ajouts/suppressions, navigation Précédent/Suivant) | 3-4 j | 🟢 | backups + diff #40-46 (`diff_backup`) ; `difflib` backend | +| 178 | Optimisation mobile — solde (audit cibles tactiles ≥ 44 px, lazy-loading images/médias, raccourcis clavier virtuel) | 4-5 j | 🟡 | base livrée #69/#83/#114 ; lazy-loading déjà partiel (#153-A9) | + +- **Critères d'acceptation transverses (liste v1.2 harmonisée avec la DoD du dépôt) :** test + desktop **et** mobile ; documentation utilisateur FR/EN dans `docs/GUIDES/` (l'appel + « Guide_MCP_ObsiGate.md » de la liste source est ramené à la cartographie documentaire du + dépôt) ; **backup automatique avant toute modification destructive** ; i18n FR/EN systématique ; + passage de l'ID à « en cours » avant codage (AGENTS.md, « Avant de commencer »). + +--- + ## ✅ Complété — index > Détail complet dans [docs/archive/COMPLETED_v1-v2.md](./archive/COMPLETED_v1-v2.md) et @@ -362,6 +495,9 @@ | 85 | Refonte architecturale — découpage du monolithe (14 routers, `main.py` 4 827 → ~750 lignes), stores JSON verrouillés, rate-limit SQLite optionnel | 2.27.2→2.27.13 | [features/archi-refonte-85.md](./features/archi-refonte-85.md) | | 157 | Recherche — facette « Extensions » dans les résultats (3ᵉ filtre avec Vaults et Tags) + panneau repliable | 2.46.0→2.47.0 | [archive](./archive/COMPLETED_v1-v2.md) | | 158 | Navigation — clic répertoire → **onglet de navigation** (sous-répertoires, facettes Vaults/Tags/Extensions, tri Pertinence/Date, Sauver), **onglet Accueil**, menus contextuels + retour Home complet (**BUG-100** → **BUG-102**) | 2.48.0→2.49.0 | [features/navigation-tab-158.md](./features/navigation-tab-158.md) | +| 166 | Assistant IA — détection & fusion de doublons (score déterministe, paires candidates, fusion backup + confirmation) | Unreleased | [features/agent-phase4-166-168-170.md](./features/agent-phase4-166-168-170.md) | +| 168 | Assistant IA — notifications externes Discord / Telegram / SMTP / webhook (déclencheurs, secrets hors config, SSRF-safe) | Unreleased | [features/agent-phase4-166-168-170.md](./features/agent-phase4-166-168-170.md) | +| 170 | Assistant IA — tâches planifiées type cron (create/append/notify, interval/daily/once, tick 60 s, échec notifié) | Unreleased | [features/agent-phase4-166-168-170.md](./features/agent-phase4-166-168-170.md) | --- @@ -371,7 +507,9 @@ |---|---|---| | ✅ Complété | #1 → #59, #61–72, #74–76, #78–86, #88–93, #94–100, #102–115, #117, #92 | ~141 jours réalisés | | 🔵 Finitions | #77 Desktop : 6 tests E2E **manuels** ([protocole](./DESKTOP_E2E_CHECKLIST.md)) — signature Windows non retenue (décision 2026-09-26) | ~0,5-1 jour | -| ⚪ P4 reporté | #73 Sync — **reporté (décision 2026-09-26)**, hors chemin critique | 6-8 jours si réactivé | +| ⚪ P4 reporté | #73 Sync — **reporté (décision 2026-09-26)**, hors chemin critique ; périmètre élargi 2026-10-04 (favoris #161 + préférences) | 6-8 jours si réactivé | +| ⚪ P1/P2 nouvelles (liste v1.2) | #161 favoris répertoires · #162 recherches fréquentes · #163 recherche Forge · #164 suggestions contextuelles | ~5-9 jours | +| ⚪ Évolutions agent IA (#165 → #178) | Git sync, doublons, Drive/Notion, notifications, chat, cron, audit, chiffrement, Kanban, thèmes dynamiques, OneNote/Evernote, permissions, diff visuel, mobile | ~45-63 jours si tout activé (à prioriser par tranches) | | ⚪ P0/P1 prioritaire | #87 CI/CD (BUG-035 → BUG-040 corrigés, #86 livré) | ~3-5 jours | | ✅ Terminé | #153 Visionneuse & édition XLSX — complétude (A1-A17 **toutes livrées**, v2.27.0 → v2.39.0) | 0 jour restant | | ✅ Terminé | #154 Refonte UI/UX tableur (A1-A5 **toutes livrées**, v2.40.0 → v2.43.1) | 0 jour restant | @@ -390,6 +528,14 @@ - **Clôture 2026-09-30 :** #156 **livré (P0 → P3)** — **A8** mise en forme en écriture (bouton **Mise en forme** : gras/italique/souligné, alignements, couleurs, formats de nombre, fusions, volets figés, largeur/hauteur — via `PUT …/xlsx/style`), **A9** décision « pas de moteur de formule, annoncée dans l'UI », **A10** undo/redo unifié conservé au re-rendu, **A11** export de la sélection / Markdown / HTML / impression + recherche sur toutes les feuilles, **A12** concurrence optimiste (`If-Match`, **409** réparable), **A13** cache des métadonnées par `mtime`, **A14** outils IA `.xlsm`/`.csv` + `search_workbook`/`analyze_range`/`edit_xlsx_structure` — détail dans [features/xlsx-editor-completeness.md](./features/xlsx-editor-completeness.md). - **Ajout 2026-09-29 :** #156 **P1 livré** — **A5** presse-papiers de plage (copier/couper/coller un bloc, presse-papiers interne + système, entrées du menu contextuel, remplissage multi-cellules), **A6** clavier complet (`Ctrl+S`/`Ctrl+A`/`Suppr`/`F2`/`Ctrl+Home|End`/`PgUp|PgDn`/`Ctrl+flèches`/`Maj+Entrée`) et **A7** zone Nom éditable + aide à la saisie ; restent P2 (mise en forme, calcul, undo unifié) et P3 (sortie, concurrence optimiste, performances, outils IA). - **Ajout 2026-09-29 :** #156 ouvert — audit de complétude de l'éditeur Excel : **4 défauts recensés** (BUG-096 enregistrement `.csv`, BUG-097 lazy-load `.xlsm`, BUG-098 délimiteur CSV, BUG-099 sonde de perte), **corrigés le jour même (P0 ✅)** avec tests de non-régression ; restent P1-P3 (presse-papiers, clavier, mise en forme, calcul, export, concurrence optimiste) puis presse-papiers de plage, clavier complet, mise en forme en écriture, calcul, undo/redo unifié, export/impression, concurrence optimiste — détail dans [features/xlsx-editor-completeness.md](./features/xlsx-editor-completeness.md). +- **Ajout 2026-10-04 :** intégration de la liste « Roadmap – Développement des Fonctionnalités » + v1.2 (phases 1-3 + suggestions agent IA). Évaluation croisée avec le code : **7 propositions déjà + livrées** (auto-complétion `autocomplete.js`, résumés/traduction/ton #106, édition collaborative + #62, scripts/plugins #61, webhooks #9, recherche dans la page en vue lecture) → pas d'ID ; + **périmètre manquant** ouvert en backlog **#161 → #178** avec **table de correspondance** en tête + de la nouvelle section ; sync favoris/préférences **fusionnée dans #73** (reporté P4). Les numéros + 161-180 de la liste source ne sont pas repris tels quels (règle dépôt : ID jamais attribué au + déjà-fait) ; la correspondance figure item par item. - **Clôture #85 (v2.27.13) :** monolithe découpé (T1→T9), stores verrouillés + rate-limit SQLite (T10), fiche `docs/features/archi-refonte-85.md`. - Les items P3/P4 ne sont pas ordonnés par priorité interne — à raffiner selon les retours utilisateurs. - L'effort inclut le développement + tests unitaires + intégration CI, mais pas la documentation utilisateur. diff --git a/docs/features/agent-phase4-166-168-170.md b/docs/features/agent-phase4-166-168-170.md new file mode 100644 index 0000000..56b898b --- /dev/null +++ b/docs/features/agent-phase4-166-168-170.md @@ -0,0 +1,98 @@ +# #166 · #168 · #170 — Agent IA phase 4 : doublons, notifications externes, tâches planifiées + +> **Statut :** ✅ livré — **Effort :** ~5 jours | **Impacts :** 🟢 +> **Références :** [Roadmap](../ROADMAP.md#--évolutions-agent-ia-liste-v12--phase-4--ids-obsigate-165--178) · [Changelog](../../CHANGELOG.md) +> **Guides :** [Assistant IA & Forge](../GUIDES/ASSISTANT_IA_FORGE.md) · [API REST](../GUIDES/API_REST.md) · [MCP](../GUIDES/MCP.md) + +## 1. Périmètre + +Trois items de la liste v1.2 « Phase 4 » livrés ensemble car ils partagent le +même socle (registre `@tool`, stores JSON verrouillés, notifications) : + +| ID | Fonctionnalité | Entrées | +|---|---|---| +| #166 | Détection & fusion de doublons | `backend/services/duplicates.py`, `backend/tools/duplicates.py`, `backend/routers/duplicates.py` | +| #168 | Notifications externes Discord / Telegram / SMTP / webhook | `backend/notify.py`, `backend/tools/notify.py`, `backend/routers/notify.py` | +| #170 | Tâches planifiées type cron | `backend/scheduler.py`, `backend/tools/scheduled.py`, `backend/routers/scheduler.py` | + +Règle transverse respectée : **tout nouvel outil = `@tool` + libellé +`labels.py` + clés i18n `ai.step.*` FR/EN + tests** (cf. `ai-tools-roadmap.md` §2). + +## 2. #166 — Doublons + +- **Score déterministe stdlib** (`similarity_score`) : Jaccard sur tokens + (frontmatter exclu, accents conservés) à 70 % + similarité du titre/first-line + (`difflib`) à 30 %. Pas de dépendance embeddings — l'index sémantique #70 + reste un raffinement optionnel, pas un prérequis. +- **Scan borné** : 500 fichiers `.md` max, 200 Ko/fichier, pré-filtre Jaccard + avant le score complet, `truncated` exposé quand le plafond est atteint. +- **Fusion jamais sans filet** : backup des 2 fichiers avant écriture, + outil `merge_duplicate_notes` en `DANGEROUS` (carte « Tout approuver »), + route `POST /api/duplicates/merge` exige `{confirm: true}`, stratégies + `append` (défaut, avec marqueur d'origine) / `prefer_target` / `prefer_source`. +- **Outils** : `find_duplicates` (READ), `merge_duplicate_notes` (DANGEROUS). + +## 3. #168 — Notifications externes + +Canaux `discord` (webhook `discord.com`), `telegram` (Bot API + `chat_id`), +`smtp` (stdlib, STARTTLS + login) et `webhook` générique (JSON +`{event, title, message, timestamp, source}`). + +- **Secrets** : jamais dans `notify_channels.json` — store `notify_secrets.json` + (0600) ou `OBSIGATE_NOTIFY_SECRET_` (même motif que #9 / BUG-026) ; + l'API n'expose que `***` + `has_secret`. Le token Telegram peut aussi venir + de `OBSIGATE_TELEGRAM_BOT_TOKEN`. +- **SSRF** : `validate_webhook_url` / `is_safe_target` réutilisés pour les + webhooks génériques et Telegram ; Discord valide son préfixe d'URL. +- **Déclencheurs** : `manual`, `schedule_failure`, `schedule_success`, + `duplicate_found` — choisis par canal. `broadcast()` n'échoue jamais en bloc + (résultat par canal, `last_error` persisté). +- **CRUD admin** (`/api/notify/channels`), test d'envoi authentifié + (`POST /api/notify/test`), outil `notify_external` (WRITE → confirmation). + +## 4. #170 — Tâches planifiées + +Store `data/scheduled_tasks.json` (RLock, écriture atomique tmp+replace). +Actions = outils existants, aucun nouveau chemin d'écriture : + +- `create_file` / `append_to_file` → `backend.services.mutations` ; +- `notify` → `backend.notify.broadcast`. + +Planifications `interval_hours` (≥ 0,25), `daily_time` (`HH:MM`) et `once_at` +(ISO-8601, one-shot désactivé après exécution). `tick()` exécute les tâches +dues, enregistre `last_status`/`last_error`/`run_count`/`next_run_at` et émet +`schedule_failure` via #168 (sauf quand l'action elle-même est `notify` — +anti-récursion). Boucle de fond dans le lifespan de `main.py` (tick 60 s via +`asyncio.to_thread`, désactivable par `OBSIGATE_SCHEDULER=0`). + +Outils : `create_scheduled_task` / `list_scheduled_tasks` / +`delete_scheduled_task` / `run_scheduled_task_now` (WRITE sauf list). +La création vérifie l'accès au vault **et** son existence (404 sinon). + +## 5. API REST (toutes avec `response_model`) + +| Route | Rôle | +|---|---| +| `GET /api/duplicates?vault&threshold&limit&subdir` | paires candidates | +| `POST /api/duplicates/merge` | fusion (`confirm: true` obligatoire) | +| `GET/POST /api/notify/channels`, `PATCH/DELETE /api/notify/channels/{id}` | CRUD admin | +| `POST /api/notify/test` | test broadcast ou canal ciblé | +| `GET/POST /api/scheduler/tasks`, `PATCH/DELETE /api/scheduler/tasks/{id}`, `POST …/run` | CRUD + exécution manuelle | + +## 6. Tests + +`tests/test_duplicates.py` (16), `tests/test_notify_channels.py` (14), +`tests/test_scheduler.py` (18) : services, routes (fixture `client`, +auth désactivée), outils (confirmation `DANGEROUS` vérifiée), stores isolés +en tmp, réseau mocké (jamais d'Internet en CI). Labels couverts par le +garde-fou `test_tool_labels.py` (tout outil IN_APP doit avoir son libellé). + +## 7. Limites assumées (V1) + +- Similarité lexicale (pas d'embeddings) — seuils réglables par l'agent. +- Pas d'UI dédiée (API + agent uniquement) ; les clés i18n des étapes + existent déjà pour la section « N étapes ». +- Scheduler in-process (pas de persistance distribuée, tick 60 s) ; + Redis/APScheduler resteraient l'option multi-workers. +- SMTP sans OAuth2 (login STARTTLS) ; Slack natif non ciblé (webhook + générique compatible `incoming-webhook` utilisable tel quel). diff --git a/frontend/locales/en.json b/frontend/locales/en.json index c0b305e..cdb0a3c 100644 --- a/frontend/locales/en.json +++ b/frontend/locales/en.json @@ -2145,6 +2145,13 @@ "ai.step.docx_create": "Word document proposed: {value}", "ai.step.csv_create": "CSV file proposed: {value}", "ai.step.pdf_create": "PDF document proposed: {value}", + "ai.step.duplicates": "Searched duplicates: {value}", + "ai.step.duplicates_merge": "Merged notes: {value}", + "ai.step.notify": "Notification sent: {value}", + "ai.step.schedule_create": "Scheduled task: {value}", + "ai.step.schedule_list": "Listed scheduled tasks", + "ai.step.schedule_delete": "Deleted task: {value}", + "ai.step.schedule_run": "Ran task: {value}", "bookslm.copied": "Copied to clipboard", "bookslm.error": "AI service error", "bookslm.regenerate": "Regenerate", diff --git a/frontend/locales/fr.json b/frontend/locales/fr.json index c6147d1..d3f2ed8 100644 --- a/frontend/locales/fr.json +++ b/frontend/locales/fr.json @@ -2145,6 +2145,13 @@ "ai.step.docx_create": "Document Word proposé : {value}", "ai.step.csv_create": "Fichier CSV proposé : {value}", "ai.step.pdf_create": "Document PDF proposé : {value}", + "ai.step.duplicates": "Doublons recherchés : {value}", + "ai.step.duplicates_merge": "Notes fusionnées : {value}", + "ai.step.notify": "Notification envoyée : {value}", + "ai.step.schedule_create": "Tâche planifiée : {value}", + "ai.step.schedule_list": "Tâches planifiées consultées", + "ai.step.schedule_delete": "Tâche supprimée : {value}", + "ai.step.schedule_run": "Tâche exécutée : {value}", "bookslm.copied": "Réponse copiée dans le presse-papiers", "bookslm.error": "Erreur du service AI", "bookslm.regenerate": "Régénérer", diff --git a/package.json b/package.json index 5fd41ed..bbe3e8a 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "obsigate", - "version": "2.51.0", + "version": "2.52.0", "description": "**Porte d'entrée web ultra-léger pour vos vaults Obsidian** — Accédez, naviguez et recherchez dans toutes vos notes Obsidian depuis n'importe quel appareil via une interface web moderne et responsive.", "main": "patch.js", "directories": { diff --git a/tests/test_duplicates.py b/tests/test_duplicates.py new file mode 100644 index 0000000..d006d3a --- /dev/null +++ b/tests/test_duplicates.py @@ -0,0 +1,166 @@ +# tests/test_duplicates.py — Duplicate detection & merge (#166) +"""Tests for backend.services.duplicates, its REST routes and AI tools. + +Fusion is destructive: the service takes backups first and the tool/route +require an explicit confirmation (DANGEROUS + ``confirm: true``). +""" + +from pathlib import Path + +import pytest + +from backend.services import duplicates as _duplicates +from backend.services.errors import ServiceError +from backend.tools.api import ToolContext, call_tool +from backend.tools.context import ToolConfirmationRequired + + +def _ctx(**kwargs) -> ToolContext: + user = {"username": "tester", "role": "admin", "vaults": ["*"]} + kwargs.setdefault("audit_enabled", False) + return ToolContext(user=user, **kwargs) + + +class TestSimilarity: + def test_identical_texts_score_one(self): + text = "# Titre\nContenu identique avec plusieurs mots significatifs." + assert _duplicates.similarity_score(text, text) == 1.0 + + def test_different_texts_score_low(self): + a = "# Recette pizza\nFarine, tomate, mozzarella, four à bois." + b = "# Config réseau\nAdresse IP, masque, passerelle, DNS du datacenter." + assert _duplicates.similarity_score(a, b) < 0.5 + + def test_empty_text_scores_zero(self): + assert _duplicates.similarity_score("", "# Titre\ncontenu") == 0.0 + + def test_frontmatter_ignored(self): + a = "---\ntitle: A\ntags: [x]\n---\n# Note\nContenu commun significatif ici." + b = "---\ntitle: B\ntags: [y]\n---\n# Note\nContenu commun significatif ici." + assert _duplicates.similarity_score(a, b) > 0.8 + + +class TestFindPairs: + def test_finds_planted_duplicates(self, client, test_vault_dir): + vault = Path(test_vault_dir) + body = "# Rapport mensuel\nVentes en hausse de vingt pour cent ce trimestre." + (vault / "rapport-a.md").write_text(body, encoding="utf-8") + (vault / "rapport-b.md").write_text(body + "\nLigne complémentaire mineure.", encoding="utf-8") + result = _duplicates.find_duplicate_pairs("TestVault", threshold=0.5, limit=20) + assert result["vault"] == "TestVault" + assert result["files_scanned"] >= 2 + pair_files = {(p["file_a"], p["file_b"]) for p in result["pairs"]} + assert ("rapport-a.md", "rapport-b.md") in pair_files + + def test_threshold_filters(self, client, test_vault_dir): + result = _duplicates.find_duplicate_pairs("TestVault", threshold=1.0, limit=20) + assert result["pairs"] == [] + + def test_invalid_threshold_rejected(self, client): + with pytest.raises(ServiceError): + _duplicates.find_duplicate_pairs("TestVault", threshold=0.1) + + def test_unknown_vault_rejected(self, client): + with pytest.raises(ServiceError): + _duplicates.find_duplicate_pairs("NoSuchVault") + + +class TestMerge: + def _plant(self, test_vault_dir: str, name: str, content: str) -> None: + Path(test_vault_dir, name).write_text(content, encoding="utf-8") + + def test_merge_append(self, client, test_vault_dir): + self._plant(test_vault_dir, "src.md", "# Source\nContenu source unique.") + self._plant(test_vault_dir, "dst.md", "# Cible\nContenu cible unique.") + result = _duplicates.merge_duplicates("TestVault", "src.md", "dst.md", strategy="append") + assert result["strategy"] == "append" + assert result["deleted"] == "src.md" + assert not Path(test_vault_dir, "src.md").exists() + merged = Path(test_vault_dir, "dst.md").read_text(encoding="utf-8") + assert "Contenu cible unique" in merged + assert "Contenu source unique" in merged + + def test_merge_prefer_target(self, client, test_vault_dir): + self._plant(test_vault_dir, "old.md", "# Vieux\nAncien contenu.") + self._plant(test_vault_dir, "new.md", "# Neuf\nNouveau contenu.") + result = _duplicates.merge_duplicates("TestVault", "old.md", "new.md", strategy="prefer_target") + assert result["strategy"] == "prefer_target" + assert Path(test_vault_dir, "new.md").read_text(encoding="utf-8") == "# Neuf\nNouveau contenu." + assert not Path(test_vault_dir, "old.md").exists() + + def test_merge_same_path_rejected(self, client, test_vault_dir): + self._plant(test_vault_dir, "same.md", "# Same\nContenu.") + with pytest.raises(ServiceError): + _duplicates.merge_duplicates("TestVault", "same.md", "same.md") + + def test_merge_missing_source_rejected(self, client, test_vault_dir): + self._plant(test_vault_dir, "exists.md", "# Exists\nContenu.") + with pytest.raises(ServiceError): + _duplicates.merge_duplicates("TestVault", "missing.md", "exists.md") + + +class TestDuplicatesApi: + def test_list_endpoint(self, client, test_vault_dir): + (Path(test_vault_dir) / "dup1.md").write_text("# Doublon\nTexte commun significatif.", encoding="utf-8") + (Path(test_vault_dir) / "dup2.md").write_text("# Doublon\nTexte commun significatif.", encoding="utf-8") + resp = client.get("/api/duplicates", params={"vault": "TestVault", "threshold": 0.5}) + assert resp.status_code == 200, resp.text + assert resp.json()["vault"] == "TestVault" + + def test_merge_requires_confirm(self, client, test_vault_dir): + (Path(test_vault_dir) / "m1.md").write_text("# M1\nContenu.", encoding="utf-8") + (Path(test_vault_dir) / "m2.md").write_text("# M2\nContenu.", encoding="utf-8") + resp = client.post( + "/api/duplicates/merge", + json={"vault": "TestVault", "source_path": "m1.md", "target_path": "m2.md"}, + ) + assert resp.status_code == 400 + + def test_merge_with_confirm(self, client, test_vault_dir): + (Path(test_vault_dir) / "c1.md").write_text("# C1\nContenu un.", encoding="utf-8") + (Path(test_vault_dir) / "c2.md").write_text("# C2\nContenu deux.", encoding="utf-8") + resp = client.post( + "/api/duplicates/merge", + json={ + "vault": "TestVault", + "source_path": "c1.md", + "target_path": "c2.md", + "strategy": "append", + "confirm": True, + }, + ) + assert resp.status_code == 200, resp.text + assert resp.json()["deleted"] == "c1.md" + + +class TestDuplicateTools: + def test_find_tool(self, client, test_vault_dir): + (Path(test_vault_dir) / "t1.md").write_text("# Outil\nRecherche de doublons.", encoding="utf-8") + result = call_tool( + "find_duplicates", + _ctx(), + {"vault": "TestVault", "threshold": 0.5, "limit": 10}, + ) + assert result.ok + assert result.data["vault"] == "TestVault" + + def test_merge_tool_requires_confirmation(self, client, test_vault_dir): + (Path(test_vault_dir) / "s1.md").write_text("# S1\nContenu.", encoding="utf-8") + (Path(test_vault_dir) / "s2.md").write_text("# S2\nContenu.", encoding="utf-8") + with pytest.raises(ToolConfirmationRequired): + call_tool( + "merge_duplicate_notes", + _ctx(), + {"vault": "TestVault", "source_path": "s1.md", "target_path": "s2.md"}, + ) + + def test_merge_tool_confirmed(self, client, test_vault_dir): + (Path(test_vault_dir) / "k1.md").write_text("# K1\nContenu k.", encoding="utf-8") + (Path(test_vault_dir) / "k2.md").write_text("# K2\nContenu l.", encoding="utf-8") + result = call_tool( + "merge_duplicate_notes", + _ctx(confirmed=True), + {"vault": "TestVault", "source_path": "k1.md", "target_path": "k2.md", "strategy": "append"}, + ) + assert result.ok + assert result.data["deleted"] == "k1.md" diff --git a/tests/test_notify_channels.py b/tests/test_notify_channels.py new file mode 100644 index 0000000..9879120 --- /dev/null +++ b/tests/test_notify_channels.py @@ -0,0 +1,202 @@ +# tests/test_notify_channels.py — External notifications (#168) +"""Tests for backend.notify (Discord, Telegram, SMTP, webhook), its REST +routes and the ``notify_external`` AI tool. + +Network calls are mocked: no test ever hits the real Internet (hermetic CI). +""" + +import pytest + +from backend import notify as _notify +from backend.tools.api import ToolContext, call_tool + + +@pytest.fixture +def isolated_store(tmp_path, monkeypatch): + """Redirect the channel + secret stores to a tmp dir.""" + channels = tmp_path / "notify_channels.json" + secrets = tmp_path / "notify_secrets.json" + monkeypatch.setattr(_notify, "CHANNELS_FILE", channels) + monkeypatch.setattr(_notify, "SECRETS_FILE", secrets) + monkeypatch.setenv("OBSIGATE_WEBHOOK_ALLOW_PRIVATE", "true") + return tmp_path + + +def _ctx(**kwargs) -> ToolContext: + user = {"username": "tester", "role": "admin", "vaults": ["*"]} + kwargs.setdefault("audit_enabled", False) + return ToolContext(user=user, **kwargs) + + +class TestValidation: + def test_unknown_type_rejected(self, isolated_store): + with pytest.raises(ValueError): + _notify.create_channel("x", "slack", {}) + + def test_discord_requires_url(self, isolated_store): + with pytest.raises(ValueError): + _notify.create_channel("d", "discord", {"webhook_url": "not-a-url"}) + + def test_telegram_requires_chat(self, isolated_store): + with pytest.raises(ValueError): + _notify.create_channel("t", "telegram", {}) + + def test_smtp_requires_fields(self, isolated_store): + with pytest.raises(ValueError): + _notify.create_channel("m", "smtp", {"host": "smtp.example.com"}) + + def test_bad_trigger_falls_back_to_manual(self, isolated_store): + channel = _notify.create_channel( + "w", + "webhook", + {"url": "https://example.com/hook", "triggers": ["nope"]}, + ) + assert channel["config"]["triggers"] == ["manual"] + + +class TestCrud: + def test_create_masks_secret(self, isolated_store): + channel = _notify.create_channel( + "disc", + "discord", + { + "webhook_url": "https://discord.com/api/webhooks/123/abcdef-secret-token", + "triggers": ["manual", "schedule_failure"], + }, + ) + assert channel["has_secret"] is True + assert channel["config"]["webhook_url"] == "***" + # Le secret vit dans le store dédié, pas dans le fichier public. + raw = isolated_store.joinpath("notify_channels.json").read_text(encoding="utf-8") + assert "abcdef-secret-token" not in raw + + def test_update_and_delete(self, isolated_store): + channel = _notify.create_channel("w", "webhook", {"url": "https://example.com/a"}) + updated = _notify.update_channel(channel["id"], {"enabled": False}) + assert updated is not None and updated["enabled"] is False + assert _notify.delete_channel(channel["id"]) is True + assert _notify.delete_channel(channel["id"]) is False + + def test_update_unknown_returns_none(self, isolated_store): + assert _notify.update_channel("nope", {"enabled": True}) is None + + +class TestDispatch: + def test_broadcast_skips_unsubscribed_trigger(self, isolated_store, monkeypatch): + _notify.create_channel("w", "webhook", {"url": "https://example.com/a", "triggers": ["manual"]}) + calls: list = [] + monkeypatch.setattr(_notify, "send_via_channel", lambda ch, t, m, trig="manual": calls.append(ch["id"])) + results = _notify.broadcast("schedule_failure", "T", "M") + assert results == [] + assert calls == [] + + def test_broadcast_delivers_and_records_failure(self, isolated_store, monkeypatch): + channel = _notify.create_channel( + "w", "webhook", {"url": "https://example.com/a", "triggers": ["manual"]} + ) + monkeypatch.setattr(_notify, "send_via_channel", lambda ch, t, m, trig="manual": None) + results = _notify.broadcast("manual", "T", "M") + assert results == [{"channel_id": channel["id"], "ok": True}] + + def _boom(ch, t, m, trig="manual"): + raise RuntimeError("down") + + monkeypatch.setattr(_notify, "send_via_channel", _boom) + results = _notify.broadcast("manual", "T", "M") + assert results[0]["ok"] is False + assert "down" in results[0]["error"] + + def test_smtp_send_uses_smtplib(self, isolated_store, monkeypatch): + _notify.create_channel( + "mail", + "smtp", + { + "host": "smtp.example.com", + "port": 587, + "from_addr": "a@example.com", + "to_addr": "b@example.com", + "username": "a", + "password": "s3cret-pwd", + "triggers": ["manual"], + }, + ) + sent: dict = {} + + class _FakeSMTP: + def __init__(self, *a, **k): + pass + + def __enter__(self): + return self + + def __exit__(self, *a): + return False + + def starttls(self): + sent["tls"] = True + + def login(self, user, pwd): + sent["login"] = (user, pwd) + + def send_message(self, msg): + sent["subject"] = msg["Subject"] + + monkeypatch.setattr(_notify.smtplib, "SMTP", _FakeSMTP) + results = _notify.broadcast("manual", "Sujet", "Corps") + assert results[0]["ok"] is True + assert sent["tls"] is True + assert sent["login"] == ("a", "s3cret-pwd") + assert "Sujet" in sent["subject"] + + +class TestNotifyTool: + def test_tool_broadcasts(self, isolated_store, monkeypatch): + monkeypatch.setattr(_notify, "broadcast", lambda trig, t, m: [{"channel_id": "c", "ok": True}]) + result = call_tool( + "notify_external", + _ctx(confirmed=True), + {"title": "Hello", "message": "World", "trigger": "manual"}, + ) + assert result.ok + assert result.data["deliveries"] == [{"channel_id": "c", "ok": True}] + + def test_tool_unknown_channel(self, isolated_store): + from backend.tools.context import ToolError + + with pytest.raises(ToolError): + call_tool( + "notify_external", + _ctx(confirmed=True), + {"title": "H", "message": "M", "channel_id": "unknown"}, + ) + + +class TestNotifyApi: + def test_channels_crud(self, client, isolated_store): + created = client.post( + "/api/notify/channels", + json={"name": "w", "type": "webhook", "config": {"url": "https://example.com/a"}}, + ) + assert created.status_code == 200, created.text + channel_id = created.json()["id"] + assert created.json()["has_secret"] is False + + listed = client.get("/api/notify/channels") + assert listed.status_code == 200 + assert any(c["id"] == channel_id for c in listed.json()) + + patched = client.patch(f"/api/notify/channels/{channel_id}", json={"enabled": False}) + assert patched.status_code == 200 + assert patched.json()["enabled"] is False + + deleted = client.delete(f"/api/notify/channels/{channel_id}") + assert deleted.status_code == 200 + + def test_create_rejects_bad_type(self, client, isolated_store): + resp = client.post("/api/notify/channels", json={"name": "x", "type": "slack", "config": {}}) + assert resp.status_code == 400 + + def test_test_endpoint_no_channel(self, client, isolated_store): + resp = client.post("/api/notify/test", json={"title": "T", "message": "M"}) + assert resp.status_code == 200 + assert resp.json()["deliveries"] == [] diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py new file mode 100644 index 0000000..f20e700 --- /dev/null +++ b/tests/test_scheduler.py @@ -0,0 +1,198 @@ +# tests/test_scheduler.py — Scheduled tasks, type cron (#170) +"""Tests for backend.scheduler, its REST routes and AI tools. + +The file store is redirected to tmp; task actions run against the TestVault +fixture vault (create/append reuse the real mutation services). +""" + +from datetime import datetime, timedelta, timezone +from pathlib import Path + +import pytest + +from backend import scheduler as _scheduler +from backend.tools.api import ToolContext, call_tool + + +@pytest.fixture +def isolated_tasks(tmp_path, monkeypatch): + """Redirect the task store to a tmp file.""" + tasks_file = tmp_path / "scheduled_tasks.json" + monkeypatch.setattr(_scheduler, "TASKS_FILE", tasks_file) + monkeypatch.setattr("backend.notify.CHANNELS_FILE", tmp_path / "notify_channels.json") + monkeypatch.setattr("backend.notify.SECRETS_FILE", tmp_path / "notify_secrets.json") + return tasks_file + + +def _ctx(**kwargs) -> ToolContext: + user = {"username": "tester", "role": "admin", "vaults": ["*"]} + kwargs.setdefault("audit_enabled", False) + return ToolContext(user=user, **kwargs) + + +class TestValidation: + def test_unknown_action_rejected(self, isolated_tasks): + with pytest.raises(ValueError): + _scheduler.create_task( + "x", + {"kind": "launch_rocket", "params": {}}, + {"kind": "interval_hours", "hours": 1}, + ) + + def test_interval_too_short_rejected(self, isolated_tasks): + with pytest.raises(ValueError): + _scheduler.create_task( + "x", + {"kind": "notify", "params": {"title": "T", "message": "M"}}, + {"kind": "interval_hours", "hours": 0.1}, + ) + + def test_bad_daily_time_rejected(self, isolated_tasks): + with pytest.raises(ValueError): + _scheduler.create_task( + "x", + {"kind": "notify", "params": {"title": "T", "message": "M"}}, + {"kind": "daily_time", "at": "25:00"}, + ) + + +class TestNextRun: + def test_interval_from_now(self, isolated_tasks): + now = datetime(2026, 10, 4, 12, 0, tzinfo=timezone.utc) + nxt = _scheduler.compute_next_run( + {"schedule": {"kind": "interval_hours", "hours": 2}, "last_run_at": None}, now + ) + assert nxt == now + timedelta(hours=2) + + def test_daily_tomorrow_when_passed(self, isolated_tasks): + now = datetime(2026, 10, 4, 12, 0, tzinfo=timezone.utc) + nxt = _scheduler.compute_next_run({"schedule": {"kind": "daily_time", "at": "08:00"}}, now) + assert (nxt - now).total_seconds() == pytest.approx(20 * 3600) + + def test_daily_today_when_upcoming(self, isolated_tasks): + now = datetime(2026, 10, 4, 7, 0, tzinfo=timezone.utc) + nxt = _scheduler.compute_next_run({"schedule": {"kind": "daily_time", "at": "08:00"}}, now) + assert (nxt - now).total_seconds() == pytest.approx(3600) + + +class TestExecution: + def test_create_and_run_append(self, client, test_vault_dir, isolated_tasks): + (Path(test_vault_dir) / "journal.md").write_text("# Journal\n", encoding="utf-8") + task = _scheduler.create_task( + "append", + {"kind": "append_to_file", "params": {"vault": "TestVault", "path": "journal.md", "content": "Ligne auto."}}, + {"kind": "once_at", "at": "2020-01-01T00:00:00"}, + ) + outcome = _scheduler.run_task(task["id"], manual=True) + assert outcome["ok"] is True + assert "Ligne auto." in Path(test_vault_dir, "journal.md").read_text(encoding="utf-8") + stored = _scheduler.get_task(task["id"]) + assert stored is not None and stored["last_status"] == "ok" + assert stored["run_count"] == 1 + assert stored["enabled"] is False # one-shot consommé + + def test_tick_runs_due_task(self, client, test_vault_dir, isolated_tasks): + task = _scheduler.create_task( + "once", + {"kind": "create_file", "params": {"vault": "TestVault", "path": "auto/tache.md", "content": "auto"}}, + {"kind": "once_at", "at": "2020-01-01T00:00:00"}, + ) + outcomes = _scheduler.tick() + assert any(o.get("task_id") == task["id"] and o.get("ok") for o in outcomes) + assert Path(test_vault_dir, "auto", "tache.md").exists() + + def test_failure_recorded(self, client, isolated_tasks): + task = _scheduler.create_task( + "fail", + {"kind": "append_to_file", "params": {"vault": "TestVault", "path": "missing/nope.md", "content": "x"}}, + {"kind": "once_at", "at": "2020-01-01T00:00:00"}, + ) + outcome = _scheduler.run_task(task["id"], manual=True) + assert outcome["ok"] is False + stored = _scheduler.get_task(task["id"]) + assert stored is not None and stored["last_status"] == "error" + assert stored["last_error"] + + def test_run_unknown_raises(self, isolated_tasks): + with pytest.raises(KeyError): + _scheduler.run_task("nope", manual=True) + + def test_delete(self, isolated_tasks): + task = _scheduler.create_task( + "del", + {"kind": "notify", "params": {"title": "T", "message": "M"}}, + {"kind": "interval_hours", "hours": 24}, + ) + assert _scheduler.delete_task(task["id"]) is True + assert _scheduler.delete_task(task["id"]) is False + + +class TestSchedulerApi: + def test_crud_and_run(self, client, test_vault_dir, isolated_tasks): + created = client.post( + "/api/scheduler/tasks", + json={ + "name": "t", + "action": { + "kind": "create_file", + "params": {"vault": "TestVault", "path": "sched/api.md", "content": "hello"}, + }, + "schedule": {"kind": "interval_hours", "hours": 24}, + }, + ) + assert created.status_code == 200, created.text + task_id = created.json()["id"] + + listed = client.get("/api/scheduler/tasks") + assert listed.status_code == 200 + assert any(t["id"] == task_id for t in listed.json()) + + run = client.post(f"/api/scheduler/tasks/{task_id}/run") + assert run.status_code == 200, run.text + assert run.json()["ok"] is True + assert Path(test_vault_dir, "sched", "api.md").exists() + + deleted = client.delete(f"/api/scheduler/tasks/{task_id}") + assert deleted.status_code == 200 + + def test_create_rejects_unknown_vault(self, client, isolated_tasks): + resp = client.post( + "/api/scheduler/tasks", + json={ + "name": "t", + "action": {"kind": "create_file", "params": {"vault": "NoVault", "path": "x.md"}}, + "schedule": {"kind": "interval_hours", "hours": 24}, + }, + ) + assert resp.status_code in (400, 403, 404) + + def test_run_unknown_404(self, client, isolated_tasks): + assert client.post("/api/scheduler/tasks/nope/run").status_code == 404 + + +class TestSchedulerTools: + def test_create_list_delete_run(self, client, test_vault_dir, isolated_tasks): + created = call_tool( + "create_scheduled_task", + _ctx(confirmed=True), + { + "name": "tool-task", + "action": { + "kind": "create_file", + "params": {"vault": "TestVault", "path": "sched/tool.md", "content": "hi"}, + }, + "schedule": {"kind": "interval_hours", "hours": 24}, + }, + ) + assert created.ok + task_id = created.data["id"] + + listed = call_tool("list_scheduled_tasks", _ctx(), {}) + assert listed.ok + assert any(t["id"] == task_id for t in listed.data) + + ran = call_tool("run_scheduled_task_now", _ctx(confirmed=True), {"task_id": task_id}) + assert ran.ok and ran.data["ok"] is True + + deleted = call_tool("delete_scheduled_task", _ctx(confirmed=True), {"task_id": task_id}) + assert deleted.ok