CI / lint (push) Successful in 1m12s
CI / security (push) Successful in 43s
CI / test (push) Successful in 1m10s
CI / build (push) Successful in 35s
CI / e2e (push) Successful in 11m47s
Desktop Build / build-windows (push) Canceled after 0s
Desktop Build / build-linux (push) Canceled after 0s
Contexte: - Repertoire (mode historique BooksLM, via menu contextuel dossier) - Documents (bouton flottant quand des documents sont ouverts) - General (bouton flottant sans document: aide app + creation de fichiers) - Le contexte est affiche dans le header; suggestions adaptees au mode - Blocs obsigate-action rendus en carte avec bouton Appliquer (pas d'execution auto) Backend: - /api/ai/bookslm/context et /chat acceptent mode + context_files - collect_files_context(), empty_context(), build_general_system_prompt() - mode documents degrade vers general si aucun fichier lisible Corrige: - bouton flottant passait le chemin d'un fichier comme repertoire -> erreur 'Aucun fichier markdown trouve dans ce dossier' - panneau precedent masque (localStorage) ne se rouvrait plus - selecteur fournisseur/modele deplace sur une ligne dediee (header actions visibles) - Entree envoie, Ctrl+Entree insere un saut de ligne Tests: test_bookslm.py (+13) et tests/frontend/ai.test.mjs (+8)
409 lines
14 KiB
Python
409 lines
14 KiB
Python
"""BooksLM — Context collection and caching for directory-scoped AI chat.
|
|
|
|
Collects markdown files from an Obsidian vault directory, applies secret
|
|
redaction, builds a system prompt with file contents, and caches results
|
|
for repeated queries.
|
|
"""
|
|
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import os
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from backend.secret_redactor import redact_file_content
|
|
|
|
logger = logging.getLogger("obsigate.bookslm")
|
|
|
|
# ── Configuration limits ──
|
|
BOOKSLM_MAX_FILES = int(os.getenv("BOOKSLM_MAX_FILES", "200"))
|
|
BOOKSLM_MAX_TOTAL_CHARS = int(os.getenv("BOOKSLM_MAX_TOTAL_CHARS", "200000"))
|
|
BOOKSLM_MAX_FILE_CHARS = int(os.getenv("BOOKSLM_MAX_FILE_CHARS", "30000"))
|
|
|
|
# ── Cache ──
|
|
_cache: dict[str, dict[str, Any]] = {}
|
|
_CACHE_TTL = 300 # seconds
|
|
|
|
|
|
def _cache_key(vault_path: Path, directory: str, file_mtimes: list[tuple[str, float]]) -> str:
|
|
"""Build a SHA-256 cache key from vault+directory+file modification times."""
|
|
raw = json.dumps({
|
|
"vault": str(vault_path),
|
|
"dir": directory,
|
|
"mtimes": sorted(file_mtimes),
|
|
}, sort_keys=True)
|
|
return hashlib.sha256(raw.encode()).hexdigest()
|
|
|
|
|
|
def _should_skip(name: str) -> bool:
|
|
"""Return True if this file/directory name should be skipped."""
|
|
skip_prefixes = (".",)
|
|
skip_names = {"_attachments", "node_modules", ".git", ".obsidian", "__pycache__"}
|
|
if name in skip_names:
|
|
return True
|
|
return bool(any(name.startswith(p) for p in skip_prefixes))
|
|
|
|
|
|
def _file_priority(path: Path) -> tuple[int, float]:
|
|
"""Sort key: README/index first, then by modification time descending.
|
|
|
|
Returns (priority_group, -mtime) so that:
|
|
- Group 0: README* and index* files (come first)
|
|
- Group 1: all other files (come after)
|
|
Within each group, newer files come first.
|
|
"""
|
|
name_lower = path.stem.lower()
|
|
if name_lower.startswith("readme") or name_lower.startswith("index"):
|
|
group = 0
|
|
else:
|
|
group = 1
|
|
try:
|
|
mtime = path.stat().st_mtime
|
|
except OSError:
|
|
mtime = 0.0
|
|
return (group, -mtime)
|
|
|
|
|
|
def collect_directory_context(vault_path: Path, directory: str) -> dict[str, Any]:
|
|
"""Walk a directory recursively, collect .md files with content.
|
|
|
|
Args:
|
|
vault_path: Absolute path to the vault root.
|
|
directory: Relative directory path within the vault (empty = root).
|
|
|
|
Returns:
|
|
Dict with keys: files, total_chars, file_count, directory_tree.
|
|
"""
|
|
target_dir = (vault_path / directory).resolve() if directory else vault_path.resolve()
|
|
vault_resolved = vault_path.resolve()
|
|
|
|
# Safety: ensure target is within vault
|
|
try:
|
|
target_dir.relative_to(vault_resolved)
|
|
except ValueError:
|
|
logger.warning(f"Directory outside vault: {target_dir}")
|
|
return {"files": [], "total_chars": 0, "file_count": 0, "directory_tree": ""}
|
|
|
|
if not target_dir.exists() or not target_dir.is_dir():
|
|
return {"files": [], "total_chars": 0, "file_count": 0, "directory_tree": ""}
|
|
|
|
# Check cache
|
|
file_mtimes: list[tuple[str, float]] = []
|
|
md_files: list[Path] = []
|
|
try:
|
|
for p in target_dir.rglob("*"):
|
|
# Skip hidden dirs/files and special dirs
|
|
parts = p.relative_to(target_dir).parts
|
|
if any(_should_skip(part) for part in parts):
|
|
continue
|
|
if p.is_file() and p.suffix.lower() == ".md":
|
|
md_files.append(p)
|
|
try:
|
|
file_mtimes.append((str(p.relative_to(target_dir)), p.stat().st_mtime))
|
|
except OSError:
|
|
file_mtimes.append((str(p.relative_to(target_dir)), 0.0))
|
|
except PermissionError:
|
|
logger.warning(f"Permission denied scanning {target_dir}")
|
|
return {"files": [], "total_chars": 0, "file_count": 0, "directory_tree": ""}
|
|
|
|
key = _cache_key(vault_resolved, directory, file_mtimes)
|
|
if key in _cache:
|
|
cached = _cache[key]
|
|
if time.time() - cached.get("_ts", 0) < _CACHE_TTL:
|
|
logger.debug(f"Cache hit for {directory}")
|
|
return {k: v for k, v in cached.items() if k != "_ts"}
|
|
|
|
# Sort by priority: README/index first, then by mtime descending
|
|
md_files.sort(key=_file_priority)
|
|
|
|
# Apply limits
|
|
collected: list[dict[str, Any]] = []
|
|
total_chars = 0
|
|
|
|
for p in md_files:
|
|
if len(collected) >= BOOKSLM_MAX_FILES:
|
|
break
|
|
if total_chars >= BOOKSLM_MAX_TOTAL_CHARS:
|
|
break
|
|
|
|
rel_path = str(p.relative_to(vault_resolved)).replace("\\", "/")
|
|
try:
|
|
content = p.read_text(encoding="utf-8", errors="replace")
|
|
except Exception as e:
|
|
logger.warning(f"Cannot read {rel_path}: {e}")
|
|
continue
|
|
|
|
# Redact secrets
|
|
content = redact_file_content(content, rel_path)
|
|
|
|
# Truncate if too long
|
|
if len(content) > BOOKSLM_MAX_FILE_CHARS:
|
|
content = content[:BOOKSLM_MAX_FILE_CHARS] + "\n\n[... tronqué]"
|
|
|
|
remaining = BOOKSLM_MAX_TOTAL_CHARS - total_chars
|
|
if len(content) > remaining:
|
|
content = content[:remaining] + "\n\n[... tronqué]"
|
|
|
|
title = p.stem.replace("-", " ").replace("_", " ").title()
|
|
file_type = "markdown"
|
|
|
|
collected.append({
|
|
"path": rel_path,
|
|
"title": title,
|
|
"content": content,
|
|
"type": file_type,
|
|
})
|
|
total_chars += len(content)
|
|
|
|
# Build directory tree
|
|
dir_tree = _build_directory_tree(target_dir, vault_resolved)
|
|
|
|
result = {
|
|
"files": collected,
|
|
"total_chars": total_chars,
|
|
"file_count": len(collected),
|
|
"directory_tree": dir_tree,
|
|
"max_total_chars": BOOKSLM_MAX_TOTAL_CHARS,
|
|
"max_files": BOOKSLM_MAX_FILES,
|
|
"scope": "directory",
|
|
}
|
|
|
|
# Store in cache
|
|
_cache[key] = {**result, "_ts": time.time()}
|
|
logger.info(f"Collected {len(collected)} files ({total_chars} chars) from {directory or '/'}")
|
|
return result
|
|
|
|
|
|
def _file_entry(target: Path, rel_path: str, remaining: int) -> dict[str, Any] | None:
|
|
"""Read, redact and truncate a single file into a context entry."""
|
|
suffix = target.suffix.lower()
|
|
try:
|
|
if suffix == ".pdf":
|
|
from backend.pdf_reader import extract_pdf_text
|
|
|
|
content = extract_pdf_text(target)
|
|
file_type = "pdf"
|
|
else:
|
|
content = target.read_text(encoding="utf-8", errors="replace")
|
|
file_type = "markdown"
|
|
except Exception as e:
|
|
logger.warning(f"Cannot read {rel_path}: {e}")
|
|
return None
|
|
|
|
content = redact_file_content(content, rel_path)
|
|
if len(content) > BOOKSLM_MAX_FILE_CHARS:
|
|
content = content[:BOOKSLM_MAX_FILE_CHARS] + "\n\n[... tronqué]"
|
|
if len(content) > remaining:
|
|
content = content[:remaining] + "\n\n[... tronqué]"
|
|
|
|
return {
|
|
"path": rel_path,
|
|
"title": target.stem.replace("-", " ").replace("_", " ").title(),
|
|
"content": content,
|
|
"type": file_type,
|
|
}
|
|
|
|
|
|
def collect_files_context(
|
|
vault_path: Path,
|
|
rel_paths: list[str],
|
|
scope: str = "documents",
|
|
) -> dict[str, Any]:
|
|
"""Collect an explicit list of files (open documents) as AI context.
|
|
|
|
Unlike :func:`collect_directory_context`, this reads only the requested
|
|
relative paths (markdown or PDF), never the whole directory. Paths outside
|
|
the vault are silently ignored.
|
|
|
|
Args:
|
|
vault_path: Absolute path to the vault root.
|
|
rel_paths: Relative file paths within the vault (in display order).
|
|
scope: Context label exposed to the UI ("documents").
|
|
|
|
Returns:
|
|
Same shape as :func:`collect_directory_context`.
|
|
"""
|
|
vault_resolved = vault_path.resolve()
|
|
collected: list[dict[str, Any]] = []
|
|
total_chars = 0
|
|
seen: set[str] = set()
|
|
|
|
for rel in rel_paths:
|
|
if len(collected) >= BOOKSLM_MAX_FILES or total_chars >= BOOKSLM_MAX_TOTAL_CHARS:
|
|
break
|
|
if not rel or rel in seen:
|
|
continue
|
|
seen.add(rel)
|
|
try:
|
|
target = (vault_resolved / rel).resolve()
|
|
target.relative_to(vault_resolved)
|
|
except (ValueError, OSError):
|
|
continue
|
|
if not target.is_file():
|
|
continue
|
|
entry = _file_entry(target, rel, BOOKSLM_MAX_TOTAL_CHARS - total_chars)
|
|
if entry is None:
|
|
continue
|
|
collected.append(entry)
|
|
total_chars += len(entry["content"])
|
|
|
|
return {
|
|
"files": collected,
|
|
"total_chars": total_chars,
|
|
"file_count": len(collected),
|
|
"directory_tree": "",
|
|
"max_total_chars": BOOKSLM_MAX_TOTAL_CHARS,
|
|
"max_files": BOOKSLM_MAX_FILES,
|
|
"scope": scope,
|
|
}
|
|
|
|
|
|
def empty_context(scope: str = "general") -> dict[str, Any]:
|
|
"""Return an empty context payload (used by the General assistant)."""
|
|
return {
|
|
"files": [],
|
|
"total_chars": 0,
|
|
"file_count": 0,
|
|
"directory_tree": "",
|
|
"max_total_chars": BOOKSLM_MAX_TOTAL_CHARS,
|
|
"max_files": BOOKSLM_MAX_FILES,
|
|
"scope": scope,
|
|
}
|
|
|
|
|
|
def _build_directory_tree(target_dir: Path, vault_root: Path) -> str:
|
|
"""Build a text representation of the directory tree (dirs + .md files)."""
|
|
lines: list[str] = []
|
|
try:
|
|
for p in sorted(target_dir.rglob("*")):
|
|
parts = p.relative_to(target_dir).parts
|
|
if any(_should_skip(part) for part in parts):
|
|
continue
|
|
if p.is_dir():
|
|
depth = len(p.relative_to(target_dir).parts)
|
|
lines.append(f"{' ' * depth}{p.name}/")
|
|
elif p.is_file() and p.suffix.lower() == ".md":
|
|
depth = len(p.relative_to(target_dir).parts)
|
|
lines.append(f"{' ' * depth}{p.name}")
|
|
except PermissionError:
|
|
pass
|
|
return "\n".join(lines)
|
|
|
|
|
|
def build_system_prompt(context: dict[str, Any], scope: str = "directory") -> str:
|
|
"""Build a system prompt for document-scoped AI chat.
|
|
|
|
Args:
|
|
context: Output of collect_directory_context()/collect_files_context().
|
|
scope: "directory" (whole folder) or "documents" (open files).
|
|
|
|
Returns:
|
|
System prompt string with file contents.
|
|
"""
|
|
files = context.get("files", [])
|
|
file_count = context.get("file_count", 0)
|
|
total_chars = context.get("total_chars", 0)
|
|
|
|
# Rough token estimate (1 token ≈ 4 chars)
|
|
est_tokens = total_chars // 4
|
|
token_warning = ""
|
|
if est_tokens > 100_000:
|
|
token_warning = f"\n⚠️ Attention : le contexte est très volumineux (~{est_tokens:,} tokens estimés). Les réponses peuvent être moins précises.\n"
|
|
|
|
if scope == "documents":
|
|
role = (
|
|
"Tu es un assistant de recherche documentaire intégré à ObsiGate. "
|
|
"Tu réponds UNIQUEMENT en te basant sur les documents ouverts par l'utilisateur ci-dessous. "
|
|
"Cite tes sources avec le nom du fichier quand tu utilises une information. "
|
|
"Si l'information ne se trouve pas dans les documents, dis-le clairement."
|
|
)
|
|
context_label = "📄 Documents ouverts"
|
|
else:
|
|
role = (
|
|
"Tu es un assistant de recherche documentaire. "
|
|
"Tu réponds UNIQUEMENT en te basant sur les documents fournis ci-dessous. "
|
|
"Cite tes sources avec le nom du fichier quand tu utilises une information. "
|
|
"Si l'information ne se trouve pas dans les documents, dis-le clairement."
|
|
)
|
|
context_label = "📚 Contexte"
|
|
|
|
prompt = (
|
|
f"{role}"
|
|
f"\n\n{context_label} : {file_count} fichier(s) ({total_chars:,} caractères)"
|
|
f"{token_warning}\n"
|
|
)
|
|
|
|
# Directory tree
|
|
tree = context.get("directory_tree", "")
|
|
if tree:
|
|
prompt += f"\n📂 Arborescence du dossier :\n```\n{tree}\n```\n"
|
|
|
|
# File contents
|
|
prompt += "\n---\n"
|
|
for f in files:
|
|
prompt += f"\n## 📄 {f['title']} (`{f['path']}`)\n\n{f['content']}\n\n---\n"
|
|
|
|
prompt += "\nFin du contexte. Réponds à la question de l'utilisateur en te basant uniquement sur ces documents."
|
|
|
|
return prompt
|
|
|
|
|
|
GENERAL_SYSTEM_PROMPT = """Tu es l'assistant intégré d'ObsiGate, une application web auto-hébergée pour consulter, rechercher et éditer des vaults Obsidian (Markdown).
|
|
|
|
Tes deux rôles :
|
|
1. **Aider sur l'application** : expliquer la navigation, la recherche (full-text, filtres `tag:`, `created:`, `path:`), l'éditeur (CodeMirror, autosave, raccourcis), les onglets et le split view, les sauvegardes et la restauration, le partage public, l'export (HTML/Markdown/ePub/PDF), Mermaid, Excalidraw, les plugins, les thèmes, le mode hors-ligne, le MFA, etc.
|
|
2. **Proposer des actions concrètes** : créer un fichier ou un dossier dans un vault.
|
|
|
|
Quand l'utilisateur demande explicitement de créer un fichier, inclus EXACTEMENT un bloc de ce type dans ta réponse (et rien d'autre à l'intérieur du bloc) :
|
|
|
|
```obsigate-action
|
|
{"action": "create_file", "vault": "<nom du vault>", "path": "<chemin/relatif.md>", "content": "<contenu markdown>"}
|
|
```
|
|
|
|
Pour créer un dossier :
|
|
|
|
```obsigate-action
|
|
{"action": "create_directory", "vault": "<nom du vault>", "path": "<chemin/relatif>"}
|
|
```
|
|
|
|
Règles :
|
|
- Ne propose une action que si l'utilisateur la demande explicitement.
|
|
- Explique en une phrase ce que fait l'action avant le bloc.
|
|
- Utilise un chemin relatif se terminant par `.md` pour un fichier.
|
|
- N'invente jamais un nom de vault : utilise l'un des vaults disponibles listés ci-dessous.
|
|
- Réponds dans la langue de l'utilisateur, de façon concise et structurée (Markdown).
|
|
"""
|
|
|
|
|
|
def build_general_system_prompt(vaults: list[str] | None = None) -> str:
|
|
"""System prompt for the General assistant (app help + actions)."""
|
|
prompt = GENERAL_SYSTEM_PROMPT
|
|
if vaults:
|
|
prompt += "\nVaults disponibles : " + ", ".join(sorted(vaults)) + "\n"
|
|
else:
|
|
prompt += "\nAucun vault n'est actuellement configuré.\n"
|
|
return prompt
|
|
|
|
|
|
def invalidate_cache(vault_path: Path | None = None, directory: str | None = None) -> int:
|
|
"""Invalidate cache entries.
|
|
|
|
Args:
|
|
vault_path: If provided, only invalidate entries for this vault.
|
|
directory: If provided, only invalidate entries for this directory.
|
|
|
|
Returns:
|
|
Number of cache entries removed.
|
|
"""
|
|
if vault_path is None and directory is None:
|
|
count = len(_cache)
|
|
_cache.clear()
|
|
return count
|
|
|
|
to_remove = list(_cache)
|
|
for k in to_remove:
|
|
del _cache[k]
|
|
return len(to_remove)
|