feat: roadmap items #7 #8 #9 #10 — recent files, semver tag, search filters, dedup

- #7: Section « Récents » dans la page d'accueil vault (5 derniers fichiers)
- #8: Git tag v1.8.0
- #9: Filtres search avancés created:/modified:/size: avec parsing dates/sizes
- #10: Déduplication IGNORED_DIRS (indexer.py source unique, watcher+main importent)
- ROADMAP: tous les items marqués FAIT, version → 1.9.0
This commit is contained in:
2026-06-04 12:11:51 -04:00
parent fc1a4e336e
commit 0b3a7d5c95
7 changed files with 353 additions and 30 deletions
+11 -4
View File
@@ -36,6 +36,7 @@ from backend.indexer import (
parse_markdown_file,
_extract_tags,
SUPPORTED_EXTENSIONS,
IGNORED_DIRS,
update_single_file,
remove_single_file,
handle_file_move,
@@ -2445,6 +2446,9 @@ async def api_advanced_search(
regex: bool = Query(False, description="Treat query as regex"),
include_paths: Optional[str] = Query(None, description="Comma-separated glob patterns to include"),
exclude_paths: Optional[str] = Query(None, description="Comma-separated glob patterns to exclude"),
created: Optional[str] = Query(None, description="Created date filter (>date, <date, date..date)"),
modified: Optional[str] = Query(None, description="Modified date filter (>date, <date, date..date, <Nd)"),
size: Optional[str] = Query(None, description="Size filter (>size, <size, size..size, e.g. >1MB, <10KB)"),
current_user=Depends(require_auth),
):
"""Advanced full-text search with TF-IDF scoring, facets, and pagination.
@@ -2455,6 +2459,9 @@ async def api_advanced_search(
- ``title:<text>`` — filter by title substring
- ``path:<text>`` — filter by path substring
- ``ext:<type>`` — filter by file extension
- ``created:>2024-01-01`` — filter by creation date
- ``modified:<7d`` or ``modified:2024-01-01..2024-06-01`` — filter by modification date
- ``size:>1MB`` or ``size:100KB..1MB`` — filter by file size
- Remaining text is scored using TF-IDF with accent normalization.
- Toggles: case_sensitive, whole_word, regex
- Path filters: include_paths, exclude_paths (glob patterns)
@@ -2467,7 +2474,8 @@ async def api_advanced_search(
partial(advanced_search, q, vault_filter=vault, tag_filter=tag,
limit=limit, offset=offset, sort_by=sort,
case_sensitive=case_sensitive, whole_word=whole_word, regex=regex,
include_paths=include_paths, exclude_paths=exclude_paths),
include_paths=include_paths, exclude_paths=exclude_paths,
created=created, modified=modified, size=size),
)
@@ -3146,7 +3154,6 @@ async def api_vault_recent_files(
raise HTTPException(status_code=404, detail=f"Directory not found: {dir}")
files = []
ignored = {'.obsidian', '.trash', '.git', '.obsigate-backup', '__pycache__', 'node_modules'}
dir_prefix = (dir or "").strip("/")
try:
@@ -3156,7 +3163,7 @@ async def api_vault_recent_files(
continue
if entry.name.startswith('.'):
continue
if entry.name in ignored:
if entry.name in IGNORED_DIRS:
continue
# Skip entries whose parent dir chain contains an ignored dir (for rglob)
@@ -3164,7 +3171,7 @@ async def api_vault_recent_files(
skip = False
try:
for part in entry.relative_to(dir_path).parts[:-1]:
if part.startswith('.') or part in ignored:
if part.startswith('.') or part in IGNORED_DIRS:
skip = True
break
except ValueError:
+165
View File
@@ -951,6 +951,146 @@ def _passes_path_filters(path: str, include: Optional[str], exclude: Optional[st
return True
# ---------------------------------------------------------------------------
# Date and size filter helpers
# ---------------------------------------------------------------------------
def _parse_date_range(raw: Optional[str]) -> Optional[tuple]:
"""Parse a date range filter string.
Supported formats:
- ">2024-01-01" → after (inclusive)
- "<2024-06-01" → before (inclusive)
- "2024-01-01..2024-06-01" → between (inclusive)
- "<7d" → within last 7 days
- ">30d" → older than 30 days
- "YYYY-MM-DD" → exact date match (created/modified on that day)
Returns (min_ts, max_ts) in Unix timestamps, or None if unparseable.
"""
if not raw:
return None
raw = raw.strip()
# Relative: <7d, >30d
rel_match = re.match(r'^([<>])(\d+)([dh])$', raw)
if rel_match:
op, num, unit = rel_match.groups()
n = int(num)
seconds = n * 3600 if unit == 'h' else n * 86400
now = time.time()
if op == '<':
return (now - seconds, now)
else:
return (0, now - seconds)
# Range: date1..date2
if '..' in raw:
parts = raw.split('..', 1)
t1 = _parse_date_to_ts(parts[0].strip())
t2 = _parse_date_to_ts(parts[1].strip())
if t1 is not None and t2 is not None:
return (min(t1, t2), max(t1, t2))
return None
# Comparator: >date, <date
if raw.startswith('>'):
ts = _parse_date_to_ts(raw[1:].strip())
return (ts, 9999999999) if ts is not None else None
if raw.startswith('<'):
ts = _parse_date_to_ts(raw[1:].strip())
return (0, ts) if ts is not None else None
# Exact date
ts = _parse_date_to_ts(raw)
if ts is not None:
return (ts, ts + 86400) # whole day
return None
def _parse_date_to_ts(s: str) -> Optional[float]:
"""Parse a date string to Unix timestamp. Supports YYYY-MM-DD."""
s = s.strip()
for fmt in ('%Y-%m-%d', '%Y/%m/%d', '%d/%m/%Y'):
try:
from datetime import datetime as dt_mod
return dt_mod.strptime(s, fmt).timestamp()
except ValueError:
continue
return None
def _matches_date_range(file_ts: Optional[float], date_range: tuple) -> bool:
"""Check if a file timestamp falls within the given range."""
if file_ts is None:
return False
return date_range[0] <= file_ts <= date_range[1]
def _parse_size_range(raw: Optional[str]) -> Optional[tuple]:
"""Parse a size range filter string.
Supported formats:
- ">1MB", ">500KB", ">1GB" → larger than
- "<100KB" → smaller than
- "500KB..2MB" → between
Returns (min_bytes, max_bytes), or None if unparseable.
"""
if not raw:
return None
raw = raw.strip()
def _parse_size(s: str) -> Optional[int]:
s = s.strip().upper()
mult = 1
if s.endswith('GB'):
mult = 1024 * 1024 * 1024
s = s[:-2]
elif s.endswith('MB'):
mult = 1024 * 1024
s = s[:-2]
elif s.endswith('KB'):
mult = 1024
s = s[:-2]
elif s.endswith('B'):
mult = 1
s = s[:-1]
elif s.endswith('O'):
mult = 1
s = s[:-1]
try:
return int(float(s) * mult)
except (ValueError, TypeError):
return None
if '..' in raw:
parts = raw.split('..', 1)
lo = _parse_size(parts[0])
hi = _parse_size(parts[1])
if lo is not None and hi is not None:
return (min(lo, hi), max(lo, hi))
return None
if raw.startswith('>'):
lo = _parse_size(raw[1:])
return (lo, 10**15) if lo is not None else None
if raw.startswith('<'):
hi = _parse_size(raw[1:])
return (0, hi) if hi is not None else None
# Exact: try parsing as a number (bytes) or with unit
exact = _parse_size(raw)
if exact is not None:
return (exact, exact * 2) # fuzzy match
return None
def _matches_size_range(file_size: int, size_range: tuple) -> bool:
"""Check if a file size falls within the given range."""
return size_range[0] <= file_size <= size_range[1]
def advanced_search(
query: str,
vault_filter: str = "all",
@@ -963,6 +1103,9 @@ def advanced_search(
regex: bool = False,
include_paths: Optional[str] = None,
exclude_paths: Optional[str] = None,
created: Optional[str] = None,
modified: Optional[str] = None,
size: Optional[str] = None,
) -> Dict[str, Any]:
"""Advanced full-text search with TF-IDF scoring, facets, and pagination.
@@ -1074,6 +1217,28 @@ def advanced_search(
)
}
# Date and size filters (from query operators or API params)
date_range_created = _parse_date_range(created or parsed.get("created"))
if date_range_created:
candidates = {
dk for dk in candidates
if _matches_date_range(inv.doc_info[dk].get("created"), date_range_created)
}
date_range_modified = _parse_date_range(modified or parsed.get("modified"))
if date_range_modified:
candidates = {
dk for dk in candidates
if _matches_date_range(inv.doc_info[dk].get("modified"), date_range_modified)
}
size_range = _parse_size_range(size or parsed.get("size"))
if size_range:
candidates = {
dk for dk in candidates
if _matches_size_range(inv.doc_info[dk].get("size", 0), size_range)
}
# ------------------------------------------------------------------
# Step 3: Score only the candidates (not all N documents)
# ------------------------------------------------------------------
+1 -16
View File
@@ -8,25 +8,10 @@ from typing import Callable, Dict, List, Optional
from watchdog.observers import Observer
from watchdog.observers.polling import PollingObserver
from watchdog.events import FileSystemEventHandler
from backend.indexer import SUPPORTED_EXTENSIONS
from backend.indexer import SUPPORTED_EXTENSIONS, IGNORED_DIRS
logger = logging.getLogger("obsigate.watcher")
# Extensions de fichiers surveillées
# Default ignored directories (can be overridden via OBSIGATE_IGNORED_DIRS env var)
_DEFAULT_IGNORED_DIRS = {'.obsidian', '.trash', '.git', '__pycache__', 'node_modules', '.obsigate-backup'}
def _load_ignored_dirs() -> set:
"""Load ignored directories from environment or use defaults."""
env_val = os.environ.get("OBSIGATE_IGNORED_DIRS", "")
if env_val:
custom = set(d.strip() for d in env_val.split(",") if d.strip())
logger.info(f"Using custom IGNORED_DIRS: {custom}")
return custom
return _DEFAULT_IGNORED_DIRS.copy()
IGNORED_DIRS = _load_ignored_dirs()
class VaultEventHandler(FileSystemEventHandler):
"""Gestionnaire d'événements filesystem pour une vault Obsidian.