feat: CI/CD pipeline + sortedcontainers for O(log n) index ops
CI/CD (.gitea/workflows/ci.yml): - Lint: ruff + mypy on every push/PR - Test: pytest with coverage report (175 tests) - Security: bandit SAST + pip-audit dependency scan - Build: Docker image verification sortedcontainers (backend/search.py): - Replace bisect with SortedList for _sorted_tokens - O(log n) add() / discard() instead of O(n) insort/pop - SortedList.bisect_left() for prefix search - Add sortedcontainers>=2.4.0 to requirements.txt
This commit is contained in:
@@ -8,4 +8,5 @@ aiohttp>=3.9.0
|
||||
watchdog>=4.0.0
|
||||
argon2-cffi>=23.1.0
|
||||
python-jose>=3.3.0
|
||||
sortedcontainers>=2.4.0
|
||||
weasyprint>=60.0
|
||||
|
||||
+6
-8
@@ -1,4 +1,4 @@
|
||||
import bisect
|
||||
from sortedcontainers import SortedList
|
||||
import logging
|
||||
import math
|
||||
import re
|
||||
@@ -325,7 +325,7 @@ class InvertedIndex:
|
||||
self.doc_vault: Dict[str, str] = {}
|
||||
self.vault_docs: Dict[str, set] = defaultdict(set)
|
||||
self.tag_docs: Dict[str, set] = defaultdict(set)
|
||||
self._sorted_tokens: List[str] = []
|
||||
self._sorted_tokens: "SortedList" = SortedList()
|
||||
self._ready: bool = False # True after initial build
|
||||
|
||||
def rebuild(self) -> None:
|
||||
@@ -394,7 +394,7 @@ class InvertedIndex:
|
||||
if tag not in self.tag_prefix_index[prefix]:
|
||||
self.tag_prefix_index[prefix].append(tag)
|
||||
|
||||
self._sorted_tokens = sorted(self.word_index.keys())
|
||||
self._sorted_tokens = SortedList(self.word_index.keys())
|
||||
self._ready = True
|
||||
logger.info(
|
||||
"Inverted index built: %d documents, %d unique tokens, %d tags",
|
||||
@@ -448,7 +448,7 @@ class InvertedIndex:
|
||||
tf[token] += 1
|
||||
for token, freq in tf.items():
|
||||
if not self.word_index.get(token):
|
||||
bisect.insort(self._sorted_tokens, token)
|
||||
self._sorted_tokens.add(token)
|
||||
self.word_index[token][doc_key] = freq
|
||||
|
||||
def remove_document(self, vault_name: str, path: str):
|
||||
@@ -510,9 +510,7 @@ class InvertedIndex:
|
||||
if not wi:
|
||||
del self.word_index[token]
|
||||
if not skip_sorted_cleanup:
|
||||
idx = bisect.bisect_left(self._sorted_tokens, token)
|
||||
if idx < len(self._sorted_tokens) and self._sorted_tokens[idx] == token:
|
||||
self._sorted_tokens.pop(idx)
|
||||
self._sorted_tokens.discard(token)
|
||||
|
||||
def idf(self, term: str) -> float:
|
||||
"""Inverse Document Frequency for a term.
|
||||
@@ -563,7 +561,7 @@ class InvertedIndex:
|
||||
"""
|
||||
if not prefix or not self._sorted_tokens:
|
||||
return []
|
||||
lo = bisect.bisect_left(self._sorted_tokens, prefix)
|
||||
lo = self._sorted_tokens.bisect_left(prefix)
|
||||
results: List[str] = []
|
||||
for i in range(lo, len(self._sorted_tokens)):
|
||||
if self._sorted_tokens[i].startswith(prefix):
|
||||
|
||||
Reference in New Issue
Block a user