- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
86 lines
2.7 KiB
Python
86 lines
2.7 KiB
Python
"""FlowDeck — forge repository file importer (v5.6.0, Phase 5).
|
|
|
|
Imports a Gitea/GitHub repository's text files as pages, preserving the folder
|
|
hierarchy. Markdown files become pages; other text files become code blocks.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
from app.services.export import _CODE_LANG, _TEXTUAL_EXTS
|
|
from app.services.importers.base import ImportPage, ImportResult
|
|
|
|
_MD_EXTS = {"md", "markdown"}
|
|
|
|
|
|
def _ext(path: str) -> str:
|
|
return Path(path).suffix.lower().lstrip(".")
|
|
|
|
|
|
def build_repo_result(
|
|
files: list[tuple[str, str]],
|
|
*,
|
|
owner: str,
|
|
repo: str,
|
|
provider: str = "",
|
|
) -> ImportResult:
|
|
"""Turn ``[(path, content)]`` into pages with folder hierarchy."""
|
|
result = ImportResult(source=f"forge-repo:{provider}" if provider else "forge-repo")
|
|
for path, content in files:
|
|
clean = path.replace("\\", "/").strip("/")
|
|
if not clean:
|
|
continue
|
|
ext = _ext(clean)
|
|
if ext in _MD_EXTS:
|
|
markdown = content
|
|
else:
|
|
lang = _CODE_LANG.get(ext, "")
|
|
markdown = f"```{lang}\n{content.rstrip()}\n```"
|
|
parts = clean.split("/")
|
|
result.pages.append(ImportPage(
|
|
title=parts[-1] or clean,
|
|
markdown=markdown,
|
|
source_path=clean,
|
|
parent_path="/".join(parts[:-1]),
|
|
external_id=f"{provider}:{owner}/{repo}:{clean}",
|
|
))
|
|
result.stats["rows"] = len(result.pages)
|
|
return result.finalize()
|
|
|
|
|
|
async def fetch_forge_repo(
|
|
adapter,
|
|
owner: str,
|
|
repo: str,
|
|
*,
|
|
provider: str = "",
|
|
path: str = "",
|
|
max_files: int = 200,
|
|
max_file_bytes: int = 512_000,
|
|
) -> ImportResult:
|
|
"""List a repo's files and fetch the textual ones."""
|
|
try:
|
|
metas = await adapter.list_repo_files(owner, repo, path)
|
|
except Exception as exc: # noqa: BLE001
|
|
result = ImportResult(source=f"forge-repo:{provider}" if provider else "forge-repo")
|
|
result.warn(f"Arborescence illisible : {exc}")
|
|
return result.finalize()
|
|
|
|
files: list[tuple[str, str]] = []
|
|
for meta in metas:
|
|
file_path = meta.get("path") or ""
|
|
if _ext(file_path) not in _TEXTUAL_EXTS:
|
|
continue
|
|
if int(meta.get("size") or 0) > max_file_bytes:
|
|
continue
|
|
if len(files) >= max_files:
|
|
break
|
|
try:
|
|
content = await adapter.get_file_content(owner, repo, file_path)
|
|
except Exception: # noqa: BLE001 - skip unreadable files
|
|
continue
|
|
if not content or content == "[binary file]":
|
|
continue
|
|
files.append((file_path, content))
|
|
return build_repo_result(files, owner=owner, repo=repo, provider=provider)
|