Files
flowdeck/app/services/importers/forge_repo.py
T
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

86 lines
2.7 KiB
Python

"""FlowDeck — forge repository file importer (v5.6.0, Phase 5).
Imports a Gitea/GitHub repository's text files as pages, preserving the folder
hierarchy. Markdown files become pages; other text files become code blocks.
"""
from __future__ import annotations
from pathlib import Path
from app.services.export import _CODE_LANG, _TEXTUAL_EXTS
from app.services.importers.base import ImportPage, ImportResult
_MD_EXTS = {"md", "markdown"}
def _ext(path: str) -> str:
return Path(path).suffix.lower().lstrip(".")
def build_repo_result(
files: list[tuple[str, str]],
*,
owner: str,
repo: str,
provider: str = "",
) -> ImportResult:
"""Turn ``[(path, content)]`` into pages with folder hierarchy."""
result = ImportResult(source=f"forge-repo:{provider}" if provider else "forge-repo")
for path, content in files:
clean = path.replace("\\", "/").strip("/")
if not clean:
continue
ext = _ext(clean)
if ext in _MD_EXTS:
markdown = content
else:
lang = _CODE_LANG.get(ext, "")
markdown = f"```{lang}\n{content.rstrip()}\n```"
parts = clean.split("/")
result.pages.append(ImportPage(
title=parts[-1] or clean,
markdown=markdown,
source_path=clean,
parent_path="/".join(parts[:-1]),
external_id=f"{provider}:{owner}/{repo}:{clean}",
))
result.stats["rows"] = len(result.pages)
return result.finalize()
async def fetch_forge_repo(
adapter,
owner: str,
repo: str,
*,
provider: str = "",
path: str = "",
max_files: int = 200,
max_file_bytes: int = 512_000,
) -> ImportResult:
"""List a repo's files and fetch the textual ones."""
try:
metas = await adapter.list_repo_files(owner, repo, path)
except Exception as exc: # noqa: BLE001
result = ImportResult(source=f"forge-repo:{provider}" if provider else "forge-repo")
result.warn(f"Arborescence illisible : {exc}")
return result.finalize()
files: list[tuple[str, str]] = []
for meta in metas:
file_path = meta.get("path") or ""
if _ext(file_path) not in _TEXTUAL_EXTS:
continue
if int(meta.get("size") or 0) > max_file_bytes:
continue
if len(files) >= max_files:
break
try:
content = await adapter.get_file_content(owner, repo, file_path)
except Exception: # noqa: BLE001 - skip unreadable files
continue
if not content or content == "[binary file]":
continue
files.append((file_path, content))
return build_repo_result(files, owner=owner, repo=repo, provider=provider)