Files
bruno 3b00cbc371 feat(import): v5.6.0 unified data import (phases 0-5)
- unified importer framework (app/services/importers/): normalized model,
  registry, common pipeline (hierarchy, attachments, collections, dedup),
  async jobs, dry-run preview, column->type mapping
- Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/
  OneNote), Google Keep, generic Markdown
- Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON
- Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders
- Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics,
  OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones)
- Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume,
  forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue,
  Notion relation resolution, exportable JSON reports
- /import wizard, API /api/import/*, migration 9 (import_items, import_jobs)
- fix: property values stored by property id (correct DB view rendering)
- deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf
- 43 import tests; full suite 491 green; ruff clean
- bump version 5.11.5
2026-09-13 10:43:20 -04:00

87 lines
2.9 KiB
Python

"""FlowDeck — OPML importer (v5.6.0, Phase 4).
Imports an OPML outline (RSS readers, feed lists) as a collection of feeds.
"""
from __future__ import annotations
import xml.etree.ElementTree as ET
from typing import Any
from app.services.importers.base import (
Importer,
ImportResult,
decode_text,
make_collection,
register_importer,
)
_SCHEMA = [
{"name": "Title", "type": "title"},
{"name": "Feed URL", "type": "url"},
{"name": "Site URL", "type": "url"},
{"name": "Type", "type": "select", "options": [
{"name": "rss", "color": "orange"},
{"name": "folder", "color": "gray"},
]},
{"name": "Folder", "type": "text"},
]
@register_importer
class OpmlImporter(Importer):
source_id = "opml"
label = "OPML (flux RSS)"
description = "Outline OPML → collection de flux (titre, URL, dossier)."
extensions = (".opml", ".xml")
order = 34
def detect(self, filename: str, data: bytes) -> bool:
if filename.lower().endswith(".opml"):
return True
head = decode_text(data)[:1000].lower()
return "<opml" in head and "<outline" in head
def parse(self, filename: str, data: bytes) -> ImportResult:
result = ImportResult(source=self.source_id)
rows: list[dict] = []
try:
root = ET.fromstring(decode_text(data))
except ET.ParseError as exc:
result.warn(f"OPML invalide : {exc}")
return result.finalize()
for outline in root.iter("outline"):
attrs = {k.lower(): v for k, v in outline.attrib.items()}
feed = attrs.get("xmlurl")
title = attrs.get("title") or attrs.get("text") or feed or ""
if not feed and not title:
continue
props: dict[str, Any] = {}
if feed:
props["Feed URL"] = feed
props["Type"] = "rss"
else:
props["Type"] = "folder"
if attrs.get("htmlurl"):
props["Site URL"] = attrs["htmlurl"]
folder = _folder_of(outline, root)
if folder:
props["Folder"] = folder
rows.append({"title": title[:200] or "Feed", "properties": props})
result.pages.append(make_collection("OPML feeds", _SCHEMA, rows, source_path=filename))
result.stats["rows"] = len(rows)
return result.finalize()
def _folder_of(node: ET.Element, root: ET.Element) -> str:
parents = {child: parent for parent in root.iter() for child in parent}
parts: list[str] = []
current = parents.get(node)
while current is not None:
attrs = {k.lower(): v for k, v in current.attrib.items()}
if not attrs.get("xmlurl"):
label = attrs.get("title") or attrs.get("text")
if label:
parts.append(label)
current = parents.get(current)
return " / ".join(reversed(parts))