- unified importer framework (app/services/importers/): normalized model, registry, common pipeline (hierarchy, attachments, collections, dedup), async jobs, dry-run preview, column->type mapping - Phase 1: Obsidian, Notion, Logseq/Roam, HTML (Apple Notes/Bear/Ulysses/ OneNote), Google Keep, generic Markdown - Phase 2: typed CSV/TSV, Excel (openpyxl), generic JSON - Phase 3: Word .docx (python-docx), PDF (pypdf), HTML folders - Phase 4: Raindrop, Pocket, Readwise, Shaarli, Netscape bookmarks, .ics, OPML, Standard Notes, Gitea/GitHub issues (+labels/milestones) - Phase 5: incremental re-sync (skip/update/duplicate), partial-error resume, forge repo files, URL web clipper (SSRF guard), batch multi-file + UI queue, Notion relation resolution, exportable JSON reports - /import wizard, API /api/import/*, migration 9 (import_items, import_jobs) - fix: property values stored by property id (correct DB view rendering) - deps: openpyxl, beautifulsoup4, PyYAML, python-docx, pypdf - 43 import tests; full suite 491 green; ruff clean - bump version 5.11.5
87 lines
2.9 KiB
Python
87 lines
2.9 KiB
Python
"""FlowDeck — OPML importer (v5.6.0, Phase 4).
|
|
|
|
Imports an OPML outline (RSS readers, feed lists) as a collection of feeds.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import xml.etree.ElementTree as ET
|
|
from typing import Any
|
|
|
|
from app.services.importers.base import (
|
|
Importer,
|
|
ImportResult,
|
|
decode_text,
|
|
make_collection,
|
|
register_importer,
|
|
)
|
|
|
|
_SCHEMA = [
|
|
{"name": "Title", "type": "title"},
|
|
{"name": "Feed URL", "type": "url"},
|
|
{"name": "Site URL", "type": "url"},
|
|
{"name": "Type", "type": "select", "options": [
|
|
{"name": "rss", "color": "orange"},
|
|
{"name": "folder", "color": "gray"},
|
|
]},
|
|
{"name": "Folder", "type": "text"},
|
|
]
|
|
|
|
|
|
@register_importer
|
|
class OpmlImporter(Importer):
|
|
source_id = "opml"
|
|
label = "OPML (flux RSS)"
|
|
description = "Outline OPML → collection de flux (titre, URL, dossier)."
|
|
extensions = (".opml", ".xml")
|
|
order = 34
|
|
|
|
def detect(self, filename: str, data: bytes) -> bool:
|
|
if filename.lower().endswith(".opml"):
|
|
return True
|
|
head = decode_text(data)[:1000].lower()
|
|
return "<opml" in head and "<outline" in head
|
|
|
|
def parse(self, filename: str, data: bytes) -> ImportResult:
|
|
result = ImportResult(source=self.source_id)
|
|
rows: list[dict] = []
|
|
try:
|
|
root = ET.fromstring(decode_text(data))
|
|
except ET.ParseError as exc:
|
|
result.warn(f"OPML invalide : {exc}")
|
|
return result.finalize()
|
|
for outline in root.iter("outline"):
|
|
attrs = {k.lower(): v for k, v in outline.attrib.items()}
|
|
feed = attrs.get("xmlurl")
|
|
title = attrs.get("title") or attrs.get("text") or feed or ""
|
|
if not feed and not title:
|
|
continue
|
|
props: dict[str, Any] = {}
|
|
if feed:
|
|
props["Feed URL"] = feed
|
|
props["Type"] = "rss"
|
|
else:
|
|
props["Type"] = "folder"
|
|
if attrs.get("htmlurl"):
|
|
props["Site URL"] = attrs["htmlurl"]
|
|
folder = _folder_of(outline, root)
|
|
if folder:
|
|
props["Folder"] = folder
|
|
rows.append({"title": title[:200] or "Feed", "properties": props})
|
|
result.pages.append(make_collection("OPML feeds", _SCHEMA, rows, source_path=filename))
|
|
result.stats["rows"] = len(rows)
|
|
return result.finalize()
|
|
|
|
|
|
def _folder_of(node: ET.Element, root: ET.Element) -> str:
|
|
parents = {child: parent for parent in root.iter() for child in parent}
|
|
parts: list[str] = []
|
|
current = parents.get(node)
|
|
while current is not None:
|
|
attrs = {k.lower(): v for k, v in current.attrib.items()}
|
|
if not attrs.get("xmlurl"):
|
|
label = attrs.get("title") or attrs.get("text")
|
|
if label:
|
|
parts.append(label)
|
|
current = parents.get(current)
|
|
return " / ".join(reversed(parts))
|