Files
flowdeck/app/services/web_clipper.py
T
bruno 0b251649e5
FlowDeck CI / lint (push) Successful in 1m11s
FlowDeck CI / test (push) Failing after 8m32s
FlowDeck CI / docker (push) Skipped
feat: v6.2.0 Web Clipper — extension navigateur (capture article/selection/bookmark/screenshot)
- Service app/services/web_clipper.py: sanitize HTML, html->blocks, extraction article, creation page workspace-aware, rate limit 50/h, device registration
- Router app/routers/web_clipper.py: POST /api/v2/web-clipper/clip, GET /status, POST /auth/verify, GET/DELETE /devices, GET /extensions (download page), auth via session ou Bearer (api_tokens / extension_devices)
- Migration 19: extension_devices + extension_clips (+ indexes)
- Extension Manifest V3: content.js (floating button, selection), background.js (clip + contextMenus), popup.html/js, clipper.css, icons
- Settings UI: onglet Extensions (liste devices, revoke, test clip, liens download), page /extensions
- Tests: 16 tests web_clipper (sanitize, blocks, article/bookmark/selection/screenshot, bearer, rate-limit, devices, extensions page)
- Bump version 6.1.0 -> 6.2.0
2026-09-19 23:26:03 -04:00

436 lines
19 KiB
Python

"""FlowDeck — Web Clipper service (v6.0.0).
Handles web content capture → FlowDeck page creation.
- Sanitizes incoming HTML (removes scripts, styles, event handlers).
- Extracts readable content via BeautifulSoup heuristics (article/main/body).
- Converts HTML → FlowDeck block list (headings, paragraphs, lists, quotes,
code, images, bookmarks).
- Creates a pages row (workspace-aware) and logs the clip.
No external network calls: images stay as remote URLs (no download in MVP);
a future iteration can download + re-host inline images.
"""
from __future__ import annotations
import hashlib
import json
import logging
import re
import time
import uuid
from bs4 import BeautifulSoup
from app.db import get_conn
logger = logging.getLogger(__name__)
MAX_CLIP_BYTES = 10 * 1024 * 1024 # 10 MB per clip
MAX_CLIPS_PER_HOUR = 50
# In-memory rate limiter per device: {device_id: [timestamps]}
_rate_store: dict[str, list[float]] = {}
def _check_rate_limit(device_id: str) -> bool:
"""Return True if allowed, False if rate-limited (50/hour)."""
now = time.time()
window = 3600
bucket = _rate_store.get(device_id, [])
bucket = [t for t in bucket if now - t < window]
if len(bucket) >= MAX_CLIPS_PER_HOUR:
_rate_store[device_id] = bucket
return False
bucket.append(now)
_rate_store[device_id] = bucket
return True
def _hash_token(token: str) -> str:
return hashlib.sha256(token.encode()).hexdigest()
def sanitize_html(html_str: str) -> str:
"""Strip dangerous tags/attributes, return sanitized HTML string."""
if not html_str:
return ""
# Cap size
if len(html_str.encode("utf-8")) > MAX_CLIP_BYTES:
html_str = html_str[: MAX_CLIP_BYTES // 2]
soup = BeautifulSoup(html_str, "html.parser")
for tag in soup(["script", "style", "noscript", "iframe"]):
tag.decompose()
# Strip event handlers and javascript: URLs
for el in soup.find_all(True):
for attr in list(el.attrs):
if attr.lower().startswith("on"):
del el.attrs[attr]
elif attr in ("href", "src", "action"):
val = str(el.attrs[attr]).strip()
if val.lower().startswith("javascript:") or val.lower().startswith("data:text/html"):
del el.attrs[attr]
return str(soup)
def _extract_main(soup: BeautifulSoup) -> BeautifulSoup:
"""Pick the most content-rich container: article > main > body."""
for sel in ["article", "main", "[role=article]", "#content", ".post-content", ".article-content"]:
el = soup.select_one(sel)
if el and len(el.get_text(strip=True)) > 120:
return el
return soup.body or soup
def _text_node(el) -> str:
return el.get_text(separator=" ", strip=True) if el else ""
def html_to_blocks(html_str: str, source_url: str = "") -> list[dict]:
"""Convert HTML → FlowDeck block list.
Covers: headings (h1-h4), paragraphs, blockquotes, code, lists,
images, links as bookmark when standalone.
"""
if not html_str or not html_str.strip():
return []
soup = BeautifulSoup(html_str, "html.parser")
main = _extract_main(soup)
blocks: list[dict] = []
def _add(b):
if b.get("content") or b.get("src") or b.get("url"):
blocks.append(b)
# Walk direct children and deeper elements
for el in main.find_all(["h1", "h2", "h3", "h4", "p", "blockquote", "pre", "ul", "ol", "img", "figure", "a"], recursive=True):
tag = el.name.lower()
if tag in ("h1", "h2", "h3", "h4"):
level = int(tag[1])
level = min(level, 4)
txt = _text_node(el)
if txt:
_add({"id": str(uuid.uuid4())[:8], "type": f"heading_{level}", "content": txt})
elif tag == "p":
txt = _text_node(el)
# Skip if parent is blockquote or li already handled
if el.find_parent(["blockquote", "li"]):
continue
# If p contains an image, emit image even when text empty
img = el.find("img")
if img and img.get("src"):
src = img.get("src", "").strip()
alt = img.get("alt", "") or ""
if src and not src.startswith("data:"):
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
# If there is also text alongside image, emit paragraph too
if txt:
# Remove image alt from paragraph? Keep text
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt})
continue
if txt:
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt})
elif tag == "blockquote":
txt = _text_node(el)
if txt:
_add({"id": str(uuid.uuid4())[:8], "type": "quote", "content": txt})
elif tag == "pre":
code_el = el.find("code")
txt = (code_el.get_text() if code_el else el.get_text())
if txt.strip():
lang = ""
if code_el and code_el.get("class"):
for c in code_el.get("class"):
if c.startswith("language-"):
lang = c.replace("language-", "")
_add({"id": str(uuid.uuid4())[:8], "type": "code", "content": txt.strip("\n"), "language": lang})
elif tag in ("ul", "ol"):
# Only top-level lists - skip nested
if el.find_parent(["ul", "ol"]):
continue
is_ordered = tag == "ol"
for li in el.find_all("li", recursive=False):
txt = _text_node(li)
if not txt:
continue
# Detect todo
if re.match(r"^\[ ?[xX] ?\]\s*", txt):
checked = bool(re.match(r"^\[ ?[xX] ?\]", txt))
txt = re.sub(r"^\[ ?[xX] ?\]\s*", "", txt)
_add({"id": str(uuid.uuid4())[:8], "type": "to_do", "content": txt, "checked": checked})
else:
_add({"id": str(uuid.uuid4())[:8], "type": "bulleted_list" if not is_ordered else "numbered_list", "content": txt})
elif tag == "img":
# Avoid double-count when inside p/figure already emitted
if el.find_parent("p") or el.find_parent("figure"):
# Still emit if parent p wasn't counted
parent_p = el.find_parent("p")
if parent_p and parent_p.find("img") == el:
continue
src = el.get("src", "").strip()
if not src:
continue
# Skip data URIs for size
if src.startswith("data:"):
continue
alt = el.get("alt", "") or ""
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
elif tag == "a":
# Standalone links as bookmark when they are the only content in p
parent = el.find_parent("p")
txt = el.get_text(strip=True)
href = el.get("href", "").strip()
if href and parent and _text_node(parent) == txt and href.startswith("http"):
# Will be handled as paragraph already; add bookmark variant if distinct
pass
elif tag == "figure":
img = el.find("img")
if img and img.get("src"):
src = img.get("src", "").strip()
alt = img.get("alt", "") or ""
if src and not src.startswith("data:"):
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
if not blocks:
# Fallback: whole text as paragraphs
texts = [t.strip() for t in main.get_text(separator="\n").split("\n") if t.strip()]
for t in texts[:30]:
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": t})
# Always ensure at least one block when source_url present
if not blocks and source_url:
_add({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": source_url, "title": source_url})
return blocks
def extract_article(html_str: str, url: str = "") -> dict:
"""High-level extraction returning title, text, images, metadata."""
soup = BeautifulSoup(html_str, "html.parser")
title = ""
if soup.title and soup.title.string:
title = soup.title.string.strip()
og_title = soup.find("meta", property="og:title")
if og_title and og_title.get("content"):
title = og_title["content"].strip() or title
# Sanitize and convert
clean = sanitize_html(html_str)
blocks = html_to_blocks(clean, source_url=url)
# Images
images = []
for b in blocks:
if b.get("type") == "image" and b.get("src"):
images.append(b["src"])
# Text
text_parts = []
for b in blocks:
if b.get("content"):
text_parts.append(b["content"])
return {"title": title or "Clipped page", "blocks": blocks, "images": images, "text": "\n\n".join(text_parts)}
def _ensure_workspace(conn, user_id: int, workspace_id: int | None) -> tuple[int, str]:
"""Return (workspace_id, workspace_key) for clip insertion."""
if workspace_id:
row = conn.execute("SELECT id, name FROM workspaces WHERE id=?", (workspace_id,)).fetchone()
if row:
# Check membership or owner
mem = conn.execute(
"SELECT 1 FROM workspace_members WHERE workspace_id=? AND user_id=?", (workspace_id, user_id)
).fetchone()
if mem or conn.execute("SELECT 1 FROM workspaces WHERE id=? AND owner_id=?", (workspace_id, user_id)).fetchone():
return workspace_id, row["name"]
# Fallback: first workspace owned or member, else create one
row = conn.execute(
"SELECT w.id, w.name FROM workspaces w LEFT JOIN workspace_members wm ON w.id=wm.workspace_id "
"WHERE w.owner_id=? OR wm.user_id=? ORDER BY w.id LIMIT 1",
(user_id, user_id),
).fetchone()
if row:
return row["id"], row["name"]
# Create default workspace
cur = conn.execute("INSERT INTO workspaces (name, owner_id) VALUES (?, ?)", ("My Workspace", user_id))
ws_id = cur.lastrowid
conn.execute("INSERT INTO workspace_members (workspace_id, user_id, role) VALUES (?, ?, 'admin')", (ws_id, user_id))
return ws_id, "My Workspace"
def create_page_from_clip(clip_data: dict, user_id: int) -> dict:
"""Create a FlowDeck page from a clip payload. Returns {page_id, title}."""
url = (clip_data.get("url") or clip_data.get("source_url") or "").strip()
title = (clip_data.get("title") or "").strip()
content = clip_data.get("content") or clip_data.get("html") or ""
content_type = clip_data.get("content_type") or clip_data.get("clip_type") or "article"
selection_html = clip_data.get("selection_html") or ""
tags = clip_data.get("tags") or []
target_ws = clip_data.get("target_workspace_id") or clip_data.get("workspace_id")
target_page_id = clip_data.get("target_page_id") or clip_data.get("parent_page_id")
# Normalize workspace id
try:
target_ws = int(target_ws) if target_ws is not None else None
except (ValueError, TypeError):
target_ws = None
# Determine title and blocks
if content_type == "screenshot":
# Screenshot: base64 image block + source bookmark (even with empty html)
img_b64 = clip_data.get("image_base64") or clip_data.get("screenshot") or ""
blocks = []
if img_b64:
src = img_b64 if img_b64.startswith("data:") else f"data:image/png;base64,{img_b64}"
blocks.append({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": title or "Screenshot"})
if url:
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url})
if not blocks:
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": title or "Screenshot"})
title = title or "Screenshot"
elif content_type == "selection" and selection_html and selection_html.strip():
clean = sanitize_html(selection_html)
blocks = html_to_blocks(clean, source_url=url)
if url:
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url})
title = title or "Clipped selection"
elif content_type == "bookmark" or not content.strip():
# Bookmark mode: no HTML body, just link card
bookmark_title = title or (url or "Bookmark")
blocks = [
{"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": bookmark_title, "description": clip_data.get("metadata", {}).get("og_description", "") or ""}
]
if url:
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": f"Source: {url}"})
title = bookmark_title
else:
# Article full
result = extract_article(content, url=url)
blocks = result["blocks"]
if not title:
title = result["title"]
# Append source bookmark if not already dominant
if url:
# avoid duplicate bookmark if last block already is bookmark to same url
if not blocks or blocks[-1].get("url") != url:
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": "Source"})
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": url})
# Cap blocks
if len(blocks) > 200:
blocks = blocks[:200]
title = (title or "Clipped page").strip()[:200] or "Clipped page"
# Persist
with get_conn() as conn:
ws_id, ws_key = _ensure_workspace(conn, user_id, target_ws)
# Validate target_page_id belongs to same workspace
parent_id = None
if target_page_id:
try:
pid = int(target_page_id)
pr = conn.execute("SELECT id, workspace_id FROM pages WHERE id=? AND deleted_at IS NULL", (pid,)).fetchone()
if pr and (pr["workspace_id"] == ws_id or pr["workspace_id"] is None):
parent_id = pid
except (ValueError, TypeError):
pass
# Determine sort order
if parent_id is not None:
next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE parent_id=?", (parent_id,)).fetchone()[0]
else:
next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE workspace_id=? AND parent_id IS NULL", (ws_id,)).fetchone()[0]
blocks_json = json.dumps(blocks, ensure_ascii=False)
cur = conn.execute(
"""INSERT INTO pages (workspace, workspace_id, title, content, content_format, parent_section, parent_id, sort_order)
VALUES (?, ?, ?, ?, 'blocks', 'Private', ?, ?)""",
(ws_key or "", ws_id, title, blocks_json, parent_id, next_order),
)
page_id = cur.lastrowid
# Tags: create per-user tags if needed and attach via page_tags
for tname in tags[:10]:
tn = tname.strip()[:50]
if not tn:
continue
conn.execute("INSERT OR IGNORE INTO tags (name, color, user_id) VALUES (?, '#787774', ?)", (tn, user_id))
tr = conn.execute("SELECT id FROM tags WHERE name=? AND user_id=?", (tn, user_id)).fetchone()
if tr:
conn.execute("INSERT OR IGNORE INTO page_tags (page_id, tag_id) VALUES (?, ?)", (page_id, tr["id"]))
conn.commit()
return {"page_id": page_id, "title": title, "blocks": blocks, "workspace_id": ws_id}
def register_device(user_id: int, device_id: str, device_name: str = "", extension_name: str = "chrome") -> dict:
"""Register or update an extension device, returns {device, token} (token shown once if new)."""
if not device_id or len(device_id) > 128:
raise ValueError("Invalid device_id")
token = f"fd_clip_{uuid.uuid4().hex}{uuid.uuid4().hex[:8]}"
thash = _hash_token(token)
with get_conn() as conn:
existing = conn.execute(
"SELECT id, token_hash FROM extension_devices WHERE user_id=? AND device_id=? AND extension_name=?",
(user_id, device_id, extension_name),
).fetchone()
if existing:
conn.execute(
"UPDATE extension_devices SET device_name=?, last_used_at=CURRENT_TIMESTAMP WHERE id=?",
(device_name[:200], existing["id"]),
)
conn.commit()
return {"id": existing["id"], "device_id": device_id, "token": None, "existing": True}
cur = conn.execute(
"""INSERT INTO extension_devices (user_id, extension_name, device_id, device_name, token_hash)
VALUES (?, ?, ?, ?, ?)""",
(user_id, extension_name, device_id, device_name[:200], thash),
)
conn.commit()
return {"id": cur.lastrowid, "device_id": device_id, "token": token, "existing": False}
def verify_device_token(device_id: str, token: str) -> dict | None:
"""Verify a device token, returns device row or None."""
thash = _hash_token(token)
with get_conn() as conn:
row = conn.execute(
"SELECT * FROM extension_devices WHERE device_id=? AND token_hash=? AND revoked=0",
(device_id, thash),
).fetchone()
if row:
conn.execute("UPDATE extension_devices SET last_used_at=CURRENT_TIMESTAMP WHERE id=?", (row["id"],))
conn.commit()
return dict(row)
return None
def log_clip(user_id: int, device_id: str, clip_type: str, source_url: str, target_page_id: int, workspace_id: int, title: str):
with get_conn() as conn:
conn.execute(
"""INSERT INTO extension_clips (user_id, device_id, clip_type, source_url, target_page_id, target_workspace_id, title)
VALUES (?, ?, ?, ?, ?, ?, ?)""",
(user_id, device_id, clip_type[:20], source_url[:2000], target_page_id, workspace_id, title[:200]),
)
conn.commit()
def list_devices(user_id: int) -> list[dict]:
with get_conn() as conn:
rows = conn.execute(
"SELECT id, extension_name, device_id, device_name, scopes, last_used_at, created_at, revoked FROM extension_devices WHERE user_id=? ORDER BY last_used_at DESC, created_at DESC",
(user_id,),
).fetchall()
# Enrich with clip counts
out = []
for r in rows:
d = dict(r)
cnt = conn.execute("SELECT COUNT(*) AS n FROM extension_clips WHERE user_id=? AND device_id=?", (user_id, r["device_id"])).fetchone()["n"]
d["clips_count"] = cnt
# Last clip
last = conn.execute("SELECT created_at FROM extension_clips WHERE user_id=? AND device_id=? ORDER BY created_at DESC LIMIT 1", (user_id, r["device_id"])).fetchone()
d["last_clip_at"] = last["created_at"] if last else None
out.append(d)
return out
def revoke_device(user_id: int, device_row_id: int) -> bool:
with get_conn() as conn:
cur = conn.execute("UPDATE extension_devices SET revoked=1 WHERE id=? AND user_id=?", (device_row_id, user_id))
conn.commit()
return cur.rowcount > 0