- Service app/services/web_clipper.py: sanitize HTML, html->blocks, extraction article, creation page workspace-aware, rate limit 50/h, device registration - Router app/routers/web_clipper.py: POST /api/v2/web-clipper/clip, GET /status, POST /auth/verify, GET/DELETE /devices, GET /extensions (download page), auth via session ou Bearer (api_tokens / extension_devices) - Migration 19: extension_devices + extension_clips (+ indexes) - Extension Manifest V3: content.js (floating button, selection), background.js (clip + contextMenus), popup.html/js, clipper.css, icons - Settings UI: onglet Extensions (liste devices, revoke, test clip, liens download), page /extensions - Tests: 16 tests web_clipper (sanitize, blocks, article/bookmark/selection/screenshot, bearer, rate-limit, devices, extensions page) - Bump version 6.1.0 -> 6.2.0
436 lines
19 KiB
Python
436 lines
19 KiB
Python
"""FlowDeck — Web Clipper service (v6.0.0).
|
|
|
|
Handles web content capture → FlowDeck page creation.
|
|
|
|
- Sanitizes incoming HTML (removes scripts, styles, event handlers).
|
|
- Extracts readable content via BeautifulSoup heuristics (article/main/body).
|
|
- Converts HTML → FlowDeck block list (headings, paragraphs, lists, quotes,
|
|
code, images, bookmarks).
|
|
- Creates a pages row (workspace-aware) and logs the clip.
|
|
|
|
No external network calls: images stay as remote URLs (no download in MVP);
|
|
a future iteration can download + re-host inline images.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import re
|
|
import time
|
|
import uuid
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
from app.db import get_conn
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
MAX_CLIP_BYTES = 10 * 1024 * 1024 # 10 MB per clip
|
|
MAX_CLIPS_PER_HOUR = 50
|
|
|
|
# In-memory rate limiter per device: {device_id: [timestamps]}
|
|
_rate_store: dict[str, list[float]] = {}
|
|
|
|
|
|
def _check_rate_limit(device_id: str) -> bool:
|
|
"""Return True if allowed, False if rate-limited (50/hour)."""
|
|
now = time.time()
|
|
window = 3600
|
|
bucket = _rate_store.get(device_id, [])
|
|
bucket = [t for t in bucket if now - t < window]
|
|
if len(bucket) >= MAX_CLIPS_PER_HOUR:
|
|
_rate_store[device_id] = bucket
|
|
return False
|
|
bucket.append(now)
|
|
_rate_store[device_id] = bucket
|
|
return True
|
|
|
|
|
|
def _hash_token(token: str) -> str:
|
|
return hashlib.sha256(token.encode()).hexdigest()
|
|
|
|
|
|
def sanitize_html(html_str: str) -> str:
|
|
"""Strip dangerous tags/attributes, return sanitized HTML string."""
|
|
if not html_str:
|
|
return ""
|
|
# Cap size
|
|
if len(html_str.encode("utf-8")) > MAX_CLIP_BYTES:
|
|
html_str = html_str[: MAX_CLIP_BYTES // 2]
|
|
soup = BeautifulSoup(html_str, "html.parser")
|
|
for tag in soup(["script", "style", "noscript", "iframe"]):
|
|
tag.decompose()
|
|
# Strip event handlers and javascript: URLs
|
|
for el in soup.find_all(True):
|
|
for attr in list(el.attrs):
|
|
if attr.lower().startswith("on"):
|
|
del el.attrs[attr]
|
|
elif attr in ("href", "src", "action"):
|
|
val = str(el.attrs[attr]).strip()
|
|
if val.lower().startswith("javascript:") or val.lower().startswith("data:text/html"):
|
|
del el.attrs[attr]
|
|
return str(soup)
|
|
|
|
|
|
def _extract_main(soup: BeautifulSoup) -> BeautifulSoup:
|
|
"""Pick the most content-rich container: article > main > body."""
|
|
for sel in ["article", "main", "[role=article]", "#content", ".post-content", ".article-content"]:
|
|
el = soup.select_one(sel)
|
|
if el and len(el.get_text(strip=True)) > 120:
|
|
return el
|
|
return soup.body or soup
|
|
|
|
|
|
def _text_node(el) -> str:
|
|
return el.get_text(separator=" ", strip=True) if el else ""
|
|
|
|
|
|
def html_to_blocks(html_str: str, source_url: str = "") -> list[dict]:
|
|
"""Convert HTML → FlowDeck block list.
|
|
|
|
Covers: headings (h1-h4), paragraphs, blockquotes, code, lists,
|
|
images, links as bookmark when standalone.
|
|
"""
|
|
if not html_str or not html_str.strip():
|
|
return []
|
|
|
|
soup = BeautifulSoup(html_str, "html.parser")
|
|
main = _extract_main(soup)
|
|
blocks: list[dict] = []
|
|
|
|
def _add(b):
|
|
if b.get("content") or b.get("src") or b.get("url"):
|
|
blocks.append(b)
|
|
|
|
# Walk direct children and deeper elements
|
|
for el in main.find_all(["h1", "h2", "h3", "h4", "p", "blockquote", "pre", "ul", "ol", "img", "figure", "a"], recursive=True):
|
|
tag = el.name.lower()
|
|
if tag in ("h1", "h2", "h3", "h4"):
|
|
level = int(tag[1])
|
|
level = min(level, 4)
|
|
txt = _text_node(el)
|
|
if txt:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": f"heading_{level}", "content": txt})
|
|
elif tag == "p":
|
|
txt = _text_node(el)
|
|
# Skip if parent is blockquote or li already handled
|
|
if el.find_parent(["blockquote", "li"]):
|
|
continue
|
|
# If p contains an image, emit image even when text empty
|
|
img = el.find("img")
|
|
if img and img.get("src"):
|
|
src = img.get("src", "").strip()
|
|
alt = img.get("alt", "") or ""
|
|
if src and not src.startswith("data:"):
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
|
|
# If there is also text alongside image, emit paragraph too
|
|
if txt:
|
|
# Remove image alt from paragraph? Keep text
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt})
|
|
continue
|
|
if txt:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt})
|
|
elif tag == "blockquote":
|
|
txt = _text_node(el)
|
|
if txt:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "quote", "content": txt})
|
|
elif tag == "pre":
|
|
code_el = el.find("code")
|
|
txt = (code_el.get_text() if code_el else el.get_text())
|
|
if txt.strip():
|
|
lang = ""
|
|
if code_el and code_el.get("class"):
|
|
for c in code_el.get("class"):
|
|
if c.startswith("language-"):
|
|
lang = c.replace("language-", "")
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "code", "content": txt.strip("\n"), "language": lang})
|
|
elif tag in ("ul", "ol"):
|
|
# Only top-level lists - skip nested
|
|
if el.find_parent(["ul", "ol"]):
|
|
continue
|
|
is_ordered = tag == "ol"
|
|
for li in el.find_all("li", recursive=False):
|
|
txt = _text_node(li)
|
|
if not txt:
|
|
continue
|
|
# Detect todo
|
|
if re.match(r"^\[ ?[xX] ?\]\s*", txt):
|
|
checked = bool(re.match(r"^\[ ?[xX] ?\]", txt))
|
|
txt = re.sub(r"^\[ ?[xX] ?\]\s*", "", txt)
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "to_do", "content": txt, "checked": checked})
|
|
else:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "bulleted_list" if not is_ordered else "numbered_list", "content": txt})
|
|
elif tag == "img":
|
|
# Avoid double-count when inside p/figure already emitted
|
|
if el.find_parent("p") or el.find_parent("figure"):
|
|
# Still emit if parent p wasn't counted
|
|
parent_p = el.find_parent("p")
|
|
if parent_p and parent_p.find("img") == el:
|
|
continue
|
|
src = el.get("src", "").strip()
|
|
if not src:
|
|
continue
|
|
# Skip data URIs for size
|
|
if src.startswith("data:"):
|
|
continue
|
|
alt = el.get("alt", "") or ""
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
|
|
elif tag == "a":
|
|
# Standalone links as bookmark when they are the only content in p
|
|
parent = el.find_parent("p")
|
|
txt = el.get_text(strip=True)
|
|
href = el.get("href", "").strip()
|
|
if href and parent and _text_node(parent) == txt and href.startswith("http"):
|
|
# Will be handled as paragraph already; add bookmark variant if distinct
|
|
pass
|
|
elif tag == "figure":
|
|
img = el.find("img")
|
|
if img and img.get("src"):
|
|
src = img.get("src", "").strip()
|
|
alt = img.get("alt", "") or ""
|
|
if src and not src.startswith("data:"):
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt})
|
|
|
|
if not blocks:
|
|
# Fallback: whole text as paragraphs
|
|
texts = [t.strip() for t in main.get_text(separator="\n").split("\n") if t.strip()]
|
|
for t in texts[:30]:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": t})
|
|
|
|
# Always ensure at least one block when source_url present
|
|
if not blocks and source_url:
|
|
_add({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": source_url, "title": source_url})
|
|
|
|
return blocks
|
|
|
|
|
|
def extract_article(html_str: str, url: str = "") -> dict:
|
|
"""High-level extraction returning title, text, images, metadata."""
|
|
soup = BeautifulSoup(html_str, "html.parser")
|
|
title = ""
|
|
if soup.title and soup.title.string:
|
|
title = soup.title.string.strip()
|
|
og_title = soup.find("meta", property="og:title")
|
|
if og_title and og_title.get("content"):
|
|
title = og_title["content"].strip() or title
|
|
# Sanitize and convert
|
|
clean = sanitize_html(html_str)
|
|
blocks = html_to_blocks(clean, source_url=url)
|
|
# Images
|
|
images = []
|
|
for b in blocks:
|
|
if b.get("type") == "image" and b.get("src"):
|
|
images.append(b["src"])
|
|
# Text
|
|
text_parts = []
|
|
for b in blocks:
|
|
if b.get("content"):
|
|
text_parts.append(b["content"])
|
|
return {"title": title or "Clipped page", "blocks": blocks, "images": images, "text": "\n\n".join(text_parts)}
|
|
|
|
|
|
def _ensure_workspace(conn, user_id: int, workspace_id: int | None) -> tuple[int, str]:
|
|
"""Return (workspace_id, workspace_key) for clip insertion."""
|
|
if workspace_id:
|
|
row = conn.execute("SELECT id, name FROM workspaces WHERE id=?", (workspace_id,)).fetchone()
|
|
if row:
|
|
# Check membership or owner
|
|
mem = conn.execute(
|
|
"SELECT 1 FROM workspace_members WHERE workspace_id=? AND user_id=?", (workspace_id, user_id)
|
|
).fetchone()
|
|
if mem or conn.execute("SELECT 1 FROM workspaces WHERE id=? AND owner_id=?", (workspace_id, user_id)).fetchone():
|
|
return workspace_id, row["name"]
|
|
# Fallback: first workspace owned or member, else create one
|
|
row = conn.execute(
|
|
"SELECT w.id, w.name FROM workspaces w LEFT JOIN workspace_members wm ON w.id=wm.workspace_id "
|
|
"WHERE w.owner_id=? OR wm.user_id=? ORDER BY w.id LIMIT 1",
|
|
(user_id, user_id),
|
|
).fetchone()
|
|
if row:
|
|
return row["id"], row["name"]
|
|
# Create default workspace
|
|
cur = conn.execute("INSERT INTO workspaces (name, owner_id) VALUES (?, ?)", ("My Workspace", user_id))
|
|
ws_id = cur.lastrowid
|
|
conn.execute("INSERT INTO workspace_members (workspace_id, user_id, role) VALUES (?, ?, 'admin')", (ws_id, user_id))
|
|
return ws_id, "My Workspace"
|
|
|
|
|
|
def create_page_from_clip(clip_data: dict, user_id: int) -> dict:
|
|
"""Create a FlowDeck page from a clip payload. Returns {page_id, title}."""
|
|
url = (clip_data.get("url") or clip_data.get("source_url") or "").strip()
|
|
title = (clip_data.get("title") or "").strip()
|
|
content = clip_data.get("content") or clip_data.get("html") or ""
|
|
content_type = clip_data.get("content_type") or clip_data.get("clip_type") or "article"
|
|
selection_html = clip_data.get("selection_html") or ""
|
|
tags = clip_data.get("tags") or []
|
|
target_ws = clip_data.get("target_workspace_id") or clip_data.get("workspace_id")
|
|
target_page_id = clip_data.get("target_page_id") or clip_data.get("parent_page_id")
|
|
# Normalize workspace id
|
|
try:
|
|
target_ws = int(target_ws) if target_ws is not None else None
|
|
except (ValueError, TypeError):
|
|
target_ws = None
|
|
|
|
# Determine title and blocks
|
|
if content_type == "screenshot":
|
|
# Screenshot: base64 image block + source bookmark (even with empty html)
|
|
img_b64 = clip_data.get("image_base64") or clip_data.get("screenshot") or ""
|
|
blocks = []
|
|
if img_b64:
|
|
src = img_b64 if img_b64.startswith("data:") else f"data:image/png;base64,{img_b64}"
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": title or "Screenshot"})
|
|
if url:
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url})
|
|
if not blocks:
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": title or "Screenshot"})
|
|
title = title or "Screenshot"
|
|
elif content_type == "selection" and selection_html and selection_html.strip():
|
|
clean = sanitize_html(selection_html)
|
|
blocks = html_to_blocks(clean, source_url=url)
|
|
if url:
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url})
|
|
title = title or "Clipped selection"
|
|
elif content_type == "bookmark" or not content.strip():
|
|
# Bookmark mode: no HTML body, just link card
|
|
bookmark_title = title or (url or "Bookmark")
|
|
blocks = [
|
|
{"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": bookmark_title, "description": clip_data.get("metadata", {}).get("og_description", "") or ""}
|
|
]
|
|
if url:
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": f"Source: {url}"})
|
|
title = bookmark_title
|
|
else:
|
|
# Article full
|
|
result = extract_article(content, url=url)
|
|
blocks = result["blocks"]
|
|
if not title:
|
|
title = result["title"]
|
|
# Append source bookmark if not already dominant
|
|
if url:
|
|
# avoid duplicate bookmark if last block already is bookmark to same url
|
|
if not blocks or blocks[-1].get("url") != url:
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": "Source"})
|
|
blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": url})
|
|
|
|
# Cap blocks
|
|
if len(blocks) > 200:
|
|
blocks = blocks[:200]
|
|
title = (title or "Clipped page").strip()[:200] or "Clipped page"
|
|
|
|
# Persist
|
|
with get_conn() as conn:
|
|
ws_id, ws_key = _ensure_workspace(conn, user_id, target_ws)
|
|
# Validate target_page_id belongs to same workspace
|
|
parent_id = None
|
|
if target_page_id:
|
|
try:
|
|
pid = int(target_page_id)
|
|
pr = conn.execute("SELECT id, workspace_id FROM pages WHERE id=? AND deleted_at IS NULL", (pid,)).fetchone()
|
|
if pr and (pr["workspace_id"] == ws_id or pr["workspace_id"] is None):
|
|
parent_id = pid
|
|
except (ValueError, TypeError):
|
|
pass
|
|
# Determine sort order
|
|
if parent_id is not None:
|
|
next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE parent_id=?", (parent_id,)).fetchone()[0]
|
|
else:
|
|
next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE workspace_id=? AND parent_id IS NULL", (ws_id,)).fetchone()[0]
|
|
blocks_json = json.dumps(blocks, ensure_ascii=False)
|
|
cur = conn.execute(
|
|
"""INSERT INTO pages (workspace, workspace_id, title, content, content_format, parent_section, parent_id, sort_order)
|
|
VALUES (?, ?, ?, ?, 'blocks', 'Private', ?, ?)""",
|
|
(ws_key or "", ws_id, title, blocks_json, parent_id, next_order),
|
|
)
|
|
page_id = cur.lastrowid
|
|
# Tags: create per-user tags if needed and attach via page_tags
|
|
for tname in tags[:10]:
|
|
tn = tname.strip()[:50]
|
|
if not tn:
|
|
continue
|
|
conn.execute("INSERT OR IGNORE INTO tags (name, color, user_id) VALUES (?, '#787774', ?)", (tn, user_id))
|
|
tr = conn.execute("SELECT id FROM tags WHERE name=? AND user_id=?", (tn, user_id)).fetchone()
|
|
if tr:
|
|
conn.execute("INSERT OR IGNORE INTO page_tags (page_id, tag_id) VALUES (?, ?)", (page_id, tr["id"]))
|
|
conn.commit()
|
|
|
|
return {"page_id": page_id, "title": title, "blocks": blocks, "workspace_id": ws_id}
|
|
|
|
|
|
def register_device(user_id: int, device_id: str, device_name: str = "", extension_name: str = "chrome") -> dict:
|
|
"""Register or update an extension device, returns {device, token} (token shown once if new)."""
|
|
if not device_id or len(device_id) > 128:
|
|
raise ValueError("Invalid device_id")
|
|
token = f"fd_clip_{uuid.uuid4().hex}{uuid.uuid4().hex[:8]}"
|
|
thash = _hash_token(token)
|
|
with get_conn() as conn:
|
|
existing = conn.execute(
|
|
"SELECT id, token_hash FROM extension_devices WHERE user_id=? AND device_id=? AND extension_name=?",
|
|
(user_id, device_id, extension_name),
|
|
).fetchone()
|
|
if existing:
|
|
conn.execute(
|
|
"UPDATE extension_devices SET device_name=?, last_used_at=CURRENT_TIMESTAMP WHERE id=?",
|
|
(device_name[:200], existing["id"]),
|
|
)
|
|
conn.commit()
|
|
return {"id": existing["id"], "device_id": device_id, "token": None, "existing": True}
|
|
cur = conn.execute(
|
|
"""INSERT INTO extension_devices (user_id, extension_name, device_id, device_name, token_hash)
|
|
VALUES (?, ?, ?, ?, ?)""",
|
|
(user_id, extension_name, device_id, device_name[:200], thash),
|
|
)
|
|
conn.commit()
|
|
return {"id": cur.lastrowid, "device_id": device_id, "token": token, "existing": False}
|
|
|
|
|
|
def verify_device_token(device_id: str, token: str) -> dict | None:
|
|
"""Verify a device token, returns device row or None."""
|
|
thash = _hash_token(token)
|
|
with get_conn() as conn:
|
|
row = conn.execute(
|
|
"SELECT * FROM extension_devices WHERE device_id=? AND token_hash=? AND revoked=0",
|
|
(device_id, thash),
|
|
).fetchone()
|
|
if row:
|
|
conn.execute("UPDATE extension_devices SET last_used_at=CURRENT_TIMESTAMP WHERE id=?", (row["id"],))
|
|
conn.commit()
|
|
return dict(row)
|
|
return None
|
|
|
|
|
|
def log_clip(user_id: int, device_id: str, clip_type: str, source_url: str, target_page_id: int, workspace_id: int, title: str):
|
|
with get_conn() as conn:
|
|
conn.execute(
|
|
"""INSERT INTO extension_clips (user_id, device_id, clip_type, source_url, target_page_id, target_workspace_id, title)
|
|
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
|
(user_id, device_id, clip_type[:20], source_url[:2000], target_page_id, workspace_id, title[:200]),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def list_devices(user_id: int) -> list[dict]:
|
|
with get_conn() as conn:
|
|
rows = conn.execute(
|
|
"SELECT id, extension_name, device_id, device_name, scopes, last_used_at, created_at, revoked FROM extension_devices WHERE user_id=? ORDER BY last_used_at DESC, created_at DESC",
|
|
(user_id,),
|
|
).fetchall()
|
|
# Enrich with clip counts
|
|
out = []
|
|
for r in rows:
|
|
d = dict(r)
|
|
cnt = conn.execute("SELECT COUNT(*) AS n FROM extension_clips WHERE user_id=? AND device_id=?", (user_id, r["device_id"])).fetchone()["n"]
|
|
d["clips_count"] = cnt
|
|
# Last clip
|
|
last = conn.execute("SELECT created_at FROM extension_clips WHERE user_id=? AND device_id=? ORDER BY created_at DESC LIMIT 1", (user_id, r["device_id"])).fetchone()
|
|
d["last_clip_at"] = last["created_at"] if last else None
|
|
out.append(d)
|
|
return out
|
|
|
|
|
|
def revoke_device(user_id: int, device_row_id: int) -> bool:
|
|
with get_conn() as conn:
|
|
cur = conn.execute("UPDATE extension_devices SET revoked=1 WHERE id=? AND user_id=?", (device_row_id, user_id))
|
|
conn.commit()
|
|
return cur.rowcount > 0
|