"""FlowDeck — Web Clipper service (v6.0.0). Handles web content capture → FlowDeck page creation. - Sanitizes incoming HTML (removes scripts, styles, event handlers). - Extracts readable content via BeautifulSoup heuristics (article/main/body). - Converts HTML → FlowDeck block list (headings, paragraphs, lists, quotes, code, images, bookmarks). - Creates a pages row (workspace-aware) and logs the clip. No external network calls: images stay as remote URLs (no download in MVP); a future iteration can download + re-host inline images. """ from __future__ import annotations import hashlib import json import logging import re import time import uuid from bs4 import BeautifulSoup from app.db import get_conn logger = logging.getLogger(__name__) MAX_CLIP_BYTES = 10 * 1024 * 1024 # 10 MB per clip MAX_CLIPS_PER_HOUR = 50 # In-memory rate limiter per device: {device_id: [timestamps]} _rate_store: dict[str, list[float]] = {} def _check_rate_limit(device_id: str) -> bool: """Return True if allowed, False if rate-limited (50/hour).""" now = time.time() window = 3600 bucket = _rate_store.get(device_id, []) bucket = [t for t in bucket if now - t < window] if len(bucket) >= MAX_CLIPS_PER_HOUR: _rate_store[device_id] = bucket return False bucket.append(now) _rate_store[device_id] = bucket return True def _hash_token(token: str) -> str: return hashlib.sha256(token.encode()).hexdigest() def sanitize_html(html_str: str) -> str: """Strip dangerous tags/attributes, return sanitized HTML string.""" if not html_str: return "" # Cap size if len(html_str.encode("utf-8")) > MAX_CLIP_BYTES: html_str = html_str[: MAX_CLIP_BYTES // 2] soup = BeautifulSoup(html_str, "html.parser") for tag in soup(["script", "style", "noscript", "iframe"]): tag.decompose() # Strip event handlers and javascript: URLs for el in soup.find_all(True): for attr in list(el.attrs): if attr.lower().startswith("on"): del el.attrs[attr] elif attr in ("href", "src", "action"): val = str(el.attrs[attr]).strip() if val.lower().startswith("javascript:") or val.lower().startswith("data:text/html"): del el.attrs[attr] return str(soup) def _extract_main(soup: BeautifulSoup) -> BeautifulSoup: """Pick the most content-rich container: article > main > body.""" for sel in ["article", "main", "[role=article]", "#content", ".post-content", ".article-content"]: el = soup.select_one(sel) if el and len(el.get_text(strip=True)) > 120: return el return soup.body or soup def _text_node(el) -> str: return el.get_text(separator=" ", strip=True) if el else "" def html_to_blocks(html_str: str, source_url: str = "") -> list[dict]: """Convert HTML → FlowDeck block list. Covers: headings (h1-h4), paragraphs, blockquotes, code, lists, images, links as bookmark when standalone. """ if not html_str or not html_str.strip(): return [] soup = BeautifulSoup(html_str, "html.parser") main = _extract_main(soup) blocks: list[dict] = [] def _add(b): if b.get("content") or b.get("src") or b.get("url"): blocks.append(b) # Walk direct children and deeper elements for el in main.find_all(["h1", "h2", "h3", "h4", "p", "blockquote", "pre", "ul", "ol", "img", "figure", "a"], recursive=True): tag = el.name.lower() if tag in ("h1", "h2", "h3", "h4"): level = int(tag[1]) level = min(level, 4) txt = _text_node(el) if txt: _add({"id": str(uuid.uuid4())[:8], "type": f"heading_{level}", "content": txt}) elif tag == "p": txt = _text_node(el) # Skip if parent is blockquote or li already handled if el.find_parent(["blockquote", "li"]): continue # If p contains an image, emit image even when text empty img = el.find("img") if img and img.get("src"): src = img.get("src", "").strip() alt = img.get("alt", "") or "" if src and not src.startswith("data:"): _add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt}) # If there is also text alongside image, emit paragraph too if txt: # Remove image alt from paragraph? Keep text _add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt}) continue if txt: _add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": txt}) elif tag == "blockquote": txt = _text_node(el) if txt: _add({"id": str(uuid.uuid4())[:8], "type": "quote", "content": txt}) elif tag == "pre": code_el = el.find("code") txt = (code_el.get_text() if code_el else el.get_text()) if txt.strip(): lang = "" if code_el and code_el.get("class"): for c in code_el.get("class"): if c.startswith("language-"): lang = c.replace("language-", "") _add({"id": str(uuid.uuid4())[:8], "type": "code", "content": txt.strip("\n"), "language": lang}) elif tag in ("ul", "ol"): # Only top-level lists - skip nested if el.find_parent(["ul", "ol"]): continue is_ordered = tag == "ol" for li in el.find_all("li", recursive=False): txt = _text_node(li) if not txt: continue # Detect todo if re.match(r"^\[ ?[xX] ?\]\s*", txt): checked = bool(re.match(r"^\[ ?[xX] ?\]", txt)) txt = re.sub(r"^\[ ?[xX] ?\]\s*", "", txt) _add({"id": str(uuid.uuid4())[:8], "type": "to_do", "content": txt, "checked": checked}) else: _add({"id": str(uuid.uuid4())[:8], "type": "bulleted_list" if not is_ordered else "numbered_list", "content": txt}) elif tag == "img": # Avoid double-count when inside p/figure already emitted if el.find_parent("p") or el.find_parent("figure"): # Still emit if parent p wasn't counted parent_p = el.find_parent("p") if parent_p and parent_p.find("img") == el: continue src = el.get("src", "").strip() if not src: continue # Skip data URIs for size if src.startswith("data:"): continue alt = el.get("alt", "") or "" _add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt}) elif tag == "a": # Standalone links as bookmark when they are the only content in p parent = el.find_parent("p") txt = el.get_text(strip=True) href = el.get("href", "").strip() if href and parent and _text_node(parent) == txt and href.startswith("http"): # Will be handled as paragraph already; add bookmark variant if distinct pass elif tag == "figure": img = el.find("img") if img and img.get("src"): src = img.get("src", "").strip() alt = img.get("alt", "") or "" if src and not src.startswith("data:"): _add({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": alt}) if not blocks: # Fallback: whole text as paragraphs texts = [t.strip() for t in main.get_text(separator="\n").split("\n") if t.strip()] for t in texts[:30]: _add({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": t}) # Always ensure at least one block when source_url present if not blocks and source_url: _add({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": source_url, "title": source_url}) return blocks def extract_article(html_str: str, url: str = "") -> dict: """High-level extraction returning title, text, images, metadata.""" soup = BeautifulSoup(html_str, "html.parser") title = "" if soup.title and soup.title.string: title = soup.title.string.strip() og_title = soup.find("meta", property="og:title") if og_title and og_title.get("content"): title = og_title["content"].strip() or title # Sanitize and convert clean = sanitize_html(html_str) blocks = html_to_blocks(clean, source_url=url) # Images images = [] for b in blocks: if b.get("type") == "image" and b.get("src"): images.append(b["src"]) # Text text_parts = [] for b in blocks: if b.get("content"): text_parts.append(b["content"]) return {"title": title or "Clipped page", "blocks": blocks, "images": images, "text": "\n\n".join(text_parts)} def _ensure_workspace(conn, user_id: int, workspace_id: int | None) -> tuple[int, str]: """Return (workspace_id, workspace_key) for clip insertion.""" if workspace_id: row = conn.execute("SELECT id, name FROM workspaces WHERE id=?", (workspace_id,)).fetchone() if row: # Check membership or owner mem = conn.execute( "SELECT 1 FROM workspace_members WHERE workspace_id=? AND user_id=?", (workspace_id, user_id) ).fetchone() if mem or conn.execute("SELECT 1 FROM workspaces WHERE id=? AND owner_id=?", (workspace_id, user_id)).fetchone(): return workspace_id, row["name"] # Fallback: first workspace owned or member, else create one row = conn.execute( "SELECT w.id, w.name FROM workspaces w LEFT JOIN workspace_members wm ON w.id=wm.workspace_id " "WHERE w.owner_id=? OR wm.user_id=? ORDER BY w.id LIMIT 1", (user_id, user_id), ).fetchone() if row: return row["id"], row["name"] # Create default workspace cur = conn.execute("INSERT INTO workspaces (name, owner_id) VALUES (?, ?)", ("My Workspace", user_id)) ws_id = cur.lastrowid conn.execute("INSERT INTO workspace_members (workspace_id, user_id, role) VALUES (?, ?, 'admin')", (ws_id, user_id)) return ws_id, "My Workspace" def create_page_from_clip(clip_data: dict, user_id: int) -> dict: """Create a FlowDeck page from a clip payload. Returns {page_id, title}.""" url = (clip_data.get("url") or clip_data.get("source_url") or "").strip() title = (clip_data.get("title") or "").strip() content = clip_data.get("content") or clip_data.get("html") or "" content_type = clip_data.get("content_type") or clip_data.get("clip_type") or "article" selection_html = clip_data.get("selection_html") or "" tags = clip_data.get("tags") or [] target_ws = clip_data.get("target_workspace_id") or clip_data.get("workspace_id") target_page_id = clip_data.get("target_page_id") or clip_data.get("parent_page_id") # Normalize workspace id try: target_ws = int(target_ws) if target_ws is not None else None except (ValueError, TypeError): target_ws = None # Determine title and blocks if content_type == "screenshot": # Screenshot: base64 image block + source bookmark (even with empty html) img_b64 = clip_data.get("image_base64") or clip_data.get("screenshot") or "" blocks = [] if img_b64: src = img_b64 if img_b64.startswith("data:") else f"data:image/png;base64,{img_b64}" blocks.append({"id": str(uuid.uuid4())[:8], "type": "image", "src": src, "alt": title or "Screenshot"}) if url: blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url}) if not blocks: blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": title or "Screenshot"}) title = title or "Screenshot" elif content_type == "selection" and selection_html and selection_html.strip(): clean = sanitize_html(selection_html) blocks = html_to_blocks(clean, source_url=url) if url: blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": title or url}) title = title or "Clipped selection" elif content_type == "bookmark" or not content.strip(): # Bookmark mode: no HTML body, just link card bookmark_title = title or (url or "Bookmark") blocks = [ {"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": bookmark_title, "description": clip_data.get("metadata", {}).get("og_description", "") or ""} ] if url: blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": f"Source: {url}"}) title = bookmark_title else: # Article full result = extract_article(content, url=url) blocks = result["blocks"] if not title: title = result["title"] # Append source bookmark if not already dominant if url: # avoid duplicate bookmark if last block already is bookmark to same url if not blocks or blocks[-1].get("url") != url: blocks.append({"id": str(uuid.uuid4())[:8], "type": "bookmark", "url": url, "title": "Source"}) blocks.append({"id": str(uuid.uuid4())[:8], "type": "paragraph", "content": url}) # Cap blocks if len(blocks) > 200: blocks = blocks[:200] title = (title or "Clipped page").strip()[:200] or "Clipped page" # Persist with get_conn() as conn: ws_id, ws_key = _ensure_workspace(conn, user_id, target_ws) # Validate target_page_id belongs to same workspace parent_id = None if target_page_id: try: pid = int(target_page_id) pr = conn.execute("SELECT id, workspace_id FROM pages WHERE id=? AND deleted_at IS NULL", (pid,)).fetchone() if pr and (pr["workspace_id"] == ws_id or pr["workspace_id"] is None): parent_id = pid except (ValueError, TypeError): pass # Determine sort order if parent_id is not None: next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE parent_id=?", (parent_id,)).fetchone()[0] else: next_order = conn.execute("SELECT COALESCE(MAX(sort_order), -1)+1 FROM pages WHERE workspace_id=? AND parent_id IS NULL", (ws_id,)).fetchone()[0] blocks_json = json.dumps(blocks, ensure_ascii=False) cur = conn.execute( """INSERT INTO pages (workspace, workspace_id, title, content, content_format, parent_section, parent_id, sort_order) VALUES (?, ?, ?, ?, 'blocks', 'Private', ?, ?)""", (ws_key or "", ws_id, title, blocks_json, parent_id, next_order), ) page_id = cur.lastrowid # Tags: create per-user tags if needed and attach via page_tags for tname in tags[:10]: tn = tname.strip()[:50] if not tn: continue conn.execute("INSERT OR IGNORE INTO tags (name, color, user_id) VALUES (?, '#787774', ?)", (tn, user_id)) tr = conn.execute("SELECT id FROM tags WHERE name=? AND user_id=?", (tn, user_id)).fetchone() if tr: conn.execute("INSERT OR IGNORE INTO page_tags (page_id, tag_id) VALUES (?, ?)", (page_id, tr["id"])) conn.commit() return {"page_id": page_id, "title": title, "blocks": blocks, "workspace_id": ws_id} def register_device(user_id: int, device_id: str, device_name: str = "", extension_name: str = "chrome") -> dict: """Register or update an extension device, returns {device, token} (token shown once if new).""" if not device_id or len(device_id) > 128: raise ValueError("Invalid device_id") token = f"fd_clip_{uuid.uuid4().hex}{uuid.uuid4().hex[:8]}" thash = _hash_token(token) with get_conn() as conn: existing = conn.execute( "SELECT id, token_hash FROM extension_devices WHERE user_id=? AND device_id=? AND extension_name=?", (user_id, device_id, extension_name), ).fetchone() if existing: conn.execute( "UPDATE extension_devices SET device_name=?, last_used_at=CURRENT_TIMESTAMP WHERE id=?", (device_name[:200], existing["id"]), ) conn.commit() return {"id": existing["id"], "device_id": device_id, "token": None, "existing": True} cur = conn.execute( """INSERT INTO extension_devices (user_id, extension_name, device_id, device_name, token_hash) VALUES (?, ?, ?, ?, ?)""", (user_id, extension_name, device_id, device_name[:200], thash), ) conn.commit() return {"id": cur.lastrowid, "device_id": device_id, "token": token, "existing": False} def verify_device_token(device_id: str, token: str) -> dict | None: """Verify a device token, returns device row or None.""" thash = _hash_token(token) with get_conn() as conn: row = conn.execute( "SELECT * FROM extension_devices WHERE device_id=? AND token_hash=? AND revoked=0", (device_id, thash), ).fetchone() if row: conn.execute("UPDATE extension_devices SET last_used_at=CURRENT_TIMESTAMP WHERE id=?", (row["id"],)) conn.commit() return dict(row) return None def log_clip(user_id: int, device_id: str, clip_type: str, source_url: str, target_page_id: int, workspace_id: int, title: str): with get_conn() as conn: conn.execute( """INSERT INTO extension_clips (user_id, device_id, clip_type, source_url, target_page_id, target_workspace_id, title) VALUES (?, ?, ?, ?, ?, ?, ?)""", (user_id, device_id, clip_type[:20], source_url[:2000], target_page_id, workspace_id, title[:200]), ) conn.commit() def list_devices(user_id: int) -> list[dict]: with get_conn() as conn: rows = conn.execute( "SELECT id, extension_name, device_id, device_name, scopes, last_used_at, created_at, revoked FROM extension_devices WHERE user_id=? ORDER BY last_used_at DESC, created_at DESC", (user_id,), ).fetchall() # Enrich with clip counts out = [] for r in rows: d = dict(r) cnt = conn.execute("SELECT COUNT(*) AS n FROM extension_clips WHERE user_id=? AND device_id=?", (user_id, r["device_id"])).fetchone()["n"] d["clips_count"] = cnt # Last clip last = conn.execute("SELECT created_at FROM extension_clips WHERE user_id=? AND device_id=? ORDER BY created_at DESC LIMIT 1", (user_id, r["device_id"])).fetchone() d["last_clip_at"] = last["created_at"] if last else None out.append(d) return out def revoke_device(user_id: int, device_row_id: int) -> bool: with get_conn() as conn: cur = conn.execute("UPDATE extension_devices SET revoked=1 WHERE id=? AND user_id=?", (device_row_id, user_id)) conn.commit() return cur.rowcount > 0