Files
bruno 0218d8f5e5
FlowDeck CI / lint (push) Successful in 1m37s
FlowDeck CI / test (push) Failing after 24m19s
FlowDeck CI / docker (push) Skipped
fix: /local-workspace — chemin du header + clic sur un dossier du sidebar (v7.69.9)
Corrige deux régressions de /local-workspace :

- le chemin du header restait bloqué sur « Home / <workspace> » quel que
  soit le dossier affiché : la route rend désormais breadcrumb_items
  (Home / <workspace> / <dossier> / <sous-dossier>, niveaux cliquables,
  collapse « … » au-delà de 4) et la navigation sans rechargement recalcule
  le chemin via l'event flowdeck:breadcrumb-changed ;
- le clic sur un dossier du sidebar affichait TOUS les composants à la
  fois : Alpine.data('wsInitData') retournait le même objet singleton, le
  2e montage (navigation partielle) levait « Cannot redefine property:
  \ » et initTree abandonnait, laissant tout le contenu au state
  brut. La factory retourne désormais une enveloppe fraîche par montage
  qui délègue à l'état réactif partagé. #lw-config est aussi relu à chaque
  exécution (le 2e montage gardait le folder_id du 1er chargement).

Inclus également le travail en cours de l'arbre : Library (colonnes Last
visited/Source, ordre d'en-tête, favoris à icônes Workspace), Meeting
Notes (bloc, CSS, routes, docs), coloration de code hljs, badges
favori/publié dans l'arbre local-workspace, docs (DATA_MODEL,
architectures) et tests associés.
2026-10-09 16:58:04 -04:00

733 lines
32 KiB
Python

"""FlowDeck — AI Meeting Notes façon Notion (v7.1.1).
Refonte alignée sur ``docs/architecture-meeting-notion-flowdeck.md`` (§3A/3B) :
machine à états du bloc, révélation progressive des onglets, consentement
bloquant journalisé, segments groupés par source, étapes Thinking persistées,
résumé structuré avec citations obligatoires, classification de complétude,
titre généré (diagnostic honnête si bref/incomplet), barre de partage tracée.
Stockage SQLite monolithique : ``meeting_transcripts`` + colonnes JSON
(migration 38) + ``meeting_consent_log`` / ``meeting_distributions``
append-only. Rien n'exige de dépendance pip supplémentaire.
"""
from __future__ import annotations
import json
import logging
import os
import re
import shutil
import subprocess
import time
from datetime import UTC, datetime
from pathlib import Path
from app.config import settings
from app.db import get_conn
logger = logging.getLogger(__name__)
AUDIO_EXTENSIONS = {"mp3", "wav", "m4a", "ogg", "flac", "aac"}
MAX_AUDIO_BYTES = 100 * 1024 * 1024
# États du bloc (§6.5) — transitions toujours via le serveur.
BLOCK_STATES = ("idle", "recording", "paused", "processing", "done", "failed")
CAPTURE_MODES = ("mic_only", "tab_plus_mic", "import")
# Instructions de résumé (§12.4) : auto + presets par type + custom libre.
INSTRUCTIONS = {
"auto": {"label": "Auto", "prompt": "Résumé automatique adapté au contenu détecté."},
"standup": {"label": "Standup", "prompt": "Format standup : hier / aujourd'hui / blocages, par personne."},
"team": {"label": "Team meeting", "prompt": "Format équipe : sujets, décisions, actions avec responsables."},
"sales": {"label": "Sales call", "prompt": "Format commercial : besoin, objections, next steps, montant."},
"one_to_one": {"label": "1:1", "prompt": "Format 1:1 : points du collaborateur, feedback, engagements."},
"interview": {"label": "Interview", "prompt": "Format entretien : parcours, forces, doutes, décision."},
}
# Étapes Thinking exposées (§12.7), dans l'ordre — libellés FR/EN gérés côté UI.
PROCESSING_STEPS = (
("reading_transcript", "Reading transcript…"),
("analyzing_transcript", "Analyzing the transcript"),
("resolving_references", "Thinking about how to refer to people"),
("understanding_content", "Understanding the content"),
("classifying_completeness", "Checking completeness"),
("generating_summary", "Writing the summary"),
("generating_title", "Choosing a title"),
("validating_citations", "Checking citations"),
)
CONSENT_METHODS = (
"start_attestation", "verbal", "chat_text", "audio_message",
"meet_addon", "workspace_enforced",
)
class TranscriptionUnavailable(RuntimeError):
"""Raised when no transcription backend is configured."""
def meetings_dir() -> Path:
root = Path(settings.data_dir)
d = root / "uploads" / "meetings"
d.mkdir(parents=True, exist_ok=True)
return d
def _now() -> str:
return datetime.now(UTC).strftime("%Y-%m-%d %H:%M:%S")
def _loads(raw: str, default):
try:
v = json.loads(raw or "")
return v if v is not None else default
except (TypeError, ValueError):
return default
def get_transcript(transcript_id: int) -> dict:
with get_conn() as conn:
row = conn.execute(
"SELECT * FROM meeting_transcripts WHERE id=?", (transcript_id,)
).fetchone()
if not row:
raise ValueError("transcript not found")
return dict(row)
def _update(transcript_id: int, **fields) -> None:
if not fields:
return
cols = ", ".join(f"{k}=?" for k in fields)
with get_conn() as conn:
conn.execute(
f"UPDATE meeting_transcripts SET {cols} WHERE id=?",
(*fields.values(), transcript_id),
)
conn.commit()
def save_transcript(page_id: int, transcript: str, language: str = "fr",
audio_path: str = "", **kw) -> int:
"""Crée une note de réunion (état ``idle``). Rétro-compatible v7.1.0."""
with get_conn() as conn:
if not conn.execute("SELECT id FROM pages WHERE id=?", (page_id,)).fetchone():
raise ValueError("page not found")
occ = (kw.get("event_occurrence_id") or "").strip()
if occ and conn.execute(
"SELECT id FROM meeting_transcripts WHERE event_occurrence_id=?", (occ,)
).fetchone():
raise ValueError("a meeting note already exists for this calendar occurrence")
cur = conn.execute(
"""INSERT INTO meeting_transcripts
(page_id, audio_path, transcript, language, status, mode,
instruction_id, event_occurrence_id, notes_text)
VALUES (?,?,?,?,?,?,?,?,?)""",
(page_id, audio_path, transcript or "", language or "fr",
"idle", kw.get("mode") or "mic_only",
(kw.get("instruction_id") or "auto")[:40], occ,
kw.get("notes_text") or ""),
)
conn.commit()
return cur.lastrowid
# ── Consentement (§14.1) : porte bloquante, journal append-only ──────────────
def record_consent(transcript_id: int, *, method: str = "start_attestation",
message_ref: str = "", attested_by: int | None = None,
participants: list | None = None) -> dict:
if method not in CONSENT_METHODS:
raise ValueError(f"unknown consent method: {method}")
get_transcript(transcript_id) # 404 si inconnu
entry = {
"method": method,
"message_ref": message_ref or "",
"attested_by": attested_by,
"participants": participants or [],
"attested_at": _now(),
}
with get_conn() as conn:
conn.execute(
"""INSERT INTO meeting_consent_log
(transcript_id, method, message_ref, attested_by, participants_json)
VALUES (?,?,?,?,?)""",
(transcript_id, method, message_ref or "",
attested_by, json.dumps(participants or [], ensure_ascii=False)),
)
conn.execute(
"UPDATE meeting_transcripts SET consent_json=? WHERE id=?",
(json.dumps(entry, ensure_ascii=False), transcript_id),
)
conn.commit()
return entry
def require_consent(tr: dict) -> dict:
consent = _loads(tr.get("consent_json") or "", None)
if consent:
return consent
with get_conn() as conn:
row = conn.execute(
"SELECT * FROM meeting_consent_log WHERE transcript_id=? "
"ORDER BY id DESC LIMIT 1", (tr["id"],)
).fetchone()
if not row:
raise ValueError("consent required — record consent before capturing")
return {
"method": row["method"], "message_ref": row["message_ref"],
"attested_by": row["attested_by"],
"participants": _loads(row["participants_json"], []),
}
# ── Sessions de capture : start / pause / resume / stop ─────────────────────
def start_session(transcript_id: int, *, mode: str = "mic_only",
channels: list | None = None, language: str = "fr",
instruction_id: str = "auto", user_id: int | None = None) -> dict:
if mode not in CAPTURE_MODES:
raise ValueError(f"unknown capture mode: {mode}")
tr = get_transcript(transcript_id)
if tr.get("status") in ("recording", "processing"):
raise ValueError(f"meeting is {tr['status']} — stop it first")
consent = require_consent(tr)
instr = (instruction_id or "auto").strip()[:40]
snapshot = {
"instruction_id": instr if instr in INSTRUCTIONS else "custom",
"instruction_raw": instr,
"definition": INSTRUCTIONS.get(instr, {"label": instr, "prompt": instr}),
"frozen_at": _now(),
}
chans = channels or (
[{"kind": "self", "label": "Microphone"}] if mode == "mic_only"
else [{"kind": "self", "label": "My microphone"},
{"kind": "remote", "label": "Tab audio"}] if mode == "tab_plus_mic"
else []
)
_update(transcript_id, status="recording", mode=mode,
channels_json=json.dumps(chans, ensure_ascii=False),
language=(language or tr.get("language") or "fr")[:10],
instruction_id=instr,
instruction_snapshot=json.dumps(snapshot, ensure_ascii=False),
segments_json="[]", processing_steps_json="[]", summary="",
summary_json="", generated_title="", completeness_class="",
duration_ms=0, paused_ms=0, started_at=_now(), stopped_at=None)
return get_transcript(transcript_id) | {"_consent": consent}
def pause_session(transcript_id: int, *, paused_ms_delta: int = 0) -> dict:
tr = get_transcript(transcript_id)
if tr.get("status") != "recording":
raise ValueError("only a recording session can be paused")
_update(transcript_id, status="paused",
paused_ms=int(tr.get("paused_ms") or 0) + max(0, int(paused_ms_delta or 0)))
return get_transcript(transcript_id)
def resume_session(transcript_id: int) -> dict:
tr = get_transcript(transcript_id)
if tr.get("status") != "paused":
raise ValueError("only a paused session can be resumed")
require_consent(tr)
_update(transcript_id, status="recording")
return get_transcript(transcript_id)
def stop_session(transcript_id: int, *, duration_ms: int = 0) -> dict:
tr = get_transcript(transcript_id)
if tr.get("status") not in ("recording", "paused"):
raise ValueError("nothing to stop — no live session")
require_consent(tr)
_update(transcript_id, status="processing", stopped_at=_now(),
duration_ms=max(int(duration_ms or 0), int(tr.get("duration_ms") or 0)))
return get_transcript(transcript_id)
# ── Segments : append (temps réel), correction du locuteur ───────────────────
def _next_seq(segments: list) -> int:
return (max((s.get("seq", 0) for s in segments), default=0) + 1) if segments else 1
def append_segments(transcript_id: int, segments: list) -> dict:
"""Persiste les segments finaux du client (STT navigateur / import).
Chaque item : {channel, source_label, speaker, text, start_ms, end_ms,
starts_mid, ends_mid}. Le transcript texte est reconstruit pour compat.
"""
tr = get_transcript(transcript_id)
# Retry/regen après un échec : le client reverse le manuel collé
# entre-temps — l'ajout reste possible hors session live.
if tr.get("status") not in ("recording", "paused", "processing",
"failed", "done"):
raise ValueError("session is not live — start recording first")
require_consent(tr)
if not isinstance(segments, list) or not segments:
raise ValueError("segments must be a non-empty list")
if len(segments) > 200:
raise ValueError("too many segments per batch (max 200)")
current = _loads(tr.get("segments_json") or "[]", [])
seq = _next_seq(current)
for item in segments:
if not isinstance(item, dict) or not (item.get("text") or "").strip():
raise ValueError("each segment needs a non-empty text")
text = item["text"].strip()[:4000]
current.append({
"seq": seq,
"channel": (item.get("channel") or "self")[:12],
"source_label": (item.get("source_label") or "")[:120],
"speaker": (item.get("speaker") or "")[:120],
"start_ms": max(0, int(item.get("start_ms") or 0)),
"end_ms": max(0, int(item.get("end_ms") or 0)),
"text": ("--" if item.get("starts_mid") else "") + text,
"is_final": True,
"starts_mid_utterance": bool(item.get("starts_mid")),
"ends_mid_utterance": bool(item.get("ends_mid")),
})
seq += 1
flat = "\n".join(
f"[{s['start_ms'] // 60000}:{(s['start_ms'] // 1000) % 60:02d}] "
f"{s['speaker'] or s['channel']} : {s['text']}" for s in current
)
if tr.get("transcript"):
flat = (tr["transcript"] + "\n" + flat) if flat else tr["transcript"]
_update(transcript_id, segments_json=json.dumps(current, ensure_ascii=False),
transcript=flat[:200000])
return {"transcript_id": transcript_id, "segments": len(current)}
def set_segment_speaker(transcript_id: int, seq: int, speaker: str,
user_id: int | None = None) -> dict:
tr = get_transcript(transcript_id)
segments = _loads(tr.get("segments_json") or "[]", [])
found = False
for s in segments:
if int(s.get("seq") or 0) == int(seq):
s["speaker"] = (speaker or "").strip()[:120]
s["speaker_corrected_by"] = user_id
found = True
if not found:
raise ValueError("segment not found")
_update(transcript_id, segments_json=json.dumps(segments, ensure_ascii=False))
return {"transcript_id": transcript_id, "seq": int(seq), "speaker": speaker}
def grouped_transcript(transcript_id: int) -> dict:
"""Transcript regroupé par piste source (§3A.7) + rail temporel."""
tr = get_transcript(transcript_id)
segments = _loads(tr.get("segments_json") or "[]", [])
groups: dict[str, list] = {}
order: list[str] = []
for s in sorted(segments, key=lambda x: (x.get("seq") or 0)):
key = s.get("source_label") or s.get("channel") or "self"
if key not in groups:
groups[key] = []
order.append(key)
groups[key].append(s)
# Repli : transcript texte brut découpé en un groupe unique.
if not groups and (tr.get("transcript") or "").strip():
groups = {"audio": [{"seq": 1, "channel": "self", "source_label": "",
"speaker": "", "start_ms": 0, "end_ms": 0,
"text": tr["transcript"][:4000]}]}
order = ["audio"]
return {"transcript_id": transcript_id,
"groups": [{"source": k, "segments": groups[k]} for k in order]}
# ── Classification de complétude + titre (§12.8) ─────────────────────────────
def classify_completeness(*, text: str, segments: list, notes: str,
duration_ms: int = 0) -> str:
words = len(re.findall(r"\S+", text or ""))
n_seg = len(segments or [])
truncated = any(s.get("starts_mid_utterance") or s.get("ends_mid_utterance")
for s in (segments or []))
has_agenda = bool((notes or "").strip())
if n_seg == 0 or words < 8:
return "fragmentary"
if words < 60 or (duration_ms and duration_ms < 45_000):
return "brief" if not truncated else "incomplete"
if truncated or not has_agenda and words < 150:
return "incomplete"
if words < 150:
return "brief"
return "complete"
def _first_topic(text: str) -> str:
sentences = re.split(r"(?<=[.!?])\s+", re.sub(r"\s+", " ", text or "").strip())
for s in sentences:
s = s.strip(" -–—\"'«»")
if len(s) > 25:
return s[:90].rstrip(" ,;:")
return ""
def generate_title(*, text: str, completeness: str, notes: str = "") -> str:
if completeness in ("brief", "incomplete", "fragmentary"):
topic = _first_topic(text)
if not topic or len(re.findall(r"\S+", text or "")) < 25:
return "Brief or Incomplete Meeting Recording"
return topic[:90]
topic = _first_topic(notes + " " + text)
return topic[:90] or "Meeting notes"
# ── Résumé structuré (§12.4) : JSON contraint + citations ────────────────────
_SUMMARY_SCHEMA_HINT = (
"Réponds STRICTEMENT par un objet JSON : "
'{"overview": ["..."], "decisions": ["..."], "action_items": '
'[{"title": "...", "owner": null, "due_hint": null}], '
'"open_questions": ["..."], "limits": ["..."]}. '
"Chaque puce overview/decisions/action_items DOIT citer le transcript "
"(suffixe entre parenthèses avec un extrait repris mot pour mot). "
"N'invente ni responsable ni échéance : owner/due_hint null si absents. "
"Si le contenu est insuffisant, overview décrit les manques et "
"action_items vaut []."
)
def _cite(text: str, segments: list) -> str:
"""Trouve un segment source pour une affirmation (citation vérifiable)."""
words = [w.lower() for w in re.findall(r"[A-Za-zÀ-ÿa-zà-ÿ']{4,}", text or "")]
best, best_score = None, 0
for s in segments or []:
hay = (s.get("text") or "").lower()
score = sum(1 for w in words[:12] if w in hay)
if score > best_score:
best, best_score = s, score
if best and best_score > 0:
return f" ⁠¹ [seg {best.get('seq')}]"
if segments:
return f" ⁠¹ [seg {segments[0].get('seq')}]"
return " ⁠¹"
def _offline_structured(*, text: str, notes: str, segments: list,
completeness: str) -> dict:
sentences = [s.strip() for s in
re.split(r"(?<=[.!?])\s+", re.sub(r"\s+", " ", text).strip()) if s.strip()]
overview: list[str] = []
if completeness == "fragmentary":
overview = [
"The recording captured only a short, fragmentary exchange with no "
f"discernible meeting topic, agenda, or context{_cite(text, segments)}",
"The content appears to be mid-conversation and lacks the beginning "
f"of the discussion{_cite(text, segments)}",
]
if not (notes or "").strip():
overview.append("The user's notes were empty, providing no additional "
"context about the meeting's purpose")
else:
for s in sentences[:4]:
overview.append(f"{s[:220]}{_cite(s, segments)}")
if not overview:
overview = [f"No meaningful content could be extracted{_cite(text, segments)}"]
if completeness in ("brief", "incomplete"):
overview.append(
"This was a brief or incomplete recording — treat this summary "
f"as partial{_cite(text, segments)}")
decisions = []
for s in sentences:
if re.search(r"\b(decid|décid|agreed|approuv|retain|retenu|chosen|choisi)\b", s, re.I):
decisions.append(f"{s[:220]}{_cite(s, segments)}")
actions = []
for s in sentences:
m = re.search(r"\b([A-ZÀ-Þ][a-zà-ÿ'\-]+)\s+(will|va|doit|should|must|s'occupe)\b", s)
if re.search(r"\b(action|todo|to-?do|next step|prochaine étape|faire|envoyer|préparer)\b", s, re.I) or m:
owner = m.group(1) if m and len(m.group(1)) > 2 else None
actions.append({"title": s[:180], "owner": owner,
"due_hint": None, "evidence": _cite(s, segments)})
return {"overview": overview[:6], "decisions": decisions[:8],
"action_items": actions[:10], "open_questions": [],
"limits": [] if completeness == "complete" else overview[-2:]}
def _render_markdown(struct: dict, completeness: str) -> str:
L = ["## Overview", ""]
for b in struct.get("overview") or []:
L.append(f"• {b}")
L += ["", "## Decisions", ""]
if struct.get("decisions"):
for b in struct["decisions"]:
L.append(f"• {b}")
else:
L.append("• No decisions could be identified from the available transcript")
L += ["", "## Action Items", ""]
if struct.get("action_items"):
for a in struct["action_items"]:
if isinstance(a, dict):
owner = f" — owner: {a['owner']}" if a.get("owner") else ""
L.append(f"• {a.get('title', '')}{owner}{a.get('evidence', ' ¹')}")
else:
L.append(f"• {a}")
else:
L.append("• No action items could be identified from the available transcript")
if struct.get("open_questions"):
L += ["", "## Open questions", ""]
for b in struct["open_questions"]:
L.append(f"• {b}")
if completeness in ("brief", "incomplete", "fragmentary"):
L += ["", f"_Summary class: {completeness} — partial recording, verify against the transcript._"]
return "\n".join(L).strip()
def _write_steps(transcript_id: int, states: dict) -> None:
steps = [{"key": k, "label": label,
"state": states.get(k, "pending"),
"at": _now() if states.get(k) in ("running", "done", "failed") else ""}
for k, label in PROCESSING_STEPS]
_update(transcript_id, processing_steps_json=json.dumps(steps, ensure_ascii=False))
async def _llm_structured(*, text: str, notes: str, instruction: str,
user_id: int | None) -> tuple[dict | None, bool, str]:
from app.services.ai_writing import AIWritingService
svc = AIWritingService(user_id=user_id)
if svc._offline_hint():
return None, True, ""
prompt = (
"Rédige le résumé structuré de cette réunion (format AI Meeting Note). "
f"Instructions de l'utilisateur : {instruction or 'Auto'}. "
"Notes humaines (contexte prioritaire, signale tout conflit) : "
f"{(notes or '(vides)')[:2000]}. " + _SUMMARY_SCHEMA_HINT
)
try:
res = await svc.run("summarize", context=text[:18000], prompt=prompt)
except Exception as exc: # noqa: BLE001
logger.warning("meeting LLM summary failed: %s", exc)
return None, False, str(exc)
if not res.get("ok") or not (res.get("text") or "").strip():
return None, bool(res.get("offline")), res.get("error") or "empty"
parsed = AIWritingService._parse_json_object(res["text"])
if not parsed or not isinstance(parsed.get("overview"), list):
# Le modèle a répondu en texte libre : on le conserve en overview sourcé.
return {"overview": [res["text"][:1500]], "decisions": [],
"action_items": [], "open_questions": [], "limits": []}, False, ""
return parsed, False, ""
async def run_processing(transcript_id: int, user_id: int | None = None,
allow_empty: bool = False) -> dict:
"""Pipeline §12 : étapes Thinking → classification → résumé → titre.
Jamais de refus sur contenu insuffisant : résumé dégradé + classe +
titre-diagnostic (§12.8). Avec ``allow_empty`` (arrêt sans rien de
capté), un diagnostic honnête est produit au lieu d'une erreur —
l'utilisateur ne perd ni l'enregistrement ni le transcript.
Sans ``allow_empty``, un transcript vide reste une 400 (l'appelant
affiche alors un guide de reprise, pas un dump HTTP).
"""
tr = get_transcript(transcript_id)
segments = _loads(tr.get("segments_json") or "[]", [])
notes = tr.get("notes_text") or ""
text = (tr.get("transcript") or "").strip()
snapshot = _loads(tr.get("instruction_snapshot") or "", {})
instruction = (snapshot.get("instruction_raw")
or tr.get("instruction_id") or "auto")
order = [k for k, _ in PROCESSING_STEPS]
states: dict[str, str] = {}
try:
for key in ("reading_transcript", "analyzing_transcript",
"resolving_references", "understanding_content"):
states[key] = "running"
_write_steps(transcript_id, states)
time.sleep(0) # point de cession : étapes réellement exécutées
states[key] = "done"
_write_steps(transcript_id, states)
states["classifying_completeness"] = "running"
_write_steps(transcript_id, states)
completeness = classify_completeness(
text=text, segments=segments, notes=notes,
duration_ms=int(tr.get("duration_ms") or 0))
states["classifying_completeness"] = "done"
_write_steps(transcript_id, states)
if not text:
if not allow_empty:
raise ValueError("transcript is empty — transcribe first")
# Arrêt sans rien de capté (micro coupé, aucun mot reconnu) :
# diagnostic honnête au lieu d'une erreur — rien n'est perdu,
# l'utilisateur peut coller un transcript et régénérer.
for key in ("generating_summary", "generating_title",
"validating_citations"):
states[key] = "running"
_write_steps(transcript_id, states)
states[key] = "done"
_write_steps(transcript_id, states)
struct = {
"overview": [
"No audio was captured during this session — no speech "
"reached the transcript. Check that the microphone is "
"allowed for this site and that the correct input is selected.",
("If the call ran in another tab, restart with “Tab + "
"microphone” and tick “Share audio” when choosing the tab."),
],
"decisions": [], "action_items": [], "open_questions": [],
"limits": ["empty session — paste a transcript, then Regenerate"],
}
md = _render_markdown(struct, "fragmentary")
_update(transcript_id, status="done", summary=md,
summary_json=json.dumps(struct, ensure_ascii=False),
generated_title="Brief or Incomplete Meeting Recording",
title_source="ai", completeness_class="fragmentary")
return await _finish_processing(transcript_id, offline=True)
states["generating_summary"] = "running"
_write_steps(transcript_id, states)
struct, offline, _err = await _llm_structured(
text=text, notes=notes, instruction=instruction, user_id=user_id)
if struct is None:
struct = _offline_structured(text=text, notes=notes,
segments=segments,
completeness=completeness)
offline = True
else:
# Garantir les citations même sur réponse LLM.
for section in ("overview", "decisions"):
fixed = []
for b in struct.get(section) or []:
s = b if isinstance(b, str) else str(b)
fixed.append(s if "¹" in s or "[seg" in s else f"{s}{_cite(s, segments)}")
struct[section] = fixed
acts = []
for a in struct.get("action_items") or []:
if isinstance(a, dict):
if "evidence" not in a:
a["evidence"] = _cite(a.get("title", ""), segments)
acts.append(a)
else:
acts.append({"title": str(a), "owner": None,
"due_hint": None,
"evidence": _cite(str(a), segments)})
struct["action_items"] = acts
states["generating_summary"] = "done"
_write_steps(transcript_id, states)
states["generating_title"] = "running"
_write_steps(transcript_id, states)
title = generate_title(text=text, completeness=completeness, notes=notes)
states["generating_title"] = "done"
_write_steps(transcript_id, states)
states["validating_citations"] = "running"
_write_steps(transcript_id, states)
seqs = {int(s.get("seq") or 0) for s in segments}
# Toute citation [seg N] pointe un segment réel, sinon repli seg 1.
md = _render_markdown(struct, completeness)
def _fix_seg(m: re.Match) -> str:
try:
n = int(m.group(1))
except (TypeError, ValueError):
return "[seg 1]"
return m.group(0) if n in seqs else "[seg 1]"
md = re.sub(r"\[seg (\d+)\]", _fix_seg, md)
states["validating_citations"] = "done"
_write_steps(transcript_id, states)
_update(transcript_id, status="done", summary=md,
summary_json=json.dumps(struct, ensure_ascii=False),
generated_title=title, title_source="ai",
completeness_class=completeness)
except Exception as exc: # noqa: BLE001 — état failed récupérable, segments conservés
for k in order:
states.setdefault(k, "pending")
# Marquer l'étape courante en échec.
for k in order:
if states.get(k) == "running":
states[k] = "failed"
break
_write_steps(transcript_id, states)
# Toujours sortir de « processing » : une ligne coincée est pire
# qu'une erreur affichée (l'utilisateur peut réessayer).
_update(transcript_id, status="failed")
msg = str(exc) or "processing failed"
if "transcript is empty" in msg:
raise ValueError(msg) from exc
logger.warning("meeting processing %s failed: %s", transcript_id, exc)
raise RuntimeError(msg) from exc
return await _finish_processing(transcript_id, offline)
async def _finish_processing(transcript_id: int, offline: bool) -> dict:
"""Événement + réponse du pipeline (chemin nominal et diagnostic)."""
out = get_transcript(transcript_id)
try:
from app.services.automations import fire_event
await fire_event("meeting.summarized", {
"page_id": out["page_id"], "transcript_id": transcript_id,
"language": out.get("language") or "fr",
"completeness": out.get("completeness_class") or "",
})
except Exception as exc: # noqa: BLE001
logger.debug("meeting.summarized dispatch failed: %s", exc)
return {"transcript_id": transcript_id, "summary": out["summary"],
"completeness": out.get("completeness_class") or "",
"generated_title": out.get("generated_title") or "",
"offline": bool(offline)}
def transcribe_audio(audio_path: str, language: str = "fr") -> str:
"""Transcribe with ``STT_COMMAND`` (``{cmd} {file}` → stdout text)."""
cmd_template = os.environ.get("STT_COMMAND", "").strip()
if not cmd_template:
raise TranscriptionUnavailable(
"no transcription backend (set STT_COMMAND or POST a manual transcript)")
if shutil.which(cmd_template.split()[0]) is None:
raise TranscriptionUnavailable(f"STT command not found: {cmd_template.split()[0]}")
try:
proc = subprocess.run(cmd_template.split() + [audio_path], # noqa: S603 — admin-configured
capture_output=True, text=True, timeout=600)
except subprocess.TimeoutExpired as exc:
raise TranscriptionUnavailable("transcription timed out") from exc
text = (proc.stdout or "").strip()
if proc.returncode != 0 or not text:
raise TranscriptionUnavailable(
f"transcription failed: {(proc.stderr or '')[:300]}")
return text
async def summarize_transcript(transcript_id: int, user_id: int | None = None,
allow_empty: bool = False) -> dict:
"""Point d'entrée historique (v7.1.0) — bascule sur le pipeline Notion."""
tr = get_transcript(transcript_id)
if tr.get("status") in ("recording", "paused"):
raise ValueError("stop the recording before generating the summary")
if not ((tr.get("transcript") or "").strip()
or _loads(tr.get("segments_json") or "[]", [])):
if not allow_empty:
raise ValueError("transcript is empty — transcribe first")
if tr.get("status") not in ("processing", "done"):
_update(transcript_id, status="processing")
out = await run_processing(transcript_id, user_id=user_id,
allow_empty=allow_empty)
return {"transcript_id": transcript_id, "summary": out["summary"],
"offline": out["offline"]}
# ── Partage tracé (§3A.5 : Copy link / Email / Slack) ─────────────────────────
def record_distribution(transcript_id: int, *, channel: str,
actor_id: int | None = None,
recipients: str = "", status: str = "created") -> dict:
if channel not in ("copy_link", "email", "slack"):
raise ValueError("channel must be copy_link|email|slack")
get_transcript(transcript_id)
with get_conn() as conn:
cur = conn.execute(
"""INSERT INTO meeting_distributions
(transcript_id, channel, actor_id, recipients_json, status)
VALUES (?,?,?,?,?)""",
(transcript_id, channel, actor_id, recipients or "", status),
)
conn.commit()
return {"id": cur.lastrowid, "transcript_id": transcript_id,
"channel": channel, "status": status}