Corrige deux régressions de /local-workspace :
- le chemin du header restait bloqué sur « Home / <workspace> » quel que
soit le dossier affiché : la route rend désormais breadcrumb_items
(Home / <workspace> / <dossier> / <sous-dossier>, niveaux cliquables,
collapse « … » au-delà de 4) et la navigation sans rechargement recalcule
le chemin via l'event flowdeck:breadcrumb-changed ;
- le clic sur un dossier du sidebar affichait TOUS les composants à la
fois : Alpine.data('wsInitData') retournait le même objet singleton, le
2e montage (navigation partielle) levait « Cannot redefine property:
\ » et initTree abandonnait, laissant tout le contenu au state
brut. La factory retourne désormais une enveloppe fraîche par montage
qui délègue à l'état réactif partagé. #lw-config est aussi relu à chaque
exécution (le 2e montage gardait le folder_id du 1er chargement).
Inclus également le travail en cours de l'arbre : Library (colonnes Last
visited/Source, ordre d'en-tête, favoris à icônes Workspace), Meeting
Notes (bloc, CSS, routes, docs), coloration de code hljs, badges
favori/publié dans l'arbre local-workspace, docs (DATA_MODEL,
architectures) et tests associés.
733 lines
32 KiB
Python
733 lines
32 KiB
Python
"""FlowDeck — AI Meeting Notes façon Notion (v7.1.1).
|
|
|
|
Refonte alignée sur ``docs/architecture-meeting-notion-flowdeck.md`` (§3A/3B) :
|
|
machine à états du bloc, révélation progressive des onglets, consentement
|
|
bloquant journalisé, segments groupés par source, étapes Thinking persistées,
|
|
résumé structuré avec citations obligatoires, classification de complétude,
|
|
titre généré (diagnostic honnête si bref/incomplet), barre de partage tracée.
|
|
|
|
Stockage SQLite monolithique : ``meeting_transcripts`` + colonnes JSON
|
|
(migration 38) + ``meeting_consent_log`` / ``meeting_distributions``
|
|
append-only. Rien n'exige de dépendance pip supplémentaire.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import time
|
|
from datetime import UTC, datetime
|
|
from pathlib import Path
|
|
|
|
from app.config import settings
|
|
from app.db import get_conn
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
AUDIO_EXTENSIONS = {"mp3", "wav", "m4a", "ogg", "flac", "aac"}
|
|
MAX_AUDIO_BYTES = 100 * 1024 * 1024
|
|
|
|
# États du bloc (§6.5) — transitions toujours via le serveur.
|
|
BLOCK_STATES = ("idle", "recording", "paused", "processing", "done", "failed")
|
|
CAPTURE_MODES = ("mic_only", "tab_plus_mic", "import")
|
|
|
|
# Instructions de résumé (§12.4) : auto + presets par type + custom libre.
|
|
INSTRUCTIONS = {
|
|
"auto": {"label": "Auto", "prompt": "Résumé automatique adapté au contenu détecté."},
|
|
"standup": {"label": "Standup", "prompt": "Format standup : hier / aujourd'hui / blocages, par personne."},
|
|
"team": {"label": "Team meeting", "prompt": "Format équipe : sujets, décisions, actions avec responsables."},
|
|
"sales": {"label": "Sales call", "prompt": "Format commercial : besoin, objections, next steps, montant."},
|
|
"one_to_one": {"label": "1:1", "prompt": "Format 1:1 : points du collaborateur, feedback, engagements."},
|
|
"interview": {"label": "Interview", "prompt": "Format entretien : parcours, forces, doutes, décision."},
|
|
}
|
|
|
|
# Étapes Thinking exposées (§12.7), dans l'ordre — libellés FR/EN gérés côté UI.
|
|
PROCESSING_STEPS = (
|
|
("reading_transcript", "Reading transcript…"),
|
|
("analyzing_transcript", "Analyzing the transcript"),
|
|
("resolving_references", "Thinking about how to refer to people"),
|
|
("understanding_content", "Understanding the content"),
|
|
("classifying_completeness", "Checking completeness"),
|
|
("generating_summary", "Writing the summary"),
|
|
("generating_title", "Choosing a title"),
|
|
("validating_citations", "Checking citations"),
|
|
)
|
|
|
|
CONSENT_METHODS = (
|
|
"start_attestation", "verbal", "chat_text", "audio_message",
|
|
"meet_addon", "workspace_enforced",
|
|
)
|
|
|
|
|
|
class TranscriptionUnavailable(RuntimeError):
|
|
"""Raised when no transcription backend is configured."""
|
|
|
|
|
|
def meetings_dir() -> Path:
|
|
root = Path(settings.data_dir)
|
|
d = root / "uploads" / "meetings"
|
|
d.mkdir(parents=True, exist_ok=True)
|
|
return d
|
|
|
|
|
|
def _now() -> str:
|
|
return datetime.now(UTC).strftime("%Y-%m-%d %H:%M:%S")
|
|
|
|
|
|
def _loads(raw: str, default):
|
|
try:
|
|
v = json.loads(raw or "")
|
|
return v if v is not None else default
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
|
|
def get_transcript(transcript_id: int) -> dict:
|
|
with get_conn() as conn:
|
|
row = conn.execute(
|
|
"SELECT * FROM meeting_transcripts WHERE id=?", (transcript_id,)
|
|
).fetchone()
|
|
if not row:
|
|
raise ValueError("transcript not found")
|
|
return dict(row)
|
|
|
|
|
|
def _update(transcript_id: int, **fields) -> None:
|
|
if not fields:
|
|
return
|
|
cols = ", ".join(f"{k}=?" for k in fields)
|
|
with get_conn() as conn:
|
|
conn.execute(
|
|
f"UPDATE meeting_transcripts SET {cols} WHERE id=?",
|
|
(*fields.values(), transcript_id),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def save_transcript(page_id: int, transcript: str, language: str = "fr",
|
|
audio_path: str = "", **kw) -> int:
|
|
"""Crée une note de réunion (état ``idle``). Rétro-compatible v7.1.0."""
|
|
with get_conn() as conn:
|
|
if not conn.execute("SELECT id FROM pages WHERE id=?", (page_id,)).fetchone():
|
|
raise ValueError("page not found")
|
|
occ = (kw.get("event_occurrence_id") or "").strip()
|
|
if occ and conn.execute(
|
|
"SELECT id FROM meeting_transcripts WHERE event_occurrence_id=?", (occ,)
|
|
).fetchone():
|
|
raise ValueError("a meeting note already exists for this calendar occurrence")
|
|
cur = conn.execute(
|
|
"""INSERT INTO meeting_transcripts
|
|
(page_id, audio_path, transcript, language, status, mode,
|
|
instruction_id, event_occurrence_id, notes_text)
|
|
VALUES (?,?,?,?,?,?,?,?,?)""",
|
|
(page_id, audio_path, transcript or "", language or "fr",
|
|
"idle", kw.get("mode") or "mic_only",
|
|
(kw.get("instruction_id") or "auto")[:40], occ,
|
|
kw.get("notes_text") or ""),
|
|
)
|
|
conn.commit()
|
|
return cur.lastrowid
|
|
|
|
|
|
# ── Consentement (§14.1) : porte bloquante, journal append-only ──────────────
|
|
|
|
def record_consent(transcript_id: int, *, method: str = "start_attestation",
|
|
message_ref: str = "", attested_by: int | None = None,
|
|
participants: list | None = None) -> dict:
|
|
if method not in CONSENT_METHODS:
|
|
raise ValueError(f"unknown consent method: {method}")
|
|
get_transcript(transcript_id) # 404 si inconnu
|
|
entry = {
|
|
"method": method,
|
|
"message_ref": message_ref or "",
|
|
"attested_by": attested_by,
|
|
"participants": participants or [],
|
|
"attested_at": _now(),
|
|
}
|
|
with get_conn() as conn:
|
|
conn.execute(
|
|
"""INSERT INTO meeting_consent_log
|
|
(transcript_id, method, message_ref, attested_by, participants_json)
|
|
VALUES (?,?,?,?,?)""",
|
|
(transcript_id, method, message_ref or "",
|
|
attested_by, json.dumps(participants or [], ensure_ascii=False)),
|
|
)
|
|
conn.execute(
|
|
"UPDATE meeting_transcripts SET consent_json=? WHERE id=?",
|
|
(json.dumps(entry, ensure_ascii=False), transcript_id),
|
|
)
|
|
conn.commit()
|
|
return entry
|
|
|
|
|
|
def require_consent(tr: dict) -> dict:
|
|
consent = _loads(tr.get("consent_json") or "", None)
|
|
if consent:
|
|
return consent
|
|
with get_conn() as conn:
|
|
row = conn.execute(
|
|
"SELECT * FROM meeting_consent_log WHERE transcript_id=? "
|
|
"ORDER BY id DESC LIMIT 1", (tr["id"],)
|
|
).fetchone()
|
|
if not row:
|
|
raise ValueError("consent required — record consent before capturing")
|
|
return {
|
|
"method": row["method"], "message_ref": row["message_ref"],
|
|
"attested_by": row["attested_by"],
|
|
"participants": _loads(row["participants_json"], []),
|
|
}
|
|
|
|
|
|
# ── Sessions de capture : start / pause / resume / stop ─────────────────────
|
|
|
|
def start_session(transcript_id: int, *, mode: str = "mic_only",
|
|
channels: list | None = None, language: str = "fr",
|
|
instruction_id: str = "auto", user_id: int | None = None) -> dict:
|
|
if mode not in CAPTURE_MODES:
|
|
raise ValueError(f"unknown capture mode: {mode}")
|
|
tr = get_transcript(transcript_id)
|
|
if tr.get("status") in ("recording", "processing"):
|
|
raise ValueError(f"meeting is {tr['status']} — stop it first")
|
|
consent = require_consent(tr)
|
|
instr = (instruction_id or "auto").strip()[:40]
|
|
snapshot = {
|
|
"instruction_id": instr if instr in INSTRUCTIONS else "custom",
|
|
"instruction_raw": instr,
|
|
"definition": INSTRUCTIONS.get(instr, {"label": instr, "prompt": instr}),
|
|
"frozen_at": _now(),
|
|
}
|
|
chans = channels or (
|
|
[{"kind": "self", "label": "Microphone"}] if mode == "mic_only"
|
|
else [{"kind": "self", "label": "My microphone"},
|
|
{"kind": "remote", "label": "Tab audio"}] if mode == "tab_plus_mic"
|
|
else []
|
|
)
|
|
_update(transcript_id, status="recording", mode=mode,
|
|
channels_json=json.dumps(chans, ensure_ascii=False),
|
|
language=(language or tr.get("language") or "fr")[:10],
|
|
instruction_id=instr,
|
|
instruction_snapshot=json.dumps(snapshot, ensure_ascii=False),
|
|
segments_json="[]", processing_steps_json="[]", summary="",
|
|
summary_json="", generated_title="", completeness_class="",
|
|
duration_ms=0, paused_ms=0, started_at=_now(), stopped_at=None)
|
|
return get_transcript(transcript_id) | {"_consent": consent}
|
|
|
|
|
|
def pause_session(transcript_id: int, *, paused_ms_delta: int = 0) -> dict:
|
|
tr = get_transcript(transcript_id)
|
|
if tr.get("status") != "recording":
|
|
raise ValueError("only a recording session can be paused")
|
|
_update(transcript_id, status="paused",
|
|
paused_ms=int(tr.get("paused_ms") or 0) + max(0, int(paused_ms_delta or 0)))
|
|
return get_transcript(transcript_id)
|
|
|
|
|
|
def resume_session(transcript_id: int) -> dict:
|
|
tr = get_transcript(transcript_id)
|
|
if tr.get("status") != "paused":
|
|
raise ValueError("only a paused session can be resumed")
|
|
require_consent(tr)
|
|
_update(transcript_id, status="recording")
|
|
return get_transcript(transcript_id)
|
|
|
|
|
|
def stop_session(transcript_id: int, *, duration_ms: int = 0) -> dict:
|
|
tr = get_transcript(transcript_id)
|
|
if tr.get("status") not in ("recording", "paused"):
|
|
raise ValueError("nothing to stop — no live session")
|
|
require_consent(tr)
|
|
_update(transcript_id, status="processing", stopped_at=_now(),
|
|
duration_ms=max(int(duration_ms or 0), int(tr.get("duration_ms") or 0)))
|
|
return get_transcript(transcript_id)
|
|
|
|
|
|
# ── Segments : append (temps réel), correction du locuteur ───────────────────
|
|
|
|
def _next_seq(segments: list) -> int:
|
|
return (max((s.get("seq", 0) for s in segments), default=0) + 1) if segments else 1
|
|
|
|
|
|
def append_segments(transcript_id: int, segments: list) -> dict:
|
|
"""Persiste les segments finaux du client (STT navigateur / import).
|
|
|
|
Chaque item : {channel, source_label, speaker, text, start_ms, end_ms,
|
|
starts_mid, ends_mid}. Le transcript texte est reconstruit pour compat.
|
|
"""
|
|
tr = get_transcript(transcript_id)
|
|
# Retry/regen après un échec : le client reverse le manuel collé
|
|
# entre-temps — l'ajout reste possible hors session live.
|
|
if tr.get("status") not in ("recording", "paused", "processing",
|
|
"failed", "done"):
|
|
raise ValueError("session is not live — start recording first")
|
|
require_consent(tr)
|
|
if not isinstance(segments, list) or not segments:
|
|
raise ValueError("segments must be a non-empty list")
|
|
if len(segments) > 200:
|
|
raise ValueError("too many segments per batch (max 200)")
|
|
current = _loads(tr.get("segments_json") or "[]", [])
|
|
seq = _next_seq(current)
|
|
for item in segments:
|
|
if not isinstance(item, dict) or not (item.get("text") or "").strip():
|
|
raise ValueError("each segment needs a non-empty text")
|
|
text = item["text"].strip()[:4000]
|
|
current.append({
|
|
"seq": seq,
|
|
"channel": (item.get("channel") or "self")[:12],
|
|
"source_label": (item.get("source_label") or "")[:120],
|
|
"speaker": (item.get("speaker") or "")[:120],
|
|
"start_ms": max(0, int(item.get("start_ms") or 0)),
|
|
"end_ms": max(0, int(item.get("end_ms") or 0)),
|
|
"text": ("--" if item.get("starts_mid") else "") + text,
|
|
"is_final": True,
|
|
"starts_mid_utterance": bool(item.get("starts_mid")),
|
|
"ends_mid_utterance": bool(item.get("ends_mid")),
|
|
})
|
|
seq += 1
|
|
flat = "\n".join(
|
|
f"[{s['start_ms'] // 60000}:{(s['start_ms'] // 1000) % 60:02d}] "
|
|
f"{s['speaker'] or s['channel']} : {s['text']}" for s in current
|
|
)
|
|
if tr.get("transcript"):
|
|
flat = (tr["transcript"] + "\n" + flat) if flat else tr["transcript"]
|
|
_update(transcript_id, segments_json=json.dumps(current, ensure_ascii=False),
|
|
transcript=flat[:200000])
|
|
return {"transcript_id": transcript_id, "segments": len(current)}
|
|
|
|
|
|
def set_segment_speaker(transcript_id: int, seq: int, speaker: str,
|
|
user_id: int | None = None) -> dict:
|
|
tr = get_transcript(transcript_id)
|
|
segments = _loads(tr.get("segments_json") or "[]", [])
|
|
found = False
|
|
for s in segments:
|
|
if int(s.get("seq") or 0) == int(seq):
|
|
s["speaker"] = (speaker or "").strip()[:120]
|
|
s["speaker_corrected_by"] = user_id
|
|
found = True
|
|
if not found:
|
|
raise ValueError("segment not found")
|
|
_update(transcript_id, segments_json=json.dumps(segments, ensure_ascii=False))
|
|
return {"transcript_id": transcript_id, "seq": int(seq), "speaker": speaker}
|
|
|
|
|
|
def grouped_transcript(transcript_id: int) -> dict:
|
|
"""Transcript regroupé par piste source (§3A.7) + rail temporel."""
|
|
tr = get_transcript(transcript_id)
|
|
segments = _loads(tr.get("segments_json") or "[]", [])
|
|
groups: dict[str, list] = {}
|
|
order: list[str] = []
|
|
for s in sorted(segments, key=lambda x: (x.get("seq") or 0)):
|
|
key = s.get("source_label") or s.get("channel") or "self"
|
|
if key not in groups:
|
|
groups[key] = []
|
|
order.append(key)
|
|
groups[key].append(s)
|
|
# Repli : transcript texte brut découpé en un groupe unique.
|
|
if not groups and (tr.get("transcript") or "").strip():
|
|
groups = {"audio": [{"seq": 1, "channel": "self", "source_label": "",
|
|
"speaker": "", "start_ms": 0, "end_ms": 0,
|
|
"text": tr["transcript"][:4000]}]}
|
|
order = ["audio"]
|
|
return {"transcript_id": transcript_id,
|
|
"groups": [{"source": k, "segments": groups[k]} for k in order]}
|
|
|
|
|
|
# ── Classification de complétude + titre (§12.8) ─────────────────────────────
|
|
|
|
def classify_completeness(*, text: str, segments: list, notes: str,
|
|
duration_ms: int = 0) -> str:
|
|
words = len(re.findall(r"\S+", text or ""))
|
|
n_seg = len(segments or [])
|
|
truncated = any(s.get("starts_mid_utterance") or s.get("ends_mid_utterance")
|
|
for s in (segments or []))
|
|
has_agenda = bool((notes or "").strip())
|
|
if n_seg == 0 or words < 8:
|
|
return "fragmentary"
|
|
if words < 60 or (duration_ms and duration_ms < 45_000):
|
|
return "brief" if not truncated else "incomplete"
|
|
if truncated or not has_agenda and words < 150:
|
|
return "incomplete"
|
|
if words < 150:
|
|
return "brief"
|
|
return "complete"
|
|
|
|
|
|
def _first_topic(text: str) -> str:
|
|
sentences = re.split(r"(?<=[.!?])\s+", re.sub(r"\s+", " ", text or "").strip())
|
|
for s in sentences:
|
|
s = s.strip(" -–—\"'«»")
|
|
if len(s) > 25:
|
|
return s[:90].rstrip(" ,;:")
|
|
return ""
|
|
|
|
|
|
def generate_title(*, text: str, completeness: str, notes: str = "") -> str:
|
|
if completeness in ("brief", "incomplete", "fragmentary"):
|
|
topic = _first_topic(text)
|
|
if not topic or len(re.findall(r"\S+", text or "")) < 25:
|
|
return "Brief or Incomplete Meeting Recording"
|
|
return topic[:90]
|
|
topic = _first_topic(notes + " " + text)
|
|
return topic[:90] or "Meeting notes"
|
|
|
|
|
|
# ── Résumé structuré (§12.4) : JSON contraint + citations ────────────────────
|
|
|
|
_SUMMARY_SCHEMA_HINT = (
|
|
"Réponds STRICTEMENT par un objet JSON : "
|
|
'{"overview": ["..."], "decisions": ["..."], "action_items": '
|
|
'[{"title": "...", "owner": null, "due_hint": null}], '
|
|
'"open_questions": ["..."], "limits": ["..."]}. '
|
|
"Chaque puce overview/decisions/action_items DOIT citer le transcript "
|
|
"(suffixe entre parenthèses avec un extrait repris mot pour mot). "
|
|
"N'invente ni responsable ni échéance : owner/due_hint null si absents. "
|
|
"Si le contenu est insuffisant, overview décrit les manques et "
|
|
"action_items vaut []."
|
|
)
|
|
|
|
|
|
def _cite(text: str, segments: list) -> str:
|
|
"""Trouve un segment source pour une affirmation (citation vérifiable)."""
|
|
words = [w.lower() for w in re.findall(r"[A-Za-zÀ-ÿa-zà-ÿ']{4,}", text or "")]
|
|
best, best_score = None, 0
|
|
for s in segments or []:
|
|
hay = (s.get("text") or "").lower()
|
|
score = sum(1 for w in words[:12] if w in hay)
|
|
if score > best_score:
|
|
best, best_score = s, score
|
|
if best and best_score > 0:
|
|
return f" ¹ [seg {best.get('seq')}]"
|
|
if segments:
|
|
return f" ¹ [seg {segments[0].get('seq')}]"
|
|
return " ¹"
|
|
|
|
|
|
def _offline_structured(*, text: str, notes: str, segments: list,
|
|
completeness: str) -> dict:
|
|
sentences = [s.strip() for s in
|
|
re.split(r"(?<=[.!?])\s+", re.sub(r"\s+", " ", text).strip()) if s.strip()]
|
|
overview: list[str] = []
|
|
if completeness == "fragmentary":
|
|
overview = [
|
|
"The recording captured only a short, fragmentary exchange with no "
|
|
f"discernible meeting topic, agenda, or context{_cite(text, segments)}",
|
|
"The content appears to be mid-conversation and lacks the beginning "
|
|
f"of the discussion{_cite(text, segments)}",
|
|
]
|
|
if not (notes or "").strip():
|
|
overview.append("The user's notes were empty, providing no additional "
|
|
"context about the meeting's purpose")
|
|
else:
|
|
for s in sentences[:4]:
|
|
overview.append(f"{s[:220]}{_cite(s, segments)}")
|
|
if not overview:
|
|
overview = [f"No meaningful content could be extracted{_cite(text, segments)}"]
|
|
if completeness in ("brief", "incomplete"):
|
|
overview.append(
|
|
"This was a brief or incomplete recording — treat this summary "
|
|
f"as partial{_cite(text, segments)}")
|
|
decisions = []
|
|
for s in sentences:
|
|
if re.search(r"\b(decid|décid|agreed|approuv|retain|retenu|chosen|choisi)\b", s, re.I):
|
|
decisions.append(f"{s[:220]}{_cite(s, segments)}")
|
|
actions = []
|
|
for s in sentences:
|
|
m = re.search(r"\b([A-ZÀ-Þ][a-zà-ÿ'\-]+)\s+(will|va|doit|should|must|s'occupe)\b", s)
|
|
if re.search(r"\b(action|todo|to-?do|next step|prochaine étape|faire|envoyer|préparer)\b", s, re.I) or m:
|
|
owner = m.group(1) if m and len(m.group(1)) > 2 else None
|
|
actions.append({"title": s[:180], "owner": owner,
|
|
"due_hint": None, "evidence": _cite(s, segments)})
|
|
return {"overview": overview[:6], "decisions": decisions[:8],
|
|
"action_items": actions[:10], "open_questions": [],
|
|
"limits": [] if completeness == "complete" else overview[-2:]}
|
|
|
|
|
|
def _render_markdown(struct: dict, completeness: str) -> str:
|
|
L = ["## Overview", ""]
|
|
for b in struct.get("overview") or []:
|
|
L.append(f"• {b}")
|
|
L += ["", "## Decisions", ""]
|
|
if struct.get("decisions"):
|
|
for b in struct["decisions"]:
|
|
L.append(f"• {b}")
|
|
else:
|
|
L.append("• No decisions could be identified from the available transcript")
|
|
L += ["", "## Action Items", ""]
|
|
if struct.get("action_items"):
|
|
for a in struct["action_items"]:
|
|
if isinstance(a, dict):
|
|
owner = f" — owner: {a['owner']}" if a.get("owner") else ""
|
|
L.append(f"• {a.get('title', '')}{owner}{a.get('evidence', ' ¹')}")
|
|
else:
|
|
L.append(f"• {a}")
|
|
else:
|
|
L.append("• No action items could be identified from the available transcript")
|
|
if struct.get("open_questions"):
|
|
L += ["", "## Open questions", ""]
|
|
for b in struct["open_questions"]:
|
|
L.append(f"• {b}")
|
|
if completeness in ("brief", "incomplete", "fragmentary"):
|
|
L += ["", f"_Summary class: {completeness} — partial recording, verify against the transcript._"]
|
|
return "\n".join(L).strip()
|
|
|
|
|
|
def _write_steps(transcript_id: int, states: dict) -> None:
|
|
steps = [{"key": k, "label": label,
|
|
"state": states.get(k, "pending"),
|
|
"at": _now() if states.get(k) in ("running", "done", "failed") else ""}
|
|
for k, label in PROCESSING_STEPS]
|
|
_update(transcript_id, processing_steps_json=json.dumps(steps, ensure_ascii=False))
|
|
|
|
|
|
async def _llm_structured(*, text: str, notes: str, instruction: str,
|
|
user_id: int | None) -> tuple[dict | None, bool, str]:
|
|
from app.services.ai_writing import AIWritingService
|
|
|
|
svc = AIWritingService(user_id=user_id)
|
|
if svc._offline_hint():
|
|
return None, True, ""
|
|
prompt = (
|
|
"Rédige le résumé structuré de cette réunion (format AI Meeting Note). "
|
|
f"Instructions de l'utilisateur : {instruction or 'Auto'}. "
|
|
"Notes humaines (contexte prioritaire, signale tout conflit) : "
|
|
f"{(notes or '(vides)')[:2000]}. " + _SUMMARY_SCHEMA_HINT
|
|
)
|
|
try:
|
|
res = await svc.run("summarize", context=text[:18000], prompt=prompt)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.warning("meeting LLM summary failed: %s", exc)
|
|
return None, False, str(exc)
|
|
if not res.get("ok") or not (res.get("text") or "").strip():
|
|
return None, bool(res.get("offline")), res.get("error") or "empty"
|
|
parsed = AIWritingService._parse_json_object(res["text"])
|
|
if not parsed or not isinstance(parsed.get("overview"), list):
|
|
# Le modèle a répondu en texte libre : on le conserve en overview sourcé.
|
|
return {"overview": [res["text"][:1500]], "decisions": [],
|
|
"action_items": [], "open_questions": [], "limits": []}, False, ""
|
|
return parsed, False, ""
|
|
|
|
|
|
async def run_processing(transcript_id: int, user_id: int | None = None,
|
|
allow_empty: bool = False) -> dict:
|
|
"""Pipeline §12 : étapes Thinking → classification → résumé → titre.
|
|
|
|
Jamais de refus sur contenu insuffisant : résumé dégradé + classe +
|
|
titre-diagnostic (§12.8). Avec ``allow_empty`` (arrêt sans rien de
|
|
capté), un diagnostic honnête est produit au lieu d'une erreur —
|
|
l'utilisateur ne perd ni l'enregistrement ni le transcript.
|
|
Sans ``allow_empty``, un transcript vide reste une 400 (l'appelant
|
|
affiche alors un guide de reprise, pas un dump HTTP).
|
|
"""
|
|
tr = get_transcript(transcript_id)
|
|
segments = _loads(tr.get("segments_json") or "[]", [])
|
|
notes = tr.get("notes_text") or ""
|
|
text = (tr.get("transcript") or "").strip()
|
|
snapshot = _loads(tr.get("instruction_snapshot") or "", {})
|
|
instruction = (snapshot.get("instruction_raw")
|
|
or tr.get("instruction_id") or "auto")
|
|
|
|
order = [k for k, _ in PROCESSING_STEPS]
|
|
states: dict[str, str] = {}
|
|
try:
|
|
for key in ("reading_transcript", "analyzing_transcript",
|
|
"resolving_references", "understanding_content"):
|
|
states[key] = "running"
|
|
_write_steps(transcript_id, states)
|
|
time.sleep(0) # point de cession : étapes réellement exécutées
|
|
states[key] = "done"
|
|
_write_steps(transcript_id, states)
|
|
|
|
states["classifying_completeness"] = "running"
|
|
_write_steps(transcript_id, states)
|
|
completeness = classify_completeness(
|
|
text=text, segments=segments, notes=notes,
|
|
duration_ms=int(tr.get("duration_ms") or 0))
|
|
states["classifying_completeness"] = "done"
|
|
_write_steps(transcript_id, states)
|
|
|
|
if not text:
|
|
if not allow_empty:
|
|
raise ValueError("transcript is empty — transcribe first")
|
|
# Arrêt sans rien de capté (micro coupé, aucun mot reconnu) :
|
|
# diagnostic honnête au lieu d'une erreur — rien n'est perdu,
|
|
# l'utilisateur peut coller un transcript et régénérer.
|
|
for key in ("generating_summary", "generating_title",
|
|
"validating_citations"):
|
|
states[key] = "running"
|
|
_write_steps(transcript_id, states)
|
|
states[key] = "done"
|
|
_write_steps(transcript_id, states)
|
|
struct = {
|
|
"overview": [
|
|
"No audio was captured during this session — no speech "
|
|
"reached the transcript. Check that the microphone is "
|
|
"allowed for this site and that the correct input is selected.",
|
|
("If the call ran in another tab, restart with “Tab + "
|
|
"microphone” and tick “Share audio” when choosing the tab."),
|
|
],
|
|
"decisions": [], "action_items": [], "open_questions": [],
|
|
"limits": ["empty session — paste a transcript, then Regenerate"],
|
|
}
|
|
md = _render_markdown(struct, "fragmentary")
|
|
_update(transcript_id, status="done", summary=md,
|
|
summary_json=json.dumps(struct, ensure_ascii=False),
|
|
generated_title="Brief or Incomplete Meeting Recording",
|
|
title_source="ai", completeness_class="fragmentary")
|
|
return await _finish_processing(transcript_id, offline=True)
|
|
|
|
states["generating_summary"] = "running"
|
|
_write_steps(transcript_id, states)
|
|
struct, offline, _err = await _llm_structured(
|
|
text=text, notes=notes, instruction=instruction, user_id=user_id)
|
|
if struct is None:
|
|
struct = _offline_structured(text=text, notes=notes,
|
|
segments=segments,
|
|
completeness=completeness)
|
|
offline = True
|
|
else:
|
|
# Garantir les citations même sur réponse LLM.
|
|
for section in ("overview", "decisions"):
|
|
fixed = []
|
|
for b in struct.get(section) or []:
|
|
s = b if isinstance(b, str) else str(b)
|
|
fixed.append(s if "¹" in s or "[seg" in s else f"{s}{_cite(s, segments)}")
|
|
struct[section] = fixed
|
|
acts = []
|
|
for a in struct.get("action_items") or []:
|
|
if isinstance(a, dict):
|
|
if "evidence" not in a:
|
|
a["evidence"] = _cite(a.get("title", ""), segments)
|
|
acts.append(a)
|
|
else:
|
|
acts.append({"title": str(a), "owner": None,
|
|
"due_hint": None,
|
|
"evidence": _cite(str(a), segments)})
|
|
struct["action_items"] = acts
|
|
states["generating_summary"] = "done"
|
|
_write_steps(transcript_id, states)
|
|
|
|
states["generating_title"] = "running"
|
|
_write_steps(transcript_id, states)
|
|
title = generate_title(text=text, completeness=completeness, notes=notes)
|
|
states["generating_title"] = "done"
|
|
_write_steps(transcript_id, states)
|
|
|
|
states["validating_citations"] = "running"
|
|
_write_steps(transcript_id, states)
|
|
seqs = {int(s.get("seq") or 0) for s in segments}
|
|
# Toute citation [seg N] pointe un segment réel, sinon repli seg 1.
|
|
md = _render_markdown(struct, completeness)
|
|
|
|
def _fix_seg(m: re.Match) -> str:
|
|
try:
|
|
n = int(m.group(1))
|
|
except (TypeError, ValueError):
|
|
return "[seg 1]"
|
|
return m.group(0) if n in seqs else "[seg 1]"
|
|
md = re.sub(r"\[seg (\d+)\]", _fix_seg, md)
|
|
states["validating_citations"] = "done"
|
|
_write_steps(transcript_id, states)
|
|
|
|
_update(transcript_id, status="done", summary=md,
|
|
summary_json=json.dumps(struct, ensure_ascii=False),
|
|
generated_title=title, title_source="ai",
|
|
completeness_class=completeness)
|
|
except Exception as exc: # noqa: BLE001 — état failed récupérable, segments conservés
|
|
for k in order:
|
|
states.setdefault(k, "pending")
|
|
# Marquer l'étape courante en échec.
|
|
for k in order:
|
|
if states.get(k) == "running":
|
|
states[k] = "failed"
|
|
break
|
|
_write_steps(transcript_id, states)
|
|
# Toujours sortir de « processing » : une ligne coincée est pire
|
|
# qu'une erreur affichée (l'utilisateur peut réessayer).
|
|
_update(transcript_id, status="failed")
|
|
msg = str(exc) or "processing failed"
|
|
if "transcript is empty" in msg:
|
|
raise ValueError(msg) from exc
|
|
logger.warning("meeting processing %s failed: %s", transcript_id, exc)
|
|
raise RuntimeError(msg) from exc
|
|
|
|
return await _finish_processing(transcript_id, offline)
|
|
|
|
|
|
async def _finish_processing(transcript_id: int, offline: bool) -> dict:
|
|
"""Événement + réponse du pipeline (chemin nominal et diagnostic)."""
|
|
out = get_transcript(transcript_id)
|
|
try:
|
|
from app.services.automations import fire_event
|
|
await fire_event("meeting.summarized", {
|
|
"page_id": out["page_id"], "transcript_id": transcript_id,
|
|
"language": out.get("language") or "fr",
|
|
"completeness": out.get("completeness_class") or "",
|
|
})
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.debug("meeting.summarized dispatch failed: %s", exc)
|
|
return {"transcript_id": transcript_id, "summary": out["summary"],
|
|
"completeness": out.get("completeness_class") or "",
|
|
"generated_title": out.get("generated_title") or "",
|
|
"offline": bool(offline)}
|
|
|
|
|
|
def transcribe_audio(audio_path: str, language: str = "fr") -> str:
|
|
"""Transcribe with ``STT_COMMAND`` (``{cmd} {file}` → stdout text)."""
|
|
cmd_template = os.environ.get("STT_COMMAND", "").strip()
|
|
if not cmd_template:
|
|
raise TranscriptionUnavailable(
|
|
"no transcription backend (set STT_COMMAND or POST a manual transcript)")
|
|
if shutil.which(cmd_template.split()[0]) is None:
|
|
raise TranscriptionUnavailable(f"STT command not found: {cmd_template.split()[0]}")
|
|
try:
|
|
proc = subprocess.run(cmd_template.split() + [audio_path], # noqa: S603 — admin-configured
|
|
capture_output=True, text=True, timeout=600)
|
|
except subprocess.TimeoutExpired as exc:
|
|
raise TranscriptionUnavailable("transcription timed out") from exc
|
|
text = (proc.stdout or "").strip()
|
|
if proc.returncode != 0 or not text:
|
|
raise TranscriptionUnavailable(
|
|
f"transcription failed: {(proc.stderr or '')[:300]}")
|
|
return text
|
|
|
|
|
|
async def summarize_transcript(transcript_id: int, user_id: int | None = None,
|
|
allow_empty: bool = False) -> dict:
|
|
"""Point d'entrée historique (v7.1.0) — bascule sur le pipeline Notion."""
|
|
tr = get_transcript(transcript_id)
|
|
if tr.get("status") in ("recording", "paused"):
|
|
raise ValueError("stop the recording before generating the summary")
|
|
if not ((tr.get("transcript") or "").strip()
|
|
or _loads(tr.get("segments_json") or "[]", [])):
|
|
if not allow_empty:
|
|
raise ValueError("transcript is empty — transcribe first")
|
|
if tr.get("status") not in ("processing", "done"):
|
|
_update(transcript_id, status="processing")
|
|
out = await run_processing(transcript_id, user_id=user_id,
|
|
allow_empty=allow_empty)
|
|
return {"transcript_id": transcript_id, "summary": out["summary"],
|
|
"offline": out["offline"]}
|
|
|
|
|
|
# ── Partage tracé (§3A.5 : Copy link / Email / Slack) ─────────────────────────
|
|
|
|
def record_distribution(transcript_id: int, *, channel: str,
|
|
actor_id: int | None = None,
|
|
recipients: str = "", status: str = "created") -> dict:
|
|
if channel not in ("copy_link", "email", "slack"):
|
|
raise ValueError("channel must be copy_link|email|slack")
|
|
get_transcript(transcript_id)
|
|
with get_conn() as conn:
|
|
cur = conn.execute(
|
|
"""INSERT INTO meeting_distributions
|
|
(transcript_id, channel, actor_id, recipients_json, status)
|
|
VALUES (?,?,?,?,?)""",
|
|
(transcript_id, channel, actor_id, recipients or "", status),
|
|
)
|
|
conn.commit()
|
|
return {"id": cur.lastrowid, "transcript_id": transcript_id,
|
|
"channel": channel, "status": status}
|