feat(groups): ratissage élargi (3 lettres, sous-chaînes, singletons, plafond, copie inclassables)

This commit is contained in:
2026-09-27 14:05:08 -04:00
parent 63a2bddf47
commit a2b4139887
3 changed files with 103 additions and 24 deletions
@@ -41,7 +41,10 @@
<div class="flex flex-wrap items-center justify-between gap-2">
<div>
<h3 class="font-semibold text-slate-100">✨ Groupes suggérés</h3>
<p class="text-xs text-slate-400">Noms de chaînes + sujets détectés. Vérifiez avant d’appliquer.</p>
<p class="text-xs text-slate-400">
Noms de chaînes + sujets détectés. Vérifiez avant d’appliquer.
<span *ngIf="ungroupedCount() > 0" class="text-amber-200/90">{{ ungroupedCount() }} sans groupe.</span>
</p>
</div>
<div class="flex flex-wrap gap-2">
<button
@@ -79,7 +82,15 @@
</div>
</div>
<div *ngIf="suggestions().length === 0" class="mt-3 rounded-lg border border-dashed border-slate-700 p-4 text-center text-sm text-slate-400">
Aucune suggestion : chaque thème a moins de 2 chaînes, ou tout est déjà rangé.
<p>Aucune suggestion : chaque thème a moins de 2 chaînes, ou tout est déjà rangé.</p>
<button
*ngIf="ungroupedCount() > 0"
type="button"
(click)="copyUngrouped()"
class="mt-2 rounded-md border border-slate-600 px-3 py-1.5 text-xs text-slate-200 transition hover:border-slate-400 hover:text-white"
>
{{ copiedUngrouped() ? 'Copié ! Collez la liste pour enrichir les règles.' : 'Copier les ' + ungroupedCount() + ' sans-groupe' }}
</button>
</div>
<ul *ngIf="suggestions().length > 0" class="mt-3 space-y-2">
<li *ngFor="let s of suggestions(); trackBy: trackBySuggestion" class="flex flex-col gap-2 rounded-lg border border-slate-800 bg-slate-900/60 p-3 sm:flex-row sm:items-center sm:justify-between">
@@ -219,6 +219,45 @@ export class SubscriptionsComponent {
}
}
copiedUngrouped = signal<boolean>(false);
/** Copie les chaînes sans groupe (une par ligne) pour analyse des règles. */
copyUngrouped(): void {
const lines = this.items()
.filter((item) => this.itemGroupIds(item).length === 0)
.map((item) => item.channel?.title || item.channel?.handle || `#${item.subscriptionId}`);
const text = lines.join('\n');
if (!text) return;
const done = () => {
this.copiedUngrouped.set(true);
setTimeout(() => this.copiedUngrouped.set(false), 2000);
};
try {
const nav = navigator as Navigator & { clipboard?: { writeText(t: string): Promise<void> } };
if (nav?.clipboard?.writeText) {
nav.clipboard.writeText(text).then(done, () => this.copyUngroupedFallback(text, done));
} else {
this.copyUngroupedFallback(text, done);
}
} catch {
this.copyUngroupedFallback(text, done);
}
}
private copyUngroupedFallback(text: string, done: () => void): void {
try {
const ta = document.createElement('textarea');
ta.value = text;
ta.style.position = 'fixed';
ta.style.opacity = '0';
document.body.appendChild(ta);
ta.select();
document.execCommand('copy');
document.body.removeChild(ta);
done();
} catch {}
}
// --- trackBy : DOM stable (pas de re-rendu complet + recharge d'images à chaque tick) ---
trackBySubscriptionId(_idx: number, item: SubscriptionItem): number {
return item.subscriptionId;
+51 -22
View File
@@ -223,23 +223,28 @@ export function normalizeChannelText(value: string | null | undefined): string {
.toLowerCase();
}
/** Mots vides FR/EN + termes YouTube génériques (évitent les méga-groupes inutiles). */
/**
* Mots vides FR/EN + jargon plateforme (jamais des thèmes utiles).
* Géographie et génériques (québec, tv, top...) restants VOLONTAIREMENT :
* ils forment des groupes bornés (plafond anti-méga-groupes ci-dessous).
*/
const TOPIC_STOPWORDS = new Set([
'avec', 'dans', 'pour', 'sans', 'sous', 'vers', 'chez', 'entre', 'contre', 'depuis',
'cette', 'celui', 'celle', 'ceux', 'celles', 'notre', 'votre', 'leur', 'leurs', 'tous',
'toute', 'toutes', 'aussi', 'encore', 'tres', 'bien', 'plus', 'moins', 'comme', 'mais',
'donc', 'alors', 'quand', 'comment', 'pourquoi', 'parce', 'sont', 'etes', 'font', 'vont',
'cest', 'nest', 'plus', 'tout', 'tous', 'aux', 'ces', 'ses', 'mon', 'mes', 'ton', 'tes',
'son', 'nous', 'vous', 'ils', 'elles', 'elle', 'celui', 'dont', 'ou', 'parce',
'the', 'and', 'for', 'with', 'from', 'your', 'you', 'your', 'this', 'that', 'these',
'cest', 'nest', 'tout', 'aux', 'ces', 'ses', 'mon', 'mes', 'ton', 'tes',
'son', 'nous', 'vous', 'ils', 'elles', 'elle', 'dont',
'the', 'and', 'for', 'with', 'from', 'your', 'you', 'this', 'that', 'these',
'those', 'are', 'was', 'were', 'been', 'have', 'has', 'will', 'would', 'there',
'their', 'what', 'when', 'where', 'which', 'while', 'about', 'into', 'over',
'after', 'before', 'your', 'videos', 'video', 'tube', 'youtube', 'channel', 'chaine',
'chanel', 'official', 'officiel', 'tv', 'media', 'network', 'studio', 'production',
'streaming', 'live', 'clips', 'shorts', 'france', 'francais', 'quebec', 'canada',
'montreal', 'paris', 'best', 'top', 'new', 'show', 'central', 'daily', 'hub', 'one',
'after', 'before', 'tube', 'youtube', 'channel', 'chaine', 'chanel',
'official', 'officiel', 'network', 'production', 'streaming',
]);
/** Plafond : au-delà, le sujet est trop générique pour être utile. */
const TOPIC_MAX_SIZE = 60;
/** Devine une icône pour un mot (recherche inverse dans les mots-clés). */
export function guessIconForWord(word: string): string {
const w = normalizeChannelText(word).trim();
@@ -257,18 +262,28 @@ export function guessIconForWord(word: string): string {
* Pur et testable : ne touche pas au réseau.
*/
export function detectEmergentTopics(
items: Array<{ subscriptionId: number; title: string; handle?: string | null; extra?: string }>,
items: Array<{ subscriptionId: number; title: string; handle?: string | null; extra?: string; groupIds?: string[] }>,
opts?: { minSize?: number; maxGroups?: number; excludeNames?: string[] },
): GroupSuggestion[] {
const minSize = Math.max(2, opts?.minSize || 2);
const maxGroups = Math.max(1, opts?.maxGroups || 10);
const maxGroups = Math.max(1, opts?.maxGroups || 15);
const excluded = new Set((opts?.excludeNames || []).map((n) => normalizeChannelText(n).trim()));
// Candidats : mots entiers ≥3 lettres (tokenisés, donc 'rap' ≠ 'drapeau').
const byWord = new Map<string, typeof items>();
for (const item of items) {
const hay = normalizeChannelText(`${item.title || ''} ${item.handle || ''} ${item.extra || ''}`);
const words = new Set(
hay.split(/[^a-z0-9]+/).filter((w) => w.length >= 4 && !/^\d+$/.test(w) && !TOPIC_STOPWORDS.has(w)),
);
const tokens = hay.split(/[^a-z0-9]+/).filter((w) => w.length >= 3 && !/^\d+$/.test(w) && !TOPIC_STOPWORDS.has(w));
const words = new Set<string>();
for (const t of tokens) {
words.add(t);
// Sous-chaînes longues : 'hockey' dans 'hockeyqc', 'tech' dans 'technology'.
if (t.length >= 8) {
for (let len = 5; len <= t.length - 2; len++) {
const sub = t.slice(0, len);
if (!TOPIC_STOPWORDS.has(sub)) words.add(sub);
}
}
}
for (const w of words) {
const list = byWord.get(w) || [];
list.push(item);
@@ -276,11 +291,23 @@ export function detectEmergentTopics(
}
}
const out: GroupSuggestion[] = [];
const sorted = [...byWord.entries()]
.filter(([, list]) => list.length >= minSize)
.sort((a, b) => b[1].length - a[1].length)
.slice(0, maxGroups);
sorted.forEach(([word, list], idx) => {
const candidates = [...byWord.entries()]
.filter(([, list]) => list.length >= minSize && list.length <= TOPIC_MAX_SIZE)
// Taille d'abord, puis mot le plus long (le préfixe redondant est éliminé ensuite).
.sort((a, b) => b[1].length - a[1].length || b[0].length - a[0].length);
// Déduplique les préfixes redondants ('docum' couvert par 'documentaires').
const kept: Array<[string, typeof items]> = [];
for (const [word, list] of candidates) {
const ids = new Set(list.map((m) => m.subscriptionId));
const covered = kept.some(([w2, l2]) => {
if (w2.length <= word.length || !w2.startsWith(word)) return false;
const ids2 = new Set(l2.map((m) => m.subscriptionId));
return [...ids].every((id) => ids2.has(id));
});
if (!covered) kept.push([word, list]);
if (kept.length >= maxGroups) break;
}
kept.forEach(([word, list], idx) => {
const name = word.charAt(0).toUpperCase() + word.slice(1);
if (excluded.has(normalizeChannelText(name))) return;
out.push({
@@ -304,7 +331,7 @@ export function suggestGroupsForChannels(
items: Array<{ subscriptionId: number; title: string; handle?: string | null; groupIds: string[]; extra?: string }>,
existingGroups: Array<{ id: string; name: string }>,
): GroupSuggestion[] {
const byRule = new Map<number, typeof items>();
const byRule = new Map<number, Array<{ item: (typeof items)[number]; score: number }>>();
// Mots courts (≤4) : mot entier uniquement (évite 'car' dans 'oscar').
// Mots longs : préfixe de mot (pluriels, 'metallica', 'documentaires'...).
const wordRe = (kw: string) => {
@@ -328,13 +355,15 @@ export function suggestGroupsForChannels(
});
if (best < 0) continue;
const list = byRule.get(best) || [];
list.push(item);
list.push({ item, score: bestScore });
byRule.set(best, list);
}
const suggestions: GroupSuggestion[] = [];
const normName = (s: string) => normalizeChannelText(s).trim();
for (const [ruleIdx, matched] of byRule) {
if (matched.length < 2) continue; // anti-bruit : 1 chaîne isolée ne fait pas un groupe
for (const [ruleIdx, scored] of byRule) {
// Singleton fort (≥3 mots-clés distincts) : signal suffisant même seul.
const matched = scored.map((s) => s.item);
if (matched.length < 2 && !scored.some((s) => s.score >= 3)) continue;
const rule = GROUP_SUGGESTION_RULES[ruleIdx];
const existing = existingGroups.find((g) => normName(g.name) === normName(rule.name));
const sample = matched.slice(0, 3).map((m) => m.title).filter(Boolean);