feat(groups): ratissage élargi (3 lettres, sous-chaînes, singletons, plafond, copie inclassables)
This commit is contained in:
@@ -41,7 +41,10 @@
|
||||
<div class="flex flex-wrap items-center justify-between gap-2">
|
||||
<div>
|
||||
<h3 class="font-semibold text-slate-100">✨ Groupes suggérés</h3>
|
||||
<p class="text-xs text-slate-400">Noms de chaînes + sujets détectés. Vérifiez avant d’appliquer.</p>
|
||||
<p class="text-xs text-slate-400">
|
||||
Noms de chaînes + sujets détectés. Vérifiez avant d’appliquer.
|
||||
<span *ngIf="ungroupedCount() > 0" class="text-amber-200/90">{{ ungroupedCount() }} sans groupe.</span>
|
||||
</p>
|
||||
</div>
|
||||
<div class="flex flex-wrap gap-2">
|
||||
<button
|
||||
@@ -79,7 +82,15 @@
|
||||
</div>
|
||||
</div>
|
||||
<div *ngIf="suggestions().length === 0" class="mt-3 rounded-lg border border-dashed border-slate-700 p-4 text-center text-sm text-slate-400">
|
||||
Aucune suggestion : chaque thème a moins de 2 chaînes, ou tout est déjà rangé.
|
||||
<p>Aucune suggestion : chaque thème a moins de 2 chaînes, ou tout est déjà rangé.</p>
|
||||
<button
|
||||
*ngIf="ungroupedCount() > 0"
|
||||
type="button"
|
||||
(click)="copyUngrouped()"
|
||||
class="mt-2 rounded-md border border-slate-600 px-3 py-1.5 text-xs text-slate-200 transition hover:border-slate-400 hover:text-white"
|
||||
>
|
||||
{{ copiedUngrouped() ? 'Copié ! Collez la liste pour enrichir les règles.' : 'Copier les ' + ungroupedCount() + ' sans-groupe' }}
|
||||
</button>
|
||||
</div>
|
||||
<ul *ngIf="suggestions().length > 0" class="mt-3 space-y-2">
|
||||
<li *ngFor="let s of suggestions(); trackBy: trackBySuggestion" class="flex flex-col gap-2 rounded-lg border border-slate-800 bg-slate-900/60 p-3 sm:flex-row sm:items-center sm:justify-between">
|
||||
|
||||
@@ -219,6 +219,45 @@ export class SubscriptionsComponent {
|
||||
}
|
||||
}
|
||||
|
||||
copiedUngrouped = signal<boolean>(false);
|
||||
|
||||
/** Copie les chaînes sans groupe (une par ligne) pour analyse des règles. */
|
||||
copyUngrouped(): void {
|
||||
const lines = this.items()
|
||||
.filter((item) => this.itemGroupIds(item).length === 0)
|
||||
.map((item) => item.channel?.title || item.channel?.handle || `#${item.subscriptionId}`);
|
||||
const text = lines.join('\n');
|
||||
if (!text) return;
|
||||
const done = () => {
|
||||
this.copiedUngrouped.set(true);
|
||||
setTimeout(() => this.copiedUngrouped.set(false), 2000);
|
||||
};
|
||||
try {
|
||||
const nav = navigator as Navigator & { clipboard?: { writeText(t: string): Promise<void> } };
|
||||
if (nav?.clipboard?.writeText) {
|
||||
nav.clipboard.writeText(text).then(done, () => this.copyUngroupedFallback(text, done));
|
||||
} else {
|
||||
this.copyUngroupedFallback(text, done);
|
||||
}
|
||||
} catch {
|
||||
this.copyUngroupedFallback(text, done);
|
||||
}
|
||||
}
|
||||
|
||||
private copyUngroupedFallback(text: string, done: () => void): void {
|
||||
try {
|
||||
const ta = document.createElement('textarea');
|
||||
ta.value = text;
|
||||
ta.style.position = 'fixed';
|
||||
ta.style.opacity = '0';
|
||||
document.body.appendChild(ta);
|
||||
ta.select();
|
||||
document.execCommand('copy');
|
||||
document.body.removeChild(ta);
|
||||
done();
|
||||
} catch {}
|
||||
}
|
||||
|
||||
// --- trackBy : DOM stable (pas de re-rendu complet + recharge d'images à chaque tick) ---
|
||||
trackBySubscriptionId(_idx: number, item: SubscriptionItem): number {
|
||||
return item.subscriptionId;
|
||||
|
||||
@@ -223,23 +223,28 @@ export function normalizeChannelText(value: string | null | undefined): string {
|
||||
.toLowerCase();
|
||||
}
|
||||
|
||||
/** Mots vides FR/EN + termes YouTube génériques (évitent les méga-groupes inutiles). */
|
||||
/**
|
||||
* Mots vides FR/EN + jargon plateforme (jamais des thèmes utiles).
|
||||
* Géographie et génériques (québec, tv, top...) restants VOLONTAIREMENT :
|
||||
* ils forment des groupes bornés (plafond anti-méga-groupes ci-dessous).
|
||||
*/
|
||||
const TOPIC_STOPWORDS = new Set([
|
||||
'avec', 'dans', 'pour', 'sans', 'sous', 'vers', 'chez', 'entre', 'contre', 'depuis',
|
||||
'cette', 'celui', 'celle', 'ceux', 'celles', 'notre', 'votre', 'leur', 'leurs', 'tous',
|
||||
'toute', 'toutes', 'aussi', 'encore', 'tres', 'bien', 'plus', 'moins', 'comme', 'mais',
|
||||
'donc', 'alors', 'quand', 'comment', 'pourquoi', 'parce', 'sont', 'etes', 'font', 'vont',
|
||||
'cest', 'nest', 'plus', 'tout', 'tous', 'aux', 'ces', 'ses', 'mon', 'mes', 'ton', 'tes',
|
||||
'son', 'nous', 'vous', 'ils', 'elles', 'elle', 'celui', 'dont', 'ou', 'parce',
|
||||
'the', 'and', 'for', 'with', 'from', 'your', 'you', 'your', 'this', 'that', 'these',
|
||||
'cest', 'nest', 'tout', 'aux', 'ces', 'ses', 'mon', 'mes', 'ton', 'tes',
|
||||
'son', 'nous', 'vous', 'ils', 'elles', 'elle', 'dont',
|
||||
'the', 'and', 'for', 'with', 'from', 'your', 'you', 'this', 'that', 'these',
|
||||
'those', 'are', 'was', 'were', 'been', 'have', 'has', 'will', 'would', 'there',
|
||||
'their', 'what', 'when', 'where', 'which', 'while', 'about', 'into', 'over',
|
||||
'after', 'before', 'your', 'videos', 'video', 'tube', 'youtube', 'channel', 'chaine',
|
||||
'chanel', 'official', 'officiel', 'tv', 'media', 'network', 'studio', 'production',
|
||||
'streaming', 'live', 'clips', 'shorts', 'france', 'francais', 'quebec', 'canada',
|
||||
'montreal', 'paris', 'best', 'top', 'new', 'show', 'central', 'daily', 'hub', 'one',
|
||||
'after', 'before', 'tube', 'youtube', 'channel', 'chaine', 'chanel',
|
||||
'official', 'officiel', 'network', 'production', 'streaming',
|
||||
]);
|
||||
|
||||
/** Plafond : au-delà, le sujet est trop générique pour être utile. */
|
||||
const TOPIC_MAX_SIZE = 60;
|
||||
|
||||
/** Devine une icône pour un mot (recherche inverse dans les mots-clés). */
|
||||
export function guessIconForWord(word: string): string {
|
||||
const w = normalizeChannelText(word).trim();
|
||||
@@ -257,18 +262,28 @@ export function guessIconForWord(word: string): string {
|
||||
* Pur et testable : ne touche pas au réseau.
|
||||
*/
|
||||
export function detectEmergentTopics(
|
||||
items: Array<{ subscriptionId: number; title: string; handle?: string | null; extra?: string }>,
|
||||
items: Array<{ subscriptionId: number; title: string; handle?: string | null; extra?: string; groupIds?: string[] }>,
|
||||
opts?: { minSize?: number; maxGroups?: number; excludeNames?: string[] },
|
||||
): GroupSuggestion[] {
|
||||
const minSize = Math.max(2, opts?.minSize || 2);
|
||||
const maxGroups = Math.max(1, opts?.maxGroups || 10);
|
||||
const maxGroups = Math.max(1, opts?.maxGroups || 15);
|
||||
const excluded = new Set((opts?.excludeNames || []).map((n) => normalizeChannelText(n).trim()));
|
||||
// Candidats : mots entiers ≥3 lettres (tokenisés, donc 'rap' ≠ 'drapeau').
|
||||
const byWord = new Map<string, typeof items>();
|
||||
for (const item of items) {
|
||||
const hay = normalizeChannelText(`${item.title || ''} ${item.handle || ''} ${item.extra || ''}`);
|
||||
const words = new Set(
|
||||
hay.split(/[^a-z0-9]+/).filter((w) => w.length >= 4 && !/^\d+$/.test(w) && !TOPIC_STOPWORDS.has(w)),
|
||||
);
|
||||
const tokens = hay.split(/[^a-z0-9]+/).filter((w) => w.length >= 3 && !/^\d+$/.test(w) && !TOPIC_STOPWORDS.has(w));
|
||||
const words = new Set<string>();
|
||||
for (const t of tokens) {
|
||||
words.add(t);
|
||||
// Sous-chaînes longues : 'hockey' dans 'hockeyqc', 'tech' dans 'technology'.
|
||||
if (t.length >= 8) {
|
||||
for (let len = 5; len <= t.length - 2; len++) {
|
||||
const sub = t.slice(0, len);
|
||||
if (!TOPIC_STOPWORDS.has(sub)) words.add(sub);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const w of words) {
|
||||
const list = byWord.get(w) || [];
|
||||
list.push(item);
|
||||
@@ -276,11 +291,23 @@ export function detectEmergentTopics(
|
||||
}
|
||||
}
|
||||
const out: GroupSuggestion[] = [];
|
||||
const sorted = [...byWord.entries()]
|
||||
.filter(([, list]) => list.length >= minSize)
|
||||
.sort((a, b) => b[1].length - a[1].length)
|
||||
.slice(0, maxGroups);
|
||||
sorted.forEach(([word, list], idx) => {
|
||||
const candidates = [...byWord.entries()]
|
||||
.filter(([, list]) => list.length >= minSize && list.length <= TOPIC_MAX_SIZE)
|
||||
// Taille d'abord, puis mot le plus long (le préfixe redondant est éliminé ensuite).
|
||||
.sort((a, b) => b[1].length - a[1].length || b[0].length - a[0].length);
|
||||
// Déduplique les préfixes redondants ('docum' couvert par 'documentaires').
|
||||
const kept: Array<[string, typeof items]> = [];
|
||||
for (const [word, list] of candidates) {
|
||||
const ids = new Set(list.map((m) => m.subscriptionId));
|
||||
const covered = kept.some(([w2, l2]) => {
|
||||
if (w2.length <= word.length || !w2.startsWith(word)) return false;
|
||||
const ids2 = new Set(l2.map((m) => m.subscriptionId));
|
||||
return [...ids].every((id) => ids2.has(id));
|
||||
});
|
||||
if (!covered) kept.push([word, list]);
|
||||
if (kept.length >= maxGroups) break;
|
||||
}
|
||||
kept.forEach(([word, list], idx) => {
|
||||
const name = word.charAt(0).toUpperCase() + word.slice(1);
|
||||
if (excluded.has(normalizeChannelText(name))) return;
|
||||
out.push({
|
||||
@@ -304,7 +331,7 @@ export function suggestGroupsForChannels(
|
||||
items: Array<{ subscriptionId: number; title: string; handle?: string | null; groupIds: string[]; extra?: string }>,
|
||||
existingGroups: Array<{ id: string; name: string }>,
|
||||
): GroupSuggestion[] {
|
||||
const byRule = new Map<number, typeof items>();
|
||||
const byRule = new Map<number, Array<{ item: (typeof items)[number]; score: number }>>();
|
||||
// Mots courts (≤4) : mot entier uniquement (évite 'car' dans 'oscar').
|
||||
// Mots longs : préfixe de mot (pluriels, 'metallica', 'documentaires'...).
|
||||
const wordRe = (kw: string) => {
|
||||
@@ -328,13 +355,15 @@ export function suggestGroupsForChannels(
|
||||
});
|
||||
if (best < 0) continue;
|
||||
const list = byRule.get(best) || [];
|
||||
list.push(item);
|
||||
list.push({ item, score: bestScore });
|
||||
byRule.set(best, list);
|
||||
}
|
||||
const suggestions: GroupSuggestion[] = [];
|
||||
const normName = (s: string) => normalizeChannelText(s).trim();
|
||||
for (const [ruleIdx, matched] of byRule) {
|
||||
if (matched.length < 2) continue; // anti-bruit : 1 chaîne isolée ne fait pas un groupe
|
||||
for (const [ruleIdx, scored] of byRule) {
|
||||
// Singleton fort (≥3 mots-clés distincts) : signal suffisant même seul.
|
||||
const matched = scored.map((s) => s.item);
|
||||
if (matched.length < 2 && !scored.some((s) => s.score >= 3)) continue;
|
||||
const rule = GROUP_SUGGESTION_RULES[ruleIdx];
|
||||
const existing = existingGroups.find((g) => normName(g.name) === normName(rule.name));
|
||||
const sample = matched.slice(0, 3).map((m) => m.title).filter(Boolean);
|
||||
|
||||
Reference in New Issue
Block a user