feat(preferences): section langues a telecharger (defaut fr+en), transcripts filtres cote serveur et watch
CI / build-and-test (push) Successful in 13m50s
CI / build-and-test (push) Successful in 13m50s
This commit is contained in:
+79
-13
@@ -81,6 +81,8 @@ import {
|
||||
resetActiveDownloadJobs,
|
||||
countActiveDownloadJobs,
|
||||
sumCompletedDownloadBytes,
|
||||
SUPPORTED_DOWNLOAD_LANGUAGES as SUPPORTED_DL_LANGS,
|
||||
DEFAULT_DOWNLOAD_LANGUAGES as DEFAULT_DL_LANGS,
|
||||
} from './db.mjs';
|
||||
import { getChannelAdapter, setTwitchTokenProvider } from './providers/channel-registry.mjs';
|
||||
import { fetchChannelContent } from './providers/channel-content.mjs';
|
||||
@@ -2589,6 +2591,55 @@ const transcriptLimiter = rateLimit({
|
||||
/** Providers known NOT to expose subtitle tracks via yt-dlp (no transcript possible). */
|
||||
const TRANSCRIPT_UNSUPPORTED_PROVIDERS = new Set(['twitch', 'odysee', 'rumble']);
|
||||
|
||||
/** Langues de transcripts autorisées : défaut fr+en, jamais de téléchargement hors liste. */
|
||||
function sanitizeAllowedTranscriptLangs(raw) {
|
||||
const supported = Array.isArray(SUPPORTED_DL_LANGS) && SUPPORTED_DL_LANGS.length
|
||||
? SUPPORTED_DL_LANGS
|
||||
: ['fr', 'en'];
|
||||
const fallback = Array.isArray(DEFAULT_DL_LANGS) && DEFAULT_DL_LANGS.length
|
||||
? DEFAULT_DL_LANGS
|
||||
: ['fr', 'en'];
|
||||
let arr = raw;
|
||||
if (typeof arr === 'string') arr = arr.split(',');
|
||||
if (!Array.isArray(arr)) return [...fallback];
|
||||
const cleaned = arr
|
||||
.map(v => String(v || '').trim().toLowerCase().replace(/_/g, '-').split('-')[0])
|
||||
.filter(v => /^[a-z]{2,3}$/.test(v) && supported.includes(v));
|
||||
return Array.from(new Set(cleaned));
|
||||
}
|
||||
|
||||
function transcriptPrimary(lang) {
|
||||
return String(lang || '').trim().toLowerCase().replace(/_/g, '-').split('-')[0];
|
||||
}
|
||||
|
||||
/** Ne jamais exposer ni télécharger une langue non activée dans les préférences. */
|
||||
function filterTranscriptLanguages(languages, allowed) {
|
||||
const set = new Set((allowed || []).map(transcriptPrimary).filter(Boolean));
|
||||
return (Array.isArray(languages) ? languages : []).filter(l => set.has(transcriptPrimary(l)));
|
||||
}
|
||||
|
||||
/** Résout les langues autorisées : `?langs=` explicite > préférence user (JWT) > défaut fr,en. */
|
||||
function resolveTranscriptAllowedLangs(req) {
|
||||
const fromQuery = sanitizeAllowedTranscriptLangs(req.query?.langs);
|
||||
const queryAsked = req.query?.langs !== undefined;
|
||||
if (queryAsked) return fromQuery;
|
||||
try {
|
||||
const hdr = req.headers?.['authorization'] || '';
|
||||
const [, token] = String(hdr).split(' ');
|
||||
if (token) {
|
||||
const payload = jwt.verify(token, JWT_SECRET);
|
||||
const uid = payload?.sub;
|
||||
if (uid) {
|
||||
const prefs = getPreferencesForApi(uid);
|
||||
if (prefs && Array.isArray(prefs.downloadLanguages)) {
|
||||
return sanitizeAllowedTranscriptLangs(prefs.downloadLanguages);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
return sanitizeAllowedTranscriptLangs(undefined);
|
||||
}
|
||||
|
||||
/** Download subtitles via yt-dlp (handles YouTube impersonation + 429-prone
|
||||
* translated tracks). Tries `langs` in order, returns the first non-empty
|
||||
* parsed lines with the language that worked, or null. Bounded by a timeout
|
||||
@@ -2630,9 +2681,10 @@ async function transcriptViaYtDlp(url, langs, netOpts = {}) {
|
||||
return null;
|
||||
};
|
||||
try {
|
||||
// Ne télécharger QUE les langues autorisées (aucun ajout forcé hors préférence).
|
||||
const wanted = Array.from(new Set((langs || []).map((l) => String(l || '').split('-')[0]).filter(Boolean)));
|
||||
if (!wanted.includes('en')) wanted.push('en');
|
||||
const subLangs = wanted.slice(0, 4).join(',');
|
||||
if (!wanted.length) return null;
|
||||
const subLangs = wanted.slice(0, 6).join(',');
|
||||
const child = youtubedl(url, {
|
||||
writeSub: true,
|
||||
writeAutoSub: true,
|
||||
@@ -2689,12 +2741,21 @@ async function fetchTimedTextLines(track) {
|
||||
r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
try {
|
||||
const { provider, videoId } = req.params;
|
||||
const lang = String(req.query.lang || 'fr').slice(0, 12) || 'fr';
|
||||
// Langues autorisées par les préférences (défaut fr,en) : rien d'autre
|
||||
// n'est affiché ni téléchargé.
|
||||
const allowed = resolveTranscriptAllowedLangs(req);
|
||||
let lang = String(req.query.lang || allowed[0] || 'fr').slice(0, 12) || 'fr';
|
||||
if (!allowed.map(transcriptPrimary).includes(transcriptPrimary(lang))) {
|
||||
lang = allowed[0] || 'fr';
|
||||
}
|
||||
const normalized = normalizeTranscriptProvider(provider);
|
||||
if (!normalized || !videoId) {
|
||||
return res.status(400).json({ available: false, error: 'invalid_provider_or_video' });
|
||||
}
|
||||
const cacheKey = `transcript:${normalized}:${videoId}:${lang.toLowerCase()}`;
|
||||
if (!allowed.length) {
|
||||
return res.json({ lang: null, available: false, languages: [], lines: [], reason: 'no_subtitles' });
|
||||
}
|
||||
const cacheKey = `transcript:${normalized}:${videoId}:${lang.toLowerCase()}:${[...allowed].sort().join(',')}`;
|
||||
const cached = transcriptCacheGet(cacheKey);
|
||||
if (cached) return res.json(cached);
|
||||
// Source de découverte des pistes : InnerTube d'abord pour YouTube
|
||||
@@ -2739,11 +2800,12 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
}
|
||||
}
|
||||
const { track, languages, lang: chosenLang } = pickTrack(meta, lang);
|
||||
if (!track) {
|
||||
const visibleLanguages = filterTranscriptLanguages(languages, allowed);
|
||||
if (!track || !allowed.map(transcriptPrimary).includes(transcriptPrimary(chosenLang))) {
|
||||
// Definitive absence: distinguish "provider never exposes subtitles"
|
||||
// (twitch/odysee/rumble) from "this video has none" — cacheable 200s.
|
||||
const reason = TRANSCRIPT_UNSUPPORTED_PROVIDERS.has(normalized) ? 'provider_unsupported' : 'no_subtitles';
|
||||
const empty = { lang: null, available: false, languages: languages || [], lines: [], reason };
|
||||
const empty = { lang: null, available: false, languages: visibleLanguages, lines: [], reason };
|
||||
transcriptCacheSet(cacheKey, empty);
|
||||
return res.json(empty);
|
||||
}
|
||||
@@ -2757,11 +2819,14 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
let lines = [];
|
||||
let workedLang = chosenLang;
|
||||
let sawRateLimit = false;
|
||||
const baseCands = normalized === 'youtube'
|
||||
const allowedSet = new Set(allowed.map(transcriptPrimary));
|
||||
const keepAllowed = (cands) => (cands || []).filter(c => allowedSet.has(transcriptPrimary(c.lang)));
|
||||
const baseCands = keepAllowed(normalized === 'youtube'
|
||||
? firstPerLanguage(orderedTracks(meta, lang), 10)
|
||||
: orderedTracks(meta, lang).slice(0, 5);
|
||||
: orderedTracks(meta, lang).slice(0, 5));
|
||||
// Server-side auto-translation only towards an allowed target language.
|
||||
const candidates = normalized === 'youtube'
|
||||
? baseCands.concat(translatedFallbacks(baseCands, lang)).slice(0, 12)
|
||||
? baseCands.concat(keepAllowed(translatedFallbacks(baseCands, lang))).slice(0, 12)
|
||||
: baseCands;
|
||||
for (const cand of candidates) {
|
||||
let attempt = 0;
|
||||
@@ -2792,9 +2857,10 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
if (!lines || lines.length === 0) {
|
||||
// Direct timedtext fetches failed (429 / Sorry pages / impersonation):
|
||||
// let yt-dlp download the subtitles instead (handles all of the above).
|
||||
// Uniquement les langues autorisées.
|
||||
try {
|
||||
const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(Boolean), ytdlpNetOpts());
|
||||
if (viaDlp && viaDlp.lines && viaDlp.lines.length) {
|
||||
const viaDlp = await transcriptViaYtDlp(transcriptUrl, [lang, chosenLang].filter(l => allowedSet.has(transcriptPrimary(l))), ytdlpNetOpts());
|
||||
if (viaDlp && viaDlp.lines && viaDlp.lines.length && allowedSet.has(transcriptPrimary(viaDlp.lang))) {
|
||||
lines = viaDlp.lines;
|
||||
if (viaDlp.lang) workedLang = viaDlp.lang;
|
||||
}
|
||||
@@ -2807,7 +2873,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
// never cached as "no subtitles".
|
||||
return res.status(502).json({
|
||||
available: false,
|
||||
languages: languages || [],
|
||||
languages: visibleLanguages,
|
||||
lines: [],
|
||||
error: 'transcript_temporarily_unavailable',
|
||||
retryable: true,
|
||||
@@ -2815,7 +2881,7 @@ r.get('/transcript/:provider/:videoId', transcriptLimiter, async (req, res) => {
|
||||
});
|
||||
}
|
||||
lines = dedupeTranscriptLines(lines);
|
||||
const result = { lang: workedLang, available: true, languages: languages || [], lines };
|
||||
const result = { lang: workedLang, available: true, languages: visibleLanguages, lines };
|
||||
transcriptCacheSet(cacheKey, result);
|
||||
return res.json(result);
|
||||
} catch (e) {
|
||||
|
||||
Reference in New Issue
Block a user