diff --git a/scripts/subtitle_to_markdown.py b/scripts/subtitle_to_markdown.py index f882e49..c0af6e0 100755 --- a/scripts/subtitle_to_markdown.py +++ b/scripts/subtitle_to_markdown.py @@ -9,6 +9,13 @@ def strip_html(text): return re.sub(r'<[^>]+>', '', text) +# YouTube auto-captions use rolling cues: the previous line is repeated as +# plain text while the new line carries inline word timings (<00:00:00.200>, +# ). Only the tagged lines contain new text. +def has_inline_timing_tags(line): + return re.search(r'<\d{2}:\d{2}:|', line) is not None + + def parse_vtt(content): blocks = re.split(r'\n{2,}', content.strip()) cues = [] @@ -23,9 +30,12 @@ def parse_vtt(content): ts_idx = next((i for i, l in enumerate(lines) if '-->' in l), None) if ts_idx is None: continue + text_lines = lines[ts_idx + 1:] + if any(has_inline_timing_tags(l) for l in text_lines): + text_lines = [l for l in text_lines if has_inline_timing_tags(l)] text = ' '.join( strip_html(l).strip() - for l in lines[ts_idx + 1:] + for l in text_lines if strip_html(l).strip() ) if text: diff --git a/src/lib/subtitle.js b/src/lib/subtitle.js index c5a58ab..3ed6022 100644 --- a/src/lib/subtitle.js +++ b/src/lib/subtitle.js @@ -5,6 +5,13 @@ function stripHtml(text) { return text.replace(/<[^>]+>/g, ''); } +// YouTube auto-captions use rolling cues: the previous line is repeated as +// plain text while the new line carries inline word timings (<00:00:00.200>, +// ). Only the tagged lines contain new text. +function hasInlineTimingTags(line) { + return /<\d{2}:\d{2}:|/.test(line); +} + function parseVtt(content) { const blocks = content.split(/\n{2,}/); const cues = []; @@ -15,8 +22,11 @@ function parseVtt(content) { if (lines[0].startsWith('NOTE') || lines[0].startsWith('STYLE')) continue; const tsIdx = lines.findIndex(l => l.includes('-->')); if (tsIdx === -1) continue; - const text = lines - .slice(tsIdx + 1) + let textLines = lines.slice(tsIdx + 1); + if (textLines.some(hasInlineTimingTags)) { + textLines = textLines.filter(hasInlineTimingTags); + } + const text = textLines .map(l => stripHtml(l).trim()) .filter(Boolean) .join(' '); diff --git a/src/routes/api/download/+server.js b/src/routes/api/download/+server.js index fc1cb28..c66ce40 100644 --- a/src/routes/api/download/+server.js +++ b/src/routes/api/download/+server.js @@ -105,12 +105,18 @@ async function runDownload(videoUrl, audioOnly, send, signal) { } } + const writtenMd = new Set(); for (const subPath of subtitlePaths) { if (!existsSync(subPath)) continue; send('info', `Converting: ${basename(subPath)}`); try { const md = convertSubtitleToMarkdown(subPath); const mdPath = subPath.replace(/\.(vtt|srt)$/, '.md'); + if (writtenMd.has(md)) { + send('info', `Skipped (identical): ${basename(mdPath)}`); + continue; + } + writtenMd.add(md); writeFileSync(mdPath, md, 'utf-8'); send('info', `Saved: ${basename(mdPath)}`); } catch (err) {