diff --git a/scripts/subtitle_to_markdown.py b/scripts/subtitle_to_markdown.py index c0af6e0..4198e3b 100755 --- a/scripts/subtitle_to_markdown.py +++ b/scripts/subtitle_to_markdown.py @@ -9,16 +9,18 @@ def strip_html(text): return re.sub(r'<[^>]+>', '', text) -# YouTube auto-captions use rolling cues: the previous line is repeated as -# plain text while the new line carries inline word timings (<00:00:00.200>, -# ). Only the tagged lines contain new text. -def has_inline_timing_tags(line): - return re.search(r'<\d{2}:\d{2}:|', line) is not None +# YouTube auto-captions use rolling cues: each cue repeats the previous +# cue's last line as plain text before the new text (which usually, but not +# always, carries inline word timings like <00:00:00.200> or ). +def has_inline_timing_tags(text): + return re.search(r'<\d{2}:\d{2}:|', text) is not None def parse_vtt(content): + rolling = has_inline_timing_tags(content) blocks = re.split(r'\n{2,}', content.strip()) cues = [] + prev_last = None for block in blocks: lines = block.strip().splitlines() if not lines: @@ -30,14 +32,14 @@ def parse_vtt(content): ts_idx = next((i for i, l in enumerate(lines) if '-->' in l), None) if ts_idx is None: continue - text_lines = lines[ts_idx + 1:] - if any(has_inline_timing_tags(l) for l in text_lines): - text_lines = [l for l in text_lines if has_inline_timing_tags(l)] - text = ' '.join( - strip_html(l).strip() - for l in text_lines - if strip_html(l).strip() - ) + text_lines = [strip_html(l).strip() for l in lines[ts_idx + 1:]] + if rolling: + while text_lines and prev_last and text_lines[0] == prev_last: + text_lines.pop(0) + non_empty = [l for l in text_lines if l] + if rolling and non_empty: + prev_last = non_empty[-1] + text = ' '.join(non_empty) if text: cues.append(text) return cues diff --git a/src/lib/subtitle.js b/src/lib/subtitle.js index 3ed6022..64d57da 100644 --- a/src/lib/subtitle.js +++ b/src/lib/subtitle.js @@ -5,16 +5,18 @@ function stripHtml(text) { return text.replace(/<[^>]+>/g, ''); } -// YouTube auto-captions use rolling cues: the previous line is repeated as -// plain text while the new line carries inline word timings (<00:00:00.200>, -// ). Only the tagged lines contain new text. -function hasInlineTimingTags(line) { - return /<\d{2}:\d{2}:|/.test(line); +// YouTube auto-captions use rolling cues: each cue repeats the previous +// cue's last line as plain text before the new text (which usually, but not +// always, carries inline word timings like <00:00:00.200> or ). +function hasInlineTimingTags(text) { + return /<\d{2}:\d{2}:|/.test(text); } function parseVtt(content) { + const rolling = hasInlineTimingTags(content); const blocks = content.split(/\n{2,}/); const cues = []; + let prevLast = null; for (const block of blocks) { const lines = block.trim().split('\n'); if (!lines.length) continue; @@ -22,14 +24,15 @@ function parseVtt(content) { if (lines[0].startsWith('NOTE') || lines[0].startsWith('STYLE')) continue; const tsIdx = lines.findIndex(l => l.includes('-->')); if (tsIdx === -1) continue; - let textLines = lines.slice(tsIdx + 1); - if (textLines.some(hasInlineTimingTags)) { - textLines = textLines.filter(hasInlineTimingTags); + const textLines = lines.slice(tsIdx + 1).map(l => stripHtml(l).trim()); + if (rolling) { + while (textLines.length && prevLast && textLines[0] === prevLast) { + textLines.shift(); + } } - const text = textLines - .map(l => stripHtml(l).trim()) - .filter(Boolean) - .join(' '); + const nonEmpty = textLines.filter(Boolean); + if (rolling && nonEmpty.length) prevLast = nonEmpty[nonEmpty.length - 1]; + const text = nonEmpty.join(' '); if (text) cues.push(text); } return cues;