feat(baoyu-youtube-transcript): add sentence segmentation and improve caching

This commit is contained in:
Jim Liu 宝玉
2026-03-21 22:42:43 -05:00
parent e52f92b193
commit e413ade164
5 changed files with 244 additions and 43 deletions
+35 -1
View File
@@ -76,7 +76,7 @@ Simply tell Claude Code:
|--------|-------------|--------| |--------|-------------|--------|
| **content-skills** | Content generation and publishing | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) | | **content-skills** | Content generation and publishing | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
| **ai-generation-skills** | AI-powered generation backends | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) | | **ai-generation-skills** | AI-powered generation backends | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
| **utility-skills** | Utility tools for content processing | [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) | | **utility-skills** | Utility tools for content processing | [youtube-transcript](#baoyu-youtube-transcript), [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
## Update Skills ## Update Skills
@@ -766,6 +766,40 @@ Interacts with Gemini Web to generate text and images.
Utility tools for content processing. Utility tools for content processing.
#### baoyu-youtube-transcript
Download YouTube video transcripts/subtitles and cover images. Supports multiple languages, translation, chapters, and speaker identification. Caches raw data for fast re-formatting.
```bash
# Default: markdown with timestamps
/baoyu-youtube-transcript https://www.youtube.com/watch?v=VIDEO_ID
# Specify languages (priority order)
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --languages zh,en,ja
# With chapters and speaker identification
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --chapters --speakers
# SRT subtitle format
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --format srt
# List available transcripts
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --list
```
**Options**:
| Option | Description | Default |
|--------|-------------|---------|
| `<url-or-id>` | YouTube URL or video ID | Required |
| `--languages <codes>` | Language codes, comma-separated | `en` |
| `--format <fmt>` | Output format: `text`, `srt` | `text` |
| `--translate <code>` | Translate to specified language | |
| `--chapters` | Chapter segmentation from video description | |
| `--speakers` | Speaker identification (requires AI post-processing) | |
| `--no-timestamps` | Disable timestamps | |
| `--list` | List available transcripts | |
| `--refresh` | Force re-fetch, ignore cache | |
#### baoyu-url-to-markdown #### baoyu-url-to-markdown
Fetch any URL via Chrome CDP and convert to clean markdown. Saves rendered HTML snapshot alongside the markdown, and automatically falls back to a legacy extractor when Defuddle fails. Fetch any URL via Chrome CDP and convert to clean markdown. Saves rendered HTML snapshot alongside the markdown, and automatically falls back to a legacy extractor when Defuddle fails.
+35 -1
View File
@@ -76,7 +76,7 @@ clawhub install baoyu-markdown-to-html
|------|------|----------| |------|------|----------|
| **content-skills** | 内容生成和发布 | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) | | **content-skills** | 内容生成和发布 | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
| **ai-generation-skills** | AI 生成后端 | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) | | **ai-generation-skills** | AI 生成后端 | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
| **utility-skills** | 内容处理工具 | [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) | | **utility-skills** | 内容处理工具 | [youtube-transcript](#baoyu-youtube-transcript), [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
## 更新技能 ## 更新技能
@@ -766,6 +766,40 @@ AI 驱动的生成后端。
内容处理工具。 内容处理工具。
#### baoyu-youtube-transcript
下载 YouTube 视频字幕/转录文本和封面图片。支持多语言、翻译、章节分段和说话人识别。缓存原始数据以便快速重新格式化。
```bash
# 默认:带时间戳的 Markdown
/baoyu-youtube-transcript https://www.youtube.com/watch?v=VIDEO_ID
# 指定语言(按优先级排列)
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --languages zh,en,ja
# 章节分段 + 说话人识别
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --chapters --speakers
# SRT 字幕格式
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --format srt
# 列出可用字幕
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --list
```
**选项**
| 选项 | 说明 | 默认值 |
|------|------|--------|
| `<url-or-id>` | YouTube URL 或视频 ID | 必填 |
| `--languages <codes>` | 语言代码,逗号分隔 | `en` |
| `--format <fmt>` | 输出格式:`text``srt` | `text` |
| `--translate <code>` | 翻译为指定语言 | |
| `--chapters` | 根据视频描述进行章节分段 | |
| `--speakers` | 说话人识别(需 AI 后处理) | |
| `--no-timestamps` | 禁用时间戳 | |
| `--list` | 列出可用字幕 | |
| `--refresh` | 强制重新获取,忽略缓存 | |
#### baoyu-url-to-markdown #### baoyu-url-to-markdown
通过 Chrome CDP 抓取任意 URL 并转换为 Markdown。同时保存渲染后的 HTML 快照,Defuddle 失败时自动回退到旧版提取器。 通过 Chrome CDP 抓取任意 URL 并转换为 Markdown。同时保存渲染后的 HTML 快照,Defuddle 失败时自动回退到旧版提取器。
+9 -5
View File
@@ -98,11 +98,12 @@ youtube-transcript/
├── .index.json # Video ID → directory path mapping (for cache lookup) ├── .index.json # Video ID → directory path mapping (for cache lookup)
└── {channel-slug}/{title-full-slug}/ └── {channel-slug}/{title-full-slug}/
├── meta.json # Video metadata (title, channel, description, duration, chapters, etc.) ├── meta.json # Video metadata (title, channel, description, duration, chapters, etc.)
├── transcript-raw.srt # Raw transcript in SRT format (cached, token-efficient for LLM) ├── transcript-raw.json # Raw transcript snippets from YouTube API (cached)
├── transcript-sentences.json # Sentence-segmented transcript (split by punctuation, merged across snippets)
├── imgs/ ├── imgs/
│ └── cover.jpg # Video thumbnail │ └── cover.jpg # Video thumbnail
├── transcript.md # Markdown transcript ├── transcript.md # Markdown transcript (generated from sentences)
└── transcript.srt # SRT subtitle (if --format srt) └── transcript.srt # SRT subtitle (generated from raw snippets, if --format srt)
``` ```
- `{channel-slug}`: Channel name in kebab-case - `{channel-slug}`: Channel name in kebab-case
@@ -114,11 +115,14 @@ The `--list` mode outputs to stdout only (no file saved).
On first fetch, the script saves: On first fetch, the script saves:
- `meta.json` — video metadata, chapters, cover image path, language info - `meta.json` — video metadata, chapters, cover image path, language info
- `transcript-raw.srt` — raw transcript in SRT format (pre-computed timestamps, token-efficient for LLM processing) - `transcript-raw.json` — raw transcript snippets from YouTube API (`{ text, start, duration }[]`)
- `transcript-sentences.json` — sentence-segmented transcript (`{ text, start: "HH:mm:ss", end: "HH:mm:ss" }[]`), split by sentence-ending punctuation (`.?!…。?!` etc.), timestamps proportionally allocated by character length, CJK-aware text merging
- `imgs/cover.jpg` — video thumbnail - `imgs/cover.jpg` — video thumbnail
Subsequent runs for the same video use cached data (no network calls). Use `--refresh` to force re-fetch. If a different language is requested, the cache is automatically refreshed. Subsequent runs for the same video use cached data (no network calls). Use `--refresh` to force re-fetch. If a different language is requested, the cache is automatically refreshed.
SRT output (`--format srt`) is generated from `transcript-raw.json`. Text/markdown output uses `transcript-sentences.json` for natural sentence boundaries.
## Workflow ## Workflow
When user provides a YouTube URL and wants the transcript: When user provides a YouTube URL and wants the transcript:
@@ -148,7 +152,7 @@ Speaker identification requires AI processing. The script outputs a raw `.md` fi
- Chapter list from description (if available) - Chapter list from description (if available)
- Raw transcript in SRT format (pre-computed start/end timestamps, token-efficient) - Raw transcript in SRT format (pre-computed start/end timestamps, token-efficient)
After the script saves the raw file: After the script saves the raw file, spawn a sub-agent (use a cheaper model like Sonnet for cost efficiency) to process speaker identification:
1. Read the saved `.md` file 1. Read the saved `.md` file
2. Read the prompt template at `{baseDir}/prompts/speaker-transcript.md` 2. Read the prompt template at `{baseDir}/prompts/speaker-transcript.md`
@@ -44,7 +44,7 @@ Use the same language as the transcription for the title and ToC.
**Chapters:** **Chapters:**
``` ```
## [HH:MM:SS] Chapter Title ## Chapter Title [HH:MM:SS]
``` ```
Two blank lines between chapters. Two blank lines between chapters.
@@ -87,14 +87,14 @@ language: en
* [00:00:12] Overview of the New Research * [00:00:12] Overview of the New Research
## [00:00:00] Introduction and Welcome ## Introduction and Welcome [00:00:00]
**Host:** Welcome back to the show. Today, we have a, uh, very special guest, Jane Doe. [00:00:00 → 00:00:03] **Host:** Welcome back to the show. Today, we have a, uh, very special guest, Jane Doe. [00:00:00 → 00:00:03]
**Jane Doe:** Thank you for having me. I'm excited to be here and discuss the findings. [00:00:03 → 00:00:07] **Jane Doe:** Thank you for having me. I'm excited to be here and discuss the findings. [00:00:03 → 00:00:07]
## [00:00:12] Overview of the New Research ## Overview of the New Research [00:00:12]
**Host:** So, Jane, before we get into the nitty-gritty, could you, you know, give us a brief overview for our audience? [00:00:12 → 00:00:16] **Host:** So, Jane, before we get into the nitty-gritty, could you, you know, give us a brief overview for our audience? [00:00:12 → 00:00:16]
+162 -33
View File
@@ -26,6 +26,12 @@ interface Snippet {
duration: number; duration: number;
} }
interface Sentence {
text: string;
start: string;
end: string;
}
interface TranscriptInfo { interface TranscriptInfo {
language: string; language: string;
languageCode: string; languageCode: string;
@@ -255,14 +261,18 @@ function parseChapters(description: string): Chapter[] {
return chapters.length >= 2 ? chapters : []; return chapters.length >= 2 ? chapters : [];
} }
function getBestThumbnailUrl(videoId: string, data: any): string { function getThumbnailUrls(videoId: string, data: any): string[] {
const urls = [
`https://i.ytimg.com/vi/${videoId}/maxresdefault.jpg`,
`https://i.ytimg.com/vi/${videoId}/hqdefault.jpg`,
];
const thumbnails = data?.videoDetails?.thumbnail?.thumbnails || const thumbnails = data?.videoDetails?.thumbnail?.thumbnails ||
data?.microformat?.playerMicroformatRenderer?.thumbnail?.thumbnails || []; data?.microformat?.playerMicroformatRenderer?.thumbnail?.thumbnails || [];
if (thumbnails.length) { if (thumbnails.length) {
const sorted = [...thumbnails].sort((a: any, b: any) => (b.width || 0) - (a.width || 0)); const sorted = [...thumbnails].sort((a: any, b: any) => (b.width || 0) - (a.width || 0));
return sorted[0].url; for (const t of sorted) if (t.url && !urls.includes(t.url)) urls.push(t.url);
} }
return `https://i.ytimg.com/vi/${videoId}/maxresdefault.jpg`; return urls;
} }
function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; name: string; isGenerated: boolean }, chapters: Chapter[]): VideoMeta { function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; name: string; isGenerated: boolean }, chapters: Chapter[]): VideoMeta {
@@ -278,15 +288,13 @@ function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; na
publishDate: mf.publishDate || mf.uploadDate || "", publishDate: mf.publishDate || mf.uploadDate || "",
url: `https://www.youtube.com/watch?v=${videoId}`, url: `https://www.youtube.com/watch?v=${videoId}`,
coverImage: "", coverImage: "",
thumbnailUrl: getBestThumbnailUrl(videoId, data), thumbnailUrl: getThumbnailUrls(videoId, data)[0],
language: langInfo, language: langInfo,
chapters, chapters,
}; };
} }
async function downloadCoverImage(url: string, outputPath: string): Promise<boolean> { async function downloadCoverImage(urls: string[], outputPath: string): Promise<boolean> {
const urls = [url];
if (url.includes("maxresdefault")) urls.push(url.replace("maxresdefault", "hqdefault"));
for (const u of urls) { for (const u of urls) {
try { try {
const r = await fetch(u); const r = await fetch(u);
@@ -356,6 +364,129 @@ function groupIntoParagraphs(snippets: Snippet[]): Paragraph[] {
return paras; return paras;
} }
// --- Sentence segmentation ---
const SENTENCE_END_RE = /[.?!…。?!⁈⁇‼‽.]/;
function isCJK(ch: string): boolean {
const code = ch.charCodeAt(0);
return (code >= 0x4E00 && code <= 0x9FFF) ||
(code >= 0x3040 && code <= 0x309F) ||
(code >= 0x30A0 && code <= 0x30FF) ||
(code >= 0xAC00 && code <= 0xD7AF) ||
(code >= 0x3400 && code <= 0x4DBF) ||
(code >= 0xF900 && code <= 0xFAFF);
}
function splitSnippetAtPunctuation(s: Snippet): { text: string; start: number; end: number }[] {
const { text, start, duration } = s;
const end = start + duration;
if (!text.length) return [];
const splitPoints: number[] = [];
for (let i = 0; i < text.length; i++) {
if (SENTENCE_END_RE.test(text[i])) {
while (i + 1 < text.length && SENTENCE_END_RE.test(text[i + 1])) i++;
if (i < text.length - 1) splitPoints.push(i);
}
}
if (!splitPoints.length) return [{ text, start, end }];
const parts: { text: string; start: number; end: number }[] = [];
let prev = 0;
for (const pos of splitPoints) {
const partText = text.slice(prev, pos + 1).trim();
if (partText) {
parts.push({
text: partText,
start: start + (prev / text.length) * duration,
end: start + ((pos + 1) / text.length) * duration,
});
}
prev = pos + 1;
}
const remaining = text.slice(prev).trim();
if (remaining) {
parts.push({ text: remaining, start: start + (prev / text.length) * duration, end });
}
return parts;
}
function mergeTexts(texts: string[]): string {
if (!texts.length) return "";
let result = texts[0];
for (let i = 1; i < texts.length; i++) {
const next = texts[i];
if (!next) continue;
const lastChar = result[result.length - 1];
const firstChar = next[0];
if (isCJK(lastChar) || isCJK(firstChar)) {
result += next;
} else {
result = result.trimEnd() + " " + next.trimStart();
}
}
return result.replace(/ {2,}/g, " ");
}
function segmentIntoSentences(snippets: Snippet[]): Sentence[] {
const parts: { text: string; start: number; end: number }[] = [];
for (const s of snippets) parts.push(...splitSnippetAtPunctuation(s));
const sentences: Sentence[] = [];
let buf: { text: string; start: number; end: number }[] = [];
for (const part of parts) {
buf.push(part);
if (SENTENCE_END_RE.test(part.text[part.text.length - 1])) {
sentences.push({
text: mergeTexts(buf.map(b => b.text)),
start: ts(buf[0].start),
end: ts(buf[buf.length - 1].end),
});
buf = [];
}
}
if (buf.length) {
sentences.push({
text: mergeTexts(buf.map(b => b.text)),
start: ts(buf[0].start),
end: ts(buf[buf.length - 1].end),
});
}
return sentences;
}
function parseTs(t: string): number {
const [h, m, s] = t.split(":").map(Number);
return h * 3600 + m * 60 + s;
}
function groupSentenceParas(sentences: Sentence[]): Paragraph[] {
if (!sentences.length) return [];
const paras: Paragraph[] = [];
let buf: Sentence[] = [];
for (let i = 0; i < sentences.length; i++) {
buf.push(sentences[i]);
const last = i === sentences.length - 1;
const gap = !last && parseTs(sentences[i + 1].start) - parseTs(sentences[i].end) > 2;
if (last || gap || buf.length >= 5) {
paras.push({
text: mergeTexts(buf.map(s => s.text)),
start: parseTs(buf[0].start),
end: parseTs(buf[buf.length - 1].end),
});
buf = [];
}
}
return paras;
}
// --- Format functions --- // --- Format functions ---
function formatSrt(snippets: Snippet[]): string { function formatSrt(snippets: Snippet[]): string {
@@ -374,7 +505,7 @@ function yamlEscape(s: string): string {
return s; return s;
} }
function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps: boolean; chapters: boolean; speakers: boolean }, rawSrt?: string): string { function formatMarkdown(sentences: Sentence[], meta: VideoMeta, opts: { timestamps: boolean; chapters: boolean; speakers: boolean }, snippets?: Snippet[]): string {
let md = "---\n"; let md = "---\n";
md += `title: ${yamlEscape(meta.title)}\n`; md += `title: ${yamlEscape(meta.title)}\n`;
md += `channel: ${yamlEscape(meta.channel)}\n`; md += `channel: ${yamlEscape(meta.channel)}\n`;
@@ -392,7 +523,7 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
md += "\n"; md += "\n";
} }
md += "# Transcript\n\n"; md += "# Transcript\n\n";
md += rawSrt || formatSrt(snippets); md += snippets ? formatSrt(snippets) : "";
return md; return md;
} }
@@ -404,8 +535,8 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
md += "\n\n"; md += "\n\n";
for (let i = 0; i < chapters.length; i++) { for (let i = 0; i < chapters.length; i++) {
const nextStart = i < chapters.length - 1 ? chapters[i + 1].start : Infinity; const nextStart = i < chapters.length - 1 ? chapters[i + 1].start : Infinity;
const segs = snippets.filter(s => s.start >= chapters[i].start && s.start < nextStart); const chSentences = sentences.filter(s => parseTs(s.start) >= chapters[i].start && parseTs(s.start) < nextStart);
const paras = groupIntoParagraphs(segs); const paras = groupSentenceParas(chSentences);
md += opts.timestamps md += opts.timestamps
? `## [${ts(chapters[i].start)}] ${chapters[i].title}\n\n` ? `## [${ts(chapters[i].start)}] ${chapters[i].title}\n\n`
: `## ${chapters[i].title}\n\n`; : `## ${chapters[i].title}\n\n`;
@@ -413,7 +544,7 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
md += "\n"; md += "\n";
} }
} else { } else {
const paras = groupIntoParagraphs(snippets); const paras = groupSentenceParas(sentences);
for (const p of paras) md += opts.timestamps ? `${p.text} [${ts(p.start)}${ts(p.end)}]\n\n` : `${p.text}\n\n`; for (const p of paras) md += opts.timestamps ? `${p.text} [${ts(p.start)}${ts(p.end)}]\n\n` : `${p.text}\n\n`;
} }
@@ -471,7 +602,7 @@ function registerVideoDir(videoId: string, channelSlug: string, titleSlug: strin
} }
function hasCachedData(videoDir: string): boolean { function hasCachedData(videoDir: string): boolean {
return existsSync(join(videoDir, "meta.json")) && existsSync(join(videoDir, "transcript-raw.srt")); return existsSync(join(videoDir, "meta.json")) && existsSync(join(videoDir, "transcript-raw.json"));
} }
function loadMeta(videoDir: string): VideoMeta { function loadMeta(videoDir: string): VideoMeta {
@@ -479,16 +610,16 @@ function loadMeta(videoDir: string): VideoMeta {
} }
function loadSnippets(videoDir: string): Snippet[] { function loadSnippets(videoDir: string): Snippet[] {
return parseSrt(readFileSync(join(videoDir, "transcript-raw.srt"), "utf-8")); return JSON.parse(readFileSync(join(videoDir, "transcript-raw.json"), "utf-8"));
} }
function loadRawSrt(videoDir: string): string { function loadSentences(videoDir: string): Sentence[] {
return readFileSync(join(videoDir, "transcript-raw.srt"), "utf-8"); return JSON.parse(readFileSync(join(videoDir, "transcript-sentences.json"), "utf-8"));
} }
// --- Main processing --- // --- Main processing ---
async function fetchAndCache(videoId: string, baseDir: string, opts: Options): Promise<{ meta: VideoMeta; snippets: Snippet[]; videoDir: string }> { async function fetchAndCache(videoId: string, baseDir: string, opts: Options): Promise<{ meta: VideoMeta; snippets: Snippet[]; sentences: Sentence[]; videoDir: string }> {
const html = await fetchHtml(videoId); const html = await fetchHtml(videoId);
const apiKey = extractApiKey(html, videoId); const apiKey = extractApiKey(html, videoId);
const data = await fetchInnertubeData(videoId, apiKey); const data = await fetchInnertubeData(videoId, apiKey);
@@ -501,23 +632,22 @@ async function fetchAndCache(videoId: string, baseDir: string, opts: Options): P
const langInfo = { code: result.languageCode, name: result.language, isGenerated: info.isGenerated }; const langInfo = { code: result.languageCode, name: result.language, isGenerated: info.isGenerated };
const meta = buildVideoMeta(data, videoId, langInfo, chapters); const meta = buildVideoMeta(data, videoId, langInfo, chapters);
// Compute directory: {baseDir}/{channel-slug}/{title-slug}/
const videoDir = registerVideoDir(videoId, slugify(meta.channel), slugify(meta.title), baseDir); const videoDir = registerVideoDir(videoId, slugify(meta.channel), slugify(meta.title), baseDir);
// Save raw data as SRT (pre-computed timestamps, token-efficient for LLM)
ensureDir(join(videoDir, "meta.json")); ensureDir(join(videoDir, "meta.json"));
writeFileSync(join(videoDir, "transcript-raw.srt"), formatSrt(result.snippets));
// Download cover image writeFileSync(join(videoDir, "transcript-raw.json"), JSON.stringify(result.snippets, null, 2));
const sentences = segmentIntoSentences(result.snippets);
writeFileSync(join(videoDir, "transcript-sentences.json"), JSON.stringify(sentences, null, 2));
const imgPath = join(videoDir, "imgs", "cover.jpg"); const imgPath = join(videoDir, "imgs", "cover.jpg");
ensureDir(imgPath); ensureDir(imgPath);
const downloaded = await downloadCoverImage(meta.thumbnailUrl, imgPath); const downloaded = await downloadCoverImage(getThumbnailUrls(videoId, data), imgPath);
meta.coverImage = downloaded ? "imgs/cover.jpg" : ""; meta.coverImage = downloaded ? "imgs/cover.jpg" : "";
// Save meta (after cover image result is known)
writeFileSync(join(videoDir, "meta.json"), JSON.stringify(meta, null, 2)); writeFileSync(join(videoDir, "meta.json"), JSON.stringify(meta, null, 2));
return { meta, snippets: result.snippets, videoDir }; return { meta, snippets: result.snippets, sentences, videoDir };
} }
async function processVideo(videoId: string, opts: Options): Promise<VideoResult> { async function processVideo(videoId: string, opts: Options): Promise<VideoResult> {
@@ -534,17 +664,16 @@ async function processVideo(videoId: string, opts: Options): Promise<VideoResult
return { videoId, title, content: formatListOutput(videoId, title, transcripts) }; return { videoId, title, content: formatListOutput(videoId, title, transcripts) };
} }
// Fetch phase: use cache via index lookup
let videoDir = lookupVideoDir(videoId, baseDir); let videoDir = lookupVideoDir(videoId, baseDir);
let meta: VideoMeta; let meta: VideoMeta;
let snippets: Snippet[]; let snippets: Snippet[];
let rawSrt: string | undefined; let sentences: Sentence[];
let needsFetch = opts.refresh || !videoDir || !hasCachedData(videoDir); let needsFetch = opts.refresh || !videoDir || !hasCachedData(videoDir);
if (!needsFetch && videoDir) { if (!needsFetch && videoDir) {
meta = loadMeta(videoDir); meta = loadMeta(videoDir);
snippets = loadSnippets(videoDir); snippets = loadSnippets(videoDir);
rawSrt = loadRawSrt(videoDir); sentences = loadSentences(videoDir);
const wantLangs = opts.translate ? [opts.translate] : opts.languages; const wantLangs = opts.translate ? [opts.translate] : opts.languages;
if (!wantLangs.includes(meta.language.code)) needsFetch = true; if (!wantLangs.includes(meta.language.code)) needsFetch = true;
} }
@@ -553,26 +682,26 @@ async function processVideo(videoId: string, opts: Options): Promise<VideoResult
const result = await fetchAndCache(videoId, baseDir, opts); const result = await fetchAndCache(videoId, baseDir, opts);
meta = result.meta; meta = result.meta;
snippets = result.snippets; snippets = result.snippets;
sentences = result.sentences;
videoDir = result.videoDir; videoDir = result.videoDir;
rawSrt = loadRawSrt(videoDir);
} else { } else {
meta = meta!; meta = meta!;
snippets = snippets!; snippets = snippets!;
sentences = sentences!;
} }
// Format phase
let content: string; let content: string;
let ext: string; let ext: string;
if (opts.format === "srt") { if (opts.format === "srt") {
content = rawSrt || formatSrt(snippets); content = formatSrt(snippets);
ext = "srt"; ext = "srt";
} else { } else {
content = formatMarkdown(snippets, meta, { content = formatMarkdown(sentences, meta, {
timestamps: opts.timestamps, timestamps: opts.timestamps,
chapters: opts.chapters, chapters: opts.chapters,
speakers: opts.speakers, speakers: opts.speakers,
}, rawSrt); }, snippets);
ext = "md"; ext = "md";
} }