mirror of
https://github.com/JimLiu/baoyu-skills.git
synced 2026-07-27 04:09:47 +08:00
feat(baoyu-youtube-transcript): add sentence segmentation and improve caching
This commit is contained in:
@@ -76,7 +76,7 @@ Simply tell Claude Code:
|
|||||||
|--------|-------------|--------|
|
|--------|-------------|--------|
|
||||||
| **content-skills** | Content generation and publishing | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
|
| **content-skills** | Content generation and publishing | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
|
||||||
| **ai-generation-skills** | AI-powered generation backends | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
|
| **ai-generation-skills** | AI-powered generation backends | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
|
||||||
| **utility-skills** | Utility tools for content processing | [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
|
| **utility-skills** | Utility tools for content processing | [youtube-transcript](#baoyu-youtube-transcript), [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
|
||||||
|
|
||||||
## Update Skills
|
## Update Skills
|
||||||
|
|
||||||
@@ -766,6 +766,40 @@ Interacts with Gemini Web to generate text and images.
|
|||||||
|
|
||||||
Utility tools for content processing.
|
Utility tools for content processing.
|
||||||
|
|
||||||
|
#### baoyu-youtube-transcript
|
||||||
|
|
||||||
|
Download YouTube video transcripts/subtitles and cover images. Supports multiple languages, translation, chapters, and speaker identification. Caches raw data for fast re-formatting.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Default: markdown with timestamps
|
||||||
|
/baoyu-youtube-transcript https://www.youtube.com/watch?v=VIDEO_ID
|
||||||
|
|
||||||
|
# Specify languages (priority order)
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --languages zh,en,ja
|
||||||
|
|
||||||
|
# With chapters and speaker identification
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --chapters --speakers
|
||||||
|
|
||||||
|
# SRT subtitle format
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --format srt
|
||||||
|
|
||||||
|
# List available transcripts
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --list
|
||||||
|
```
|
||||||
|
|
||||||
|
**Options**:
|
||||||
|
| Option | Description | Default |
|
||||||
|
|--------|-------------|---------|
|
||||||
|
| `<url-or-id>` | YouTube URL or video ID | Required |
|
||||||
|
| `--languages <codes>` | Language codes, comma-separated | `en` |
|
||||||
|
| `--format <fmt>` | Output format: `text`, `srt` | `text` |
|
||||||
|
| `--translate <code>` | Translate to specified language | |
|
||||||
|
| `--chapters` | Chapter segmentation from video description | |
|
||||||
|
| `--speakers` | Speaker identification (requires AI post-processing) | |
|
||||||
|
| `--no-timestamps` | Disable timestamps | |
|
||||||
|
| `--list` | List available transcripts | |
|
||||||
|
| `--refresh` | Force re-fetch, ignore cache | |
|
||||||
|
|
||||||
#### baoyu-url-to-markdown
|
#### baoyu-url-to-markdown
|
||||||
|
|
||||||
Fetch any URL via Chrome CDP and convert to clean markdown. Saves rendered HTML snapshot alongside the markdown, and automatically falls back to a legacy extractor when Defuddle fails.
|
Fetch any URL via Chrome CDP and convert to clean markdown. Saves rendered HTML snapshot alongside the markdown, and automatically falls back to a legacy extractor when Defuddle fails.
|
||||||
|
|||||||
+35
-1
@@ -76,7 +76,7 @@ clawhub install baoyu-markdown-to-html
|
|||||||
|------|------|----------|
|
|------|------|----------|
|
||||||
| **content-skills** | 内容生成和发布 | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
|
| **content-skills** | 内容生成和发布 | [xhs-images](#baoyu-xhs-images), [infographic](#baoyu-infographic), [cover-image](#baoyu-cover-image), [slide-deck](#baoyu-slide-deck), [comic](#baoyu-comic), [article-illustrator](#baoyu-article-illustrator), [post-to-x](#baoyu-post-to-x), [post-to-wechat](#baoyu-post-to-wechat), [post-to-weibo](#baoyu-post-to-weibo) |
|
||||||
| **ai-generation-skills** | AI 生成后端 | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
|
| **ai-generation-skills** | AI 生成后端 | [image-gen](#baoyu-image-gen), [danger-gemini-web](#baoyu-danger-gemini-web) |
|
||||||
| **utility-skills** | 内容处理工具 | [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
|
| **utility-skills** | 内容处理工具 | [youtube-transcript](#baoyu-youtube-transcript), [url-to-markdown](#baoyu-url-to-markdown), [danger-x-to-markdown](#baoyu-danger-x-to-markdown), [compress-image](#baoyu-compress-image), [format-markdown](#baoyu-format-markdown), [markdown-to-html](#baoyu-markdown-to-html), [translate](#baoyu-translate) |
|
||||||
|
|
||||||
## 更新技能
|
## 更新技能
|
||||||
|
|
||||||
@@ -766,6 +766,40 @@ AI 驱动的生成后端。
|
|||||||
|
|
||||||
内容处理工具。
|
内容处理工具。
|
||||||
|
|
||||||
|
#### baoyu-youtube-transcript
|
||||||
|
|
||||||
|
下载 YouTube 视频字幕/转录文本和封面图片。支持多语言、翻译、章节分段和说话人识别。缓存原始数据以便快速重新格式化。
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 默认:带时间戳的 Markdown
|
||||||
|
/baoyu-youtube-transcript https://www.youtube.com/watch?v=VIDEO_ID
|
||||||
|
|
||||||
|
# 指定语言(按优先级排列)
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --languages zh,en,ja
|
||||||
|
|
||||||
|
# 章节分段 + 说话人识别
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --chapters --speakers
|
||||||
|
|
||||||
|
# SRT 字幕格式
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --format srt
|
||||||
|
|
||||||
|
# 列出可用字幕
|
||||||
|
/baoyu-youtube-transcript https://youtu.be/VIDEO_ID --list
|
||||||
|
```
|
||||||
|
|
||||||
|
**选项**:
|
||||||
|
| 选项 | 说明 | 默认值 |
|
||||||
|
|------|------|--------|
|
||||||
|
| `<url-or-id>` | YouTube URL 或视频 ID | 必填 |
|
||||||
|
| `--languages <codes>` | 语言代码,逗号分隔 | `en` |
|
||||||
|
| `--format <fmt>` | 输出格式:`text`、`srt` | `text` |
|
||||||
|
| `--translate <code>` | 翻译为指定语言 | |
|
||||||
|
| `--chapters` | 根据视频描述进行章节分段 | |
|
||||||
|
| `--speakers` | 说话人识别(需 AI 后处理) | |
|
||||||
|
| `--no-timestamps` | 禁用时间戳 | |
|
||||||
|
| `--list` | 列出可用字幕 | |
|
||||||
|
| `--refresh` | 强制重新获取,忽略缓存 | |
|
||||||
|
|
||||||
#### baoyu-url-to-markdown
|
#### baoyu-url-to-markdown
|
||||||
|
|
||||||
通过 Chrome CDP 抓取任意 URL 并转换为 Markdown。同时保存渲染后的 HTML 快照,Defuddle 失败时自动回退到旧版提取器。
|
通过 Chrome CDP 抓取任意 URL 并转换为 Markdown。同时保存渲染后的 HTML 快照,Defuddle 失败时自动回退到旧版提取器。
|
||||||
|
|||||||
@@ -98,11 +98,12 @@ youtube-transcript/
|
|||||||
├── .index.json # Video ID → directory path mapping (for cache lookup)
|
├── .index.json # Video ID → directory path mapping (for cache lookup)
|
||||||
└── {channel-slug}/{title-full-slug}/
|
└── {channel-slug}/{title-full-slug}/
|
||||||
├── meta.json # Video metadata (title, channel, description, duration, chapters, etc.)
|
├── meta.json # Video metadata (title, channel, description, duration, chapters, etc.)
|
||||||
├── transcript-raw.srt # Raw transcript in SRT format (cached, token-efficient for LLM)
|
├── transcript-raw.json # Raw transcript snippets from YouTube API (cached)
|
||||||
|
├── transcript-sentences.json # Sentence-segmented transcript (split by punctuation, merged across snippets)
|
||||||
├── imgs/
|
├── imgs/
|
||||||
│ └── cover.jpg # Video thumbnail
|
│ └── cover.jpg # Video thumbnail
|
||||||
├── transcript.md # Markdown transcript
|
├── transcript.md # Markdown transcript (generated from sentences)
|
||||||
└── transcript.srt # SRT subtitle (if --format srt)
|
└── transcript.srt # SRT subtitle (generated from raw snippets, if --format srt)
|
||||||
```
|
```
|
||||||
|
|
||||||
- `{channel-slug}`: Channel name in kebab-case
|
- `{channel-slug}`: Channel name in kebab-case
|
||||||
@@ -114,11 +115,14 @@ The `--list` mode outputs to stdout only (no file saved).
|
|||||||
|
|
||||||
On first fetch, the script saves:
|
On first fetch, the script saves:
|
||||||
- `meta.json` — video metadata, chapters, cover image path, language info
|
- `meta.json` — video metadata, chapters, cover image path, language info
|
||||||
- `transcript-raw.srt` — raw transcript in SRT format (pre-computed timestamps, token-efficient for LLM processing)
|
- `transcript-raw.json` — raw transcript snippets from YouTube API (`{ text, start, duration }[]`)
|
||||||
|
- `transcript-sentences.json` — sentence-segmented transcript (`{ text, start: "HH:mm:ss", end: "HH:mm:ss" }[]`), split by sentence-ending punctuation (`.?!…。?!` etc.), timestamps proportionally allocated by character length, CJK-aware text merging
|
||||||
- `imgs/cover.jpg` — video thumbnail
|
- `imgs/cover.jpg` — video thumbnail
|
||||||
|
|
||||||
Subsequent runs for the same video use cached data (no network calls). Use `--refresh` to force re-fetch. If a different language is requested, the cache is automatically refreshed.
|
Subsequent runs for the same video use cached data (no network calls). Use `--refresh` to force re-fetch. If a different language is requested, the cache is automatically refreshed.
|
||||||
|
|
||||||
|
SRT output (`--format srt`) is generated from `transcript-raw.json`. Text/markdown output uses `transcript-sentences.json` for natural sentence boundaries.
|
||||||
|
|
||||||
## Workflow
|
## Workflow
|
||||||
|
|
||||||
When user provides a YouTube URL and wants the transcript:
|
When user provides a YouTube URL and wants the transcript:
|
||||||
@@ -148,7 +152,7 @@ Speaker identification requires AI processing. The script outputs a raw `.md` fi
|
|||||||
- Chapter list from description (if available)
|
- Chapter list from description (if available)
|
||||||
- Raw transcript in SRT format (pre-computed start/end timestamps, token-efficient)
|
- Raw transcript in SRT format (pre-computed start/end timestamps, token-efficient)
|
||||||
|
|
||||||
After the script saves the raw file:
|
After the script saves the raw file, spawn a sub-agent (use a cheaper model like Sonnet for cost efficiency) to process speaker identification:
|
||||||
|
|
||||||
1. Read the saved `.md` file
|
1. Read the saved `.md` file
|
||||||
2. Read the prompt template at `{baseDir}/prompts/speaker-transcript.md`
|
2. Read the prompt template at `{baseDir}/prompts/speaker-transcript.md`
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ Use the same language as the transcription for the title and ToC.
|
|||||||
|
|
||||||
**Chapters:**
|
**Chapters:**
|
||||||
```
|
```
|
||||||
## [HH:MM:SS] Chapter Title
|
## Chapter Title [HH:MM:SS]
|
||||||
```
|
```
|
||||||
Two blank lines between chapters.
|
Two blank lines between chapters.
|
||||||
|
|
||||||
@@ -87,14 +87,14 @@ language: en
|
|||||||
* [00:00:12] Overview of the New Research
|
* [00:00:12] Overview of the New Research
|
||||||
|
|
||||||
|
|
||||||
## [00:00:00] Introduction and Welcome
|
## Introduction and Welcome [00:00:00]
|
||||||
|
|
||||||
**Host:** Welcome back to the show. Today, we have a, uh, very special guest, Jane Doe. [00:00:00 → 00:00:03]
|
**Host:** Welcome back to the show. Today, we have a, uh, very special guest, Jane Doe. [00:00:00 → 00:00:03]
|
||||||
|
|
||||||
**Jane Doe:** Thank you for having me. I'm excited to be here and discuss the findings. [00:00:03 → 00:00:07]
|
**Jane Doe:** Thank you for having me. I'm excited to be here and discuss the findings. [00:00:03 → 00:00:07]
|
||||||
|
|
||||||
|
|
||||||
## [00:00:12] Overview of the New Research
|
## Overview of the New Research [00:00:12]
|
||||||
|
|
||||||
**Host:** So, Jane, before we get into the nitty-gritty, could you, you know, give us a brief overview for our audience? [00:00:12 → 00:00:16]
|
**Host:** So, Jane, before we get into the nitty-gritty, could you, you know, give us a brief overview for our audience? [00:00:12 → 00:00:16]
|
||||||
|
|
||||||
|
|||||||
@@ -26,6 +26,12 @@ interface Snippet {
|
|||||||
duration: number;
|
duration: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
interface Sentence {
|
||||||
|
text: string;
|
||||||
|
start: string;
|
||||||
|
end: string;
|
||||||
|
}
|
||||||
|
|
||||||
interface TranscriptInfo {
|
interface TranscriptInfo {
|
||||||
language: string;
|
language: string;
|
||||||
languageCode: string;
|
languageCode: string;
|
||||||
@@ -255,14 +261,18 @@ function parseChapters(description: string): Chapter[] {
|
|||||||
return chapters.length >= 2 ? chapters : [];
|
return chapters.length >= 2 ? chapters : [];
|
||||||
}
|
}
|
||||||
|
|
||||||
function getBestThumbnailUrl(videoId: string, data: any): string {
|
function getThumbnailUrls(videoId: string, data: any): string[] {
|
||||||
|
const urls = [
|
||||||
|
`https://i.ytimg.com/vi/${videoId}/maxresdefault.jpg`,
|
||||||
|
`https://i.ytimg.com/vi/${videoId}/hqdefault.jpg`,
|
||||||
|
];
|
||||||
const thumbnails = data?.videoDetails?.thumbnail?.thumbnails ||
|
const thumbnails = data?.videoDetails?.thumbnail?.thumbnails ||
|
||||||
data?.microformat?.playerMicroformatRenderer?.thumbnail?.thumbnails || [];
|
data?.microformat?.playerMicroformatRenderer?.thumbnail?.thumbnails || [];
|
||||||
if (thumbnails.length) {
|
if (thumbnails.length) {
|
||||||
const sorted = [...thumbnails].sort((a: any, b: any) => (b.width || 0) - (a.width || 0));
|
const sorted = [...thumbnails].sort((a: any, b: any) => (b.width || 0) - (a.width || 0));
|
||||||
return sorted[0].url;
|
for (const t of sorted) if (t.url && !urls.includes(t.url)) urls.push(t.url);
|
||||||
}
|
}
|
||||||
return `https://i.ytimg.com/vi/${videoId}/maxresdefault.jpg`;
|
return urls;
|
||||||
}
|
}
|
||||||
|
|
||||||
function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; name: string; isGenerated: boolean }, chapters: Chapter[]): VideoMeta {
|
function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; name: string; isGenerated: boolean }, chapters: Chapter[]): VideoMeta {
|
||||||
@@ -278,15 +288,13 @@ function buildVideoMeta(data: any, videoId: string, langInfo: { code: string; na
|
|||||||
publishDate: mf.publishDate || mf.uploadDate || "",
|
publishDate: mf.publishDate || mf.uploadDate || "",
|
||||||
url: `https://www.youtube.com/watch?v=${videoId}`,
|
url: `https://www.youtube.com/watch?v=${videoId}`,
|
||||||
coverImage: "",
|
coverImage: "",
|
||||||
thumbnailUrl: getBestThumbnailUrl(videoId, data),
|
thumbnailUrl: getThumbnailUrls(videoId, data)[0],
|
||||||
language: langInfo,
|
language: langInfo,
|
||||||
chapters,
|
chapters,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
async function downloadCoverImage(url: string, outputPath: string): Promise<boolean> {
|
async function downloadCoverImage(urls: string[], outputPath: string): Promise<boolean> {
|
||||||
const urls = [url];
|
|
||||||
if (url.includes("maxresdefault")) urls.push(url.replace("maxresdefault", "hqdefault"));
|
|
||||||
for (const u of urls) {
|
for (const u of urls) {
|
||||||
try {
|
try {
|
||||||
const r = await fetch(u);
|
const r = await fetch(u);
|
||||||
@@ -356,6 +364,129 @@ function groupIntoParagraphs(snippets: Snippet[]): Paragraph[] {
|
|||||||
return paras;
|
return paras;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- Sentence segmentation ---
|
||||||
|
|
||||||
|
const SENTENCE_END_RE = /[.?!…。?!⁈⁇‼‽.]/;
|
||||||
|
|
||||||
|
function isCJK(ch: string): boolean {
|
||||||
|
const code = ch.charCodeAt(0);
|
||||||
|
return (code >= 0x4E00 && code <= 0x9FFF) ||
|
||||||
|
(code >= 0x3040 && code <= 0x309F) ||
|
||||||
|
(code >= 0x30A0 && code <= 0x30FF) ||
|
||||||
|
(code >= 0xAC00 && code <= 0xD7AF) ||
|
||||||
|
(code >= 0x3400 && code <= 0x4DBF) ||
|
||||||
|
(code >= 0xF900 && code <= 0xFAFF);
|
||||||
|
}
|
||||||
|
|
||||||
|
function splitSnippetAtPunctuation(s: Snippet): { text: string; start: number; end: number }[] {
|
||||||
|
const { text, start, duration } = s;
|
||||||
|
const end = start + duration;
|
||||||
|
if (!text.length) return [];
|
||||||
|
|
||||||
|
const splitPoints: number[] = [];
|
||||||
|
for (let i = 0; i < text.length; i++) {
|
||||||
|
if (SENTENCE_END_RE.test(text[i])) {
|
||||||
|
while (i + 1 < text.length && SENTENCE_END_RE.test(text[i + 1])) i++;
|
||||||
|
if (i < text.length - 1) splitPoints.push(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!splitPoints.length) return [{ text, start, end }];
|
||||||
|
|
||||||
|
const parts: { text: string; start: number; end: number }[] = [];
|
||||||
|
let prev = 0;
|
||||||
|
for (const pos of splitPoints) {
|
||||||
|
const partText = text.slice(prev, pos + 1).trim();
|
||||||
|
if (partText) {
|
||||||
|
parts.push({
|
||||||
|
text: partText,
|
||||||
|
start: start + (prev / text.length) * duration,
|
||||||
|
end: start + ((pos + 1) / text.length) * duration,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
prev = pos + 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const remaining = text.slice(prev).trim();
|
||||||
|
if (remaining) {
|
||||||
|
parts.push({ text: remaining, start: start + (prev / text.length) * duration, end });
|
||||||
|
}
|
||||||
|
|
||||||
|
return parts;
|
||||||
|
}
|
||||||
|
|
||||||
|
function mergeTexts(texts: string[]): string {
|
||||||
|
if (!texts.length) return "";
|
||||||
|
let result = texts[0];
|
||||||
|
for (let i = 1; i < texts.length; i++) {
|
||||||
|
const next = texts[i];
|
||||||
|
if (!next) continue;
|
||||||
|
const lastChar = result[result.length - 1];
|
||||||
|
const firstChar = next[0];
|
||||||
|
if (isCJK(lastChar) || isCJK(firstChar)) {
|
||||||
|
result += next;
|
||||||
|
} else {
|
||||||
|
result = result.trimEnd() + " " + next.trimStart();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result.replace(/ {2,}/g, " ");
|
||||||
|
}
|
||||||
|
|
||||||
|
function segmentIntoSentences(snippets: Snippet[]): Sentence[] {
|
||||||
|
const parts: { text: string; start: number; end: number }[] = [];
|
||||||
|
for (const s of snippets) parts.push(...splitSnippetAtPunctuation(s));
|
||||||
|
|
||||||
|
const sentences: Sentence[] = [];
|
||||||
|
let buf: { text: string; start: number; end: number }[] = [];
|
||||||
|
|
||||||
|
for (const part of parts) {
|
||||||
|
buf.push(part);
|
||||||
|
if (SENTENCE_END_RE.test(part.text[part.text.length - 1])) {
|
||||||
|
sentences.push({
|
||||||
|
text: mergeTexts(buf.map(b => b.text)),
|
||||||
|
start: ts(buf[0].start),
|
||||||
|
end: ts(buf[buf.length - 1].end),
|
||||||
|
});
|
||||||
|
buf = [];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (buf.length) {
|
||||||
|
sentences.push({
|
||||||
|
text: mergeTexts(buf.map(b => b.text)),
|
||||||
|
start: ts(buf[0].start),
|
||||||
|
end: ts(buf[buf.length - 1].end),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
return sentences;
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseTs(t: string): number {
|
||||||
|
const [h, m, s] = t.split(":").map(Number);
|
||||||
|
return h * 3600 + m * 60 + s;
|
||||||
|
}
|
||||||
|
|
||||||
|
function groupSentenceParas(sentences: Sentence[]): Paragraph[] {
|
||||||
|
if (!sentences.length) return [];
|
||||||
|
const paras: Paragraph[] = [];
|
||||||
|
let buf: Sentence[] = [];
|
||||||
|
for (let i = 0; i < sentences.length; i++) {
|
||||||
|
buf.push(sentences[i]);
|
||||||
|
const last = i === sentences.length - 1;
|
||||||
|
const gap = !last && parseTs(sentences[i + 1].start) - parseTs(sentences[i].end) > 2;
|
||||||
|
if (last || gap || buf.length >= 5) {
|
||||||
|
paras.push({
|
||||||
|
text: mergeTexts(buf.map(s => s.text)),
|
||||||
|
start: parseTs(buf[0].start),
|
||||||
|
end: parseTs(buf[buf.length - 1].end),
|
||||||
|
});
|
||||||
|
buf = [];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return paras;
|
||||||
|
}
|
||||||
|
|
||||||
// --- Format functions ---
|
// --- Format functions ---
|
||||||
|
|
||||||
function formatSrt(snippets: Snippet[]): string {
|
function formatSrt(snippets: Snippet[]): string {
|
||||||
@@ -374,7 +505,7 @@ function yamlEscape(s: string): string {
|
|||||||
return s;
|
return s;
|
||||||
}
|
}
|
||||||
|
|
||||||
function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps: boolean; chapters: boolean; speakers: boolean }, rawSrt?: string): string {
|
function formatMarkdown(sentences: Sentence[], meta: VideoMeta, opts: { timestamps: boolean; chapters: boolean; speakers: boolean }, snippets?: Snippet[]): string {
|
||||||
let md = "---\n";
|
let md = "---\n";
|
||||||
md += `title: ${yamlEscape(meta.title)}\n`;
|
md += `title: ${yamlEscape(meta.title)}\n`;
|
||||||
md += `channel: ${yamlEscape(meta.channel)}\n`;
|
md += `channel: ${yamlEscape(meta.channel)}\n`;
|
||||||
@@ -392,7 +523,7 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
|
|||||||
md += "\n";
|
md += "\n";
|
||||||
}
|
}
|
||||||
md += "# Transcript\n\n";
|
md += "# Transcript\n\n";
|
||||||
md += rawSrt || formatSrt(snippets);
|
md += snippets ? formatSrt(snippets) : "";
|
||||||
return md;
|
return md;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -404,8 +535,8 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
|
|||||||
md += "\n\n";
|
md += "\n\n";
|
||||||
for (let i = 0; i < chapters.length; i++) {
|
for (let i = 0; i < chapters.length; i++) {
|
||||||
const nextStart = i < chapters.length - 1 ? chapters[i + 1].start : Infinity;
|
const nextStart = i < chapters.length - 1 ? chapters[i + 1].start : Infinity;
|
||||||
const segs = snippets.filter(s => s.start >= chapters[i].start && s.start < nextStart);
|
const chSentences = sentences.filter(s => parseTs(s.start) >= chapters[i].start && parseTs(s.start) < nextStart);
|
||||||
const paras = groupIntoParagraphs(segs);
|
const paras = groupSentenceParas(chSentences);
|
||||||
md += opts.timestamps
|
md += opts.timestamps
|
||||||
? `## [${ts(chapters[i].start)}] ${chapters[i].title}\n\n`
|
? `## [${ts(chapters[i].start)}] ${chapters[i].title}\n\n`
|
||||||
: `## ${chapters[i].title}\n\n`;
|
: `## ${chapters[i].title}\n\n`;
|
||||||
@@ -413,7 +544,7 @@ function formatMarkdown(snippets: Snippet[], meta: VideoMeta, opts: { timestamps
|
|||||||
md += "\n";
|
md += "\n";
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
const paras = groupIntoParagraphs(snippets);
|
const paras = groupSentenceParas(sentences);
|
||||||
for (const p of paras) md += opts.timestamps ? `${p.text} [${ts(p.start)} → ${ts(p.end)}]\n\n` : `${p.text}\n\n`;
|
for (const p of paras) md += opts.timestamps ? `${p.text} [${ts(p.start)} → ${ts(p.end)}]\n\n` : `${p.text}\n\n`;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -471,7 +602,7 @@ function registerVideoDir(videoId: string, channelSlug: string, titleSlug: strin
|
|||||||
}
|
}
|
||||||
|
|
||||||
function hasCachedData(videoDir: string): boolean {
|
function hasCachedData(videoDir: string): boolean {
|
||||||
return existsSync(join(videoDir, "meta.json")) && existsSync(join(videoDir, "transcript-raw.srt"));
|
return existsSync(join(videoDir, "meta.json")) && existsSync(join(videoDir, "transcript-raw.json"));
|
||||||
}
|
}
|
||||||
|
|
||||||
function loadMeta(videoDir: string): VideoMeta {
|
function loadMeta(videoDir: string): VideoMeta {
|
||||||
@@ -479,16 +610,16 @@ function loadMeta(videoDir: string): VideoMeta {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function loadSnippets(videoDir: string): Snippet[] {
|
function loadSnippets(videoDir: string): Snippet[] {
|
||||||
return parseSrt(readFileSync(join(videoDir, "transcript-raw.srt"), "utf-8"));
|
return JSON.parse(readFileSync(join(videoDir, "transcript-raw.json"), "utf-8"));
|
||||||
}
|
}
|
||||||
|
|
||||||
function loadRawSrt(videoDir: string): string {
|
function loadSentences(videoDir: string): Sentence[] {
|
||||||
return readFileSync(join(videoDir, "transcript-raw.srt"), "utf-8");
|
return JSON.parse(readFileSync(join(videoDir, "transcript-sentences.json"), "utf-8"));
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- Main processing ---
|
// --- Main processing ---
|
||||||
|
|
||||||
async function fetchAndCache(videoId: string, baseDir: string, opts: Options): Promise<{ meta: VideoMeta; snippets: Snippet[]; videoDir: string }> {
|
async function fetchAndCache(videoId: string, baseDir: string, opts: Options): Promise<{ meta: VideoMeta; snippets: Snippet[]; sentences: Sentence[]; videoDir: string }> {
|
||||||
const html = await fetchHtml(videoId);
|
const html = await fetchHtml(videoId);
|
||||||
const apiKey = extractApiKey(html, videoId);
|
const apiKey = extractApiKey(html, videoId);
|
||||||
const data = await fetchInnertubeData(videoId, apiKey);
|
const data = await fetchInnertubeData(videoId, apiKey);
|
||||||
@@ -501,23 +632,22 @@ async function fetchAndCache(videoId: string, baseDir: string, opts: Options): P
|
|||||||
const langInfo = { code: result.languageCode, name: result.language, isGenerated: info.isGenerated };
|
const langInfo = { code: result.languageCode, name: result.language, isGenerated: info.isGenerated };
|
||||||
const meta = buildVideoMeta(data, videoId, langInfo, chapters);
|
const meta = buildVideoMeta(data, videoId, langInfo, chapters);
|
||||||
|
|
||||||
// Compute directory: {baseDir}/{channel-slug}/{title-slug}/
|
|
||||||
const videoDir = registerVideoDir(videoId, slugify(meta.channel), slugify(meta.title), baseDir);
|
const videoDir = registerVideoDir(videoId, slugify(meta.channel), slugify(meta.title), baseDir);
|
||||||
|
|
||||||
// Save raw data as SRT (pre-computed timestamps, token-efficient for LLM)
|
|
||||||
ensureDir(join(videoDir, "meta.json"));
|
ensureDir(join(videoDir, "meta.json"));
|
||||||
writeFileSync(join(videoDir, "transcript-raw.srt"), formatSrt(result.snippets));
|
|
||||||
|
|
||||||
// Download cover image
|
writeFileSync(join(videoDir, "transcript-raw.json"), JSON.stringify(result.snippets, null, 2));
|
||||||
|
|
||||||
|
const sentences = segmentIntoSentences(result.snippets);
|
||||||
|
writeFileSync(join(videoDir, "transcript-sentences.json"), JSON.stringify(sentences, null, 2));
|
||||||
|
|
||||||
const imgPath = join(videoDir, "imgs", "cover.jpg");
|
const imgPath = join(videoDir, "imgs", "cover.jpg");
|
||||||
ensureDir(imgPath);
|
ensureDir(imgPath);
|
||||||
const downloaded = await downloadCoverImage(meta.thumbnailUrl, imgPath);
|
const downloaded = await downloadCoverImage(getThumbnailUrls(videoId, data), imgPath);
|
||||||
meta.coverImage = downloaded ? "imgs/cover.jpg" : "";
|
meta.coverImage = downloaded ? "imgs/cover.jpg" : "";
|
||||||
|
|
||||||
// Save meta (after cover image result is known)
|
|
||||||
writeFileSync(join(videoDir, "meta.json"), JSON.stringify(meta, null, 2));
|
writeFileSync(join(videoDir, "meta.json"), JSON.stringify(meta, null, 2));
|
||||||
|
|
||||||
return { meta, snippets: result.snippets, videoDir };
|
return { meta, snippets: result.snippets, sentences, videoDir };
|
||||||
}
|
}
|
||||||
|
|
||||||
async function processVideo(videoId: string, opts: Options): Promise<VideoResult> {
|
async function processVideo(videoId: string, opts: Options): Promise<VideoResult> {
|
||||||
@@ -534,17 +664,16 @@ async function processVideo(videoId: string, opts: Options): Promise<VideoResult
|
|||||||
return { videoId, title, content: formatListOutput(videoId, title, transcripts) };
|
return { videoId, title, content: formatListOutput(videoId, title, transcripts) };
|
||||||
}
|
}
|
||||||
|
|
||||||
// Fetch phase: use cache via index lookup
|
|
||||||
let videoDir = lookupVideoDir(videoId, baseDir);
|
let videoDir = lookupVideoDir(videoId, baseDir);
|
||||||
let meta: VideoMeta;
|
let meta: VideoMeta;
|
||||||
let snippets: Snippet[];
|
let snippets: Snippet[];
|
||||||
let rawSrt: string | undefined;
|
let sentences: Sentence[];
|
||||||
let needsFetch = opts.refresh || !videoDir || !hasCachedData(videoDir);
|
let needsFetch = opts.refresh || !videoDir || !hasCachedData(videoDir);
|
||||||
|
|
||||||
if (!needsFetch && videoDir) {
|
if (!needsFetch && videoDir) {
|
||||||
meta = loadMeta(videoDir);
|
meta = loadMeta(videoDir);
|
||||||
snippets = loadSnippets(videoDir);
|
snippets = loadSnippets(videoDir);
|
||||||
rawSrt = loadRawSrt(videoDir);
|
sentences = loadSentences(videoDir);
|
||||||
const wantLangs = opts.translate ? [opts.translate] : opts.languages;
|
const wantLangs = opts.translate ? [opts.translate] : opts.languages;
|
||||||
if (!wantLangs.includes(meta.language.code)) needsFetch = true;
|
if (!wantLangs.includes(meta.language.code)) needsFetch = true;
|
||||||
}
|
}
|
||||||
@@ -553,26 +682,26 @@ async function processVideo(videoId: string, opts: Options): Promise<VideoResult
|
|||||||
const result = await fetchAndCache(videoId, baseDir, opts);
|
const result = await fetchAndCache(videoId, baseDir, opts);
|
||||||
meta = result.meta;
|
meta = result.meta;
|
||||||
snippets = result.snippets;
|
snippets = result.snippets;
|
||||||
|
sentences = result.sentences;
|
||||||
videoDir = result.videoDir;
|
videoDir = result.videoDir;
|
||||||
rawSrt = loadRawSrt(videoDir);
|
|
||||||
} else {
|
} else {
|
||||||
meta = meta!;
|
meta = meta!;
|
||||||
snippets = snippets!;
|
snippets = snippets!;
|
||||||
|
sentences = sentences!;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Format phase
|
|
||||||
let content: string;
|
let content: string;
|
||||||
let ext: string;
|
let ext: string;
|
||||||
|
|
||||||
if (opts.format === "srt") {
|
if (opts.format === "srt") {
|
||||||
content = rawSrt || formatSrt(snippets);
|
content = formatSrt(snippets);
|
||||||
ext = "srt";
|
ext = "srt";
|
||||||
} else {
|
} else {
|
||||||
content = formatMarkdown(snippets, meta, {
|
content = formatMarkdown(sentences, meta, {
|
||||||
timestamps: opts.timestamps,
|
timestamps: opts.timestamps,
|
||||||
chapters: opts.chapters,
|
chapters: opts.chapters,
|
||||||
speakers: opts.speakers,
|
speakers: opts.speakers,
|
||||||
}, rawSrt);
|
}, snippets);
|
||||||
ext = "md";
|
ext = "md";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user