feat(baoyu-url-to-markdown): add URL-specific parser layer for X/Twitter and archive.ph

- New parsers/ module with pluggable rule system for site-specific HTML extraction
- X status parser: extract tweet text, media, quotes, author from data-testid elements
- X article parser: extract long-form article content with inline media
- archive.ph parser: restore original URL and prefer #CONTENT container
- Improved slug generation with stop words and content-aware slugs
- Output path uses subdirectory structure (domain/slug/slug.md)
- Fix: preserve anchor elements containing media in legacy converter
- Fix: smarter title deduplication in markdown document builder
This commit is contained in:
Jim Liu 宝玉
2026-03-22 15:18:46 -05:00
parent 6a4b312146
commit e5d6c8ec68
13 changed files with 953 additions and 26 deletions
@@ -300,6 +300,24 @@ export function createMarkdownDocument(result: ConversionResult): string {
const escapedTitle = result.metadata.title.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const titleRegex = new RegExp(`^#\\s+${escapedTitle}\\s*(\\n|$)`, "i");
const hasTitle = titleRegex.test(result.markdown.trimStart());
const title = result.metadata.title && !hasTitle ? `\n\n# ${result.metadata.title}\n\n` : "\n\n";
const firstMeaningfulLine = result.markdown
.replace(/\r\n/g, "\n")
.split("\n")
.map((line) => line.trim())
.find((line) => line && !/^!?\[[^\]]*\]\([^)]+\)$/.test(line))
?.replace(/^>\s*/, "")
?.replace(/^#+\s+/, "")
?.trim();
const comparableTitle = result.metadata.title.toLowerCase().replace(/(?:\.{3}|…)\s*$/, "");
const comparableFirstLine = firstMeaningfulLine?.toLowerCase() ?? "";
const titleRepeatsContent =
comparableTitle !== "" &&
comparableFirstLine !== "" &&
(comparableFirstLine === comparableTitle ||
comparableFirstLine.startsWith(comparableTitle) ||
comparableTitle.startsWith(comparableFirstLine));
const title = result.metadata.title && !hasTitle && !titleRepeatsContent
? `\n\n# ${result.metadata.title}\n\n`
: "\n\n";
return yaml + title + result.markdown;
}