feat(baoyu-url-to-markdown): add URL-specific parser layer for X/Twitter and archive.ph

- New parsers/ module with pluggable rule system for site-specific HTML extraction
- X status parser: extract tweet text, media, quotes, author from data-testid elements
- X article parser: extract long-form article content with inline media
- archive.ph parser: restore original URL and prefer #CONTENT container
- Improved slug generation with stop words and content-aware slugs
- Output path uses subdirectory structure (domain/slug/slug.md)
- Fix: preserve anchor elements containing media in legacy converter
- Fix: smarter title deduplication in markdown document builder
This commit is contained in:
Jim Liu 宝玉
2026-03-22 15:18:46 -05:00
parent 6a4b312146
commit e5d6c8ec68
13 changed files with 953 additions and 26 deletions
@@ -521,14 +521,18 @@ turndown.addRule("collapseFigure", {
turndown.addRule("dropInvisibleAnchors", {
filter(node) {
return node.nodeName === "A" && !(node as Element).textContent?.trim();
return (
node.nodeName === "A" &&
!(node as Element).textContent?.trim() &&
!(node as Element).querySelector("img, video, picture, source")
);
},
replacement() {
return "";
},
});
function convertHtmlToMarkdown(html: string): string {
export function convertHtmlFragmentToMarkdown(html: string): string {
if (!html || !html.trim()) return "";
try {
@@ -609,7 +613,7 @@ export function shouldCompareWithLegacy(markdown: string): boolean {
export function convertWithLegacyExtractor(html: string, baseMetadata: PageMetadata): ConversionResult {
const extracted = extractFromHtml(html);
let markdown = extracted?.html ? convertHtmlToMarkdown(extracted.html) : "";
let markdown = extracted?.html ? convertHtmlFragmentToMarkdown(extracted.html) : "";
if (!markdown.trim()) {
markdown = extracted?.textContent?.trim() || fallbackPlainText(html);
}