mirror of
https://github.com/JimLiu/baoyu-skills.git
synced 2026-08-08 01:43:03 +08:00
feat(baoyu-url-to-markdown): add URL-specific parser layer for X/Twitter and archive.ph
- New parsers/ module with pluggable rule system for site-specific HTML extraction - X status parser: extract tweet text, media, quotes, author from data-testid elements - X article parser: extract long-form article content with inline media - archive.ph parser: restore original URL and prefer #CONTENT container - Improved slug generation with stop words and content-aware slugs - Output path uses subdirectory structure (domain/slug/slug.md) - Fix: preserve anchor elements containing media in legacy converter - Fix: smarter title deduplication in markdown document builder
This commit is contained in:
@@ -0,0 +1,97 @@
|
||||
import { convertHtmlFragmentToMarkdown } from "../../legacy-converter.js";
|
||||
import {
|
||||
normalizeMarkdown,
|
||||
pickString,
|
||||
type ConversionResult,
|
||||
} from "../../markdown-conversion-shared.js";
|
||||
import type { UrlRuleParser, UrlRuleParserContext } from "../types.js";
|
||||
|
||||
const ARCHIVE_HOSTS = new Set([
|
||||
"archive.ph",
|
||||
"archive.is",
|
||||
"archive.today",
|
||||
"archive.md",
|
||||
"archive.vn",
|
||||
"archive.li",
|
||||
"archive.fo",
|
||||
]);
|
||||
|
||||
function isArchiveHost(url: string): boolean {
|
||||
try {
|
||||
return ARCHIVE_HOSTS.has(new URL(url).hostname.toLowerCase());
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function readOriginalUrl(document: Document): string | undefined {
|
||||
const value = document.querySelector("input[name='q']")?.getAttribute("value")?.trim();
|
||||
if (!value) return undefined;
|
||||
|
||||
try {
|
||||
return new URL(value).href;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
function summarize(text: string, maxLength: number): string | undefined {
|
||||
const normalized = text.replace(/\s+/g, " ").trim();
|
||||
if (!normalized) return undefined;
|
||||
if (normalized.length <= maxLength) return normalized;
|
||||
return `${normalized.slice(0, Math.max(0, maxLength - 1)).trimEnd()}…`;
|
||||
}
|
||||
|
||||
function pickContentRoot(document: Document): Element | null {
|
||||
return (
|
||||
document.querySelector("#CONTENT") ??
|
||||
document.querySelector("#content") ??
|
||||
document.body
|
||||
);
|
||||
}
|
||||
|
||||
function pickContentTitle(root: Element, fallbackTitle: string): string {
|
||||
const contentTitle = pickString(
|
||||
root.querySelector("h1")?.textContent,
|
||||
root.querySelector("[itemprop='headline']")?.textContent,
|
||||
root.querySelector("article h2")?.textContent
|
||||
);
|
||||
if (contentTitle) return contentTitle;
|
||||
if (fallbackTitle && !/^archive\./i.test(fallbackTitle.trim())) return fallbackTitle;
|
||||
return "";
|
||||
}
|
||||
|
||||
function parseArchivePage(context: UrlRuleParserContext): ConversionResult | null {
|
||||
const root = pickContentRoot(context.document);
|
||||
if (!root) return null;
|
||||
|
||||
const markdown = normalizeMarkdown(convertHtmlFragmentToMarkdown(root.innerHTML));
|
||||
if (!markdown) return null;
|
||||
|
||||
const originalUrl = readOriginalUrl(context.document) ?? context.baseMetadata.url;
|
||||
const bodyText = root.textContent?.replace(/\s+/g, " ").trim() ?? "";
|
||||
const published = root.querySelector("time[datetime]")?.getAttribute("datetime") ?? undefined;
|
||||
const coverImage = root.querySelector("img[src]")?.getAttribute("src") ?? undefined;
|
||||
|
||||
return {
|
||||
metadata: {
|
||||
...context.baseMetadata,
|
||||
url: originalUrl,
|
||||
title: pickContentTitle(root, context.baseMetadata.title),
|
||||
description: summarize(bodyText, 220) ?? context.baseMetadata.description,
|
||||
published: pickString(published, context.baseMetadata.published) ?? undefined,
|
||||
coverImage: pickString(coverImage, context.baseMetadata.coverImage) ?? undefined,
|
||||
},
|
||||
markdown,
|
||||
rawHtml: context.html,
|
||||
conversionMethod: "parser:archive-ph",
|
||||
};
|
||||
}
|
||||
|
||||
export const archivePhRuleParser: UrlRuleParser = {
|
||||
id: "archive-ph",
|
||||
supports(context) {
|
||||
return isArchiveHost(context.url);
|
||||
},
|
||||
parse: parseArchivePage,
|
||||
};
|
||||
@@ -0,0 +1,10 @@
|
||||
import { archivePhRuleParser } from "./archive-ph.js";
|
||||
import { xArticleRuleParser } from "./x-article.js";
|
||||
import { xStatusRuleParser } from "./x-status.js";
|
||||
import type { UrlRuleParser } from "../types.js";
|
||||
|
||||
export const URL_RULE_PARSERS: UrlRuleParser[] = [
|
||||
archivePhRuleParser,
|
||||
xArticleRuleParser,
|
||||
xStatusRuleParser,
|
||||
];
|
||||
@@ -0,0 +1,137 @@
|
||||
import {
|
||||
normalizeMarkdown,
|
||||
pickString,
|
||||
type ConversionResult,
|
||||
} from "../../markdown-conversion-shared.js";
|
||||
import type { UrlRuleParser, UrlRuleParserContext } from "../types.js";
|
||||
import {
|
||||
cleanText,
|
||||
collectMediaMarkdown,
|
||||
convertXRichTextElementToMarkdown,
|
||||
extractPublishedForCurrentUrl,
|
||||
inferLanguage,
|
||||
isXArticlePath,
|
||||
isXHost,
|
||||
normalizeXMarkdown,
|
||||
parseUrl,
|
||||
pickFirstValidLinkText,
|
||||
sanitizeCoverImage,
|
||||
summarizeText,
|
||||
} from "./x-shared.js";
|
||||
|
||||
function collectArticleMarkdown(root: Element): { markdown: string; mediaUrls: string[] } {
|
||||
const parts: string[] = [];
|
||||
const seenMedia = new Set<string>();
|
||||
const mediaUrls: string[] = [];
|
||||
|
||||
function pushPart(value: string): void {
|
||||
const normalized = normalizeMarkdown(value);
|
||||
if (!normalized) return;
|
||||
parts.push(normalized);
|
||||
}
|
||||
|
||||
function walk(node: Element): void {
|
||||
const testId = node.getAttribute("data-testid");
|
||||
|
||||
if (testId === "twitterArticleRichTextView" || testId === "longformRichTextComponent") {
|
||||
const bodyMedia = collectMediaMarkdown(node, seenMedia);
|
||||
mediaUrls.push(...bodyMedia.urls.filter((url) => !mediaUrls.includes(url)));
|
||||
pushPart(convertXRichTextElementToMarkdown(node));
|
||||
return;
|
||||
}
|
||||
|
||||
if (testId === "tweetPhoto") {
|
||||
const media = collectMediaMarkdown(node, seenMedia);
|
||||
mediaUrls.push(...media.urls.filter((url) => !mediaUrls.includes(url)));
|
||||
for (const line of media.lines) pushPart(line);
|
||||
return;
|
||||
}
|
||||
|
||||
if (
|
||||
testId === "twitter-article-title" ||
|
||||
testId === "User-Name" ||
|
||||
testId === "Tweet-User-Avatar" ||
|
||||
testId === "reply" ||
|
||||
testId === "retweet" ||
|
||||
testId === "like" ||
|
||||
testId === "bookmark" ||
|
||||
testId === "caret" ||
|
||||
testId === "app-text-transition-container"
|
||||
) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (node.tagName === "TIME" || node.tagName === "BUTTON") {
|
||||
return;
|
||||
}
|
||||
|
||||
for (const child of Array.from(node.children)) {
|
||||
walk(child);
|
||||
}
|
||||
}
|
||||
|
||||
for (const child of Array.from(root.children)) {
|
||||
walk(child);
|
||||
}
|
||||
|
||||
return {
|
||||
markdown: normalizeXMarkdown(parts.join("\n\n")),
|
||||
mediaUrls,
|
||||
};
|
||||
}
|
||||
|
||||
function parseXArticle(context: UrlRuleParserContext): ConversionResult | null {
|
||||
const articleRoot = context.document.querySelector("[data-testid='twitterArticleReadView']") as Element | null;
|
||||
if (!articleRoot) return null;
|
||||
|
||||
const title = cleanText(
|
||||
context.document.querySelector("[data-testid='twitter-article-title']")?.textContent
|
||||
);
|
||||
const identity = pickFirstValidLinkText(
|
||||
context.document.querySelector("[data-testid='User-Name']")
|
||||
);
|
||||
const published = extractPublishedForCurrentUrl(articleRoot, context.url);
|
||||
const { markdown, mediaUrls } = collectArticleMarkdown(articleRoot);
|
||||
if (!markdown) return null;
|
||||
|
||||
const bodyText = cleanText(
|
||||
context.document.querySelector("[data-testid='twitterArticleRichTextView']")?.textContent ??
|
||||
context.document.querySelector("[data-testid='longformRichTextComponent']")?.textContent
|
||||
);
|
||||
|
||||
return {
|
||||
metadata: {
|
||||
...context.baseMetadata,
|
||||
title: pickString(title, context.baseMetadata.title) ?? "",
|
||||
description: summarizeText(bodyText, 220) ?? context.baseMetadata.description,
|
||||
author: pickString(identity.author, context.baseMetadata.author) ?? undefined,
|
||||
published: pickString(published, context.baseMetadata.published) ?? undefined,
|
||||
coverImage: sanitizeCoverImage(mediaUrls[0], context.baseMetadata.coverImage),
|
||||
language: inferLanguage(bodyText, context.baseMetadata.language),
|
||||
},
|
||||
markdown,
|
||||
rawHtml: context.html,
|
||||
conversionMethod: "parser:x-article",
|
||||
};
|
||||
}
|
||||
|
||||
export const xArticleRuleParser: UrlRuleParser = {
|
||||
id: "x-article",
|
||||
supports(context) {
|
||||
const parsed = parseUrl(context.url);
|
||||
if (!parsed || !isXHost(parsed.hostname)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return (
|
||||
isXArticlePath(parsed.pathname) ||
|
||||
Boolean(
|
||||
context.document.querySelector("[data-testid='twitterArticleReadView']") ||
|
||||
context.document.querySelector("[data-testid='twitterArticleRichTextView']")
|
||||
)
|
||||
);
|
||||
},
|
||||
parse(context) {
|
||||
return parseXArticle(context);
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,249 @@
|
||||
import { convertHtmlFragmentToMarkdown } from "../../legacy-converter.js";
|
||||
import { normalizeMarkdown } from "../../markdown-conversion-shared.js";
|
||||
|
||||
export const DEFAULT_X_OG_IMAGE = "https://abs.twimg.com/rweb/ssr/default/v2/og/image.png";
|
||||
|
||||
export type MediaResult = {
|
||||
lines: string[];
|
||||
urls: string[];
|
||||
};
|
||||
|
||||
export function isXHost(hostname: string): boolean {
|
||||
const normalized = hostname.toLowerCase();
|
||||
return (
|
||||
normalized === "x.com" ||
|
||||
normalized === "twitter.com" ||
|
||||
normalized.endsWith(".x.com") ||
|
||||
normalized.endsWith(".twitter.com")
|
||||
);
|
||||
}
|
||||
|
||||
export function parseUrl(input: string): URL | null {
|
||||
try {
|
||||
return new URL(input);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export function isXStatusPath(pathname: string): boolean {
|
||||
return /^\/[^/]+\/status(?:es)?\/\d+$/i.test(pathname) || /^\/i\/web\/status\/\d+$/i.test(pathname);
|
||||
}
|
||||
|
||||
export function isXArticlePath(pathname: string): boolean {
|
||||
return /^\/[^/]+\/article\/\d+$/i.test(pathname) || /^\/(?:i\/)?article\/\d+$/i.test(pathname);
|
||||
}
|
||||
|
||||
export function cleanText(value: string | null | undefined): string {
|
||||
return (value ?? "").replace(/\s+/g, " ").trim();
|
||||
}
|
||||
|
||||
export function cleanUserLabel(value: string | null | undefined): string {
|
||||
return cleanText(value).replace(/\bVerified account\b/gi, "").replace(/\s{2,}/g, " ").trim();
|
||||
}
|
||||
|
||||
export function escapeMarkdownAlt(text: string): string {
|
||||
return text.replace(/[\[\]]/g, "\\$&");
|
||||
}
|
||||
|
||||
export function normalizeAlt(text: string | null | undefined): string {
|
||||
const cleaned = cleanText(text);
|
||||
if (!cleaned || /^(image|photo)$/i.test(cleaned)) return "";
|
||||
return escapeMarkdownAlt(cleaned);
|
||||
}
|
||||
|
||||
export function summarizeText(text: string, maxLength: number): string | undefined {
|
||||
const normalized = cleanText(text);
|
||||
if (!normalized) return undefined;
|
||||
return normalized.length > maxLength
|
||||
? `${normalized.slice(0, maxLength - 3)}...`
|
||||
: normalized;
|
||||
}
|
||||
|
||||
export function buildTweetTitle(text: string, fallback: string): string {
|
||||
return summarizeText(text, 80) ?? fallback;
|
||||
}
|
||||
|
||||
export function normalizeXMarkdown(markdown: string): string {
|
||||
return normalizeMarkdown(markdown.replace(/^(#{1,6})\s*\n+([^\n])/gm, "$1 $2"));
|
||||
}
|
||||
|
||||
export function inferLanguage(text: string, fallback?: string): string | undefined {
|
||||
const normalized = cleanText(text);
|
||||
if (!normalized) return fallback;
|
||||
|
||||
const han = (normalized.match(/\p{Script=Han}/gu) || []).length;
|
||||
const hiragana = (normalized.match(/\p{Script=Hiragana}/gu) || []).length;
|
||||
const katakana = (normalized.match(/\p{Script=Katakana}/gu) || []).length;
|
||||
const hangul = (normalized.match(/\p{Script=Hangul}/gu) || []).length;
|
||||
|
||||
if (hangul >= 8) return "ko";
|
||||
if (hiragana + katakana >= 8) return "ja";
|
||||
if (han >= 16) return "zh";
|
||||
return fallback;
|
||||
}
|
||||
|
||||
export function buildQuoteMarkdown(markdown: string, author?: string): string {
|
||||
const normalized = normalizeMarkdown(markdown);
|
||||
if (!normalized) return "";
|
||||
|
||||
const lines = normalized.split("\n");
|
||||
const prefixed = lines.map((line) => (line ? `> ${line}` : ">")).join("\n");
|
||||
const header = author ? `> Quote from ${author}` : "> Quote";
|
||||
return `${header}\n${prefixed}`;
|
||||
}
|
||||
|
||||
export function pickFirstValidLinkText(userNameEl: Element | null | undefined): {
|
||||
name?: string;
|
||||
username?: string;
|
||||
author?: string;
|
||||
} {
|
||||
if (!userNameEl) return {};
|
||||
|
||||
const linkTexts = Array.from(userNameEl.querySelectorAll("a[href]"))
|
||||
.map((link) => cleanUserLabel(link.textContent))
|
||||
.filter(Boolean);
|
||||
|
||||
let username = linkTexts.find((text) => text.startsWith("@"));
|
||||
let name = linkTexts.find((text) => !text.startsWith("@") && !/^(promote|more)$/i.test(text));
|
||||
|
||||
if (!username || !name) {
|
||||
const text = cleanUserLabel(userNameEl.textContent);
|
||||
const fallbackMatch = text.match(/^(.*?)\s*(@[A-Za-z0-9_]+)(?:\s*·.*)?$/);
|
||||
if (fallbackMatch) {
|
||||
name = name ?? cleanText(fallbackMatch[1]);
|
||||
username = username ?? cleanText(fallbackMatch[2]);
|
||||
}
|
||||
}
|
||||
|
||||
const author = name && username ? `${name} (${username})` : username ?? name;
|
||||
return { name, username, author };
|
||||
}
|
||||
|
||||
export function extractPublishedForCurrentUrl(root: ParentNode, url: string): string | undefined {
|
||||
const parsed = parseUrl(url);
|
||||
if (!parsed) return undefined;
|
||||
const currentPath = parsed.pathname.toLowerCase();
|
||||
|
||||
for (const timeElement of root.querySelectorAll("a[href] time[datetime]")) {
|
||||
const href = timeElement.closest("a")?.getAttribute("href");
|
||||
const hrefUrl = href ? parseUrl(href.startsWith("http") ? href : `${parsed.origin}${href}`) : null;
|
||||
if (hrefUrl?.pathname.toLowerCase() === currentPath) {
|
||||
return timeElement.getAttribute("datetime") ?? undefined;
|
||||
}
|
||||
}
|
||||
|
||||
return root.querySelector("time[datetime]")?.getAttribute("datetime") ?? undefined;
|
||||
}
|
||||
|
||||
export function collectMediaMarkdown(root: ParentNode, seen: Set<string>): MediaResult {
|
||||
const lines: string[] = [];
|
||||
const urls: string[] = [];
|
||||
const rootElement = root as Element & {
|
||||
getAttribute?: (name: string) => string | null;
|
||||
};
|
||||
const photoNodes = [
|
||||
...(typeof rootElement.getAttribute === "function" &&
|
||||
rootElement.getAttribute("data-testid") === "tweetPhoto"
|
||||
? [rootElement]
|
||||
: []),
|
||||
...Array.from(root.querySelectorAll("[data-testid='tweetPhoto']")),
|
||||
];
|
||||
|
||||
for (const node of photoNodes) {
|
||||
const img = node.querySelector("img");
|
||||
const imageUrl = img?.getAttribute("src");
|
||||
if (imageUrl && !seen.has(imageUrl)) {
|
||||
seen.add(imageUrl);
|
||||
urls.push(imageUrl);
|
||||
lines.push(``);
|
||||
}
|
||||
|
||||
const video = node.querySelector("video");
|
||||
const posterUrl = video?.getAttribute("poster");
|
||||
if (posterUrl && !seen.has(posterUrl)) {
|
||||
seen.add(posterUrl);
|
||||
urls.push(posterUrl);
|
||||
lines.push(``);
|
||||
}
|
||||
|
||||
const videoUrl = video?.getAttribute("src") ?? video?.querySelector("source")?.getAttribute("src");
|
||||
if (videoUrl && !seen.has(videoUrl)) {
|
||||
seen.add(videoUrl);
|
||||
urls.push(videoUrl);
|
||||
lines.push(`[video](${videoUrl})`);
|
||||
}
|
||||
}
|
||||
|
||||
return { lines, urls };
|
||||
}
|
||||
|
||||
export function materializeTweetPhotoNodes(root: Element): void {
|
||||
for (const photo of Array.from(root.querySelectorAll("[data-testid='tweetPhoto']"))) {
|
||||
const document = photo.ownerDocument;
|
||||
const container = document.createElement("span");
|
||||
|
||||
const img = photo.querySelector("img");
|
||||
const imageUrl = img?.getAttribute("src");
|
||||
if (imageUrl) {
|
||||
const image = document.createElement("img");
|
||||
image.setAttribute("src", imageUrl);
|
||||
const alt = normalizeAlt(img?.getAttribute("alt"));
|
||||
if (alt) {
|
||||
image.setAttribute("alt", alt);
|
||||
}
|
||||
container.appendChild(image);
|
||||
}
|
||||
|
||||
const video = photo.querySelector("video");
|
||||
const posterUrl = video?.getAttribute("poster");
|
||||
if (posterUrl) {
|
||||
const poster = document.createElement("img");
|
||||
poster.setAttribute("src", posterUrl);
|
||||
poster.setAttribute("alt", "video");
|
||||
container.appendChild(poster);
|
||||
}
|
||||
|
||||
const videoUrl = video?.getAttribute("src") ?? video?.querySelector("source")?.getAttribute("src");
|
||||
if (videoUrl) {
|
||||
if (container.childNodes.length > 0) {
|
||||
container.appendChild(document.createTextNode(" "));
|
||||
}
|
||||
const link = document.createElement("a");
|
||||
link.setAttribute("href", videoUrl);
|
||||
link.textContent = "video";
|
||||
container.appendChild(link);
|
||||
}
|
||||
|
||||
if (container.childNodes.length === 0) {
|
||||
photo.remove();
|
||||
continue;
|
||||
}
|
||||
|
||||
photo.replaceWith(container);
|
||||
}
|
||||
}
|
||||
|
||||
function collapseLinkedMediaContainers(root: Element): void {
|
||||
for (const anchor of Array.from(root.querySelectorAll("a[href]"))) {
|
||||
const images = Array.from(anchor.querySelectorAll("img"));
|
||||
if (images.length !== 1) continue;
|
||||
if (cleanText(anchor.textContent)) continue;
|
||||
|
||||
const image = images[0].cloneNode(true);
|
||||
anchor.replaceChildren(image);
|
||||
}
|
||||
}
|
||||
|
||||
export function convertXRichTextElementToMarkdown(node: Element): string {
|
||||
const clone = node.cloneNode(true) as Element;
|
||||
materializeTweetPhotoNodes(clone);
|
||||
collapseLinkedMediaContainers(clone);
|
||||
return normalizeXMarkdown(convertHtmlFragmentToMarkdown(clone.innerHTML));
|
||||
}
|
||||
|
||||
export function sanitizeCoverImage(primary?: string, fallback?: string): string | undefined {
|
||||
if (primary) return primary;
|
||||
if (!fallback || fallback === DEFAULT_X_OG_IMAGE) return undefined;
|
||||
return fallback;
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
import type { ConversionResult } from "../../markdown-conversion-shared.js";
|
||||
import type { UrlRuleParser, UrlRuleParserContext } from "../types.js";
|
||||
import {
|
||||
buildQuoteMarkdown,
|
||||
buildTweetTitle,
|
||||
cleanText,
|
||||
collectMediaMarkdown,
|
||||
convertXRichTextElementToMarkdown,
|
||||
extractPublishedForCurrentUrl,
|
||||
inferLanguage,
|
||||
isXHost,
|
||||
isXStatusPath,
|
||||
normalizeXMarkdown,
|
||||
parseUrl,
|
||||
pickFirstValidLinkText,
|
||||
sanitizeCoverImage,
|
||||
summarizeText,
|
||||
} from "./x-shared.js";
|
||||
|
||||
function parseXStatus(context: UrlRuleParserContext): ConversionResult | null {
|
||||
const article = context.document.querySelector("article[data-testid='tweet'], article") as Element | null;
|
||||
if (!article) return null;
|
||||
|
||||
const tweetTextElements = Array.from(article.querySelectorAll("[data-testid='tweetText']")) as Element[];
|
||||
if (tweetTextElements.length === 0) return null;
|
||||
|
||||
const userNameElements = Array.from(article.querySelectorAll("[data-testid='User-Name']")) as Element[];
|
||||
const mainTextElement = tweetTextElements[0];
|
||||
const mainIdentity = pickFirstValidLinkText(userNameElements[0]);
|
||||
const published = extractPublishedForCurrentUrl(article, context.url);
|
||||
const mainMarkdown = normalizeXMarkdown(convertXRichTextElementToMarkdown(mainTextElement));
|
||||
if (!mainMarkdown) return null;
|
||||
|
||||
const parts = [mainMarkdown];
|
||||
const quotedTextElements = tweetTextElements.slice(1);
|
||||
const quotedUserNameElements = userNameElements.slice(1);
|
||||
|
||||
quotedTextElements.forEach((element, index) => {
|
||||
const quoteMarkdown = normalizeXMarkdown(convertXRichTextElementToMarkdown(element));
|
||||
if (!quoteMarkdown) return;
|
||||
const quoteIdentity = pickFirstValidLinkText(quotedUserNameElements[index]);
|
||||
parts.push(buildQuoteMarkdown(quoteMarkdown, quoteIdentity.author));
|
||||
});
|
||||
|
||||
const media = collectMediaMarkdown(article, new Set<string>());
|
||||
if (media.lines.length > 0) {
|
||||
parts.push(media.lines.join("\n\n"));
|
||||
}
|
||||
|
||||
const mainText = cleanText(mainTextElement.textContent);
|
||||
const markdown = normalizeXMarkdown(parts.join("\n\n"));
|
||||
|
||||
return {
|
||||
metadata: {
|
||||
...context.baseMetadata,
|
||||
title: buildTweetTitle(mainText, context.baseMetadata.title),
|
||||
description: summarizeText(mainText, 220) ?? context.baseMetadata.description,
|
||||
author: mainIdentity.author ?? context.baseMetadata.author,
|
||||
published: published ?? context.baseMetadata.published,
|
||||
coverImage: sanitizeCoverImage(media.urls[0], context.baseMetadata.coverImage),
|
||||
language: inferLanguage(mainText, context.baseMetadata.language),
|
||||
},
|
||||
markdown,
|
||||
rawHtml: context.html,
|
||||
conversionMethod: "parser:x-status",
|
||||
};
|
||||
}
|
||||
|
||||
export const xStatusRuleParser: UrlRuleParser = {
|
||||
id: "x-status",
|
||||
supports(context) {
|
||||
const parsed = parseUrl(context.url);
|
||||
if (!parsed || !isXHost(parsed.hostname)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return isXStatusPath(parsed.pathname) && Boolean(context.document.querySelector("[data-testid='tweetText']"));
|
||||
},
|
||||
parse(context): ConversionResult | null {
|
||||
return parseXStatus(context);
|
||||
},
|
||||
};
|
||||
Reference in New Issue
Block a user