mirror of
https://github.com/JimLiu/baoyu-skills.git
synced 2026-08-02 07:09:47 +08:00
feat(baoyu-fetch): add URL reader CLI with Chrome CDP and site adapters
This commit is contained in:
@@ -0,0 +1,433 @@
|
||||
import type { ExtractedDocument } from "../../extract/document";
|
||||
import {
|
||||
findTweetNode,
|
||||
findTweetNodeById,
|
||||
formatMediaList,
|
||||
formatTweetAuthor,
|
||||
getTweetAuthorMetadata,
|
||||
getTweetText,
|
||||
getUser,
|
||||
isRecord,
|
||||
normalizeTitle,
|
||||
toHighResXImageUrl,
|
||||
toXTweet,
|
||||
} from "./shared";
|
||||
import type { JsonObject } from "./types";
|
||||
|
||||
function resolveArticleMediaUrl(mediaInfo: JsonObject): string {
|
||||
const rawUrl =
|
||||
(typeof mediaInfo.original_img_url === "string" && mediaInfo.original_img_url) ||
|
||||
(typeof mediaInfo.url === "string" && mediaInfo.url) ||
|
||||
"";
|
||||
|
||||
return rawUrl ? toHighResXImageUrl(rawUrl) : "";
|
||||
}
|
||||
|
||||
function normalizeEntityMap(entityMap: unknown): Map<string, JsonObject> {
|
||||
const normalized = new Map<string, JsonObject>();
|
||||
|
||||
if (Array.isArray(entityMap)) {
|
||||
for (const entry of entityMap) {
|
||||
if (!isRecord(entry)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const key =
|
||||
typeof entry.key === "string" || typeof entry.key === "number"
|
||||
? String(entry.key)
|
||||
: undefined;
|
||||
const value = isRecord(entry.value) ? entry.value : undefined;
|
||||
if (!key || !value) {
|
||||
continue;
|
||||
}
|
||||
normalized.set(key, value);
|
||||
}
|
||||
|
||||
return normalized;
|
||||
}
|
||||
|
||||
if (!isRecord(entityMap)) {
|
||||
return normalized;
|
||||
}
|
||||
|
||||
for (const [key, value] of Object.entries(entityMap)) {
|
||||
if (!isRecord(value)) {
|
||||
continue;
|
||||
}
|
||||
normalized.set(key, value);
|
||||
}
|
||||
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function getEntityMarkdown(entityMap: Map<string, JsonObject>, entityKey: unknown): string | null {
|
||||
const key =
|
||||
typeof entityKey === "string" || typeof entityKey === "number"
|
||||
? String(entityKey)
|
||||
: undefined;
|
||||
if (!key) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const entity = entityMap.get(key);
|
||||
if (!entity || entity.type !== "MARKDOWN") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const data = isRecord(entity.data) ? entity.data : {};
|
||||
if (typeof data.markdown !== "string") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const markdown = data.markdown.trim();
|
||||
return markdown || null;
|
||||
}
|
||||
|
||||
function getLinkUrl(entityMap: Map<string, JsonObject>, entityKey: unknown): string | null {
|
||||
const key =
|
||||
typeof entityKey === "string" || typeof entityKey === "number"
|
||||
? String(entityKey)
|
||||
: undefined;
|
||||
if (!key) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const entity = entityMap.get(key);
|
||||
if (!entity || entity.type !== "LINK") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const data = isRecord(entity.data) ? entity.data : {};
|
||||
const candidates = [
|
||||
data.expanded_url,
|
||||
data.expandedUrl,
|
||||
data.original_url,
|
||||
data.originalUrl,
|
||||
data.url,
|
||||
data.display_url,
|
||||
data.displayUrl,
|
||||
];
|
||||
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === "string" && candidate.trim()) {
|
||||
return candidate.trim();
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function getTweetId(entityMap: Map<string, JsonObject>, entityKey: unknown): string | null {
|
||||
const key =
|
||||
typeof entityKey === "string" || typeof entityKey === "number"
|
||||
? String(entityKey)
|
||||
: undefined;
|
||||
if (!key) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const entity = entityMap.get(key);
|
||||
if (!entity || entity.type !== "TWEET") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const data = isRecord(entity.data) ? entity.data : {};
|
||||
if (typeof data.tweetId !== "string") {
|
||||
return null;
|
||||
}
|
||||
|
||||
return data.tweetId;
|
||||
}
|
||||
|
||||
function buildMediaUrlMap(articleResult: JsonObject): Map<string, string> {
|
||||
const mediaMap = new Map<string, string>();
|
||||
const mediaEntities = Array.isArray(articleResult.media_entities) ? articleResult.media_entities : [];
|
||||
|
||||
for (const entity of mediaEntities) {
|
||||
if (!isRecord(entity) || typeof entity.media_id !== "string" || !isRecord(entity.media_info)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const mediaInfo = entity.media_info;
|
||||
const url = resolveArticleMediaUrl(mediaInfo);
|
||||
if (url) {
|
||||
mediaMap.set(entity.media_id, url);
|
||||
}
|
||||
}
|
||||
|
||||
const coverMedia = isRecord(articleResult.cover_media) ? articleResult.cover_media : null;
|
||||
if (coverMedia && typeof coverMedia.media_id === "string" && isRecord(coverMedia.media_info)) {
|
||||
const url = resolveArticleMediaUrl(coverMedia.media_info);
|
||||
if (url) {
|
||||
mediaMap.set(coverMedia.media_id, url);
|
||||
}
|
||||
}
|
||||
|
||||
return mediaMap;
|
||||
}
|
||||
|
||||
function getMediaMarkdown(entityMap: Map<string, JsonObject>, entityKey: unknown, mediaMap: Map<string, string>): string[] {
|
||||
const key =
|
||||
typeof entityKey === "string" || typeof entityKey === "number"
|
||||
? String(entityKey)
|
||||
: undefined;
|
||||
if (!key) {
|
||||
return [];
|
||||
}
|
||||
|
||||
const entity = entityMap.get(key);
|
||||
if (!entity || entity.type !== "MEDIA") {
|
||||
return [];
|
||||
}
|
||||
|
||||
const data = isRecord(entity.data) ? entity.data : {};
|
||||
const mediaItems = Array.isArray(data.mediaItems) ? data.mediaItems : [];
|
||||
const urls: string[] = [];
|
||||
|
||||
for (const item of mediaItems) {
|
||||
if (!isRecord(item) || typeof item.mediaId !== "string") {
|
||||
continue;
|
||||
}
|
||||
const url = mediaMap.get(item.mediaId);
|
||||
if (url && !urls.includes(url)) {
|
||||
urls.push(url);
|
||||
}
|
||||
}
|
||||
|
||||
return urls.map((url) => ``);
|
||||
}
|
||||
|
||||
function resolveTweetMarkdown(payloads: unknown[], tweetId: string, pageUrl: string): string | null {
|
||||
for (const payload of payloads) {
|
||||
const tweet = findTweetNodeById(payload, tweetId);
|
||||
if (!tweet) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const xTweet = toXTweet(tweet, pageUrl);
|
||||
const author = formatTweetAuthor(xTweet) ?? xTweet.url;
|
||||
const lines = [`> ${author}`, ...xTweet.text.split("\n").map((line) => `> ${line}`)];
|
||||
|
||||
const media = formatMediaList(xTweet.media).map((line) =>
|
||||
line.startsWith("photo: ") ? `> })` : `> - ${line}`,
|
||||
);
|
||||
|
||||
const parts = [lines.join("\n")];
|
||||
if (media.length > 0) {
|
||||
parts.push([">", ...media].join("\n"));
|
||||
}
|
||||
parts.push(`> ${xTweet.url}`);
|
||||
|
||||
return parts.join("\n").trim();
|
||||
}
|
||||
|
||||
return `> Embedded tweet: https://x.com/i/status/${tweetId}`;
|
||||
}
|
||||
|
||||
function replaceLinkEntities(text: string, block: JsonObject, entityMap: Map<string, JsonObject>): string {
|
||||
const entityRanges = Array.isArray(block.entityRanges) ? block.entityRanges : [];
|
||||
const replacements = entityRanges
|
||||
.filter((range): range is JsonObject => isRecord(range))
|
||||
.map((range) => {
|
||||
const offset = typeof range.offset === "number" ? range.offset : -1;
|
||||
const length = typeof range.length === "number" ? range.length : -1;
|
||||
const url = getLinkUrl(entityMap, range.key);
|
||||
return { offset, length, url };
|
||||
})
|
||||
.filter((range) => range.offset >= 0 && range.length > 0 && range.url)
|
||||
.sort((left, right) => right.offset - left.offset);
|
||||
|
||||
let next = text;
|
||||
for (const replacement of replacements) {
|
||||
next =
|
||||
next.slice(0, replacement.offset) +
|
||||
replacement.url +
|
||||
next.slice(replacement.offset + replacement.length);
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
function renderAtomicBlock(
|
||||
block: JsonObject,
|
||||
entityMap: Map<string, JsonObject>,
|
||||
mediaMap: Map<string, string>,
|
||||
payloads: unknown[],
|
||||
pageUrl: string,
|
||||
): string | null {
|
||||
const entityRanges = Array.isArray(block.entityRanges) ? block.entityRanges : [];
|
||||
const parts: string[] = [];
|
||||
|
||||
for (const range of entityRanges) {
|
||||
if (!isRecord(range)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const markdown = getEntityMarkdown(entityMap, range.key);
|
||||
if (markdown) {
|
||||
parts.push(markdown);
|
||||
continue;
|
||||
}
|
||||
|
||||
const mediaMarkdown = getMediaMarkdown(entityMap, range.key, mediaMap);
|
||||
if (mediaMarkdown.length > 0) {
|
||||
parts.push(mediaMarkdown.join("\n\n"));
|
||||
continue;
|
||||
}
|
||||
|
||||
const tweetId = getTweetId(entityMap, range.key);
|
||||
if (tweetId) {
|
||||
const tweetMarkdown = resolveTweetMarkdown(payloads, tweetId, pageUrl);
|
||||
if (tweetMarkdown) {
|
||||
parts.push(tweetMarkdown);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (parts.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return parts.join("\n\n");
|
||||
}
|
||||
|
||||
function renderArticleBlocks(
|
||||
blocks: unknown[],
|
||||
entityMap: Map<string, JsonObject>,
|
||||
mediaMap: Map<string, string>,
|
||||
payloads: unknown[],
|
||||
pageUrl: string,
|
||||
): string {
|
||||
const parts: string[] = [];
|
||||
let orderedCounter = 0;
|
||||
|
||||
for (const block of blocks) {
|
||||
if (!isRecord(block)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const blockType = typeof block.type === "string" ? block.type : "unstyled";
|
||||
const rawText = typeof block.text === "string" ? block.text : "";
|
||||
const text = replaceLinkEntities(rawText, block, entityMap).trim();
|
||||
if (!text && blockType !== "atomic") {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (blockType !== "ordered-list-item") {
|
||||
orderedCounter = 0;
|
||||
}
|
||||
|
||||
switch (blockType) {
|
||||
case "header-one":
|
||||
parts.push(`# ${text}`);
|
||||
break;
|
||||
case "header-two":
|
||||
parts.push(`## ${text}`);
|
||||
break;
|
||||
case "header-three":
|
||||
parts.push(`### ${text}`);
|
||||
break;
|
||||
case "blockquote":
|
||||
parts.push(`> ${text}`);
|
||||
break;
|
||||
case "unordered-list-item":
|
||||
parts.push(`- ${text}`);
|
||||
break;
|
||||
case "ordered-list-item":
|
||||
orderedCounter += 1;
|
||||
parts.push(`${orderedCounter}. ${text}`);
|
||||
break;
|
||||
case "code-block":
|
||||
parts.push(`\`\`\`\n${text}\n\`\`\``);
|
||||
break;
|
||||
case "atomic": {
|
||||
const markdown = renderAtomicBlock(block, entityMap, mediaMap, payloads, pageUrl);
|
||||
if (markdown) {
|
||||
parts.push(markdown);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
parts.push(text);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return parts.join("\n\n").trim();
|
||||
}
|
||||
|
||||
function getArticleResult(tweet: JsonObject): JsonObject | null {
|
||||
if (
|
||||
isRecord(tweet.article) &&
|
||||
isRecord(tweet.article.article_results) &&
|
||||
isRecord(tweet.article.article_results.result)
|
||||
) {
|
||||
return tweet.article.article_results.result as JsonObject;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractSummary(markdown: string): string | undefined {
|
||||
const segments = markdown
|
||||
.split(/\n\n+/)
|
||||
.map((segment) => segment.trim())
|
||||
.filter(Boolean);
|
||||
|
||||
const preferred = segments.find((segment) => !/^(#|>|- |\d+\. |\`\`\`)/.test(segment));
|
||||
return preferred?.slice(0, 220);
|
||||
}
|
||||
|
||||
export function extractArticleDocumentFromPayload(
|
||||
payload: unknown,
|
||||
statusId: string,
|
||||
pageUrl: string,
|
||||
payloads: unknown[] = [payload],
|
||||
): ExtractedDocument | null {
|
||||
const tweet = findTweetNode(payload, statusId);
|
||||
if (!tweet) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const articleResult = getArticleResult(tweet);
|
||||
if (!articleResult) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const title = typeof articleResult.title === "string" ? articleResult.title.trim() : undefined;
|
||||
const contentState = isRecord(articleResult.content_state) ? articleResult.content_state : {};
|
||||
const blocks = Array.isArray(contentState.blocks) ? contentState.blocks : [];
|
||||
const entityMap = normalizeEntityMap(contentState.entityMap);
|
||||
const mediaMap = buildMediaUrlMap(articleResult);
|
||||
const richMarkdown = renderArticleBlocks(blocks, entityMap, mediaMap, payloads, pageUrl);
|
||||
const plainText = typeof articleResult.plain_text === "string" ? articleResult.plain_text.trim() : "";
|
||||
const markdown = richMarkdown || plainText || getTweetText(tweet);
|
||||
if (!markdown) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const xTweet = toXTweet(tweet, pageUrl);
|
||||
const user = getUser(tweet);
|
||||
const coverMedia = isRecord(articleResult.cover_media) ? articleResult.cover_media : null;
|
||||
const coverMediaInfo = coverMedia && isRecord(coverMedia.media_info) ? coverMedia.media_info : null;
|
||||
const coverImage = coverMediaInfo ? resolveArticleMediaUrl(coverMediaInfo) || undefined : undefined;
|
||||
|
||||
return {
|
||||
url: pageUrl,
|
||||
canonicalUrl: xTweet.url,
|
||||
title: title || normalizeTitle(xTweet.text, "X Article"),
|
||||
author: formatTweetAuthor(xTweet),
|
||||
siteName: "X",
|
||||
publishedAt: xTweet.createdAt,
|
||||
summary: extractSummary(markdown) || xTweet.text.slice(0, 200) || undefined,
|
||||
adapter: "x",
|
||||
metadata: {
|
||||
kind: "x/article",
|
||||
tweetId: xTweet.id,
|
||||
coverImage,
|
||||
authorName: xTweet.authorName ?? user.name,
|
||||
authorUsername: xTweet.author ?? user.screenName,
|
||||
authorUrl: (xTweet.author ?? user.screenName) ? `https://x.com/${xTweet.author ?? user.screenName}` : undefined,
|
||||
...getTweetAuthorMetadata(xTweet),
|
||||
},
|
||||
content: [{ type: "markdown", markdown }],
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
import type { Adapter, AdapterLoginInfo } from "../types";
|
||||
import { detectInteractionGate } from "../../browser/interaction-gates";
|
||||
import type { ExtractedDocument } from "../../extract/document";
|
||||
import { collectMediaFromDocument } from "../../media/markdown-media";
|
||||
import { extractArticleDocumentFromPayload } from "./article";
|
||||
import { buildNeedsLoginResult, detectXLogin } from "./login";
|
||||
import { extractStatusId, isXHost } from "./match";
|
||||
import { collectXJsonPayloads, waitForInitialXPayload } from "./payloads";
|
||||
import { extractSingleTweetDocumentFromPayload } from "./single";
|
||||
import { extractThreadDocumentFromPayloads } from "./thread";
|
||||
import { loadFullXThread } from "./thread-loader";
|
||||
|
||||
function extractDocumentFromPayloads(
|
||||
payloads: unknown[],
|
||||
statusId: string,
|
||||
pageUrl: string,
|
||||
): ExtractedDocument | null {
|
||||
for (const payload of payloads) {
|
||||
const articleDocument = extractArticleDocumentFromPayload(payload, statusId, pageUrl, payloads);
|
||||
if (articleDocument) {
|
||||
return articleDocument;
|
||||
}
|
||||
}
|
||||
|
||||
const threadDocument = extractThreadDocumentFromPayloads(payloads, statusId, pageUrl);
|
||||
if (threadDocument) {
|
||||
return threadDocument;
|
||||
}
|
||||
|
||||
for (const payload of payloads) {
|
||||
const singleDocument = extractSingleTweetDocumentFromPayload(payload, statusId, pageUrl);
|
||||
if (singleDocument) {
|
||||
return singleDocument;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
async function ensureXLoginState(context: Parameters<Adapter["process"]>[0]): Promise<AdapterLoginInfo> {
|
||||
return detectXLogin(context);
|
||||
}
|
||||
|
||||
export const xAdapter: Adapter = {
|
||||
name: "x",
|
||||
match(input) {
|
||||
return isXHost(input.url.hostname);
|
||||
},
|
||||
async checkLogin(context) {
|
||||
return detectXLogin(context);
|
||||
},
|
||||
async process(context) {
|
||||
const statusId = extractStatusId(context.input.url);
|
||||
if (!statusId) {
|
||||
return {
|
||||
status: "no_document",
|
||||
};
|
||||
}
|
||||
|
||||
context.log.info(`Loading ${context.input.url.toString()} with x adapter`);
|
||||
await context.browser.goto(context.input.url.toString(), context.timeoutMs);
|
||||
|
||||
const interaction = await detectInteractionGate(context.browser);
|
||||
if (interaction) {
|
||||
return {
|
||||
status: "needs_interaction",
|
||||
interaction,
|
||||
};
|
||||
}
|
||||
|
||||
let login = await ensureXLoginState(context);
|
||||
if (login.state === "logged_out") {
|
||||
return buildNeedsLoginResult(login);
|
||||
}
|
||||
|
||||
await waitForInitialXPayload(context);
|
||||
await loadFullXThread(context, statusId);
|
||||
|
||||
const pageUrl = await context.browser.getURL();
|
||||
const postLoadInteraction = await detectInteractionGate(context.browser);
|
||||
if (postLoadInteraction) {
|
||||
return {
|
||||
status: "needs_interaction",
|
||||
interaction: postLoadInteraction,
|
||||
login,
|
||||
};
|
||||
}
|
||||
|
||||
login = await ensureXLoginState(context).catch(() => login);
|
||||
if (login.state === "logged_out") {
|
||||
return buildNeedsLoginResult(login);
|
||||
}
|
||||
|
||||
const payloads = await collectXJsonPayloads(context);
|
||||
if (payloads.length === 0) {
|
||||
return {
|
||||
status: "no_document",
|
||||
login,
|
||||
};
|
||||
}
|
||||
|
||||
const document = extractDocumentFromPayloads(payloads, statusId, pageUrl);
|
||||
if (document) {
|
||||
return {
|
||||
status: "ok",
|
||||
document,
|
||||
media: collectMediaFromDocument(document),
|
||||
login,
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
status: "no_document",
|
||||
login,
|
||||
};
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,80 @@
|
||||
import type { AdapterContext, AdapterLoginInfo, AdapterProcessResult } from "../types";
|
||||
|
||||
interface XLoginSnapshot {
|
||||
currentUrl: string;
|
||||
hasAccountMenu: boolean;
|
||||
hasLoginInputs: boolean;
|
||||
bodyText: string;
|
||||
}
|
||||
|
||||
export async function detectXLogin(context: AdapterContext): Promise<AdapterLoginInfo> {
|
||||
const snapshot = await context.browser.evaluate<XLoginSnapshot>(`
|
||||
(() => {
|
||||
const bodyText = (document.body?.innerText ?? "").slice(0, 2500);
|
||||
return {
|
||||
currentUrl: window.location.href,
|
||||
hasAccountMenu: Boolean(
|
||||
document.querySelector(
|
||||
'[data-testid="SideNav_AccountSwitcher_Button"], [data-testid="AppTabBar_Profile_Link"], [aria-label="Account menu"]'
|
||||
)
|
||||
),
|
||||
hasLoginInputs: Boolean(
|
||||
document.querySelector(
|
||||
'input[name="text"], input[name="password"], input[autocomplete="username"], input[autocomplete="current-password"]'
|
||||
)
|
||||
),
|
||||
bodyText,
|
||||
};
|
||||
})()
|
||||
`).catch(async () => ({
|
||||
currentUrl: await context.browser.getURL().catch(() => context.input.url.toString()),
|
||||
hasAccountMenu: false,
|
||||
hasLoginInputs: false,
|
||||
bodyText: "",
|
||||
}));
|
||||
|
||||
if (
|
||||
/\/i\/flow\/login|\/login/i.test(snapshot.currentUrl) ||
|
||||
snapshot.hasLoginInputs ||
|
||||
/sign in to x|join x today|登录 x|注册 x|登录到 x/i.test(snapshot.bodyText)
|
||||
) {
|
||||
return {
|
||||
provider: "x",
|
||||
state: "logged_out",
|
||||
required: true,
|
||||
reason: "X login page detected",
|
||||
};
|
||||
}
|
||||
|
||||
if (snapshot.hasAccountMenu) {
|
||||
return {
|
||||
provider: "x",
|
||||
state: "logged_in",
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
provider: "x",
|
||||
state: "unknown",
|
||||
};
|
||||
}
|
||||
|
||||
export function buildNeedsLoginResult(login: AdapterLoginInfo): AdapterProcessResult {
|
||||
return {
|
||||
status: "needs_interaction",
|
||||
login: {
|
||||
...login,
|
||||
provider: "x",
|
||||
state: login.state === "logged_in" ? "unknown" : login.state,
|
||||
required: true,
|
||||
},
|
||||
interaction: {
|
||||
type: "wait_for_interaction",
|
||||
kind: "login",
|
||||
provider: "x",
|
||||
reason: login.reason,
|
||||
prompt: "Please sign in to X in the opened Chrome window. Extraction will continue automatically once login is detected.",
|
||||
requiresVisibleBrowser: true,
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
export function isXHost(hostname: string): boolean {
|
||||
return ["x.com", "www.x.com", "twitter.com", "www.twitter.com"].includes(hostname);
|
||||
}
|
||||
|
||||
export function extractStatusId(url: URL): string | undefined {
|
||||
const match = url.pathname.match(/\/(?:status|article)\/(\d+)/);
|
||||
return match?.[1];
|
||||
}
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
import type { AdapterContext } from "../types";
|
||||
import { filterXGraphQlEntries } from "./shared";
|
||||
|
||||
export function getRelevantXThreadEntries(context: AdapterContext) {
|
||||
return filterXGraphQlEntries(context.network.getEntries()).filter(
|
||||
(entry) =>
|
||||
entry.method === "GET" &&
|
||||
entry.finished &&
|
||||
(
|
||||
entry.url.includes("TweetDetail") ||
|
||||
entry.url.includes("TweetResultByRestId") ||
|
||||
entry.url.includes("TweetResultsByRestIds")
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
export async function prefetchRelevantXThreadBodies(context: AdapterContext): Promise<void> {
|
||||
const entries = getRelevantXThreadEntries(context).filter((entry) => entry.body === undefined && !entry.bodyError);
|
||||
for (const entry of entries) {
|
||||
await context.network.ensureBody(entry);
|
||||
}
|
||||
}
|
||||
|
||||
export async function collectXJsonPayloads(context: AdapterContext): Promise<unknown[]> {
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
const entries = getRelevantXThreadEntries(context);
|
||||
|
||||
const payloads: unknown[] = [];
|
||||
for (const entry of entries) {
|
||||
const payload = await context.network.getJsonBody(entry);
|
||||
if (payload) {
|
||||
payloads.push(payload);
|
||||
}
|
||||
}
|
||||
return payloads;
|
||||
}
|
||||
|
||||
export async function waitForInitialXPayload(context: AdapterContext): Promise<void> {
|
||||
try {
|
||||
await context.network.waitForResponse(
|
||||
(entry) =>
|
||||
entry.url.includes("/graphql/") &&
|
||||
(entry.url.includes("TweetDetail") || entry.url.includes("TweetResultByRestId")),
|
||||
{ timeoutMs: Math.min(context.timeoutMs, 15_000) },
|
||||
);
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
} catch {
|
||||
context.log.debug("No tweet GraphQL response observed before timeout.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,386 @@
|
||||
import path from "node:path";
|
||||
import type { NetworkEntry } from "../../browser/network-journal";
|
||||
import type { XMedia, XQuotedTweet, XTweet, XUser, JsonObject } from "./types";
|
||||
|
||||
const X_IMAGE_EXTENSIONS = new Set(["jpg", "jpeg", "png", "webp", "gif", "bmp", "avif"]);
|
||||
|
||||
function emptyObject(): JsonObject {
|
||||
return {};
|
||||
}
|
||||
|
||||
export function isRecord(value: unknown): value is JsonObject {
|
||||
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
||||
}
|
||||
|
||||
export function walk(value: unknown, visitor: (node: unknown) => boolean | void): boolean {
|
||||
if (visitor(value)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if (Array.isArray(value)) {
|
||||
for (const item of value) {
|
||||
if (walk(item, visitor)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
if (isRecord(value)) {
|
||||
for (const child of Object.values(value)) {
|
||||
if (walk(child, visitor)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
function hasTweetText(node: JsonObject): boolean {
|
||||
const legacy = isRecord(node.legacy) ? node.legacy : emptyObject();
|
||||
return (
|
||||
typeof legacy.full_text === "string" ||
|
||||
typeof getNoteTweetText(node) === "string"
|
||||
);
|
||||
}
|
||||
|
||||
export function findTweetNodeById(payload: unknown, tweetId: string): JsonObject | null {
|
||||
let match: JsonObject | null = null;
|
||||
|
||||
walk(payload, (node) => {
|
||||
if (!isRecord(node) || typeof node.rest_id !== "string" || !isRecord(node.legacy)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!hasTweetText(node)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (node.rest_id === tweetId) {
|
||||
match = node;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
});
|
||||
|
||||
return match;
|
||||
}
|
||||
|
||||
export function findTweetNode(payload: unknown, statusId: string): JsonObject | null {
|
||||
let firstMatch: JsonObject | null = null;
|
||||
const exactMatch = findTweetNodeById(payload, statusId);
|
||||
if (exactMatch) {
|
||||
return exactMatch;
|
||||
}
|
||||
|
||||
walk(payload, (node) => {
|
||||
if (!isRecord(node) || typeof node.rest_id !== "string" || !isRecord(node.legacy)) {
|
||||
return false;
|
||||
}
|
||||
if (!hasTweetText(node)) {
|
||||
return false;
|
||||
}
|
||||
if (!firstMatch) {
|
||||
firstMatch = node;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
|
||||
return firstMatch;
|
||||
}
|
||||
|
||||
export function getLegacy(tweet: JsonObject): JsonObject {
|
||||
return isRecord(tweet.legacy) ? tweet.legacy : emptyObject();
|
||||
}
|
||||
|
||||
export function unwrapTweetResult(node: unknown): JsonObject | null {
|
||||
if (!isRecord(node)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (node.__typename === "TweetWithVisibilityResults" && isRecord(node.tweet)) {
|
||||
return unwrapTweetResult(node.tweet);
|
||||
}
|
||||
|
||||
const tweet = isRecord(node.tweet) ? (node.tweet as JsonObject) : node;
|
||||
if (typeof tweet.rest_id !== "string" || !isRecord(tweet.legacy)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return tweet;
|
||||
}
|
||||
|
||||
export function getUser(tweet: JsonObject): XUser {
|
||||
const result =
|
||||
isRecord(tweet.core) &&
|
||||
isRecord(tweet.core.user_results) &&
|
||||
isRecord(tweet.core.user_results.result)
|
||||
? (tweet.core.user_results.result as JsonObject)
|
||||
: emptyObject();
|
||||
const legacy = isRecord(result.legacy) ? result.legacy : emptyObject();
|
||||
const core = isRecord(result.core) ? result.core : emptyObject();
|
||||
return {
|
||||
name:
|
||||
(typeof legacy.name === "string" ? legacy.name : undefined) ??
|
||||
(typeof core.name === "string" ? core.name : undefined),
|
||||
screenName:
|
||||
(typeof legacy.screen_name === "string" ? legacy.screen_name : undefined) ??
|
||||
(typeof core.screen_name === "string" ? core.screen_name : undefined),
|
||||
};
|
||||
}
|
||||
|
||||
function getNoteTweetResult(tweet: JsonObject): JsonObject | null {
|
||||
if (
|
||||
!isRecord(tweet.note_tweet) ||
|
||||
!isRecord(tweet.note_tweet.note_tweet_results) ||
|
||||
!isRecord(tweet.note_tweet.note_tweet_results.result)
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return tweet.note_tweet.note_tweet_results.result as JsonObject;
|
||||
}
|
||||
|
||||
function getNoteTweetText(tweet: JsonObject): string | undefined {
|
||||
const noteTweet = getNoteTweetResult(tweet);
|
||||
return typeof noteTweet?.text === "string" ? noteTweet.text : undefined;
|
||||
}
|
||||
|
||||
interface TweetUrlEntity {
|
||||
url: string;
|
||||
expandedUrl?: string;
|
||||
displayUrl?: string;
|
||||
}
|
||||
|
||||
function collectTweetUrlEntities(values: unknown[]): TweetUrlEntity[] {
|
||||
return values.reduce<TweetUrlEntity[]>((entities, value) => {
|
||||
if (!isRecord(value) || typeof value.url !== "string" || !value.url) {
|
||||
return entities;
|
||||
}
|
||||
|
||||
entities.push({
|
||||
url: value.url,
|
||||
expandedUrl: typeof value.expanded_url === "string" ? value.expanded_url : undefined,
|
||||
displayUrl: typeof value.display_url === "string" ? value.display_url : undefined,
|
||||
});
|
||||
|
||||
return entities;
|
||||
}, []);
|
||||
}
|
||||
|
||||
function getTweetUrlEntities(tweet: JsonObject): TweetUrlEntity[] {
|
||||
const noteTweet = getNoteTweetResult(tweet);
|
||||
const noteTweetEntitySet = noteTweet && isRecord(noteTweet.entity_set) ? noteTweet.entity_set : emptyObject();
|
||||
const noteTweetUrls = collectTweetUrlEntities(Array.isArray(noteTweetEntitySet.urls) ? noteTweetEntitySet.urls : []);
|
||||
|
||||
const legacy = getLegacy(tweet);
|
||||
const legacyEntities = isRecord(legacy.entities) ? legacy.entities : emptyObject();
|
||||
const legacyUrls = collectTweetUrlEntities(Array.isArray(legacyEntities.urls) ? legacyEntities.urls : []);
|
||||
|
||||
const seen = new Set<string>();
|
||||
return [...noteTweetUrls, ...legacyUrls].filter((value) => {
|
||||
if (seen.has(value.url)) {
|
||||
return false;
|
||||
}
|
||||
seen.add(value.url);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export function getTweetText(tweet: JsonObject): string {
|
||||
const legacy = getLegacy(tweet);
|
||||
let text =
|
||||
getNoteTweetText(tweet) ?? (typeof legacy.full_text === "string" ? legacy.full_text : "");
|
||||
|
||||
for (const value of getTweetUrlEntities(tweet)) {
|
||||
const replacement =
|
||||
(typeof value.expandedUrl === "string" && value.expandedUrl) ||
|
||||
(typeof value.displayUrl === "string" && value.displayUrl) ||
|
||||
value.url;
|
||||
text = text.replaceAll(value.url, replacement);
|
||||
}
|
||||
|
||||
const extendedEntities = isRecord(legacy.extended_entities) ? legacy.extended_entities : emptyObject();
|
||||
const media = Array.isArray(extendedEntities.media) ? extendedEntities.media : [];
|
||||
for (const value of media) {
|
||||
if (isRecord(value) && typeof value.url === "string") {
|
||||
text = text.replaceAll(value.url, "").trim();
|
||||
}
|
||||
}
|
||||
|
||||
return text.replace(/\n{3,}/g, "\n\n").trim();
|
||||
}
|
||||
|
||||
function normalizeXImageExtension(raw: string | undefined | null): string | undefined {
|
||||
if (!raw) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const normalized = raw.replace(/^\./, "").trim().toLowerCase();
|
||||
if (!normalized) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return normalized === "jpeg" ? "jpg" : normalized;
|
||||
}
|
||||
|
||||
export function toHighResXImageUrl(rawUrl: string): string {
|
||||
try {
|
||||
const parsed = new URL(rawUrl);
|
||||
if (parsed.hostname.toLowerCase() !== "pbs.twimg.com") {
|
||||
return rawUrl;
|
||||
}
|
||||
|
||||
const pathExtension = normalizeXImageExtension(path.posix.extname(parsed.pathname));
|
||||
const format = normalizeXImageExtension(parsed.searchParams.get("format")) ?? pathExtension;
|
||||
if (!format || !X_IMAGE_EXTENSIONS.has(format)) {
|
||||
return rawUrl;
|
||||
}
|
||||
|
||||
if (pathExtension) {
|
||||
parsed.pathname = parsed.pathname.replace(new RegExp(`\\.${pathExtension}$`, "i"), "");
|
||||
}
|
||||
|
||||
parsed.searchParams.set("format", format);
|
||||
parsed.searchParams.set("name", "4096x4096");
|
||||
return parsed.toString();
|
||||
} catch {
|
||||
return rawUrl;
|
||||
}
|
||||
}
|
||||
|
||||
export function getTweetMedia(tweet: JsonObject): XMedia[] {
|
||||
const legacy = getLegacy(tweet);
|
||||
const extendedEntities = isRecord(legacy.extended_entities) ? legacy.extended_entities : emptyObject();
|
||||
const media = Array.isArray(extendedEntities.media) ? extendedEntities.media : [];
|
||||
|
||||
return media
|
||||
.map((value) => {
|
||||
if (!isRecord(value) || typeof value.type !== "string") {
|
||||
return null;
|
||||
}
|
||||
if (value.type === "photo" && typeof value.media_url_https === "string") {
|
||||
return {
|
||||
type: value.type,
|
||||
url: toHighResXImageUrl(value.media_url_https),
|
||||
alt: typeof value.ext_alt_text === "string" ? value.ext_alt_text : undefined,
|
||||
};
|
||||
}
|
||||
if ((value.type === "video" || value.type === "animated_gif") && typeof value.media_url_https === "string") {
|
||||
return {
|
||||
type: value.type,
|
||||
url: value.media_url_https,
|
||||
};
|
||||
}
|
||||
return null;
|
||||
})
|
||||
.filter((value): value is XMedia => value !== null);
|
||||
}
|
||||
|
||||
export function getTweetUrl(tweet: JsonObject, fallbackUrl: string): string {
|
||||
const user = getUser(tweet);
|
||||
const fallbackScreenName = extractScreenNameFromUrl(fallbackUrl);
|
||||
const id = typeof tweet.rest_id === "string" ? tweet.rest_id : "";
|
||||
const screenName = user.screenName ?? fallbackScreenName;
|
||||
if (screenName && id) {
|
||||
return `https://x.com/${screenName}/status/${id}`;
|
||||
}
|
||||
return fallbackUrl;
|
||||
}
|
||||
|
||||
export function getQuotedTweet(tweet: JsonObject, fallbackUrl: string): XQuotedTweet | undefined {
|
||||
const quoted = unwrapTweetResult(
|
||||
isRecord(tweet.quoted_status_result) ? tweet.quoted_status_result.result : null,
|
||||
);
|
||||
if (!quoted) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const user = getUser(quoted);
|
||||
return {
|
||||
id: typeof quoted.rest_id === "string" ? quoted.rest_id : "",
|
||||
author: user.screenName,
|
||||
authorName: user.name,
|
||||
text: getTweetText(quoted),
|
||||
url: getTweetUrl(quoted, fallbackUrl),
|
||||
media: getTweetMedia(quoted),
|
||||
};
|
||||
}
|
||||
|
||||
export function extractScreenNameFromUrl(url: string): string | undefined {
|
||||
try {
|
||||
const parsed = new URL(url);
|
||||
const match = parsed.pathname.match(/^\/([^/]+)\/(?:status|article)\//);
|
||||
if (!match) {
|
||||
return undefined;
|
||||
}
|
||||
if (match[1] === "i") {
|
||||
return undefined;
|
||||
}
|
||||
return match[1];
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
export function toXTweet(tweet: JsonObject, fallbackUrl: string): XTweet {
|
||||
const legacy = getLegacy(tweet);
|
||||
const user = getUser(tweet);
|
||||
const fallbackScreenName = extractScreenNameFromUrl(fallbackUrl);
|
||||
const screenName = user.screenName ?? fallbackScreenName;
|
||||
return {
|
||||
id: typeof tweet.rest_id === "string" ? tweet.rest_id : "",
|
||||
author: screenName,
|
||||
authorName: user.name,
|
||||
text: getTweetText(tweet),
|
||||
likes: typeof legacy.favorite_count === "number" ? legacy.favorite_count : 0,
|
||||
retweets: typeof legacy.retweet_count === "number" ? legacy.retweet_count : 0,
|
||||
replies: typeof legacy.reply_count === "number" ? legacy.reply_count : 0,
|
||||
createdAt: typeof legacy.created_at === "string" ? legacy.created_at : undefined,
|
||||
inReplyTo: typeof legacy.in_reply_to_status_id_str === "string" ? legacy.in_reply_to_status_id_str : undefined,
|
||||
url: getTweetUrl(tweet, fallbackUrl),
|
||||
media: getTweetMedia(tweet),
|
||||
quotedTweet: getQuotedTweet(tweet, fallbackUrl),
|
||||
};
|
||||
}
|
||||
|
||||
export function normalizeTitle(text: string, fallback: string): string {
|
||||
const firstLine = text.split("\n")[0]?.trim();
|
||||
if (!firstLine) {
|
||||
return fallback;
|
||||
}
|
||||
return firstLine.slice(0, 120);
|
||||
}
|
||||
|
||||
export function formatTweetAuthor(tweet: XTweet): string | undefined {
|
||||
if (tweet.author && tweet.authorName) {
|
||||
return `${tweet.authorName} (@${tweet.author})`;
|
||||
}
|
||||
if (tweet.author) {
|
||||
return `@${tweet.author}`;
|
||||
}
|
||||
return tweet.authorName;
|
||||
}
|
||||
|
||||
export function getTweetAuthorMetadata(tweet: XTweet): Record<string, unknown> {
|
||||
return {
|
||||
authorName: tweet.authorName,
|
||||
authorUsername: tweet.author,
|
||||
authorUrl: tweet.author ? `https://x.com/${tweet.author}` : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
export function formatMediaList(media: XMedia[]): string[] {
|
||||
return media.map((item) => {
|
||||
if (item.type === "photo") {
|
||||
return `photo: ${item.url}`;
|
||||
}
|
||||
return `${item.type}: ${item.url}`;
|
||||
});
|
||||
}
|
||||
|
||||
export function filterXGraphQlEntries(entries: NetworkEntry[]): NetworkEntry[] {
|
||||
return entries.filter((entry) => entry.url.includes("/graphql/"));
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
import type { ExtractedDocument, ContentBlock } from "../../extract/document";
|
||||
import { findTweetNode, formatMediaList, formatTweetAuthor, getTweetAuthorMetadata, normalizeTitle, toXTweet } from "./shared";
|
||||
|
||||
export function extractSingleTweetDocumentFromPayload(
|
||||
payload: unknown,
|
||||
statusId: string,
|
||||
pageUrl: string,
|
||||
): ExtractedDocument | null {
|
||||
const tweet = findTweetNode(payload, statusId);
|
||||
if (!tweet) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const xTweet = toXTweet(tweet, pageUrl);
|
||||
const content: ContentBlock[] = [];
|
||||
|
||||
if (xTweet.text) {
|
||||
content.push({ type: "paragraph", text: xTweet.text });
|
||||
}
|
||||
|
||||
for (const mediaLine of formatMediaList(xTweet.media)) {
|
||||
if (mediaLine.startsWith("photo: ")) {
|
||||
content.push({
|
||||
type: "image",
|
||||
url: mediaLine.slice("photo: ".length),
|
||||
});
|
||||
} else {
|
||||
content.push({
|
||||
type: "list",
|
||||
ordered: false,
|
||||
items: [mediaLine],
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (xTweet.quotedTweet) {
|
||||
const quotedLines: string[] = [];
|
||||
const quotedAuthor =
|
||||
xTweet.quotedTweet.author && xTweet.quotedTweet.authorName
|
||||
? `${xTweet.quotedTweet.authorName} (@${xTweet.quotedTweet.author})`
|
||||
: xTweet.quotedTweet.author
|
||||
? `@${xTweet.quotedTweet.author}`
|
||||
: xTweet.quotedTweet.authorName;
|
||||
|
||||
if (quotedAuthor) {
|
||||
quotedLines.push(quotedAuthor);
|
||||
}
|
||||
if (xTweet.quotedTweet.text) {
|
||||
quotedLines.push(xTweet.quotedTweet.text);
|
||||
}
|
||||
quotedLines.push(...formatMediaList(xTweet.quotedTweet.media));
|
||||
|
||||
if (quotedLines.length > 0) {
|
||||
content.push({ type: "heading", depth: 2, text: "Quoted Tweet" });
|
||||
content.push({ type: "quote", text: quotedLines.join("\n\n") });
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
url: pageUrl,
|
||||
canonicalUrl: xTweet.url,
|
||||
title: normalizeTitle(
|
||||
xTweet.author ? `@${xTweet.author}: ${xTweet.text}` : xTweet.text,
|
||||
"Tweet",
|
||||
),
|
||||
author: formatTweetAuthor(xTweet),
|
||||
siteName: "X",
|
||||
publishedAt: xTweet.createdAt,
|
||||
summary: xTweet.text.slice(0, 200) || undefined,
|
||||
adapter: "x",
|
||||
metadata: {
|
||||
kind: "x/post",
|
||||
tweetId: xTweet.id,
|
||||
...getTweetAuthorMetadata(xTweet),
|
||||
conversationId:
|
||||
typeof tweet.legacy === "object" &&
|
||||
tweet.legacy !== null &&
|
||||
typeof (tweet.legacy as Record<string, unknown>).conversation_id_str === "string"
|
||||
? (tweet.legacy as Record<string, unknown>).conversation_id_str
|
||||
: undefined,
|
||||
favoriteCount: xTweet.likes,
|
||||
replyCount: xTweet.replies,
|
||||
retweetCount: xTweet.retweets,
|
||||
},
|
||||
content,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,286 @@
|
||||
import type { AdapterContext } from "../types";
|
||||
import { extractThreadTweetsFromPayloads } from "./thread";
|
||||
import { collectXJsonPayloads, getRelevantXThreadEntries, prefetchRelevantXThreadBodies } from "./payloads";
|
||||
|
||||
interface ClickTextResult {
|
||||
clicked: boolean;
|
||||
text?: string;
|
||||
}
|
||||
|
||||
interface ScrollStepResult {
|
||||
moved: boolean;
|
||||
atTop: boolean;
|
||||
atBottom: boolean;
|
||||
}
|
||||
|
||||
interface ThreadProgress {
|
||||
tweetCount: number;
|
||||
firstTweetId?: string;
|
||||
lastTweetId?: string;
|
||||
requestCount: number;
|
||||
tweetDetailCount: number;
|
||||
}
|
||||
|
||||
interface TopProbeState {
|
||||
requestCount: number;
|
||||
tweetDetailCount: number;
|
||||
scrollHeight: number;
|
||||
}
|
||||
|
||||
function sleep(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
async function waitForXNetworkSettle(context: AdapterContext, reason: string): Promise<void> {
|
||||
try {
|
||||
await context.network.waitForIdle({
|
||||
idleMs: 650,
|
||||
timeoutMs: Math.min(context.timeoutMs, 5_000),
|
||||
});
|
||||
} catch {
|
||||
context.log.debug(`Network idle timed out after ${reason}.`);
|
||||
}
|
||||
}
|
||||
|
||||
async function captureTopProbeState(context: AdapterContext): Promise<TopProbeState> {
|
||||
const entries = getRelevantXThreadEntries(context);
|
||||
const scrollHeight = await context.browser.evaluate<number>(`
|
||||
(() => {
|
||||
const scrollRoot = document.scrollingElement ?? document.documentElement ?? document.body;
|
||||
return scrollRoot.scrollHeight;
|
||||
})()
|
||||
`);
|
||||
|
||||
return {
|
||||
requestCount: entries.length,
|
||||
tweetDetailCount: entries.filter((entry) => entry.url.includes("TweetDetail")).length,
|
||||
scrollHeight,
|
||||
};
|
||||
}
|
||||
|
||||
async function waitForTopProbe(context: AdapterContext): Promise<boolean> {
|
||||
const initial = await captureTopProbeState(context);
|
||||
const deadline = Date.now() + 1_200;
|
||||
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
await context.network.waitForIdle({
|
||||
idleMs: 250,
|
||||
timeoutMs: 350,
|
||||
});
|
||||
} catch {
|
||||
// Keep polling until the shorter top-probe budget expires.
|
||||
}
|
||||
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
const next = await captureTopProbeState(context);
|
||||
if (
|
||||
next.requestCount > initial.requestCount ||
|
||||
next.tweetDetailCount > initial.tweetDetailCount ||
|
||||
next.scrollHeight > initial.scrollHeight + 4
|
||||
) {
|
||||
context.log.debug("Observed additional X thread activity while probing the page top.");
|
||||
return true;
|
||||
}
|
||||
|
||||
await sleep(120);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
async function scrollThreadToTop(context: AdapterContext): Promise<void> {
|
||||
let settledTopChecks = 0;
|
||||
|
||||
while (settledTopChecks < 2) {
|
||||
const scroll = await context.browser.evaluate<ScrollStepResult>(`
|
||||
(() => {
|
||||
const scrollRoot = document.scrollingElement ?? document.documentElement ?? document.body;
|
||||
const before = window.scrollY;
|
||||
window.scrollTo({ top: 0, left: 0, behavior: "instant" });
|
||||
const after = window.scrollY;
|
||||
return {
|
||||
moved: after !== before,
|
||||
atTop: after <= 4,
|
||||
atBottom: window.innerHeight + after >= scrollRoot.scrollHeight - 4,
|
||||
};
|
||||
})()
|
||||
`);
|
||||
await sleep(140);
|
||||
await waitForXNetworkSettle(context, "scrolling X thread to top");
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
|
||||
if (scroll.moved) {
|
||||
settledTopChecks = 0;
|
||||
continue;
|
||||
}
|
||||
|
||||
const observedTopActivity = await waitForTopProbe(context);
|
||||
if (observedTopActivity) {
|
||||
settledTopChecks = 0;
|
||||
continue;
|
||||
}
|
||||
|
||||
settledTopChecks += 1;
|
||||
}
|
||||
}
|
||||
|
||||
async function clickVisibleShowReplies(context: AdapterContext): Promise<ClickTextResult> {
|
||||
return context.browser.evaluate<ClickTextResult>(`
|
||||
(() => {
|
||||
const normalize = (value) => value.replace(/\\s+/g, " ").trim();
|
||||
const matches = [
|
||||
/^Show replies$/i,
|
||||
/^Show more replies$/i,
|
||||
/^Show additional replies$/i,
|
||||
/^显示回复$/,
|
||||
/^展开回复$/,
|
||||
];
|
||||
const isVisible = (element) => {
|
||||
if (!(element instanceof HTMLElement)) {
|
||||
return false;
|
||||
}
|
||||
const rect = element.getBoundingClientRect();
|
||||
const style = window.getComputedStyle(element);
|
||||
return (
|
||||
rect.width > 0 &&
|
||||
rect.height > 0 &&
|
||||
style.visibility !== "hidden" &&
|
||||
style.display !== "none"
|
||||
);
|
||||
};
|
||||
|
||||
const selectors = [
|
||||
"a",
|
||||
"button",
|
||||
'[role="button"]',
|
||||
'[role="link"]',
|
||||
];
|
||||
|
||||
for (const element of document.querySelectorAll(selectors.join(","))) {
|
||||
if (!isVisible(element)) {
|
||||
continue;
|
||||
}
|
||||
const text = normalize(element.textContent ?? "");
|
||||
if (!text || !matches.some((pattern) => pattern.test(text))) {
|
||||
continue;
|
||||
}
|
||||
element.scrollIntoView({ block: "center", inline: "nearest" });
|
||||
if (element instanceof HTMLElement) {
|
||||
element.click();
|
||||
return { clicked: true, text };
|
||||
}
|
||||
}
|
||||
|
||||
return { clicked: false };
|
||||
})()
|
||||
`);
|
||||
}
|
||||
|
||||
async function expandVisibleShowReplies(context: AdapterContext): Promise<number> {
|
||||
let clickCount = 0;
|
||||
|
||||
while (clickCount < 8) {
|
||||
const result = await clickVisibleShowReplies(context).catch<ClickTextResult>(() => ({ clicked: false }));
|
||||
if (!result.clicked) {
|
||||
break;
|
||||
}
|
||||
|
||||
clickCount += 1;
|
||||
context.log.debug(`Expanded X thread replies via "${result.text ?? "Show replies"}".`);
|
||||
await sleep(250);
|
||||
await waitForXNetworkSettle(context, "expanding Show replies");
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
}
|
||||
|
||||
return clickCount;
|
||||
}
|
||||
|
||||
async function scrollThreadBy(context: AdapterContext, stepPx: number): Promise<ScrollStepResult> {
|
||||
const result = await context.browser.evaluate<ScrollStepResult>(`
|
||||
(() => {
|
||||
const scrollRoot = document.scrollingElement ?? document.documentElement ?? document.body;
|
||||
const before = window.scrollY;
|
||||
window.scrollBy({ top: ${stepPx}, left: 0, behavior: "instant" });
|
||||
const after = window.scrollY;
|
||||
return {
|
||||
moved: after !== before,
|
||||
atTop: after <= 4,
|
||||
atBottom: window.innerHeight + after >= scrollRoot.scrollHeight - 4,
|
||||
};
|
||||
})()
|
||||
`);
|
||||
|
||||
await sleep(140);
|
||||
await waitForXNetworkSettle(context, "scrolling X thread");
|
||||
await prefetchRelevantXThreadBodies(context);
|
||||
return result;
|
||||
}
|
||||
|
||||
async function captureThreadProgress(context: AdapterContext, statusId: string): Promise<ThreadProgress> {
|
||||
const entries = getRelevantXThreadEntries(context);
|
||||
const payloads = await collectXJsonPayloads(context);
|
||||
const tweets = extractThreadTweetsFromPayloads(payloads, statusId, context.input.url.toString());
|
||||
return {
|
||||
tweetCount: tweets.length,
|
||||
firstTweetId: tweets[0]?.id,
|
||||
lastTweetId: tweets[tweets.length - 1]?.id,
|
||||
requestCount: entries.length,
|
||||
tweetDetailCount: entries.filter((entry) => entry.url.includes("TweetDetail")).length,
|
||||
};
|
||||
}
|
||||
|
||||
export async function loadFullXThread(context: AdapterContext, statusId: string): Promise<void> {
|
||||
await scrollThreadToTop(context);
|
||||
|
||||
let progress = await captureThreadProgress(context, statusId);
|
||||
let stagnantRounds = 0;
|
||||
let roundsWithoutMovement = 0;
|
||||
let distanceWithoutThreadActivityPx = 0;
|
||||
|
||||
for (let round = 0; ; round += 1) {
|
||||
const stepPx = round < 12 ? 1_200 : 1_600;
|
||||
let expandedCount = await expandVisibleShowReplies(context);
|
||||
const scroll = await scrollThreadBy(context, stepPx);
|
||||
expandedCount += await expandVisibleShowReplies(context);
|
||||
const nextProgress = await captureThreadProgress(context, statusId);
|
||||
const grew =
|
||||
nextProgress.tweetCount > progress.tweetCount ||
|
||||
nextProgress.firstTweetId !== progress.firstTweetId ||
|
||||
nextProgress.lastTweetId !== progress.lastTweetId ||
|
||||
nextProgress.requestCount > progress.requestCount ||
|
||||
nextProgress.tweetDetailCount > progress.tweetDetailCount;
|
||||
|
||||
if (grew) {
|
||||
context.log.debug(
|
||||
`X thread progress: ${nextProgress.tweetCount} tweets (${nextProgress.firstTweetId ?? "unknown"} -> ${nextProgress.lastTweetId ?? "unknown"}), ${nextProgress.requestCount} requests, ${nextProgress.tweetDetailCount} TweetDetail.`,
|
||||
);
|
||||
stagnantRounds = 0;
|
||||
distanceWithoutThreadActivityPx = 0;
|
||||
} else if (expandedCount > 0) {
|
||||
stagnantRounds = 0;
|
||||
distanceWithoutThreadActivityPx = 0;
|
||||
} else {
|
||||
stagnantRounds += 1;
|
||||
distanceWithoutThreadActivityPx += stepPx;
|
||||
}
|
||||
|
||||
roundsWithoutMovement = scroll.moved ? 0 : roundsWithoutMovement + 1;
|
||||
progress = nextProgress;
|
||||
|
||||
if (scroll.atBottom && stagnantRounds >= 6) {
|
||||
context.log.debug("Stopping X thread scroll after reaching page bottom with no further thread progress.");
|
||||
break;
|
||||
}
|
||||
|
||||
if (roundsWithoutMovement >= 2 && stagnantRounds >= 4) {
|
||||
context.log.debug("Stopping X thread scroll after repeated downward scrolls no longer move the page.");
|
||||
break;
|
||||
}
|
||||
|
||||
if (distanceWithoutThreadActivityPx >= 24_000 && stagnantRounds >= 12) {
|
||||
context.log.debug("Stopping X thread scroll after a long stretch with no thread-related progress.");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,316 @@
|
||||
import type { ExtractedDocument } from "../../extract/document";
|
||||
import {
|
||||
formatMediaList,
|
||||
formatTweetAuthor,
|
||||
getLegacy,
|
||||
getTweetAuthorMetadata,
|
||||
isRecord,
|
||||
normalizeTitle,
|
||||
toXTweet,
|
||||
unwrapTweetResult,
|
||||
} from "./shared";
|
||||
import type { JsonObject, XQuotedTweet, XTweet } from "./types";
|
||||
|
||||
interface ParsedThreadTweet extends XTweet {
|
||||
userId?: string;
|
||||
conversationId?: string;
|
||||
inReplyToUserId?: string;
|
||||
sortTimestamp: number;
|
||||
}
|
||||
|
||||
function compareTweetIds(left: string, right: string): number {
|
||||
try {
|
||||
const leftId = BigInt(left);
|
||||
const rightId = BigInt(right);
|
||||
if (leftId === rightId) {
|
||||
return 0;
|
||||
}
|
||||
return leftId < rightId ? -1 : 1;
|
||||
} catch {
|
||||
return left.localeCompare(right);
|
||||
}
|
||||
}
|
||||
|
||||
function toTimestamp(value: string | undefined): number {
|
||||
if (!value) {
|
||||
return 0;
|
||||
}
|
||||
const parsed = Date.parse(value);
|
||||
return Number.isNaN(parsed) ? 0 : parsed;
|
||||
}
|
||||
|
||||
function scoreParsedTweet(tweet: ParsedThreadTweet): number {
|
||||
return (
|
||||
(tweet.text ? 4 : 0) +
|
||||
(tweet.author ? 2 : 0) +
|
||||
(tweet.authorName ? 2 : 0) +
|
||||
(tweet.media.length > 0 ? 1 : 0)
|
||||
);
|
||||
}
|
||||
|
||||
function toParsedThreadTweet(tweet: JsonObject, pageUrl: string): ParsedThreadTweet {
|
||||
const legacy = getLegacy(tweet);
|
||||
const xTweet = toXTweet(tweet, pageUrl);
|
||||
|
||||
return {
|
||||
...xTweet,
|
||||
userId: typeof legacy.user_id_str === "string" ? legacy.user_id_str : undefined,
|
||||
conversationId: typeof legacy.conversation_id_str === "string" ? legacy.conversation_id_str : undefined,
|
||||
inReplyToUserId: typeof legacy.in_reply_to_user_id_str === "string" ? legacy.in_reply_to_user_id_str : undefined,
|
||||
sortTimestamp: toTimestamp(xTweet.createdAt),
|
||||
};
|
||||
}
|
||||
|
||||
function collectTweetFromItemContent(
|
||||
itemContent: unknown,
|
||||
pageUrl: string,
|
||||
tweets: Map<string, ParsedThreadTweet>,
|
||||
): void {
|
||||
if (!isRecord(itemContent)) {
|
||||
return;
|
||||
}
|
||||
|
||||
const tweet = unwrapTweetResult(
|
||||
isRecord(itemContent.tweet_results) ? itemContent.tweet_results.result : null,
|
||||
);
|
||||
if (!tweet || typeof tweet.rest_id !== "string") {
|
||||
return;
|
||||
}
|
||||
|
||||
const parsed = toParsedThreadTweet(tweet, pageUrl);
|
||||
const existing = tweets.get(parsed.id);
|
||||
if (!existing || scoreParsedTweet(parsed) >= scoreParsedTweet(existing)) {
|
||||
tweets.set(parsed.id, parsed);
|
||||
}
|
||||
}
|
||||
|
||||
function collectTweetsFromItems(
|
||||
items: unknown,
|
||||
pageUrl: string,
|
||||
tweets: Map<string, ParsedThreadTweet>,
|
||||
): void {
|
||||
if (!Array.isArray(items)) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (const item of items) {
|
||||
if (!isRecord(item)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (isRecord(item.item) && isRecord(item.item.itemContent)) {
|
||||
collectTweetFromItemContent(item.item.itemContent, pageUrl, tweets);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (isRecord(item.itemContent)) {
|
||||
collectTweetFromItemContent(item.itemContent, pageUrl, tweets);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function getInstructions(payload: unknown): unknown[] {
|
||||
if (!isRecord(payload) || !isRecord(payload.data)) {
|
||||
return [];
|
||||
}
|
||||
|
||||
const { data } = payload;
|
||||
return (
|
||||
(isRecord(data.threaded_conversation_with_injections_v2) &&
|
||||
Array.isArray(data.threaded_conversation_with_injections_v2.instructions)
|
||||
? data.threaded_conversation_with_injections_v2.instructions
|
||||
: undefined) ??
|
||||
(isRecord(data.threaded_conversation_with_injections) &&
|
||||
Array.isArray(data.threaded_conversation_with_injections.instructions)
|
||||
? data.threaded_conversation_with_injections.instructions
|
||||
: undefined) ??
|
||||
(isRecord(data.tweetResult) &&
|
||||
isRecord(data.tweetResult.result) &&
|
||||
isRecord(data.tweetResult.result.timeline) &&
|
||||
Array.isArray(data.tweetResult.result.timeline.instructions)
|
||||
? data.tweetResult.result.timeline.instructions
|
||||
: [])
|
||||
);
|
||||
}
|
||||
|
||||
function parseTweetDetailPayload(payload: unknown, pageUrl: string): ParsedThreadTweet[] {
|
||||
const tweets = new Map<string, ParsedThreadTweet>();
|
||||
|
||||
const instructions = getInstructions(payload);
|
||||
for (const instruction of instructions) {
|
||||
if (!isRecord(instruction)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
collectTweetsFromItems(instruction.moduleItems, pageUrl, tweets);
|
||||
|
||||
if (!Array.isArray(instruction.entries)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (const entry of instruction.entries) {
|
||||
if (!isRecord(entry)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const content = isRecord(entry.content) ? entry.content : {};
|
||||
collectTweetFromItemContent(content.itemContent, pageUrl, tweets);
|
||||
collectTweetsFromItems(content.items, pageUrl, tweets);
|
||||
}
|
||||
}
|
||||
|
||||
return Array.from(tweets.values());
|
||||
}
|
||||
|
||||
function buildContinuousThread(tweets: ParsedThreadTweet[], statusId: string): ParsedThreadTweet[] {
|
||||
const byId = new Map<string, ParsedThreadTweet>();
|
||||
for (const tweet of tweets) {
|
||||
const existing = byId.get(tweet.id);
|
||||
if (!existing || scoreParsedTweet(tweet) >= scoreParsedTweet(existing)) {
|
||||
byId.set(tweet.id, tweet);
|
||||
}
|
||||
}
|
||||
|
||||
const rootTweet = byId.get(statusId);
|
||||
if (!rootTweet?.userId || !rootTweet.conversationId) {
|
||||
return [];
|
||||
}
|
||||
|
||||
const candidates = Array.from(byId.values()).filter(
|
||||
(tweet) =>
|
||||
tweet.id === statusId ||
|
||||
(tweet.userId === rootTweet.userId && tweet.conversationId === rootTweet.conversationId),
|
||||
);
|
||||
|
||||
const repliesByParent = new Map<string, ParsedThreadTweet[]>();
|
||||
for (const tweet of candidates) {
|
||||
if (!tweet.inReplyTo || tweet.id === statusId) {
|
||||
continue;
|
||||
}
|
||||
const bucket = repliesByParent.get(tweet.inReplyTo) ?? [];
|
||||
bucket.push(tweet);
|
||||
bucket.sort((left, right) => {
|
||||
if (left.sortTimestamp !== right.sortTimestamp) {
|
||||
return left.sortTimestamp - right.sortTimestamp;
|
||||
}
|
||||
return compareTweetIds(left.id, right.id);
|
||||
});
|
||||
repliesByParent.set(tweet.inReplyTo, bucket);
|
||||
}
|
||||
|
||||
const ancestorPath: ParsedThreadTweet[] = [rootTweet];
|
||||
const ancestorSeen = new Set<string>([rootTweet.id]);
|
||||
let currentAncestor = rootTweet;
|
||||
|
||||
while (currentAncestor.inReplyTo) {
|
||||
const parent = byId.get(currentAncestor.inReplyTo);
|
||||
if (!parent || ancestorSeen.has(parent.id)) {
|
||||
break;
|
||||
}
|
||||
ancestorPath.unshift(parent);
|
||||
ancestorSeen.add(parent.id);
|
||||
currentAncestor = parent;
|
||||
}
|
||||
|
||||
const chain = ancestorPath.slice();
|
||||
const seen = new Set<string>(chain.map((tweet) => tweet.id));
|
||||
let currentId = rootTweet.id;
|
||||
|
||||
while (true) {
|
||||
const next = (repliesByParent.get(currentId) ?? []).find((tweet) => !seen.has(tweet.id));
|
||||
if (!next) {
|
||||
break;
|
||||
}
|
||||
chain.push(next);
|
||||
seen.add(next.id);
|
||||
currentId = next.id;
|
||||
}
|
||||
|
||||
return chain;
|
||||
}
|
||||
|
||||
export function extractThreadTweetsFromPayloads(
|
||||
payloads: unknown[],
|
||||
statusId: string,
|
||||
pageUrl: string,
|
||||
): XTweet[] {
|
||||
const parsedTweets: ParsedThreadTweet[] = [];
|
||||
|
||||
for (const payload of payloads) {
|
||||
parsedTweets.push(...parseTweetDetailPayload(payload, pageUrl));
|
||||
}
|
||||
|
||||
return buildContinuousThread(parsedTweets, statusId).map(({ sortTimestamp: _sortTimestamp, ...tweet }) => tweet);
|
||||
}
|
||||
|
||||
function buildQuotedTweetMarkdown(quotedTweet: XQuotedTweet): string {
|
||||
const author = quotedTweet.author ? `@${quotedTweet.author}` : "Unknown";
|
||||
const name = quotedTweet.authorName ? `${quotedTweet.authorName} ` : "";
|
||||
const lines: string[] = [`Quoted Tweet${quotedTweet.author || quotedTweet.authorName ? `: ${name}${author}`.trim() : ""}`];
|
||||
|
||||
if (quotedTweet.text) {
|
||||
lines.push(...quotedTweet.text.split("\n"));
|
||||
}
|
||||
|
||||
for (const mediaLine of formatMediaList(quotedTweet.media)) {
|
||||
lines.push(mediaLine);
|
||||
}
|
||||
|
||||
return lines.map((line) => (line ? `> ${line}` : ">")).join("\n");
|
||||
}
|
||||
|
||||
function buildThreadMarkdown(tweets: XTweet[]): string {
|
||||
return tweets
|
||||
.map((tweet, index) => {
|
||||
const lines: string[] = [];
|
||||
const author = tweet.author ? `@${tweet.author}` : "Unknown";
|
||||
const name = tweet.authorName ? `${tweet.authorName} ` : "";
|
||||
lines.push(`## ${index + 1}. ${name}${author}`.trim());
|
||||
if (tweet.createdAt) {
|
||||
lines.push(`_Published: ${tweet.createdAt}_`);
|
||||
}
|
||||
lines.push(tweet.text || "(No text)");
|
||||
const mediaLines = formatMediaList(tweet.media);
|
||||
if (mediaLines.length > 0) {
|
||||
lines.push(mediaLines.map((line) => `- ${line}`).join("\n"));
|
||||
}
|
||||
if (tweet.quotedTweet) {
|
||||
lines.push(buildQuotedTweetMarkdown(tweet.quotedTweet));
|
||||
}
|
||||
return lines.join("\n\n");
|
||||
})
|
||||
.join("\n\n");
|
||||
}
|
||||
|
||||
export function extractThreadDocumentFromPayloads(
|
||||
payloads: unknown[],
|
||||
statusId: string,
|
||||
pageUrl: string,
|
||||
): ExtractedDocument | null {
|
||||
const tweets = extractThreadTweetsFromPayloads(payloads, statusId, pageUrl);
|
||||
if (tweets.length <= 1) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const rootTweet = tweets[0];
|
||||
const rootAuthor = formatTweetAuthor(rootTweet);
|
||||
|
||||
return {
|
||||
url: pageUrl,
|
||||
canonicalUrl: rootTweet.url,
|
||||
title: normalizeTitle(rootTweet.text, "X Thread"),
|
||||
author: rootAuthor,
|
||||
siteName: "X",
|
||||
publishedAt: rootTweet.createdAt,
|
||||
summary: rootTweet.text.slice(0, 200) || undefined,
|
||||
adapter: "x",
|
||||
metadata: {
|
||||
kind: "x/thread",
|
||||
tweetId: rootTweet.id,
|
||||
tweetCount: tweets.length,
|
||||
lastTweetId: tweets[tweets.length - 1]?.id,
|
||||
...getTweetAuthorMetadata(rootTweet),
|
||||
},
|
||||
content: [{ type: "markdown", markdown: buildThreadMarkdown(tweets) }],
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
export type JsonObject = Record<string, unknown>;
|
||||
|
||||
export interface XUser {
|
||||
name?: string;
|
||||
screenName?: string;
|
||||
}
|
||||
|
||||
export interface XMedia {
|
||||
type: string;
|
||||
url: string;
|
||||
alt?: string;
|
||||
}
|
||||
|
||||
export interface XQuotedTweet {
|
||||
id: string;
|
||||
author?: string;
|
||||
authorName?: string;
|
||||
text: string;
|
||||
url: string;
|
||||
media: XMedia[];
|
||||
}
|
||||
|
||||
export interface XTweet {
|
||||
id: string;
|
||||
author?: string;
|
||||
authorName?: string;
|
||||
text: string;
|
||||
likes: number;
|
||||
retweets: number;
|
||||
replies: number;
|
||||
createdAt?: string;
|
||||
inReplyTo?: string;
|
||||
url: string;
|
||||
media: XMedia[];
|
||||
quotedTweet?: XQuotedTweet;
|
||||
}
|
||||
Reference in New Issue
Block a user