import assert from "node:assert/strict"; import test from "node:test"; import { cleanContent } from "./content-cleaner.js"; import { convertWithLegacyExtractor } from "./legacy-converter.js"; import { extractMetadataFromHtml } from "./markdown-conversion-shared.js"; const CAPTURED_AT = "2026-03-24T03:00:00.000Z"; const NEXT_DATA_HTML = ` Hydrated Story

Short teaser text that should not win over the structured article payload.

`; test("legacy extractor still uses original __NEXT_DATA__ after HTML cleaning", () => { const url = "https://example.com/posts/hydrated-story"; const baseMetadata = extractMetadataFromHtml(NEXT_DATA_HTML, url, CAPTURED_AT); const cleanedHtml = cleanContent(NEXT_DATA_HTML, url); const result = convertWithLegacyExtractor(NEXT_DATA_HTML, baseMetadata, cleanedHtml); assert.equal(result.conversionMethod, "legacy:next-data"); assert.match(result.markdown, /The full article lives in .*NEXT.*DATA/); assert.match(result.markdown, /A second paragraph keeps the content comfortably above the minimum extraction threshold/); assert.doesNotMatch(result.markdown, /Short teaser text that should not win/); assert.equal(result.rawHtml, NEXT_DATA_HTML); });