import assert from "node:assert/strict";
import test from "node:test";
import { cleanContent } from "./content-cleaner.js";
import { convertWithLegacyExtractor } from "./legacy-converter.js";
import { extractMetadataFromHtml } from "./markdown-conversion-shared.js";
const CAPTURED_AT = "2026-03-24T03:00:00.000Z";
const NEXT_DATA_HTML = `
Hydrated Story
Accept cookies
Short teaser text that should not win over the structured article payload.
`;
test("legacy extractor still uses original __NEXT_DATA__ after HTML cleaning", () => {
const url = "https://example.com/posts/hydrated-story";
const baseMetadata = extractMetadataFromHtml(NEXT_DATA_HTML, url, CAPTURED_AT);
const cleanedHtml = cleanContent(NEXT_DATA_HTML, url);
const result = convertWithLegacyExtractor(NEXT_DATA_HTML, baseMetadata, cleanedHtml);
assert.equal(result.conversionMethod, "legacy:next-data");
assert.match(result.markdown, /The full article lives in .*NEXT.*DATA/);
assert.match(result.markdown, /A second paragraph keeps the content comfortably above the minimum extraction threshold/);
assert.doesNotMatch(result.markdown, /Short teaser text that should not win/);
assert.equal(result.rawHtml, NEXT_DATA_HTML);
});