From b68151a179fe9a9e02388b2aad9352a34cffec75 Mon Sep 17 00:00:00 2001 From: seaHi Date: Wed, 16 Sep 2026 21:32:47 +0800 Subject: [PATCH] =?UTF-8?q?refactor:=20=E5=B0=86=E6=96=87=E7=AB=A0?= =?UTF-8?q?=E6=91=98=E8=A6=81=E7=94=9F=E6=88=90=E7=AE=80=E5=8C=96=E4=B8=BA?= =?UTF-8?q?=E7=BA=AF=E6=96=87=E6=9C=AC=E5=A4=84=E7=90=86=E5=B9=B6=E7=A7=BB?= =?UTF-8?q?=E9=99=A4=20HTML=20=E6=B8=B2=E6=9F=93?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/lib/markdown/excerpt.test.ts | 62 +++++------ src/lib/markdown/excerpt.ts | 109 ++----------------- src/routes/+page.server.ts | 6 +- src/routes/+page.svelte | 3 +- src/routes/categories/[slug]/+page.server.ts | 4 +- src/routes/categories/[slug]/+page.svelte | 3 +- src/routes/tags/[tag]/+page.server.ts | 4 +- src/routes/tags/[tag]/+page.svelte | 3 +- 8 files changed, 47 insertions(+), 147 deletions(-) diff --git a/src/lib/markdown/excerpt.test.ts b/src/lib/markdown/excerpt.test.ts index 3edfedf..0dbfd49 100644 --- a/src/lib/markdown/excerpt.test.ts +++ b/src/lib/markdown/excerpt.test.ts @@ -1,63 +1,57 @@ import { describe, expect, it } from 'vitest'; -import { createExcerptHtml } from './excerpt'; +import { createExcerpt } from './excerpt'; -describe('createExcerptHtml', () => { - it('renders inline styles instead of stripping them', () => { - expect(createExcerptHtml('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe( - '这是粗体与斜体,还有删除线。...' +describe('createExcerpt', () => { + it('strips inline markers instead of rendering them', () => { + expect(createExcerpt('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe( + '这是粗体与斜体,还有删除线。' ); }); - it('keeps inline code and drops the backticks', () => { - expect(createExcerptHtml('先执行 `pnpm run build` 再发布')).toBe( - '先执行 pnpm run build 再发布...' - ); + it('drops the backticks of inline code', () => { + expect(createExcerpt('先执行 `pnpm run build` 再发布')).toBe('先执行 pnpm run build 再发布'); }); it('reduces links and images to readable text', () => { - expect(createExcerptHtml('参考[官方文档](https://example.com)与![配图](cover.png)说明')).toBe( - '参考官方文档与配图说明...' + expect(createExcerpt('参考[官方文档](https://example.com)与![配图](cover.png)说明')).toBe( + '参考官方文档与配图说明' ); }); it('removes block level markers so the excerpt stays a single line', () => { - expect(createExcerptHtml('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe( - '标题 第一项 第二项 引用内容...' + expect(createExcerpt('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe( + '标题 第一项 第二项 引用内容' ); }); it('drops Hugo shortcodes and raw HTML', () => { expect( - createExcerptHtml('{{< admonition note "提示" >}}正文{{< /admonition >}}片段') - ).toBe('正文片段...'); + createExcerpt('{{< admonition note "提示" >}}正文{{< /admonition >}}片段') + ).toBe('正文片段'); }); - it('closes inline tags that are cut off by the length limit', () => { - const html = createExcerptHtml(`**${'a'.repeat(200)}**`, 10); + it('truncates to the length limit without appending an ellipsis', () => { + const excerpt = createExcerpt('a'.repeat(200), 10); - expect(html).toBe(`${'a'.repeat(10)}...`); + expect(excerpt).toBe('a'.repeat(10)); + expect(excerpt).not.toContain('...'); }); - it('counts HTML entities as a single visible character', () => { - expect(createExcerptHtml('&'.repeat(30), 5)).toBe(`${'&'.repeat(5)}...`); + it('never returns markdown markers or markup', () => { + const excerpt = createExcerpt('**粗体** 和 `code` 以及 html'); + + expect(excerpt).toBe('粗体 和 code 以及 html'); + expect(excerpt).not.toMatch(/[*`~<>]/); }); - it('never keeps a dangling inline tag when the cut lands on it', () => { - const html = createExcerptHtml(`${'a'.repeat(9)}**bc**`, 10); - - expect(html).toBe(`aaaaaaaaab...`); - }); - - it('escapes characters that would otherwise break the markup', () => { - const html = createExcerptHtml('对比 1 < 2 的结果 '); - - expect(html).toContain('1 < 2'); - expect(html).not.toContain(' { + expect(createExcerpt('对比 1 < 2 的结果是安全的')).toBe('对比 1 < 2 的结果是安全的'); }); it('returns an empty string when there is nothing to show', () => { - expect(createExcerptHtml('')).toBe(''); - expect(createExcerptHtml(' \n\n ')).toBe(''); - expect(createExcerptHtml('{{< admonition note >}}{{< /admonition >}}')).toBe(''); + expect(createExcerpt('')).toBe(''); + expect(createExcerpt(' \n\n ')).toBe(''); + expect(createExcerpt('{{< admonition note >}}{{< /admonition >}}')).toBe(''); + expect(createExcerpt('***')).toBe(''); }); }); diff --git a/src/lib/markdown/excerpt.ts b/src/lib/markdown/excerpt.ts index 3e98f1d..f88f332 100644 --- a/src/lib/markdown/excerpt.ts +++ b/src/lib/markdown/excerpt.ts @@ -1,32 +1,12 @@ -import { Marked } from 'marked'; -import sanitizeHtml from 'sanitize-html'; - -/** 摘要默认保留的可见字符数 */ +/** 摘要默认保留的字符数 */ export const EXCERPT_LENGTH = 150; -/** 摘要中允许保留的行内样式标签(粗体、斜体、删除线、行内代码) */ -export const EXCERPT_INLINE_TAGS = ['strong', 'em', 'del', 'code']; - -const excerptMarked = new Marked(); - -/** 仅匹配摘要白名单内的行内标签,用于截断时配对 */ -const INLINE_TAG_PATTERN = new RegExp( - `^<\\s*(/?)\\s*(${EXCERPT_INLINE_TAGS.join('|')})\\s*>$`, - 'i' -); - -/** 匹配没有任何内容的行内标签,例如 `` */ -const EMPTY_TAG_PATTERN = new RegExp(`<(${EXCERPT_INLINE_TAGS.join('|')})>\\s*`, 'gi'); - /** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */ const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g; -/** 匹配 HTML 实体,截断时按一个可见字符计算 */ -const HTML_ENTITY_PATTERN = /^&(?:#\d+|#x[0-9a-fA-F]+|[a-zA-Z]+);/; - /** - * 把文章内容整理成适合做摘要的单行文本: - * 去掉 shortcode / HTML / 块级 Markdown 标记,保留粗体、斜体等行内标记。 + * 把文章内容整理成适合做摘要的单行纯文本: + * 去掉 shortcode、HTML 与 Markdown 标记,只保留正文文字。 */ function normalizeExcerptSource(source: string): string { return source @@ -42,88 +22,17 @@ function normalizeExcerptSource(source: string): string { .replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行 .replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线 .replace(/\|/g, ' ') // 表格竖线 + .replace(/[*`~]/g, '') // 行内标记:粗体、斜体、删除线、行内代码 .replace(/\s+/g, ' ') .trim(); } /** - * 按可见字符数截断行内 HTML,并补齐被截断的标签。 - * 只接受白名单内的行内标签,遇到其它标签立即停止,保证输出安全。 + * 由文章内容生成纯文本摘要:不保留任何 Markdown 标记,也不追加省略号。 + * 无内容时返回空字符串。 */ -function truncateInlineHtml(html: string, maxLength: number): string { - let result = ''; - let visible = 0; - const openTags: string[] = []; +export function createExcerpt(source: string, maxLength: number = EXCERPT_LENGTH): string { + const plainText = normalizeExcerptSource(source || ''); - for (let index = 0; index < html.length && visible < maxLength;) { - const char = html[index]; - - if (char === '<') { - const end = html.indexOf('>', index); - if (end === -1) break; - - const tag = html.slice(index, end + 1); - const match = INLINE_TAG_PATTERN.exec(tag); - if (!match) break; - - const name = match[2].toLowerCase(); - if (match[1]) { - const position = openTags.lastIndexOf(name); - if (position !== -1) openTags.splice(position, 1); - } else { - openTags.push(name); - } - - result += tag; - index = end + 1; - continue; - } - - if (char === '&') { - const entity = HTML_ENTITY_PATTERN.exec(html.slice(index)); - if (entity) { - result += entity[0]; - visible += 1; - index += entity[0].length; - continue; - } - } - - result += char; - visible += 1; - index += 1; - } - - for (const name of openTags.reverse()) { - result += ``; - } - - let cleaned = result; - for (;;) { - const next = cleaned.replace(EMPTY_TAG_PATTERN, '').replace(/\s+$/, ''); - if (next === cleaned) break; - cleaned = next; - } - - return cleaned; -} - -/** - * 由文章内容生成带行内样式的摘要 HTML(粗体、斜体、删除线、行内代码)。 - * 返回值已做白名单净化,可安全地用 `{@html}` 渲染;无内容时返回空字符串。 - */ -export function createExcerptHtml(source: string, maxLength: number = EXCERPT_LENGTH): string { - const normalized = normalizeExcerptSource(source || ''); - if (!normalized) return ''; - - // 先解析再截断:源文本必须先完整解析,否则成对的行内标记会被切断而失效 - const html = excerptMarked.parseInline(normalized) as string; - const sanitized = sanitizeHtml(html, { - allowedTags: EXCERPT_INLINE_TAGS, - allowedAttributes: {}, - disallowedTagsMode: 'discard' - }); - - const truncated = truncateInlineHtml(sanitized, maxLength).trim(); - return truncated ? `${truncated}...` : ''; + return plainText.slice(0, maxLength).trim(); } diff --git a/src/routes/+page.server.ts b/src/routes/+page.server.ts index 092034c..cbedebf 100644 --- a/src/routes/+page.server.ts +++ b/src/routes/+page.server.ts @@ -1,6 +1,6 @@ /* eslint-disable @typescript-eslint/no-explicit-any */ -import { createExcerptHtml } from '$lib/markdown/excerpt'; +import { createExcerpt } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; export const load: PageServerLoad = async ({ platform }) => { @@ -51,8 +51,8 @@ export const load: PageServerLoad = async ({ platform }) => { // Prefer the manually marked summary, otherwise derive it from the beginning const contentStr = ((post.content as string) || '').split('')[0]; - // Inline styles (bold, italic, ...) are kept as safe HTML for the cards - const excerpt = createExcerptHtml(contentStr); + // The excerpt is plain text: no Markdown markers and no trailing ellipsis + const excerpt = createExcerpt(contentStr); // Do not send full content to the list view to save bandwidth diff --git a/src/routes/+page.svelte b/src/routes/+page.svelte index 5049c9f..4bfc3f3 100644 --- a/src/routes/+page.svelte +++ b/src/routes/+page.svelte @@ -24,8 +24,7 @@ {/if}

{post.title}

- -

{@html post.excerpt}

+

{post.excerpt}

diff --git a/src/routes/categories/[slug]/+page.server.ts b/src/routes/categories/[slug]/+page.server.ts index a45a068..f377a53 100644 --- a/src/routes/categories/[slug]/+page.server.ts +++ b/src/routes/categories/[slug]/+page.server.ts @@ -1,5 +1,5 @@ import { error, isHttpError } from '@sveltejs/kit'; -import { createExcerptHtml } from '$lib/markdown/excerpt'; +import { createExcerpt } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; const PAGE_SIZE = 20; @@ -72,7 +72,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) => year: 'numeric', timeZone: 'UTC' }), - excerpt: createExcerptHtml(post.excerpt_source || '') + excerpt: createExcerpt(post.excerpt_source || '') }; }); diff --git a/src/routes/categories/[slug]/+page.svelte b/src/routes/categories/[slug]/+page.svelte index 01427c7..85f7be4 100644 --- a/src/routes/categories/[slug]/+page.svelte +++ b/src/routes/categories/[slug]/+page.svelte @@ -38,8 +38,7 @@ {/if}

{post.title}

- -

{@html post.excerpt}

+

{post.excerpt}

diff --git a/src/routes/tags/[tag]/+page.server.ts b/src/routes/tags/[tag]/+page.server.ts index 38649a8..057ff5c 100644 --- a/src/routes/tags/[tag]/+page.server.ts +++ b/src/routes/tags/[tag]/+page.server.ts @@ -1,5 +1,5 @@ import { error, isHttpError } from '@sveltejs/kit'; -import { createExcerptHtml } from '$lib/markdown/excerpt'; +import { createExcerpt } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; const PAGE_SIZE = 20; @@ -73,7 +73,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) => year: 'numeric', timeZone: 'UTC' }), - excerpt: createExcerptHtml(post.excerpt_source || '') + excerpt: createExcerpt(post.excerpt_source || '') }; }); diff --git a/src/routes/tags/[tag]/+page.svelte b/src/routes/tags/[tag]/+page.svelte index 4d41671..f3686fc 100644 --- a/src/routes/tags/[tag]/+page.svelte +++ b/src/routes/tags/[tag]/+page.svelte @@ -37,8 +37,7 @@ {/if}

{post.title}

- -

{@html post.excerpt}

+

{post.excerpt}