refactor: 将文章摘要生成简化为纯文本处理并移除 HTML 渲染
continuous-integration/drone/push Build is passing

This commit is contained in:
seahi committed 2026-09-16 21:32:47 +08:00
1 parent 3831171c06
commit b68151a179
8 files changed
+47 -147

No files matched your search

+28 -34
View File
@@ -1,63 +1,57 @@
import { describe, expect, it } from 'vitest'; import { describe, expect, it } from 'vitest';
import { createExcerptHtml } from './excerpt'; import { createExcerpt } from './excerpt';
describe('createExcerptHtml', () => { describe('createExcerpt', () => {
it('renders inline styles instead of stripping them', () => { it('strips inline markers instead of rendering them', () => {
expect(createExcerptHtml('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe( expect(createExcerpt('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe(
'这是<strong>粗体</strong>与<em>斜体</em>,还有<del>删除线</del>。...' '这是粗体与斜体,还有删除线。'
); );
}); });
it('keeps inline code and drops the backticks', () => { it('drops the backticks of inline code', () => {
expect(createExcerptHtml('先执行 `pnpm run build` 再发布')).toBe( expect(createExcerpt('先执行 `pnpm run build` 再发布')).toBe('先执行 pnpm run build 再发布');
'先执行 <code>pnpm run build</code> 再发布...'
);
}); });
it('reduces links and images to readable text', () => { it('reduces links and images to readable text', () => {
expect(createExcerptHtml('参考[官方文档](https://example.com)与![配图](cover.png)说明')).toBe( expect(createExcerpt('参考[官方文档](https://example.com)与![配图](cover.png)说明')).toBe(
'参考官方文档与配图说明...' '参考官方文档与配图说明'
); );
}); });
it('removes block level markers so the excerpt stays a single line', () => { it('removes block level markers so the excerpt stays a single line', () => {
expect(createExcerptHtml('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe( expect(createExcerpt('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe(
'标题 第一项 第二项 引用内容...' '标题 第一项 第二项 引用内容'
); );
}); });
it('drops Hugo shortcodes and raw HTML', () => { it('drops Hugo shortcodes and raw HTML', () => {
expect( expect(
createExcerptHtml('{{< admonition note "提示" >}}正文{{< /admonition >}}<span>片段</span>') createExcerpt('{{< admonition note "提示" >}}正文{{< /admonition >}}<span>片段</span>')
).toBe('正文片段...'); ).toBe('正文片段');
}); });
it('closes inline tags that are cut off by the length limit', () => { it('truncates to the length limit without appending an ellipsis', () => {
const html = createExcerptHtml(`**${'a'.repeat(200)}**`, 10); const excerpt = createExcerpt('a'.repeat(200), 10);
expect(html).toBe(`<strong>${'a'.repeat(10)}</strong>...`); expect(excerpt).toBe('a'.repeat(10));
expect(excerpt).not.toContain('...');
}); });
it('counts HTML entities as a single visible character', () => { it('never returns markdown markers or markup', () => {
expect(createExcerptHtml('&'.repeat(30), 5)).toBe(`${'&amp;'.repeat(5)}...`); const excerpt = createExcerpt('**粗体** 和 `code` 以及 <b>html</b>');
expect(excerpt).toBe('粗体 和 code 以及 html');
expect(excerpt).not.toMatch(/[*`~<>]/);
}); });
it('never keeps a dangling inline tag when the cut lands on it', () => { it('keeps plain text that only looks like markup', () => {
const html = createExcerptHtml(`${'a'.repeat(9)}**bc**`, 10); expect(createExcerpt('对比 1 < 2 的结果是安全的')).toBe('对比 1 < 2 的结果是安全的');
expect(html).toBe(`aaaaaaaaa<strong>b</strong>...`);
});
it('escapes characters that would otherwise break the markup', () => {
const html = createExcerptHtml('对比 1 < 2 的结果 <script>alert(1)</script>');
expect(html).toContain('1 &lt; 2');
expect(html).not.toContain('<script');
}); });
it('returns an empty string when there is nothing to show', () => { it('returns an empty string when there is nothing to show', () => {
expect(createExcerptHtml('')).toBe(''); expect(createExcerpt('')).toBe('');
expect(createExcerptHtml(' \n\n ')).toBe(''); expect(createExcerpt(' \n\n ')).toBe('');
expect(createExcerptHtml('{{< admonition note >}}{{< /admonition >}}')).toBe(''); expect(createExcerpt('{{< admonition note >}}{{< /admonition >}}')).toBe('');
expect(createExcerpt('***')).toBe('');
}); });
}); });
+9 -100
View File
@@ -1,32 +1,12 @@
import { Marked } from 'marked'; /** 摘要默认保留的字符数 */
import sanitizeHtml from 'sanitize-html';
/** 摘要默认保留的可见字符数 */
export const EXCERPT_LENGTH = 150; export const EXCERPT_LENGTH = 150;
/** 摘要中允许保留的行内样式标签(粗体、斜体、删除线、行内代码) */
export const EXCERPT_INLINE_TAGS = ['strong', 'em', 'del', 'code'];
const excerptMarked = new Marked();
/** 仅匹配摘要白名单内的行内标签,用于截断时配对 */
const INLINE_TAG_PATTERN = new RegExp(
`^<\\s*(/?)\\s*(${EXCERPT_INLINE_TAGS.join('|')})\\s*>$`,
'i'
);
/** 匹配没有任何内容的行内标签,例如 `<strong></strong>` */
const EMPTY_TAG_PATTERN = new RegExp(`<(${EXCERPT_INLINE_TAGS.join('|')})>\\s*</\\1>`, 'gi');
/** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */ /** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */
const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g; const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g;
/** 匹配 HTML 实体,截断时按一个可见字符计算 */
const HTML_ENTITY_PATTERN = /^&(?:#\d+|#x[0-9a-fA-F]+|[a-zA-Z]+);/;
/** /**
* 把文章内容整理成适合做摘要的单行文本: * 把文章内容整理成适合做摘要的单行纯文本:
* 去掉 shortcode / HTML / 块级 Markdown 标记,保留粗体、斜体等行内标记。 * 去掉 shortcode、HTML 与 Markdown 标记,只保留正文文字。
*/ */
function normalizeExcerptSource(source: string): string { function normalizeExcerptSource(source: string): string {
return source return source
@@ -42,88 +22,17 @@ function normalizeExcerptSource(source: string): string {
.replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行 .replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行
.replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线 .replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线
.replace(/\|/g, ' ') // 表格竖线 .replace(/\|/g, ' ') // 表格竖线
.replace(/[*`~]/g, '') // 行内标记:粗体、斜体、删除线、行内代码
.replace(/\s+/g, ' ') .replace(/\s+/g, ' ')
.trim(); .trim();
} }
/** /**
* 按可见字符数截断行内 HTML,并补齐被截断的标签。 * 由文章内容生成纯文本摘要:不保留任何 Markdown 标记,也不追加省略号。
* 只接受白名单内的行内标签,遇到其它标签立即停止,保证输出安全。 * 无内容时返回空字符串。
*/ */
function truncateInlineHtml(html: string, maxLength: number): string { export function createExcerpt(source: string, maxLength: number = EXCERPT_LENGTH): string {
let result = ''; const plainText = normalizeExcerptSource(source || '');
let visible = 0;
const openTags: string[] = [];
for (let index = 0; index < html.length && visible < maxLength;) { return plainText.slice(0, maxLength).trim();
const char = html[index];
if (char === '<') {
const end = html.indexOf('>', index);
if (end === -1) break;
const tag = html.slice(index, end + 1);
const match = INLINE_TAG_PATTERN.exec(tag);
if (!match) break;
const name = match[2].toLowerCase();
if (match[1]) {
const position = openTags.lastIndexOf(name);
if (position !== -1) openTags.splice(position, 1);
} else {
openTags.push(name);
}
result += tag;
index = end + 1;
continue;
}
if (char === '&') {
const entity = HTML_ENTITY_PATTERN.exec(html.slice(index));
if (entity) {
result += entity[0];
visible += 1;
index += entity[0].length;
continue;
}
}
result += char;
visible += 1;
index += 1;
}
for (const name of openTags.reverse()) {
result += `</${name}>`;
}
let cleaned = result;
for (;;) {
const next = cleaned.replace(EMPTY_TAG_PATTERN, '').replace(/\s+$/, '');
if (next === cleaned) break;
cleaned = next;
}
return cleaned;
}
/**
* 由文章内容生成带行内样式的摘要 HTML(粗体、斜体、删除线、行内代码)。
* 返回值已做白名单净化,可安全地用 `{@html}` 渲染;无内容时返回空字符串。
*/
export function createExcerptHtml(source: string, maxLength: number = EXCERPT_LENGTH): string {
const normalized = normalizeExcerptSource(source || '');
if (!normalized) return '';
// 先解析再截断:源文本必须先完整解析,否则成对的行内标记会被切断而失效
const html = excerptMarked.parseInline(normalized) as string;
const sanitized = sanitizeHtml(html, {
allowedTags: EXCERPT_INLINE_TAGS,
allowedAttributes: {},
disallowedTagsMode: 'discard'
});
const truncated = truncateInlineHtml(sanitized, maxLength).trim();
return truncated ? `${truncated}...` : '';
} }
+3 -3
View File
@@ -1,6 +1,6 @@
/* eslint-disable @typescript-eslint/no-explicit-any */ /* eslint-disable @typescript-eslint/no-explicit-any */
import { createExcerptHtml } from '$lib/markdown/excerpt'; import { createExcerpt } from '$lib/markdown/excerpt';
import type { PageServerLoad } from './$types'; import type { PageServerLoad } from './$types';
export const load: PageServerLoad = async ({ platform }) => { export const load: PageServerLoad = async ({ platform }) => {
@@ -51,8 +51,8 @@ export const load: PageServerLoad = async ({ platform }) => {
// Prefer the manually marked summary, otherwise derive it from the beginning // Prefer the manually marked summary, otherwise derive it from the beginning
const contentStr = ((post.content as string) || '').split('<!--more-->')[0]; const contentStr = ((post.content as string) || '').split('<!--more-->')[0];
// Inline styles (bold, italic, ...) are kept as safe HTML for the cards // The excerpt is plain text: no Markdown markers and no trailing ellipsis
const excerpt = createExcerptHtml(contentStr); const excerpt = createExcerpt(contentStr);
// Do not send full content to the list view to save bandwidth // Do not send full content to the list view to save bandwidth
+1 -2
View File
@@ -24,8 +24,7 @@
{/if} {/if}
<div class="content-wrap"> <div class="content-wrap">
<h1 class="title">{post.title}</h1> <h1 class="title">{post.title}</h1>
<!-- eslint-disable-next-line svelte/no-at-html-tags --> <p class="excerpt">{post.excerpt}</p>
<p class="excerpt">{@html post.excerpt}</p>
</div> </div>
</article> </article>
</a> </a>
+2 -2
View File
@@ -1,5 +1,5 @@
import { error, isHttpError } from '@sveltejs/kit'; import { error, isHttpError } from '@sveltejs/kit';
import { createExcerptHtml } from '$lib/markdown/excerpt'; import { createExcerpt } from '$lib/markdown/excerpt';
import type { PageServerLoad } from './$types'; import type { PageServerLoad } from './$types';
const PAGE_SIZE = 20; const PAGE_SIZE = 20;
@@ -72,7 +72,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
year: 'numeric', year: 'numeric',
timeZone: 'UTC' timeZone: 'UTC'
}), }),
excerpt: createExcerptHtml(post.excerpt_source || '') excerpt: createExcerpt(post.excerpt_source || '')
}; };
}); });
+1 -2
View File
@@ -38,8 +38,7 @@
{/if} {/if}
<div class="content-wrap"> <div class="content-wrap">
<h1 class="title">{post.title}</h1> <h1 class="title">{post.title}</h1>
<!-- eslint-disable-next-line svelte/no-at-html-tags --> <p class="excerpt">{post.excerpt}</p>
<p class="excerpt">{@html post.excerpt}</p>
</div> </div>
</article> </article>
</a> </a>
+2 -2
View File
@@ -1,5 +1,5 @@
import { error, isHttpError } from '@sveltejs/kit'; import { error, isHttpError } from '@sveltejs/kit';
import { createExcerptHtml } from '$lib/markdown/excerpt'; import { createExcerpt } from '$lib/markdown/excerpt';
import type { PageServerLoad } from './$types'; import type { PageServerLoad } from './$types';
const PAGE_SIZE = 20; const PAGE_SIZE = 20;
@@ -73,7 +73,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
year: 'numeric', year: 'numeric',
timeZone: 'UTC' timeZone: 'UTC'
}), }),
excerpt: createExcerptHtml(post.excerpt_source || '') excerpt: createExcerpt(post.excerpt_source || '')
}; };
}); });
+1 -2
View File
@@ -37,8 +37,7 @@
{/if} {/if}
<div class="content-wrap"> <div class="content-wrap">
<h1 class="title">{post.title}</h1> <h1 class="title">{post.title}</h1>
<!-- eslint-disable-next-line svelte/no-at-html-tags --> <p class="excerpt">{post.excerpt}</p>
<p class="excerpt">{@html post.excerpt}</p>
</div> </div>
</article> </article>
</a> </a>