From 3831171c06d154fa7d4c444302110ed787baeb32 Mon Sep 17 00:00:00 2001 From: seaHi Date: Wed, 16 Sep 2026 21:22:04 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E6=94=AF=E6=8C=81=E5=9C=A8=E6=96=87?= =?UTF-8?q?=E7=AB=A0=E6=91=98=E8=A6=81=E4=B8=AD=E4=BF=9D=E7=95=99=E5=B9=B6?= =?UTF-8?q?=E6=AD=A3=E7=A1=AE=E6=B8=B2=E6=9F=93=E8=A1=8C=E5=86=85=E6=A0=B7?= =?UTF-8?q?=E5=BC=8F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/lib/markdown/excerpt.test.ts | 63 +++++++++ src/lib/markdown/excerpt.ts | 129 +++++++++++++++++++ src/routes/+page.server.ts | 23 +--- src/routes/+page.svelte | 3 +- src/routes/categories/[slug]/+page.server.ts | 16 +-- src/routes/categories/[slug]/+page.svelte | 3 +- src/routes/tags/[tag]/+page.server.ts | 16 +-- src/routes/tags/[tag]/+page.svelte | 3 +- 8 files changed, 207 insertions(+), 49 deletions(-) create mode 100644 src/lib/markdown/excerpt.test.ts create mode 100644 src/lib/markdown/excerpt.ts diff --git a/src/lib/markdown/excerpt.test.ts b/src/lib/markdown/excerpt.test.ts new file mode 100644 index 0000000..3edfedf --- /dev/null +++ b/src/lib/markdown/excerpt.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from 'vitest'; +import { createExcerptHtml } from './excerpt'; + +describe('createExcerptHtml', () => { + it('renders inline styles instead of stripping them', () => { + expect(createExcerptHtml('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe( + '这是粗体与斜体,还有删除线。...' + ); + }); + + it('keeps inline code and drops the backticks', () => { + expect(createExcerptHtml('先执行 `pnpm run build` 再发布')).toBe( + '先执行 pnpm run build 再发布...' + ); + }); + + it('reduces links and images to readable text', () => { + expect(createExcerptHtml('参考[官方文档](https://example.com)与![配图](cover.png)说明')).toBe( + '参考官方文档与配图说明...' + ); + }); + + it('removes block level markers so the excerpt stays a single line', () => { + expect(createExcerptHtml('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe( + '标题 第一项 第二项 引用内容...' + ); + }); + + it('drops Hugo shortcodes and raw HTML', () => { + expect( + createExcerptHtml('{{< admonition note "提示" >}}正文{{< /admonition >}}片段') + ).toBe('正文片段...'); + }); + + it('closes inline tags that are cut off by the length limit', () => { + const html = createExcerptHtml(`**${'a'.repeat(200)}**`, 10); + + expect(html).toBe(`${'a'.repeat(10)}...`); + }); + + it('counts HTML entities as a single visible character', () => { + expect(createExcerptHtml('&'.repeat(30), 5)).toBe(`${'&'.repeat(5)}...`); + }); + + it('never keeps a dangling inline tag when the cut lands on it', () => { + const html = createExcerptHtml(`${'a'.repeat(9)}**bc**`, 10); + + expect(html).toBe(`aaaaaaaaab...`); + }); + + it('escapes characters that would otherwise break the markup', () => { + const html = createExcerptHtml('对比 1 < 2 的结果 '); + + expect(html).toContain('1 < 2'); + expect(html).not.toContain(' { + expect(createExcerptHtml('')).toBe(''); + expect(createExcerptHtml(' \n\n ')).toBe(''); + expect(createExcerptHtml('{{< admonition note >}}{{< /admonition >}}')).toBe(''); + }); +}); diff --git a/src/lib/markdown/excerpt.ts b/src/lib/markdown/excerpt.ts new file mode 100644 index 0000000..3e98f1d --- /dev/null +++ b/src/lib/markdown/excerpt.ts @@ -0,0 +1,129 @@ +import { Marked } from 'marked'; +import sanitizeHtml from 'sanitize-html'; + +/** 摘要默认保留的可见字符数 */ +export const EXCERPT_LENGTH = 150; + +/** 摘要中允许保留的行内样式标签(粗体、斜体、删除线、行内代码) */ +export const EXCERPT_INLINE_TAGS = ['strong', 'em', 'del', 'code']; + +const excerptMarked = new Marked(); + +/** 仅匹配摘要白名单内的行内标签,用于截断时配对 */ +const INLINE_TAG_PATTERN = new RegExp( + `^<\\s*(/?)\\s*(${EXCERPT_INLINE_TAGS.join('|')})\\s*>$`, + 'i' +); + +/** 匹配没有任何内容的行内标签,例如 `` */ +const EMPTY_TAG_PATTERN = new RegExp(`<(${EXCERPT_INLINE_TAGS.join('|')})>\\s*`, 'gi'); + +/** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */ +const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g; + +/** 匹配 HTML 实体,截断时按一个可见字符计算 */ +const HTML_ENTITY_PATTERN = /^&(?:#\d+|#x[0-9a-fA-F]+|[a-zA-Z]+);/; + +/** + * 把文章内容整理成适合做摘要的单行文本: + * 去掉 shortcode / HTML / 块级 Markdown 标记,保留粗体、斜体等行内标记。 + */ +function normalizeExcerptSource(source: string): string { + return source + .replace(/\{\{[<%][\s\S]*?[>%]\}\}/g, '') // Hugo / Docsy shortcode + .replace(//g, '') // HTML 注释 + .replace(HTML_TAG_PATTERN, '') // HTML 标签 + .replace(/!\[([^\]]*)\]\([^)]*\)/g, '$1') // 图片 → alt 文本 + .replace(/\[([^\]]+)\]\([^)]*\)/g, '$1') // 链接 → 链接文本 + .replace(/^[ \t]{0,3}(?:`{3,}|~{3,})[^\n]*$/gm, '') // 代码块围栏 + .replace(/^[ \t]{0,3}#{1,6}[ \t]+/gm, '') // ATX 标题 + .replace(/^[ \t]{0,3}>[ \t]?/gm, '') // 引用 + .replace(/^[ \t]{0,3}(?:[-*+]|\d{1,9}[.)])[ \t]+/gm, '') // 列表标记 + .replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行 + .replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线 + .replace(/\|/g, ' ') // 表格竖线 + .replace(/\s+/g, ' ') + .trim(); +} + +/** + * 按可见字符数截断行内 HTML,并补齐被截断的标签。 + * 只接受白名单内的行内标签,遇到其它标签立即停止,保证输出安全。 + */ +function truncateInlineHtml(html: string, maxLength: number): string { + let result = ''; + let visible = 0; + const openTags: string[] = []; + + for (let index = 0; index < html.length && visible < maxLength;) { + const char = html[index]; + + if (char === '<') { + const end = html.indexOf('>', index); + if (end === -1) break; + + const tag = html.slice(index, end + 1); + const match = INLINE_TAG_PATTERN.exec(tag); + if (!match) break; + + const name = match[2].toLowerCase(); + if (match[1]) { + const position = openTags.lastIndexOf(name); + if (position !== -1) openTags.splice(position, 1); + } else { + openTags.push(name); + } + + result += tag; + index = end + 1; + continue; + } + + if (char === '&') { + const entity = HTML_ENTITY_PATTERN.exec(html.slice(index)); + if (entity) { + result += entity[0]; + visible += 1; + index += entity[0].length; + continue; + } + } + + result += char; + visible += 1; + index += 1; + } + + for (const name of openTags.reverse()) { + result += ``; + } + + let cleaned = result; + for (;;) { + const next = cleaned.replace(EMPTY_TAG_PATTERN, '').replace(/\s+$/, ''); + if (next === cleaned) break; + cleaned = next; + } + + return cleaned; +} + +/** + * 由文章内容生成带行内样式的摘要 HTML(粗体、斜体、删除线、行内代码)。 + * 返回值已做白名单净化,可安全地用 `{@html}` 渲染;无内容时返回空字符串。 + */ +export function createExcerptHtml(source: string, maxLength: number = EXCERPT_LENGTH): string { + const normalized = normalizeExcerptSource(source || ''); + if (!normalized) return ''; + + // 先解析再截断:源文本必须先完整解析,否则成对的行内标记会被切断而失效 + const html = excerptMarked.parseInline(normalized) as string; + const sanitized = sanitizeHtml(html, { + allowedTags: EXCERPT_INLINE_TAGS, + allowedAttributes: {}, + disallowedTagsMode: 'discard' + }); + + const truncated = truncateInlineHtml(sanitized, maxLength).trim(); + return truncated ? `${truncated}...` : ''; +} diff --git a/src/routes/+page.server.ts b/src/routes/+page.server.ts index 301cfcb..092034c 100644 --- a/src/routes/+page.server.ts +++ b/src/routes/+page.server.ts @@ -1,5 +1,6 @@ /* eslint-disable @typescript-eslint/no-explicit-any */ +import { createExcerptHtml } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; export const load: PageServerLoad = async ({ platform }) => { @@ -47,25 +48,11 @@ export const load: PageServerLoad = async ({ platform }) => { const day = d.getDate(); const year = d.getFullYear(); - let excerpt; - let contentStr = (post.content as string) || ''; + // Prefer the manually marked summary, otherwise derive it from the beginning + const contentStr = ((post.content as string) || '').split('')[0]; - // Filter out Hugo shortcodes like {{< admonition abstract "标题" >}} - contentStr = contentStr.replace(/\{\{<[\s\S]*?>\}\}/g, ''); - - const parts = contentStr.split(''); - if (parts.length > 1) { - excerpt = parts[0].trim(); - } else { - // Strip basic markdown syntax for the automatic excerpt - const plainText = contentStr.replace(/[#*`>~[\]()-]/g, '').trim(); - excerpt = plainText.substring(0, 150).trim(); - } - - // Append ... to indicate there is more to read - if (excerpt && !excerpt.endsWith('...')) { - excerpt += '...'; - } + // Inline styles (bold, italic, ...) are kept as safe HTML for the cards + const excerpt = createExcerptHtml(contentStr); // Do not send full content to the list view to save bandwidth diff --git a/src/routes/+page.svelte b/src/routes/+page.svelte index 4bfc3f3..5049c9f 100644 --- a/src/routes/+page.svelte +++ b/src/routes/+page.svelte @@ -24,7 +24,8 @@ {/if}

{post.title}

-

{post.excerpt}

+ +

{@html post.excerpt}

diff --git a/src/routes/categories/[slug]/+page.server.ts b/src/routes/categories/[slug]/+page.server.ts index 984bd42..a45a068 100644 --- a/src/routes/categories/[slug]/+page.server.ts +++ b/src/routes/categories/[slug]/+page.server.ts @@ -1,4 +1,5 @@ import { error, isHttpError } from '@sveltejs/kit'; +import { createExcerptHtml } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; const PAGE_SIZE = 20; @@ -12,19 +13,6 @@ type CategoryPostRow = { excerpt_source: string; }; -function createExcerpt(source: string): string { - const plainText = source - .replace(/\{\{<[\s\S]*?>\}\}/g, '') - .replace(/<[^>]+>/g, '') - .replace(/[#*`>~[\]()-]/g, '') - .replace(/\s+/g, ' ') - .trim() - .slice(0, 150) - .trim(); - - return plainText ? `${plainText}...` : ''; -} - export const load: PageServerLoad = async ({ platform, params, url, parent }) => { const db = platform?.env.DB; const { aliases } = await parent(); @@ -84,7 +72,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) => year: 'numeric', timeZone: 'UTC' }), - excerpt: createExcerpt(post.excerpt_source || '') + excerpt: createExcerptHtml(post.excerpt_source || '') }; }); diff --git a/src/routes/categories/[slug]/+page.svelte b/src/routes/categories/[slug]/+page.svelte index 85f7be4..01427c7 100644 --- a/src/routes/categories/[slug]/+page.svelte +++ b/src/routes/categories/[slug]/+page.svelte @@ -38,7 +38,8 @@ {/if}

{post.title}

-

{post.excerpt}

+ +

{@html post.excerpt}

diff --git a/src/routes/tags/[tag]/+page.server.ts b/src/routes/tags/[tag]/+page.server.ts index a981d02..38649a8 100644 --- a/src/routes/tags/[tag]/+page.server.ts +++ b/src/routes/tags/[tag]/+page.server.ts @@ -1,4 +1,5 @@ import { error, isHttpError } from '@sveltejs/kit'; +import { createExcerptHtml } from '$lib/markdown/excerpt'; import type { PageServerLoad } from './$types'; const PAGE_SIZE = 20; @@ -12,19 +13,6 @@ type TagPostRow = { excerpt_source: string; }; -function createExcerpt(source: string): string { - const plainText = source - .replace(/\{\{<[\s\S]*?>\}\}/g, '') - .replace(/<[^>]+>/g, '') - .replace(/[#*`>~[\]()-]/g, '') - .replace(/\s+/g, ' ') - .trim() - .slice(0, 150) - .trim(); - - return plainText ? `${plainText}...` : ''; -} - export const load: PageServerLoad = async ({ platform, params, url, parent }) => { const db = platform?.env.DB; const { aliases } = await parent(); @@ -85,7 +73,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) => year: 'numeric', timeZone: 'UTC' }), - excerpt: createExcerpt(post.excerpt_source || '') + excerpt: createExcerptHtml(post.excerpt_source || '') }; }); diff --git a/src/routes/tags/[tag]/+page.svelte b/src/routes/tags/[tag]/+page.svelte index f3686fc..4d41671 100644 --- a/src/routes/tags/[tag]/+page.svelte +++ b/src/routes/tags/[tag]/+page.svelte @@ -37,7 +37,8 @@ {/if}

{post.title}

-

{post.excerpt}

+ +

{@html post.excerpt}