This commit is contained in:
1 parent
22bdfc259f
commit
3831171c06
8 files changed
+207
-49
No files matched your search
@@ -0,0 +1,63 @@
|
|||||||
|
import { describe, expect, it } from 'vitest';
|
||||||
|
import { createExcerptHtml } from './excerpt';
|
||||||
|
|
||||||
|
describe('createExcerptHtml', () => {
|
||||||
|
it('renders inline styles instead of stripping them', () => {
|
||||||
|
expect(createExcerptHtml('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe(
|
||||||
|
'这是<strong>粗体</strong>与<em>斜体</em>,还有<del>删除线</del>。...'
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('keeps inline code and drops the backticks', () => {
|
||||||
|
expect(createExcerptHtml('先执行 `pnpm run build` 再发布')).toBe(
|
||||||
|
'先执行 <code>pnpm run build</code> 再发布...'
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('reduces links and images to readable text', () => {
|
||||||
|
expect(createExcerptHtml('参考[官方文档](https://example.com)与说明')).toBe(
|
||||||
|
'参考官方文档与配图说明...'
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('removes block level markers so the excerpt stays a single line', () => {
|
||||||
|
expect(createExcerptHtml('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe(
|
||||||
|
'标题 第一项 第二项 引用内容...'
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('drops Hugo shortcodes and raw HTML', () => {
|
||||||
|
expect(
|
||||||
|
createExcerptHtml('{{< admonition note "提示" >}}正文{{< /admonition >}}<span>片段</span>')
|
||||||
|
).toBe('正文片段...');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('closes inline tags that are cut off by the length limit', () => {
|
||||||
|
const html = createExcerptHtml(`**${'a'.repeat(200)}**`, 10);
|
||||||
|
|
||||||
|
expect(html).toBe(`<strong>${'a'.repeat(10)}</strong>...`);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('counts HTML entities as a single visible character', () => {
|
||||||
|
expect(createExcerptHtml('&'.repeat(30), 5)).toBe(`${'&'.repeat(5)}...`);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('never keeps a dangling inline tag when the cut lands on it', () => {
|
||||||
|
const html = createExcerptHtml(`${'a'.repeat(9)}**bc**`, 10);
|
||||||
|
|
||||||
|
expect(html).toBe(`aaaaaaaaa<strong>b</strong>...`);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('escapes characters that would otherwise break the markup', () => {
|
||||||
|
const html = createExcerptHtml('对比 1 < 2 的结果 <script>alert(1)</script>');
|
||||||
|
|
||||||
|
expect(html).toContain('1 < 2');
|
||||||
|
expect(html).not.toContain('<script');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('returns an empty string when there is nothing to show', () => {
|
||||||
|
expect(createExcerptHtml('')).toBe('');
|
||||||
|
expect(createExcerptHtml(' \n\n ')).toBe('');
|
||||||
|
expect(createExcerptHtml('{{< admonition note >}}{{< /admonition >}}')).toBe('');
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -0,0 +1,129 @@
|
|||||||
|
import { Marked } from 'marked';
|
||||||
|
import sanitizeHtml from 'sanitize-html';
|
||||||
|
|
||||||
|
/** 摘要默认保留的可见字符数 */
|
||||||
|
export const EXCERPT_LENGTH = 150;
|
||||||
|
|
||||||
|
/** 摘要中允许保留的行内样式标签(粗体、斜体、删除线、行内代码) */
|
||||||
|
export const EXCERPT_INLINE_TAGS = ['strong', 'em', 'del', 'code'];
|
||||||
|
|
||||||
|
const excerptMarked = new Marked();
|
||||||
|
|
||||||
|
/** 仅匹配摘要白名单内的行内标签,用于截断时配对 */
|
||||||
|
const INLINE_TAG_PATTERN = new RegExp(
|
||||||
|
`^<\\s*(/?)\\s*(${EXCERPT_INLINE_TAGS.join('|')})\\s*>$`,
|
||||||
|
'i'
|
||||||
|
);
|
||||||
|
|
||||||
|
/** 匹配没有任何内容的行内标签,例如 `<strong></strong>` */
|
||||||
|
const EMPTY_TAG_PATTERN = new RegExp(`<(${EXCERPT_INLINE_TAGS.join('|')})>\\s*</\\1>`, 'gi');
|
||||||
|
|
||||||
|
/** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */
|
||||||
|
const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g;
|
||||||
|
|
||||||
|
/** 匹配 HTML 实体,截断时按一个可见字符计算 */
|
||||||
|
const HTML_ENTITY_PATTERN = /^&(?:#\d+|#x[0-9a-fA-F]+|[a-zA-Z]+);/;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 把文章内容整理成适合做摘要的单行文本:
|
||||||
|
* 去掉 shortcode / HTML / 块级 Markdown 标记,保留粗体、斜体等行内标记。
|
||||||
|
*/
|
||||||
|
function normalizeExcerptSource(source: string): string {
|
||||||
|
return source
|
||||||
|
.replace(/\{\{[<%][\s\S]*?[>%]\}\}/g, '') // Hugo / Docsy shortcode
|
||||||
|
.replace(/<!--[\s\S]*?-->/g, '') // HTML 注释
|
||||||
|
.replace(HTML_TAG_PATTERN, '') // HTML 标签
|
||||||
|
.replace(/!\[([^\]]*)\]\([^)]*\)/g, '$1') // 图片 → alt 文本
|
||||||
|
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1') // 链接 → 链接文本
|
||||||
|
.replace(/^[ \t]{0,3}(?:`{3,}|~{3,})[^\n]*$/gm, '') // 代码块围栏
|
||||||
|
.replace(/^[ \t]{0,3}#{1,6}[ \t]+/gm, '') // ATX 标题
|
||||||
|
.replace(/^[ \t]{0,3}>[ \t]?/gm, '') // 引用
|
||||||
|
.replace(/^[ \t]{0,3}(?:[-*+]|\d{1,9}[.)])[ \t]+/gm, '') // 列表标记
|
||||||
|
.replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行
|
||||||
|
.replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线
|
||||||
|
.replace(/\|/g, ' ') // 表格竖线
|
||||||
|
.replace(/\s+/g, ' ')
|
||||||
|
.trim();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 按可见字符数截断行内 HTML,并补齐被截断的标签。
|
||||||
|
* 只接受白名单内的行内标签,遇到其它标签立即停止,保证输出安全。
|
||||||
|
*/
|
||||||
|
function truncateInlineHtml(html: string, maxLength: number): string {
|
||||||
|
let result = '';
|
||||||
|
let visible = 0;
|
||||||
|
const openTags: string[] = [];
|
||||||
|
|
||||||
|
for (let index = 0; index < html.length && visible < maxLength;) {
|
||||||
|
const char = html[index];
|
||||||
|
|
||||||
|
if (char === '<') {
|
||||||
|
const end = html.indexOf('>', index);
|
||||||
|
if (end === -1) break;
|
||||||
|
|
||||||
|
const tag = html.slice(index, end + 1);
|
||||||
|
const match = INLINE_TAG_PATTERN.exec(tag);
|
||||||
|
if (!match) break;
|
||||||
|
|
||||||
|
const name = match[2].toLowerCase();
|
||||||
|
if (match[1]) {
|
||||||
|
const position = openTags.lastIndexOf(name);
|
||||||
|
if (position !== -1) openTags.splice(position, 1);
|
||||||
|
} else {
|
||||||
|
openTags.push(name);
|
||||||
|
}
|
||||||
|
|
||||||
|
result += tag;
|
||||||
|
index = end + 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (char === '&') {
|
||||||
|
const entity = HTML_ENTITY_PATTERN.exec(html.slice(index));
|
||||||
|
if (entity) {
|
||||||
|
result += entity[0];
|
||||||
|
visible += 1;
|
||||||
|
index += entity[0].length;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
result += char;
|
||||||
|
visible += 1;
|
||||||
|
index += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const name of openTags.reverse()) {
|
||||||
|
result += `</${name}>`;
|
||||||
|
}
|
||||||
|
|
||||||
|
let cleaned = result;
|
||||||
|
for (;;) {
|
||||||
|
const next = cleaned.replace(EMPTY_TAG_PATTERN, '').replace(/\s+$/, '');
|
||||||
|
if (next === cleaned) break;
|
||||||
|
cleaned = next;
|
||||||
|
}
|
||||||
|
|
||||||
|
return cleaned;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* 由文章内容生成带行内样式的摘要 HTML(粗体、斜体、删除线、行内代码)。
|
||||||
|
* 返回值已做白名单净化,可安全地用 `{@html}` 渲染;无内容时返回空字符串。
|
||||||
|
*/
|
||||||
|
export function createExcerptHtml(source: string, maxLength: number = EXCERPT_LENGTH): string {
|
||||||
|
const normalized = normalizeExcerptSource(source || '');
|
||||||
|
if (!normalized) return '';
|
||||||
|
|
||||||
|
// 先解析再截断:源文本必须先完整解析,否则成对的行内标记会被切断而失效
|
||||||
|
const html = excerptMarked.parseInline(normalized) as string;
|
||||||
|
const sanitized = sanitizeHtml(html, {
|
||||||
|
allowedTags: EXCERPT_INLINE_TAGS,
|
||||||
|
allowedAttributes: {},
|
||||||
|
disallowedTagsMode: 'discard'
|
||||||
|
});
|
||||||
|
|
||||||
|
const truncated = truncateInlineHtml(sanitized, maxLength).trim();
|
||||||
|
return truncated ? `${truncated}...` : '';
|
||||||
|
}
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
/* eslint-disable @typescript-eslint/no-explicit-any */
|
/* eslint-disable @typescript-eslint/no-explicit-any */
|
||||||
|
|
||||||
|
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
export const load: PageServerLoad = async ({ platform }) => {
|
export const load: PageServerLoad = async ({ platform }) => {
|
||||||
@@ -47,25 +48,11 @@ export const load: PageServerLoad = async ({ platform }) => {
|
|||||||
const day = d.getDate();
|
const day = d.getDate();
|
||||||
const year = d.getFullYear();
|
const year = d.getFullYear();
|
||||||
|
|
||||||
let excerpt;
|
// Prefer the manually marked summary, otherwise derive it from the beginning
|
||||||
let contentStr = (post.content as string) || '';
|
const contentStr = ((post.content as string) || '').split('<!--more-->')[0];
|
||||||
|
|
||||||
// Filter out Hugo shortcodes like {{< admonition abstract "标题" >}}
|
// Inline styles (bold, italic, ...) are kept as safe HTML for the cards
|
||||||
contentStr = contentStr.replace(/\{\{<[\s\S]*?>\}\}/g, '');
|
const excerpt = createExcerptHtml(contentStr);
|
||||||
|
|
||||||
const parts = contentStr.split('<!--more-->');
|
|
||||||
if (parts.length > 1) {
|
|
||||||
excerpt = parts[0].trim();
|
|
||||||
} else {
|
|
||||||
// Strip basic markdown syntax for the automatic excerpt
|
|
||||||
const plainText = contentStr.replace(/[#*`>~[\]()-]/g, '').trim();
|
|
||||||
excerpt = plainText.substring(0, 150).trim();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Append ... to indicate there is more to read
|
|
||||||
if (excerpt && !excerpt.endsWith('...')) {
|
|
||||||
excerpt += '...';
|
|
||||||
}
|
|
||||||
|
|
||||||
// Do not send full content to the list view to save bandwidth
|
// Do not send full content to the list view to save bandwidth
|
||||||
|
|
||||||
|
|||||||
@@ -24,7 +24,8 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<p class="excerpt">{post.excerpt}</p>
|
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
||||||
|
<p class="excerpt">{@html post.excerpt}</p>
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { error, isHttpError } from '@sveltejs/kit';
|
import { error, isHttpError } from '@sveltejs/kit';
|
||||||
|
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
const PAGE_SIZE = 20;
|
const PAGE_SIZE = 20;
|
||||||
@@ -12,19 +13,6 @@ type CategoryPostRow = {
|
|||||||
excerpt_source: string;
|
excerpt_source: string;
|
||||||
};
|
};
|
||||||
|
|
||||||
function createExcerpt(source: string): string {
|
|
||||||
const plainText = source
|
|
||||||
.replace(/\{\{<[\s\S]*?>\}\}/g, '')
|
|
||||||
.replace(/<[^>]+>/g, '')
|
|
||||||
.replace(/[#*`>~[\]()-]/g, '')
|
|
||||||
.replace(/\s+/g, ' ')
|
|
||||||
.trim()
|
|
||||||
.slice(0, 150)
|
|
||||||
.trim();
|
|
||||||
|
|
||||||
return plainText ? `${plainText}...` : '';
|
|
||||||
}
|
|
||||||
|
|
||||||
export const load: PageServerLoad = async ({ platform, params, url, parent }) => {
|
export const load: PageServerLoad = async ({ platform, params, url, parent }) => {
|
||||||
const db = platform?.env.DB;
|
const db = platform?.env.DB;
|
||||||
const { aliases } = await parent();
|
const { aliases } = await parent();
|
||||||
@@ -84,7 +72,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
|
|||||||
year: 'numeric',
|
year: 'numeric',
|
||||||
timeZone: 'UTC'
|
timeZone: 'UTC'
|
||||||
}),
|
}),
|
||||||
excerpt: createExcerpt(post.excerpt_source || '')
|
excerpt: createExcerptHtml(post.excerpt_source || '')
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -38,7 +38,8 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<p class="excerpt">{post.excerpt}</p>
|
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
||||||
|
<p class="excerpt">{@html post.excerpt}</p>
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { error, isHttpError } from '@sveltejs/kit';
|
import { error, isHttpError } from '@sveltejs/kit';
|
||||||
|
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
const PAGE_SIZE = 20;
|
const PAGE_SIZE = 20;
|
||||||
@@ -12,19 +13,6 @@ type TagPostRow = {
|
|||||||
excerpt_source: string;
|
excerpt_source: string;
|
||||||
};
|
};
|
||||||
|
|
||||||
function createExcerpt(source: string): string {
|
|
||||||
const plainText = source
|
|
||||||
.replace(/\{\{<[\s\S]*?>\}\}/g, '')
|
|
||||||
.replace(/<[^>]+>/g, '')
|
|
||||||
.replace(/[#*`>~[\]()-]/g, '')
|
|
||||||
.replace(/\s+/g, ' ')
|
|
||||||
.trim()
|
|
||||||
.slice(0, 150)
|
|
||||||
.trim();
|
|
||||||
|
|
||||||
return plainText ? `${plainText}...` : '';
|
|
||||||
}
|
|
||||||
|
|
||||||
export const load: PageServerLoad = async ({ platform, params, url, parent }) => {
|
export const load: PageServerLoad = async ({ platform, params, url, parent }) => {
|
||||||
const db = platform?.env.DB;
|
const db = platform?.env.DB;
|
||||||
const { aliases } = await parent();
|
const { aliases } = await parent();
|
||||||
@@ -85,7 +73,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
|
|||||||
year: 'numeric',
|
year: 'numeric',
|
||||||
timeZone: 'UTC'
|
timeZone: 'UTC'
|
||||||
}),
|
}),
|
||||||
excerpt: createExcerpt(post.excerpt_source || '')
|
excerpt: createExcerptHtml(post.excerpt_source || '')
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -37,7 +37,8 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<p class="excerpt">{post.excerpt}</p>
|
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
||||||
|
<p class="excerpt">{@html post.excerpt}</p>
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
Reference in new issue
Block a user