This commit is contained in:
1 parent
3831171c06
commit
b68151a179
8 files changed
+47
-147
No files matched your search
@@ -1,63 +1,57 @@
|
|||||||
import { describe, expect, it } from 'vitest';
|
import { describe, expect, it } from 'vitest';
|
||||||
import { createExcerptHtml } from './excerpt';
|
import { createExcerpt } from './excerpt';
|
||||||
|
|
||||||
describe('createExcerptHtml', () => {
|
describe('createExcerpt', () => {
|
||||||
it('renders inline styles instead of stripping them', () => {
|
it('strips inline markers instead of rendering them', () => {
|
||||||
expect(createExcerptHtml('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe(
|
expect(createExcerpt('这是**粗体**与*斜体*,还有~~删除线~~。')).toBe(
|
||||||
'这是<strong>粗体</strong>与<em>斜体</em>,还有<del>删除线</del>。...'
|
'这是粗体与斜体,还有删除线。'
|
||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it('keeps inline code and drops the backticks', () => {
|
it('drops the backticks of inline code', () => {
|
||||||
expect(createExcerptHtml('先执行 `pnpm run build` 再发布')).toBe(
|
expect(createExcerpt('先执行 `pnpm run build` 再发布')).toBe('先执行 pnpm run build 再发布');
|
||||||
'先执行 <code>pnpm run build</code> 再发布...'
|
|
||||||
);
|
|
||||||
});
|
});
|
||||||
|
|
||||||
it('reduces links and images to readable text', () => {
|
it('reduces links and images to readable text', () => {
|
||||||
expect(createExcerptHtml('参考[官方文档](https://example.com)与说明')).toBe(
|
expect(createExcerpt('参考[官方文档](https://example.com)与说明')).toBe(
|
||||||
'参考官方文档与配图说明...'
|
'参考官方文档与配图说明'
|
||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it('removes block level markers so the excerpt stays a single line', () => {
|
it('removes block level markers so the excerpt stays a single line', () => {
|
||||||
expect(createExcerptHtml('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe(
|
expect(createExcerpt('# 标题\n\n- 第一项\n- 第二项\n\n> 引用内容')).toBe(
|
||||||
'标题 第一项 第二项 引用内容...'
|
'标题 第一项 第二项 引用内容'
|
||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
it('drops Hugo shortcodes and raw HTML', () => {
|
it('drops Hugo shortcodes and raw HTML', () => {
|
||||||
expect(
|
expect(
|
||||||
createExcerptHtml('{{< admonition note "提示" >}}正文{{< /admonition >}}<span>片段</span>')
|
createExcerpt('{{< admonition note "提示" >}}正文{{< /admonition >}}<span>片段</span>')
|
||||||
).toBe('正文片段...');
|
).toBe('正文片段');
|
||||||
});
|
});
|
||||||
|
|
||||||
it('closes inline tags that are cut off by the length limit', () => {
|
it('truncates to the length limit without appending an ellipsis', () => {
|
||||||
const html = createExcerptHtml(`**${'a'.repeat(200)}**`, 10);
|
const excerpt = createExcerpt('a'.repeat(200), 10);
|
||||||
|
|
||||||
expect(html).toBe(`<strong>${'a'.repeat(10)}</strong>...`);
|
expect(excerpt).toBe('a'.repeat(10));
|
||||||
|
expect(excerpt).not.toContain('...');
|
||||||
});
|
});
|
||||||
|
|
||||||
it('counts HTML entities as a single visible character', () => {
|
it('never returns markdown markers or markup', () => {
|
||||||
expect(createExcerptHtml('&'.repeat(30), 5)).toBe(`${'&'.repeat(5)}...`);
|
const excerpt = createExcerpt('**粗体** 和 `code` 以及 <b>html</b>');
|
||||||
|
|
||||||
|
expect(excerpt).toBe('粗体 和 code 以及 html');
|
||||||
|
expect(excerpt).not.toMatch(/[*`~<>]/);
|
||||||
});
|
});
|
||||||
|
|
||||||
it('never keeps a dangling inline tag when the cut lands on it', () => {
|
it('keeps plain text that only looks like markup', () => {
|
||||||
const html = createExcerptHtml(`${'a'.repeat(9)}**bc**`, 10);
|
expect(createExcerpt('对比 1 < 2 的结果是安全的')).toBe('对比 1 < 2 的结果是安全的');
|
||||||
|
|
||||||
expect(html).toBe(`aaaaaaaaa<strong>b</strong>...`);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('escapes characters that would otherwise break the markup', () => {
|
|
||||||
const html = createExcerptHtml('对比 1 < 2 的结果 <script>alert(1)</script>');
|
|
||||||
|
|
||||||
expect(html).toContain('1 < 2');
|
|
||||||
expect(html).not.toContain('<script');
|
|
||||||
});
|
});
|
||||||
|
|
||||||
it('returns an empty string when there is nothing to show', () => {
|
it('returns an empty string when there is nothing to show', () => {
|
||||||
expect(createExcerptHtml('')).toBe('');
|
expect(createExcerpt('')).toBe('');
|
||||||
expect(createExcerptHtml(' \n\n ')).toBe('');
|
expect(createExcerpt(' \n\n ')).toBe('');
|
||||||
expect(createExcerptHtml('{{< admonition note >}}{{< /admonition >}}')).toBe('');
|
expect(createExcerpt('{{< admonition note >}}{{< /admonition >}}')).toBe('');
|
||||||
|
expect(createExcerpt('***')).toBe('');
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
+9
-100
@@ -1,32 +1,12 @@
|
|||||||
import { Marked } from 'marked';
|
/** 摘要默认保留的字符数 */
|
||||||
import sanitizeHtml from 'sanitize-html';
|
|
||||||
|
|
||||||
/** 摘要默认保留的可见字符数 */
|
|
||||||
export const EXCERPT_LENGTH = 150;
|
export const EXCERPT_LENGTH = 150;
|
||||||
|
|
||||||
/** 摘要中允许保留的行内样式标签(粗体、斜体、删除线、行内代码) */
|
|
||||||
export const EXCERPT_INLINE_TAGS = ['strong', 'em', 'del', 'code'];
|
|
||||||
|
|
||||||
const excerptMarked = new Marked();
|
|
||||||
|
|
||||||
/** 仅匹配摘要白名单内的行内标签,用于截断时配对 */
|
|
||||||
const INLINE_TAG_PATTERN = new RegExp(
|
|
||||||
`^<\\s*(/?)\\s*(${EXCERPT_INLINE_TAGS.join('|')})\\s*>$`,
|
|
||||||
'i'
|
|
||||||
);
|
|
||||||
|
|
||||||
/** 匹配没有任何内容的行内标签,例如 `<strong></strong>` */
|
|
||||||
const EMPTY_TAG_PATTERN = new RegExp(`<(${EXCERPT_INLINE_TAGS.join('|')})>\\s*</\\1>`, 'gi');
|
|
||||||
|
|
||||||
/** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */
|
/** 匹配 HTML 标签,要求以字母开头,避免误伤 `1 < 2` 这类正文 */
|
||||||
const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g;
|
const HTML_TAG_PATTERN = /<\/?[a-zA-Z][^>]*>/g;
|
||||||
|
|
||||||
/** 匹配 HTML 实体,截断时按一个可见字符计算 */
|
|
||||||
const HTML_ENTITY_PATTERN = /^&(?:#\d+|#x[0-9a-fA-F]+|[a-zA-Z]+);/;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 把文章内容整理成适合做摘要的单行文本:
|
* 把文章内容整理成适合做摘要的单行纯文本:
|
||||||
* 去掉 shortcode / HTML / 块级 Markdown 标记,保留粗体、斜体等行内标记。
|
* 去掉 shortcode、HTML 与 Markdown 标记,只保留正文文字。
|
||||||
*/
|
*/
|
||||||
function normalizeExcerptSource(source: string): string {
|
function normalizeExcerptSource(source: string): string {
|
||||||
return source
|
return source
|
||||||
@@ -42,88 +22,17 @@ function normalizeExcerptSource(source: string): string {
|
|||||||
.replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行
|
.replace(/^[ \t]{0,3}\|?[ \t]*:?-{3,}:?[ \t]*(?:\|[ \t]*:?-{3,}:?[ \t]*)*\|?[ \t]*$/gm, '') // 表格分隔行
|
||||||
.replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线
|
.replace(/^[ \t]{0,3}(?:[*_-][ \t]*){3,}$/gm, '') // 分隔线
|
||||||
.replace(/\|/g, ' ') // 表格竖线
|
.replace(/\|/g, ' ') // 表格竖线
|
||||||
|
.replace(/[*`~]/g, '') // 行内标记:粗体、斜体、删除线、行内代码
|
||||||
.replace(/\s+/g, ' ')
|
.replace(/\s+/g, ' ')
|
||||||
.trim();
|
.trim();
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 按可见字符数截断行内 HTML,并补齐被截断的标签。
|
* 由文章内容生成纯文本摘要:不保留任何 Markdown 标记,也不追加省略号。
|
||||||
* 只接受白名单内的行内标签,遇到其它标签立即停止,保证输出安全。
|
* 无内容时返回空字符串。
|
||||||
*/
|
*/
|
||||||
function truncateInlineHtml(html: string, maxLength: number): string {
|
export function createExcerpt(source: string, maxLength: number = EXCERPT_LENGTH): string {
|
||||||
let result = '';
|
const plainText = normalizeExcerptSource(source || '');
|
||||||
let visible = 0;
|
|
||||||
const openTags: string[] = [];
|
|
||||||
|
|
||||||
for (let index = 0; index < html.length && visible < maxLength;) {
|
return plainText.slice(0, maxLength).trim();
|
||||||
const char = html[index];
|
|
||||||
|
|
||||||
if (char === '<') {
|
|
||||||
const end = html.indexOf('>', index);
|
|
||||||
if (end === -1) break;
|
|
||||||
|
|
||||||
const tag = html.slice(index, end + 1);
|
|
||||||
const match = INLINE_TAG_PATTERN.exec(tag);
|
|
||||||
if (!match) break;
|
|
||||||
|
|
||||||
const name = match[2].toLowerCase();
|
|
||||||
if (match[1]) {
|
|
||||||
const position = openTags.lastIndexOf(name);
|
|
||||||
if (position !== -1) openTags.splice(position, 1);
|
|
||||||
} else {
|
|
||||||
openTags.push(name);
|
|
||||||
}
|
|
||||||
|
|
||||||
result += tag;
|
|
||||||
index = end + 1;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (char === '&') {
|
|
||||||
const entity = HTML_ENTITY_PATTERN.exec(html.slice(index));
|
|
||||||
if (entity) {
|
|
||||||
result += entity[0];
|
|
||||||
visible += 1;
|
|
||||||
index += entity[0].length;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
result += char;
|
|
||||||
visible += 1;
|
|
||||||
index += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
for (const name of openTags.reverse()) {
|
|
||||||
result += `</${name}>`;
|
|
||||||
}
|
|
||||||
|
|
||||||
let cleaned = result;
|
|
||||||
for (;;) {
|
|
||||||
const next = cleaned.replace(EMPTY_TAG_PATTERN, '').replace(/\s+$/, '');
|
|
||||||
if (next === cleaned) break;
|
|
||||||
cleaned = next;
|
|
||||||
}
|
|
||||||
|
|
||||||
return cleaned;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* 由文章内容生成带行内样式的摘要 HTML(粗体、斜体、删除线、行内代码)。
|
|
||||||
* 返回值已做白名单净化,可安全地用 `{@html}` 渲染;无内容时返回空字符串。
|
|
||||||
*/
|
|
||||||
export function createExcerptHtml(source: string, maxLength: number = EXCERPT_LENGTH): string {
|
|
||||||
const normalized = normalizeExcerptSource(source || '');
|
|
||||||
if (!normalized) return '';
|
|
||||||
|
|
||||||
// 先解析再截断:源文本必须先完整解析,否则成对的行内标记会被切断而失效
|
|
||||||
const html = excerptMarked.parseInline(normalized) as string;
|
|
||||||
const sanitized = sanitizeHtml(html, {
|
|
||||||
allowedTags: EXCERPT_INLINE_TAGS,
|
|
||||||
allowedAttributes: {},
|
|
||||||
disallowedTagsMode: 'discard'
|
|
||||||
});
|
|
||||||
|
|
||||||
const truncated = truncateInlineHtml(sanitized, maxLength).trim();
|
|
||||||
return truncated ? `${truncated}...` : '';
|
|
||||||
}
|
}
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
/* eslint-disable @typescript-eslint/no-explicit-any */
|
/* eslint-disable @typescript-eslint/no-explicit-any */
|
||||||
|
|
||||||
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
import { createExcerpt } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
export const load: PageServerLoad = async ({ platform }) => {
|
export const load: PageServerLoad = async ({ platform }) => {
|
||||||
@@ -51,8 +51,8 @@ export const load: PageServerLoad = async ({ platform }) => {
|
|||||||
// Prefer the manually marked summary, otherwise derive it from the beginning
|
// Prefer the manually marked summary, otherwise derive it from the beginning
|
||||||
const contentStr = ((post.content as string) || '').split('<!--more-->')[0];
|
const contentStr = ((post.content as string) || '').split('<!--more-->')[0];
|
||||||
|
|
||||||
// Inline styles (bold, italic, ...) are kept as safe HTML for the cards
|
// The excerpt is plain text: no Markdown markers and no trailing ellipsis
|
||||||
const excerpt = createExcerptHtml(contentStr);
|
const excerpt = createExcerpt(contentStr);
|
||||||
|
|
||||||
// Do not send full content to the list view to save bandwidth
|
// Do not send full content to the list view to save bandwidth
|
||||||
|
|
||||||
|
|||||||
@@ -24,8 +24,7 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
<p class="excerpt">{post.excerpt}</p>
|
||||||
<p class="excerpt">{@html post.excerpt}</p>
|
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
import { error, isHttpError } from '@sveltejs/kit';
|
import { error, isHttpError } from '@sveltejs/kit';
|
||||||
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
import { createExcerpt } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
const PAGE_SIZE = 20;
|
const PAGE_SIZE = 20;
|
||||||
@@ -72,7 +72,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
|
|||||||
year: 'numeric',
|
year: 'numeric',
|
||||||
timeZone: 'UTC'
|
timeZone: 'UTC'
|
||||||
}),
|
}),
|
||||||
excerpt: createExcerptHtml(post.excerpt_source || '')
|
excerpt: createExcerpt(post.excerpt_source || '')
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -38,8 +38,7 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
<p class="excerpt">{post.excerpt}</p>
|
||||||
<p class="excerpt">{@html post.excerpt}</p>
|
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
import { error, isHttpError } from '@sveltejs/kit';
|
import { error, isHttpError } from '@sveltejs/kit';
|
||||||
import { createExcerptHtml } from '$lib/markdown/excerpt';
|
import { createExcerpt } from '$lib/markdown/excerpt';
|
||||||
import type { PageServerLoad } from './$types';
|
import type { PageServerLoad } from './$types';
|
||||||
|
|
||||||
const PAGE_SIZE = 20;
|
const PAGE_SIZE = 20;
|
||||||
@@ -73,7 +73,7 @@ export const load: PageServerLoad = async ({ platform, params, url, parent }) =>
|
|||||||
year: 'numeric',
|
year: 'numeric',
|
||||||
timeZone: 'UTC'
|
timeZone: 'UTC'
|
||||||
}),
|
}),
|
||||||
excerpt: createExcerptHtml(post.excerpt_source || '')
|
excerpt: createExcerpt(post.excerpt_source || '')
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -37,8 +37,7 @@
|
|||||||
{/if}
|
{/if}
|
||||||
<div class="content-wrap">
|
<div class="content-wrap">
|
||||||
<h1 class="title">{post.title}</h1>
|
<h1 class="title">{post.title}</h1>
|
||||||
<!-- eslint-disable-next-line svelte/no-at-html-tags -->
|
<p class="excerpt">{post.excerpt}</p>
|
||||||
<p class="excerpt">{@html post.excerpt}</p>
|
|
||||||
</div>
|
</div>
|
||||||
</article>
|
</article>
|
||||||
</a>
|
</a>
|
||||||
|
|||||||
Reference in new issue
Block a user