| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225 |
- // lib/markdown.ts:由 lib/utils.ts 按职责拆分(实现未改)
- import { marked, Renderer, walkTokens } from 'marked';
- import type { Tokens } from 'marked';
- import katex from 'katex';
- import { sanitizeHtml } from '/@/utils/markdownSafe';
- const mdRenderer = new Renderer();
- const defaultLinkRenderer = mdRenderer.link.bind(mdRenderer);
- mdRenderer.link = (token: Tokens.Link): string => {
- const html = defaultLinkRenderer(token);
- return html.replace(/^<a\s/, '<a target="_blank" rel="noopener" ');
- };
- const renderLatexInHtml = (html: string): string => {
- // 匹配块级公式($$...$$)
- html = html.replace(/\$\$(.*?)\$\$/gs, (_match, formula) => {
- return katex.renderToString(formula.trim(), {
- displayMode: true,
- throwOnError: false,
- strict: false,
- trust: true,
- });
- });
- // 匹配行内公式($...$)
- html = html.replace(/\$(.*?)\$/g, (_match, formula) => {
- return katex.renderToString(formula.trim(), {
- displayMode: false,
- throwOnError: false,
- strict: false,
- });
- });
- return html;
- };
- const processFootnotes = (html: string, maxId?: number): string => {
- if (maxId === undefined) {
- // 无 citations 上下文,剥离所有 [^n] 标记
- return html.replace(/\[\^\d+\]/g, '');
- }
- // 有效编号范围 [1, maxId],范围外的剥离
- return html.replace(/\[\^(\d+)\]/g, (_match, n) => {
- const id = parseInt(n, 10);
- if (id >= 1 && id <= maxId) {
- return `<sup class="footnote-ref" data-n="${id}">[${n}]</sup>`;
- }
- return '';
- });
- };
- export const renderMarkdown = (
- text: string,
- icons?: { copy: string; download: string; preview: string; copySuccess: string },
- maxCitationId?: number
- ): string => {
- if (!text) return '';
- const processed = text.replace(/<br\s*\/?>/gi, ' \n');
- let html = marked.parse(processed, { renderer: mdRenderer }) as string;
- html = processFootnotes(html, maxCitationId);
- html = renderLatexInHtml(html);
- // Sanitize HTML to prevent XSS attacks
- // [perf-base-v1] 卡6: 转调公共 sanitizeHtml(DOMPurify,ADD_TAGS:['img']),保留本模块额外交互属性白名单
- html = sanitizeHtml(html, ['target', 'data-action', 'data-table', 'data-icon', 'data-icon-success', 'data-n']);
- // 原始 HTML 锚点不经过 marked link renderer,统一补 target/rel
- html = html.replace(/<a\b[^>]*>/gi, (tag) => {
- let attrs = '';
- if (!/\btarget\s*=/i.test(tag)) attrs += ' target="_blank"';
- if (!/\brel\s*=/i.test(tag)) attrs += ' rel="noopener"';
- return attrs ? tag.replace(/^<a\b/i, '<a' + attrs) : tag;
- });
- return icons ? wrapTables(html, icons) : html;
- };
- // renderMarkdown 是纯函数(入参相同则输出相同),但 marked 解析 + KaTeX + DOMPurify 全流程开销大。
- // 流式回答时每个 token 都会让消息列表重渲染,此时未变化的历史消息也会被重复解析,
- // 这里按「引用上限 + 图标 + 原文」记忆化:命中缓存的直接返回上次结果。
- // 内容每帧都在变的流式消息由调用方传 cacheable=false 排除,避免缓存被中间态刷满。
- const MD_CACHE_MAX = 200;
- const mdCache = new Map<string, string>();
- export const renderMarkdownCached = (
- text: string,
- icons?: { copy: string; download: string; preview: string; copySuccess: string },
- maxCitationId?: number,
- cacheable = true
- ): string => {
- if (!cacheable || !text) return renderMarkdown(text, icons, maxCitationId);
- // \u0001 作分隔符:不会出现在正文与图标地址里,避免不同入参拼出同一个 key
- const iconKey = icons ? `${icons.copy}\u0001${icons.download}\u0001${icons.preview}\u0001${icons.copySuccess}` : '';
- const key = `${maxCitationId ?? -1}\u0001${iconKey}\u0001${text}`;
- const cached = mdCache.get(key);
- if (cached !== undefined) return cached;
- const html = renderMarkdown(text, icons, maxCitationId);
- // 简单 FIFO 淘汰,长会话下缓存条目数有上限
- if (mdCache.size >= MD_CACHE_MAX) {
- const oldestKey = mdCache.keys().next().value as string | undefined;
- if (oldestKey !== undefined) mdCache.delete(oldestKey);
- }
- mdCache.set(key, html);
- return html;
- };
- // 从 markdown 文本中提取链接(与 renderMarkdown 共用同一套 marked 解析,
- // 仅收集 link token,代码块/图片等不会被误收)
- export const extractMarkdownLinks = (text: string): Array<{ title: string; href: string }> => {
- if (!text) return [];
- const links: Array<{ title: string; href: string }> = [];
- try {
- const tokens = marked.lexer(text);
- walkTokens(tokens, (token) => {
- if (token.type === 'link') {
- const linkToken = token as Tokens.Link;
- const href = linkToken.href || '';
- if (href) {
- links.push({ title: linkToken.text || href, href });
- }
- }
- });
- } catch {
- // 文本解析异常时返回已收集到的部分链接
- }
- return links;
- };
- export const wrapTables = (html: string, icons: { copy: string; download: string; preview: string; copySuccess: string }): string => {
- return html.replace(/<table>([\s\S]*?)<\/table>/g, (match) => {
- const encodedTable = encodeURIComponent(match);
- return `<div class="markdown-table-wrapper">
- <div class="table-actions">
- <span class="table-action-btn" data-action="copy" data-table="${encodedTable}" data-icon="${icons.copy}" data-icon-success="${icons.copySuccess}" title="复制Markdown">
- <img src="${icons.copy}" width="14" height="14" />
- </span>
- <span class="table-action-btn" data-action="csv" data-table="${encodedTable}" title="下载">
- <img src="${icons.download}" width="14" height="14" />
- </span>
- <span class="table-action-btn" data-action="preview" data-table="${encodedTable}" title="预览">
- <img src="${icons.preview}" width="14" height="14" />
- </span>
- </div>
- ${match}
- </div>`;
- });
- };
- const decodeHtmlEntities = (escaped: string): string => {
- const textarea = document.createElement('textarea');
- textarea.innerHTML = escaped;
- return textarea.value;
- };
- // 给渲染后 HTML 中的 ```markdown / ```md 代码块加上操作工具栏(复制/下载/预览)
- export const wrapMarkdownCodeBlocks = (html: string, icons: { copy: string; download: string; preview: string; copySuccess: string }): string => {
- return html.replace(/<pre><code class="[^"]*language-(?:markdown|md)[^"]*">([\s\S]*?)<\/code><\/pre>/g, (match, escaped: string) => {
- const encoded = encodeURIComponent(decodeHtmlEntities(escaped));
- return `<div class="markdown-code-wrapper">
- <div class="table-actions">
- <span class="table-action-btn" data-action="copy-md" data-content="${encoded}" data-icon="${icons.copy}" data-icon-success="${icons.copySuccess}" title="复制内容">
- <img src="${icons.copy}" width="14" height="14" />
- </span>
- <span class="table-action-btn" data-action="download-md" data-content="${encoded}" title="下载.md文件">
- <img src="${icons.download}" width="14" height="14" />
- </span>
- <span class="table-action-btn" data-action="preview-md" data-content="${encoded}" title="预览渲染效果">
- <img src="${icons.preview}" width="14" height="14" />
- </span>
- </div>
- ${match}
- </div>`;
- });
- };
- export const tableHtmlToMarkdown = (tableHtml: string): string => {
- const parser = new DOMParser();
- const doc = parser.parseFromString(tableHtml, 'text/html');
- const table = doc.querySelector('table');
- if (!table) return tableHtml;
- const rows: string[][] = [];
- table.querySelectorAll('tr').forEach((tr) => {
- const cells: string[] = [];
- tr.querySelectorAll('th, td').forEach((cell) => {
- cells.push(cell.textContent?.trim() || '');
- });
- rows.push(cells);
- });
- if (rows.length === 0) return tableHtml;
- const colCount = Math.max(...rows.map((r) => r.length));
- const normalized = rows.map((r) => {
- while (r.length < colCount) r.push('');
- return r;
- });
- const lines: string[] = [];
- lines.push('| ' + normalized[0].join(' | ') + ' |');
- lines.push('| ' + normalized[0].map(() => '---').join(' | ') + ' |');
- for (let i = 1; i < normalized.length; i++) {
- lines.push('| ' + normalized[i].join(' | ') + ' |');
- }
- return lines.join('\n');
- };
- export const tableHtmlToCsv = (tableHtml: string): string => {
- const parser = new DOMParser();
- const doc = parser.parseFromString(tableHtml, 'text/html');
- const table = doc.querySelector('table');
- if (!table) return '';
- const rows: string[] = [];
- table.querySelectorAll('tr').forEach((tr) => {
- const cells: string[] = [];
- tr.querySelectorAll('th, td').forEach((cell) => {
- const text = (cell.textContent?.trim() || '').replace(/"/g, '""');
- cells.push(`"${text}"`);
- });
- rows.push(cells.join(','));
- });
- return '' + rows.join('\n');
- };
- // 大数字按「万」展示(上下文容量与今日用量共用)
|