// Recognize HTML syntax (including quoted >), not clinical comparisons such as <1 month. var HTML_TAG_PATTERN = /|<\/[A-Za-z][A-Za-z0-9-]*\s*>|<[A-Za-z][A-Za-z0-9-]*(?:\s+[A-Za-z_:][A-Za-z0-9_.:-]*(?:\s*=\s*(?:"[^"]*"|'[^']*'|[^\s"'=<>`\x00-\x20]+))?)*\s*\/?>/; // Mirrors the Open WebUI safeImageUrl allowlist: data:image/* and same-origin // URLs only. External/protocol-relative image URLs never render as ; // they fall back to their alt text instead. export function safeImageUrl(src) { var s = String(src == null ? '' : src).trim(); if (!s) return ''; if (/^data:image\/[a-z0-9.+-]+;/i.test(s)) return s; if (/^blob:/i.test(s)) return s; try { var base = (typeof window !== 'undefined' && window.location && window.location.href) || 'https://synthetic.invalid/'; var url = new URL(s, base); if (url.protocol === 'http:' || url.protocol === 'https:') return url.origin === new URL(base).origin ? s : ''; if (url.protocol === 'file:' || url.protocol === 'javascript:' || url.protocol === 'data:') return ''; return url.origin === new URL(base).origin ? s : ''; // relative paths resolve to our own origin only } catch (e) { return ''; } } // ─── Citations as tokens ─────────────────────────────────────────────── // // A citation is recognised where markdown-it runs inline rules — in prose — // and nowhere else. A bracket inside a code span, a fence, an HTML tag, a link // destination or a held math expression never reaches this rule, because the // tokenizer never runs inline rules there. That is the whole design: the parser // already knows where prose is, and nothing here re-derives it with patterns. // // What it replaced was five hundred lines that did exactly that re-deriving: // hold code, hold links, hold HTML, hold math, rewrite the text, parse, then // regex the HTML for what survived. Every literal context was another branch, // and each one found was another that had been missed. The last two found were // HTML attributes and array indexing in code. // // The stored text is never rewritten. `[7]` in the answer stays `[7]`, and it // is the identity of that source everywhere — the chip's data-source-number, // the card's id, a saved chat, an export. What changes is only what the reader // sees: chips are numbered in order of first appearance, and the sources list // is ordered the same way, both derived from the same walk of the same tokens. var CITE_RE = /^\\?\[((?:\d+\s*,\s*)*\d+)\\?\]/; function citationRule(state, silent) { var src = state.src, pos = state.pos, c = src.charCodeAt(pos); if (c !== 0x5B && !(c === 0x5C && src.charCodeAt(pos + 1) === 0x5B)) return false; var m = CITE_RE.exec(src.slice(pos, pos + 64)); if (!m) return false; // [3](url) is a link and [3]: a definition. [3][2] is two citations — // reference links with all-digit labels do not occur in model output, and // adjacent clusters do, constantly. var after = src.charAt(pos + m[0].length); if (after === '(' || after === ':') return false; var env = state.env || {}; var sources = env.assistantSources || []; var nums = m[1].split(',').map(function (n) { return Number(n.trim()); }); // Every number must name a retrieved source. An invented one is not a // citation: it stays as the text the model wrote, visibly unlinked, and the // server's citation audit records it. Nothing here guesses. if (!nums.every(function (n) { return n > 0 && sourceByNumber(sources, n); })) return false; if (!silent) { nums.forEach(function (n) { var token = state.push('assistant_cite', 'a', 0); token.meta = { source: n, display: displayNumber(env, n) }; }); } state.pos += m[0].length; return true; } // First appearance wins and never moves: a later citation only appends, so a // number assigned while streaming is the number it keeps. function displayNumber(env, n) { var order = env.assistantCiteOrder || (env.assistantCiteOrder = []); var i = order.indexOf(n); if (i === -1) { order.push(n); i = order.length - 1; } return i + 1; } function renderCitation(tokens, idx, o, env) { var meta = tokens[idx].meta; var opts = (env && env.assistantOpts) || {}; var source = sourceByNumber((env && env.assistantSources) || [], meta.source); // A list ordered by citation already says what the reader should see, and // the panel shows the same number. Otherwise, first appearance in this text. var display = source && source.sourceNumber != null ? source.number : meta.display; var title = source ? source.title || source.resource || 'Source' : 'Source'; var page = source && (source.page || source.page_number || source.pageNumber); var label = 'Source ' + display + ': ' + title + (page ? ', page ' + page : ''); var text = opts.citationLabel === 'text' ? 'src' : String(display); // Chips in one cluster, or back to back, are separated by a space rather // than run together. var gap = idx > 0 && tokens[idx - 1].type === 'assistant_cite' ? ' ' : ''; return gap + '' + text + ''; } /** * Teach a markdown-it instance the assistant's three renderer rules. Idempotent, * so the same instance can be handed in on every render. * * Fences render mermaid and chart blocks to the placeholders the page then * draws, and everything else to a plain
. Images render only from
 * the safe allowlist; anything else falls back to its alt text. Both used to be
 * done by holding the block out of the text before parsing and splicing HTML
 * back in afterwards, which is what a renderer rule is for.
 */
export function installCitations(md) {
  if (!md || md.__assistantCitations) return md;
  md.__assistantCitations = true;
  // Before 'escape', or \[1\] is consumed as an escaped bracket first.
  md.inline.ruler.before('escape', 'assistant_cite', citationRule);
  md.renderer.rules.assistant_cite = renderCitation;
  var image = md.renderer.rules.image || function (t, i, o, e, self) { return self.renderToken(t, i, o); };
  md.renderer.rules.image = function (tokens, idx, o, env, self) {
    if (!safeImageUrl(tokens[idx].attrGet('src'))) return escapeHtml(tokens[idx].content || 'Image');
    return image(tokens, idx, o, env, self);
  };
  // markdown-it writes column alignment as style="text-align:…"; the export,
  // the stylesheet and a decade of email clients read the align attribute.
  // Both are emitted so nothing that looks for either is wrong.
  ['th_open', 'td_open'].forEach(function (name) {
    var base = md.renderer.rules[name] || function (t, i, o, e, self) { return self.renderToken(t, i, o); };
    md.renderer.rules[name] = function (tokens, idx, o, env, self) {
      var style = tokens[idx].attrGet('style') || '';
      var m = /text-align:\s*(left|right|center)/.exec(style);
      if (m && !tokens[idx].attrGet('align')) tokens[idx].attrSet('align', m[1]);
      return base(tokens, idx, o, env, self);
    };
  });
  // Math is not markdown, so the parser needs telling where it is — and once
  // told, a $ inside a URL, a code span or an HTML attribute is never math,
  // by construction, which is what a text pre-pass could not promise.
  md.inline.ruler.before('escape', 'assistant_math', mathInlineRule);
  md.block.ruler.before('fence', 'assistant_math_block', mathBlockRule, { alt: ['paragraph', 'reference', 'blockquote', 'list'] });
  md.renderer.rules.assistant_math = function (tokens, idx, o, env) { return renderMath(env, tokens[idx], false); };
  md.renderer.rules.assistant_math_block = function (tokens, idx, o, env) { return renderMath(env, tokens[idx], true) + '\n'; };
  md.renderer.rules.fence = function (tokens, idx) {
    var lang = String(tokens[idx].info || '').trim().toLowerCase();
    var code = String(tokens[idx].content || '').replace(/\n$/, '');
    // Percent-encoding keeps Mermaid arrows intact through DOMPurify.
    if (lang === 'mermaid') return '
Rendering graph...
\n'; if (lang === 'chart' || lang === 'chartjs') return '\n'; return '
' + escapeHtml(code) + '
\n'; }; return md; } // ─── Math as tokens ──────────────────────────────────────────────────── // $…$ and \(…\) inline; $$…$$ and \[…\] display, on one line or spanning // several; \ce{…} and \pu{…} for mhchem. Rendered by KaTeX when the page has // it, otherwise escaped and shown as written. var MATH_INLINE_RE = /^(?:\$\$([\s\S]+?)\$\$|\\\[([\s\S]+?)\\\]|\$([^$\n]+?)\$|\\\((.+?)\\\)|\\(ce|pu)\{([^{}\n]*)\})/; function mathInlineRule(state, silent) { var c = state.src.charCodeAt(state.pos); if (c !== 0x24 /* $ */ && c !== 0x5C /* \ */) return false; var m = MATH_INLINE_RE.exec(state.src.slice(state.pos)); if (!m) return false; // A lone $ with a space either side is a dollar sign, not an equation. if (m[3] !== undefined && (/^\s/.test(m[3]) || /\s$/.test(m[3]))) return false; if (!silent) { var token = state.push('assistant_math', 'span', 0); var display = m[1] !== undefined || m[2] !== undefined; token.content = m[1] !== undefined ? m[1] : m[2] !== undefined ? m[2] : m[3] !== undefined ? m[3] : m[4] !== undefined ? m[4] : '\\' + m[5] + '{' + m[6] + '}'; // Without KaTeX the expression is shown exactly as written — \(a\) stays // \(a\), not re-dressed as $a$ — so raw is kept beside the content. token.meta = { display: display, raw: m[0] }; if (display) token.type = 'assistant_math_block'; } state.pos += m[0].length; return true; } // $$ or \[ opening a line, closing on the same or a later line. function mathBlockRule(state, startLine, endLine, silent) { var start = state.bMarks[startLine] + state.tShift[startLine]; var max = state.eMarks[startLine]; var first = state.src.slice(start, max); var open = /^\$\$/.test(first) ? '$$' : /^\\\[/.test(first) ? '\\[' : null; if (!open) return false; var close = open === '$$' ? '$$' : '\\]'; var rest = first.slice(open.length); var content, next = startLine; var closeAt = rest.indexOf(close); if (closeAt !== -1 && rest.slice(closeAt + close.length).trim() === '') { content = rest.slice(0, closeAt); } else { var lines = [rest]; for (next = startLine + 1; next < endLine; next++) { var line = state.src.slice(state.bMarks[next] + state.tShift[next], state.eMarks[next]); var at = line.indexOf(close); if (at !== -1) { lines.push(line.slice(0, at)); break; } lines.push(line); } if (next >= endLine) return false; // never closed: leave it to the paragraph rule content = lines.join('\n'); } if (silent) return true; var token = state.push('assistant_math_block', 'div', 0); token.content = content.trim(); token.meta = { display: true, raw: state.getLines(startLine, next + 1, 0, false).trim() }; token.map = [startLine, next + 1]; state.line = next + 1; return true; } function renderMath(env, token, display) { var katex = env && env.assistantOpts && env.assistantOpts.katex; if (!katex) return escapeHtml((token.meta && token.meta.raw) || token.content); return safeKatex(katex, token.content, display); } var defaultRenderer = null; function resolveRenderer(opts) { if (opts && opts.markdownIt && typeof opts.markdownIt.render === 'function') return installCitations(opts.markdownIt); if (!defaultRenderer) { // The browser's window is globalThis. A test sandbox keeps window as a // property of its own global, so both are looked at. var g = typeof globalThis !== 'undefined' ? globalThis : {}; var factory = typeof g.markdownit === 'function' ? g.markdownit : (g.window && typeof g.window.markdownit === 'function' ? g.window.markdownit : null); // html:true, as marked rendered before it: an answer may carry a or // a
, and the single sanitisation boundary is what makes that safe. if (factory) defaultRenderer = installCitations(factory({ html: true, linkify: true, typographer: true, breaks: true })); } return defaultRenderer; } // The text as the parser should see it: placeholders dropped, math rendered and // held, fences and inline code held while the legacy table repair runs, then put // back so the parser's own fence rule renders them. Shared by rendering and by // ordering, so the two cannot disagree about what is a citation. function prepareMarkdown(md, opts, renderer) { // The legacy table repair and the fence hold need a lexer to know where // blocks are. It is the same parser that will render — one, not two. opts = Object.assign({}, opts, { markdownIt: renderer }); var text = String(md || '') // The model occasionally writes [src]/[source] placeholders instead of // real citation numbers. Never invent numbers — drop the tokens; the // sources panel still lists the actual sources. .replace(/\[(?:src|source)\]/gi, '') // A cluster the model ran out of room for: "[1][4][2][3" at the very end. // Only when it directly follows complete brackets, so a lone "[3" is left. .replace(/((?:\[\d+\]\s*)+)\[(\d+)\s*$/, '$1[$2]'); var embedded = textProtection(text, 'html'); var fences = textProtection(text, 'fence'); text = protectBlocks(text, opts, function (code, lang) { return fences.hold('```' + (lang || '') + '\n' + code + '\n```'); }, true); var inlineCode = textProtection(text, 'code'); text = text.replace(/(`+)[\s\S]*?\1/g, inlineCode.hold); // \[1\] is a citation the model escaped, not display math — but \[ ... \] is // the display-math delimiter, and the math pass runs next. A bracket holding // nothing but numbers is never an equation, so it is unescaped here, before // the math pass can hold it. This is the one place text syntax is genuinely // ambiguous, and the parser cannot resolve it because math is not markdown. text = text.replace(/\\\[((?:\d+\s*,\s*)*\d+)\\\]/g, '[$1]'); var limited = false; text = normalizeMarkdownText(text, Object.assign({}, opts, { onUnrecoverable: function (raw) { limited = true; raw = inlineCode.restore(fences.restore(embedded.restore(raw, true))); return embedded.hold('
' + escapeHtml(raw) + '
', raw); } })); text = fences.restore(inlineCode.restore(text)); return { text: text, embedded: embedded, limited: limited }; } export function renderAssistantMarkdown(md, sources, options) { var opts = options || {}; var renderer = resolveRenderer(opts); if (!renderer) throw new Error('renderAssistantMarkdown: no markdown-it renderer is available'); var prepared = prepareMarkdown(md, opts, renderer); var env = { assistantSources: sources || [], assistantOpts: opts, assistantCiteOrder: [] }; var html = renderer.render(prepared.text, env); html = prepared.embedded.restore(html); html = wrapTables(html); if (prepared.limited) html += '

Could not safely recover this flattened table. Stored text is shown unchanged; missing cells or rows may be present.

'; if (opts.notice) html += '

' + escapeHtml(opts.notice) + '

'; return typeof opts.sanitize === 'function' ? opts.sanitize(html) : html; } /** * Link `[n]` markers in HTML that did not come from our renderer. * * The translation service is handed HTML and returns HTML, with the markers * carried across as plain text. That cannot go through the tokenizer, but the * document has structure of its own: only text nodes are visited, and none * inside code, pre, a or script. An attribute value is not a text node, so it is * never touched — the case that defeated the pattern approach. */ export function linkCitationsInHtml(html, sources, options) { if (typeof document === 'undefined') return String(html || ''); var opts = options || {}; var tpl = document.createElement('template'); tpl.innerHTML = String(html || ''); var walker = document.createTreeWalker(tpl.content, 4 /* NodeFilter.SHOW_TEXT */); var texts = []; for (var node = walker.nextNode(); node; node = walker.nextNode()) { var skip = false; for (var el = node.parentNode; el && el !== tpl.content; el = el.parentNode) { if (/^(code|pre|a|script|style)$/i.test(el.nodeName)) { skip = true; break; } } if (!skip && node.nodeValue.indexOf('[') !== -1) texts.push(node); } var env = { assistantSources: sources || [], assistantOpts: opts, assistantCiteOrder: [] }; texts.forEach(function (node) { var parts = node.nodeValue.split(/(\\?\[(?:\d+\s*,\s*)*\d+\\?\])/); if (parts.length === 1) return; var frag = document.createDocumentFragment(); parts.forEach(function (part) { var m = /^\\?\[((?:\d+\s*,\s*)*\d+)\\?\]$/.exec(part); var nums = m ? m[1].split(',').map(function (n) { return Number(n.trim()); }) : null; if (!nums || !nums.every(function (n) { return n > 0 && sourceByNumber(env.assistantSources, n); })) { frag.appendChild(document.createTextNode(part)); return; } var span = document.createElement('span'); span.innerHTML = nums.map(function (n, i) { return renderCitation([{ type: 'assistant_cite', meta: { source: n, display: displayNumber(env, n) } }], 0, {}, env) .replace(/^ /, i ? ' ' : ''); }).join(''); while (span.firstChild) frag.appendChild(span.firstChild); }); node.parentNode.replaceChild(frag, node); }); return tpl.innerHTML; } export function wrapTables(html) { return String(html || '') .replace(/]*)?>/g, '
') .replace(/<\/table>/g, '
'); } // A citation marker names a source by its `number`, which dedupeSources assigns // server-side. Resolving it by array position happens to work today because the // two agree, but any future filtering or reordering of the list between the // server and here would silently point citations at the wrong source. Matching // on the number cannot drift. function sourceByNumber(sources, n) { var list = sources || []; // A list ordered by citation carries each source's original number as // sourceNumber and its display position as number. `[7]` in the text names // the original, always. for (var i = 0; i < list.length; i++) { if (list[i] && list[i].sourceNumber != null && Number(list[i].sourceNumber) === Number(n)) return list[i]; } if (list.some(function (s) { return s && s.sourceNumber != null; })) return undefined; for (var j = 0; j < list.length; j++) { if (list[j] && Number(list[j].number) === Number(n)) return list[j]; } // Sources predating the numbering, or a caller passing a bare list. return list[n - 1]; } /** * Sources in the order the answer cites them. * * Derived from the same token walk that renders the chips, so the list and the * numbers on screen cannot disagree. The text is returned exactly as given: * `[7]` stays `[7]`, because that number is the identity of the source in the * saved chat, the export and every chip's data-source-number. Only `number` — * what the reader sees — is by first appearance; `sourceNumber` keeps the * original. * * Cited sources first. Anything retrieved and not cited follows, marked, and * still numbered so the list stays contiguous — it is evidence of what the * search returned, which is what the panel is for. */ export function orderSourcesByCitation(text, sources, options) { var list = Array.isArray(sources) ? sources : []; var plain = String(text || ''); var renderer = resolveRenderer(options || {}); if (!list.length || !renderer) return { text: plain, sources: list }; var env = { assistantSources: list, assistantOpts: {}, assistantCiteOrder: [] }; renderer.parse(prepareMarkdown(plain, Object.assign({}, options || {}, { katex: null }), renderer).text, env); var order = env.assistantCiteOrder; if (!order.length) return { text: plain, sources: list }; var out = order.map(function (n, i) { return Object.assign({}, sourceByNumber(list, n), { number: i + 1, sourceNumber: n }); }); list.forEach(function (source, i) { var n = Number(source && source.number); if (order.indexOf(n) !== -1) return; // Identity is whatever `number` was — kept raw, never coerced. A legacy // save can carry a non-numeric value here, and inventing a clean one in // its place would change which card a saved chip points at. var identity = source && source.number != null ? source.number : i + 1; out.push(Object.assign({}, source, { number: out.length + 1, sourceNumber: identity, uncited: true })); }); return { text: plain, sources: out }; } function textProtection(text, kind) { var prefix = '\uE000' + (kind || 'markdown') + ':'; while (text.indexOf(prefix) !== -1) prefix += '\uE000'; var values = []; var originals = []; var pattern = new RegExp(prefix + '(\\d+)\uE001', 'g'); return { hold: function(value, original) { originals.push(typeof original === 'string' ? original : value); return prefix + (values.push(value) - 1) + '\uE001'; }, restore: function(value, raw) { var previous; do { previous = value; value = value.replace(pattern, function(_, i) { return (raw ? originals : values)[Number(i)]; }); } while (value !== previous); return value; } }; } function protectBlocks(text, opts, hold, codeOnly) { if (opts.marked && typeof opts.marked.lexer === 'function') { return opts.marked.lexer(text, { gfm: true }).map(function(token) { if (codeOnly) return token.type === 'code' ? hold(token.text, token.lang) + '\n\n' : token.raw; if (['space', 'paragraph', 'heading'].includes(token.type)) return token.raw; return hold(token.type === 'table' ? normalizeTableSourceCitationCells(stripTrailingMarkerLine(token.raw)) : token.type === 'code' || token.type === 'html' ? token.raw : stripTrailingMarkerLine(token.raw)) + '\n\n'; }).join(''); } if (opts.markdownIt && typeof opts.markdownIt.parse === 'function') { var lines = text.split('\n'); var ranges = opts.markdownIt.parse(text, {}).filter(function(token) { return token.map && (codeOnly ? ['fence', 'code_block'] : ['table_open', 'fence', 'code_block', 'html_block', 'blockquote_open', 'bullet_list_open', 'ordered_list_open']).includes(token.type); }); var end = 0; ranges.forEach(function(token) { if (token.map[0] < end) return; end = token.map[1]; var raw = lines.slice(token.map[0], end).join('\n'); lines[token.map[0]] = codeOnly ? hold(token.content, token.info) : hold(token.type === 'table_open' ? normalizeTableSourceCitationCells(stripTrailingMarkerLine(raw)) : ['fence', 'code_block', 'html_block'].includes(token.type) ? raw : stripTrailingMarkerLine(raw)); for (var i = token.map[0] + 1; i < end; i++) lines[i] = ''; }); return lines.join('\n'); } text = text.replace(/^ {0,3}(`{3,}|~{3,})([^\n]*)\n([\s\S]*?)(?:^ {0,3}\1[^\n]*(?:\n|$)|(?![\s\S]))/gm, function(raw, fence, lang, code) { return codeOnly ? hold(code, lang.trim()) : hold(raw); }); return text.replace(/(?:^(?: {4}|\t)[^\n]*(?:\n|$))+/gm, function(raw) { return codeOnly ? hold(raw.replace(/^(?: {4}|\t)/gm, ''), '') : hold(raw); }); } function stripTrailingMarkerLine(text) { return text.replace(/\n[ \t]*(?:\*\*|__|\*|_)[ \t]*\n?$/, '\n'); } export function normalizeMarkdownText(text, options) { var opts = options || {}; text = String(text || '').replace(/\r\n/g, '\n'); var protectedText = textProtection(text); text = protectBlocks(text, opts, protectedText.hold); // These spans must not become headings/lists or apparent legacy row boundaries. text = text.replace(/(`+)[\s\S]*?\1|\$\$[\s\S]*?\$\$|\\\[[\s\S]*?\\\]|\\\([\s\S]*?\\\)|\$[^$\n]+\$|!?\[[^\]\n]*\]\([^\n]*?\)|https?:\/\/(?:\\[^\s]|[^\s<>|\\])+|\\\|/gi, protectedText.hold); text = text.split('\n').map(function(line) { if (!/\|[ \t]*:?-{2,}:?[ \t]*\|/.test(line)) return line; var recovered = recoverLegacyTableLine(line); if (recovered !== null) return recovered; // A standalone delimiter row belongs to existing multiline Markdown. if (/^[ \t]*\|?[ :|\t-]+$/.test(line)) return line; var start = line.indexOf('|'); var raw = protectedText.restore(line.slice(start)); var literal = typeof opts.onUnrecoverable === 'function' ? opts.onUnrecoverable(raw) : raw; return line.slice(0, start) + '\n\n' + protectedText.hold(literal); }).join('\n'); text = protectBlocks(text, opts, protectedText.hold); // Also leave incomplete/non-GFM pipe structures alone, rather than guessing. text = text.replace(/^[^\n]*\|[^\n]*(?:\n[^\n]*\|[^\n]*)*/gm, function(block) { return protectedText.hold(normalizeTableSourceCitationCells(block)); }); return protectedText.restore(normalizeProse(text)); } // Only outer-pipe rows separated by "| |", with a header + alignment row and // identical nonempty cell counts, survive whitespace collapse unambiguously. // Empty cells, trailing prose, partial rows and mixed separators are not guessed. function recoverLegacyTableLine(line) { var start = line.indexOf('|'); var prefix = line.slice(0, start); var candidate = line.slice(start).trimEnd(); if (start < 0 || !candidate.endsWith('|')) return null; var rows = candidate.split(/(?<=\|)[ \t]+(?=\|)/); if (rows.length < 3) return null; var cells = rows.map(tableCells); if (cells.some(function(row) { return row.length < 2 || row.some(function(cell) { return !cell; }); })) return null; function separator(row) { return row.every(function(cell) { return /^:?-{2,}:?$/.test(cell); }); } var output = []; for (var i = 0; i < rows.length;) { if (!cells[i + 1] || !separator(cells[i + 1]) || cells[i].length !== cells[i + 1].length || separator(cells[i])) return null; var width = cells[i].length; var end = i + 2; while (end < rows.length && !(cells[end + 1] && separator(cells[end + 1]))) { if (cells[end].length !== width || separator(cells[end])) return null; end++; } if (end === i + 2) return null; output.push(rows.slice(i, end).join('\n')); i = end; } return (prefix.trim() ? prefix + '\n\n' : '') + output.join('\n\n'); } function normalizeProse(text) { return stripOrphanMarkdownMarkers(String(text || '') .replace(/([^\n])\n+\s*(\[(?:\d+\s*,\s*)*\d+\])\s*(?:\n+\s*([.,;:]))?(?=\s*(?:\n|$))/g, '$1 $2$3') .replace(/([.!?])\s*[-–—]\s+(\*\*)?/g, '$1\n- $2') .replace(/(:)\s*[-–—]\s+(\*\*)?/g, '$1\n- $2') .replace(/([.!?])\s+(\d+\.\s+[A-Z][A-Za-z][^\n]{0,80})/g, '$1\n$2') .replace(/([^\n])\s+(#{1,4}\s+)/g, '$1\n\n$2') .replace(/(#{1,4}\s+[^\n]+?)\s+(-\s+)/g, '$1\n\n$2') .replace(/(#{1,4}\s+[^\n]+)\n(-\s+)/g, '$1\n\n$2') .replace(/([^\n])\s+(-\s+(?:Mainstay|Medications|Hospitalization|Other therapies|Prevention|Short-acting|Anticholinergics|Systemic|Adjuncts|Long-term|Infants|Differentiating|Persistent|Severe|Need for|Inadequate)\b)/g, '$1\n$2') .trim()); } export function normalizeTableSourceCitationCells(text) { var lines = String(text || '').split('\n'); for (var i = 0; i < lines.length - 1; i++) { if (!/^\s*\|.*\|\s*$/.test(lines[i]) || !/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)+\|?\s*$/.test(lines[i + 1])) continue; var header = tableCells(lines[i]); var sourceCols = []; header.forEach(function(cell, idx) { if (/^(?:source|sources|source\(s\)|citation|citations|citation\(s\)|reference|references|ref|refs)$/i.test(cell.trim())) sourceCols.push(idx); }); if (!sourceCols.length) continue; var j = i + 2; while (j < lines.length && /^\s*\|.*\|\s*$/.test(lines[j])) { lines[j] = rewriteTableCells(lines[j], sourceCols, function(cell) { return normalizeBareCitationCell(cell); }); j++; } i = j - 1; } return lines.join('\n'); } function rewriteTableCells(line, indexes, fn) { var trimmed = String(line || '').trim(); var leading = /^\|/.test(trimmed); var trailing = /\|$/.test(trimmed); var cells = tableCells(line); indexes.forEach(function(idx) { if (idx < cells.length) cells[idx] = fn(cells[idx]); }); return (leading ? '| ' : '') + cells.join(' | ') + (trailing ? ' |' : ''); } function normalizeBareCitationCell(cell) { var text = String(cell || '').trim(); if (/^\[(?:\d+\s*,\s*)*\d+\]$/.test(text)) return text; if (/^\d+(?:\s*,\s*\d+)*$/.test(text)) return '[' + text.replace(/\s*,\s*/g, ', ') + ']'; return text.replace(/(^|\s)(\d+(?:\s*,\s*\d+)+)(?=$|\s)/g, function(match, prefix, nums) { return prefix + '[' + nums.replace(/\s*,\s*/g, ', ') + ']'; }); } export function stripOrphanMarkdownMarkers(text) { return String(text || '') .replace(/\s*(?:\*\*|__|\*|_)\s*$/g, '') .replace(/\s*(?:\*\*|__)?\s*(?:Figure|Fig\.)\s*(?:\*\*|__)?\s*$/i, '') .trim(); } function tableCells(line) { return String(line || '').trim().replace(/^\|/, '').replace(/\|$/, '').split(/(?/g, '>').replace(/"/g, '"'); } export function escapeAttr(s) { return escapeHtml(s).replace(/'/g, '''); }