// ============================================================ // DOCUMENT EXPORT // Markdown in, PowerPoint / Word / PDF out. // // Markdown is the artifact everywhere: it is what the model writes, what a // refinement edits, and what is stored. Every format here is rendered from it // on demand, so nothing ever has to patch a binary to change a slide. // // pandoc does pptx and docx in the image. PDF needs a renderer pandoc does not // ship, so it goes to Gotenberg, which is LibreOffice behind an HTTP API. That // is a network call, and it is the one export allowed to fail: if Gotenberg is // down the deck and the document still download, and only PDF is unavailable. // ============================================================ var fsp = require('fs/promises'); var os = require('os'); var pathMod = require('path'); var { execFile, spawn } = require('child_process'); var JSZip = require('jszip'); var slideSpec = require('./slideSpec'); var docSpec = require('./docSpec'); var REFERENCE_DECK = pathMod.join(__dirname, '..', '..', 'assets', 'learning', 'slides-reference.pptx'); var DECK_RENDERER = pathMod.join(__dirname, '..', '..', 'scripts', 'render_pptx.py'); var DOC_RENDERER = pathMod.join(__dirname, '..', '..', 'scripts', 'render_docx.py'); var GOTENBERG = process.env.GOTENBERG_URL || 'http://gotenberg:3000'; var FORMATS = { pptx: { ext: 'pptx', mime: 'application/vnd.openxmlformats-officedocument.presentationml.presentation' }, docx: { ext: 'docx', mime: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' }, pdf: { ext: 'pdf', mime: 'application/pdf' } }; function isSupported(format) { return Object.hasOwn(FORMATS, String(format)); } function mimeFor(format) { return (FORMATS[format] || {}).mime; } /** * Let a slide shrink its own text rather than spilling off the bottom. * * pandoc writes a bare on every shape, which leaves the body with no * autofit even though the slide master has one — so a slide with too much on it * is simply cut off mid-sentence, and the remaining bullets are not rendered at * all. Verified by rendering one: three of eight bullets survived. * * with no scale asks the renderer to work out the reduction * itself, which means a slide that already fits is untouched. A fixed * fontScale would shrink every slide whether it needed it or not. * * This is a floor, not a substitute for slides that are the right length — the * prompt still asks for one idea per slide. It stops a long one becoming * unreadable rather than making overcrowding acceptable. */ async function fitSlideText(bytes) { try { var zip = await JSZip.loadAsync(bytes); var slides = Object.keys(zip.files).filter(function (name) { return /^ppt\/slides\/slide\d+\.xml$/.test(name); }); if (!slides.length) return bytes; for (var i = 0; i < slides.length; i++) { var xml = await zip.file(slides[i]).async('string'); if (xml.indexOf('normAutofit') !== -1) continue; zip.file(slides[i], xml.replace(//g, '')); } return await zip.generateAsync({ type: 'nodebuffer', compression: 'DEFLATE' }); } catch (e) { // A deck that renders imperfectly beats no deck at all. console.warn('[export] could not apply slide autofit:', e.message); return bytes; } } function runPandoc(args, cwd) { return new Promise(function (resolve, reject) { execFile('pandoc', args, { cwd: cwd, timeout: 60000, maxBuffer: 1024 * 1024 }, function (err, stdout, stderr) { if (err) return reject(new Error(String(stderr || err.message).slice(0, 400))); resolve(); }); }); } /** * Render markdown to one format. * * `kind` decides how a document is laid out: a presentation becomes slides with * the reference deck's fonts and layouts, an article becomes a Word document. * PDF is produced by converting whichever of those two applies, so a PDF of a * presentation looks like the presentation rather than like a long page. * * Returns a Buffer. Throws with a readable message; the caller decides whether * a failed PDF is fatal. */ async function render(markdown, kind, format, options) { if (!isSupported(format)) throw new Error('Unsupported format: ' + format); options = options || {}; var workdir = await fsp.mkdtemp(pathMod.join(os.tmpdir(), 'export-')); try { await fsp.writeFile(pathMod.join(workdir, 'doc.md'), String(markdown || ''), 'utf8'); var office = kind === 'presentation' ? 'pptx' : 'docx'; if (office === 'docx' || format === 'docx') { await buildDoc(markdown, workdir, options); } if (office === 'pptx' || format === 'pptx') { await buildDeck(markdown, workdir, options.images || [], options); } if (format !== 'pdf') { return await fsp.readFile(pathMod.join(workdir, 'doc.' + format)); } // PDF: hand the office file to Gotenberg. Its LibreOffice keeps the deck's // layout, which is why this is not rendered from the markdown directly. var bytes = await fsp.readFile(pathMod.join(workdir, 'doc.' + office)); var form = new FormData(); form.append('files', new File([bytes], 'doc.' + office, { type: FORMATS[office].mime })); var response = await fetch(GOTENBERG + '/forms/libreoffice/convert', { method: 'POST', body: form, signal: AbortSignal.timeout(90000) }); if (!response.ok) throw new Error('PDF conversion failed (' + response.status + ')'); return Buffer.from(await response.arrayBuffer()); } finally { try { await fsp.rm(workdir, { recursive: true, force: true }); } catch (e) { console.warn('[export] could not clean', workdir, e.message); } } } // ── Deck ──────────────────────────────────────────────────── // Built by scripts/render_pptx.py rather than pandoc. Pandoc's pptx writer maps // markdown onto a handful of reference layouts and gives no control over // per-slide layout, positioning, or how large an image is drawn — the reason // every generated deck came out as bullets on a template, and the reason // autofit had to be injected into its output by hand afterwards. // // If the renderer fails for any reason, pandoc still produces a deck. A plainer // deck beats a failed download. function runRenderer(script, out, spec, workdir) { return new Promise(function (resolve, reject) { var child = spawn('python3', [script, out], { cwd: workdir }); var stderr = ''; child.stderr.on('data', function (chunk) { stderr += chunk.toString().slice(0, 2000); }); child.on('error', reject); child.on('close', function (code) { if (code === 0) return resolve(); reject(new Error(pathMod.basename(script) + ' exited ' + code + (stderr ? ': ' + stderr.trim() : ''))); }); child.stdin.end(JSON.stringify(spec)); }); } // ── Word ──────────────────────────────────────────────────── // Also python, for the same reason as the deck. Pandoc reads markdown, so a deck // had to be flattened to markdown first — and a deck flattened to markdown stops // being one: a comparison became two headings and two lists, a callout became // bold text, and a figure became nothing at all. Built from the typed spec, a // comparison is a two-column table, a callout is a shaded box, and a figure // keeps its caption. // // Pandoc remains the fallback. A plainer document beats a failed download. async function buildDoc(markdown, workdir, options) { var out = pathMod.join(workdir, 'doc.docx'); try { var spec = docSpec.build({ deck: options.deck, markdown: markdown, images: figuresByJob(options.images, options.figureIds) }); if (!spec.blocks.length) throw new Error('nothing to render'); await runRenderer(DOC_RENDERER, out, spec, workdir); if (!(await fsp.readFile(out)).length) throw new Error('the renderer produced an empty file'); } catch (err) { console.warn('[export] document renderer failed, falling back to pandoc:', err.message); await runPandoc(['doc.md', '-o', 'doc.docx'], workdir); } } function figuresByJob(files, figureIds) { var byJob = {}; (figureIds || []).forEach(function (id, index) { if (files && files[index]) byJob[id] = files[index]; }); return byJob; } async function buildDeck(markdown, workdir, images, options) { options = options || {}; var out = pathMod.join(workdir, 'doc.pptx'); try { // A deck the model designed is rendered as designed. Only a resource made // before decks existed, or an article being forced into slides, falls back // to inferring a layout from markdown. var spec = options.deck ? attachFigures(options.deck, images, options.figureIds) : slideSpec.build(markdown, { images: images }); await runRenderer(DECK_RENDERER, out, spec, workdir); var built = await fsp.readFile(out); if (built.length) return; throw new Error('deck renderer produced an empty file'); } catch (err) { console.warn('[export] deck renderer failed, falling back to pandoc:', err.message); await runPandoc(['doc.md', '--reference-doc=' + REFERENCE_DECK, '-o', 'doc.pptx'], workdir); // pandoc leaves a bare on every shape, so its decks still need // autofit injecting or they overflow. await fsp.writeFile(out, await fitSlideText(await fsp.readFile(out))); } } // Put the drawn figures back on the slides that asked for them. The files // arrive in the order the jobs were created, which is the order the slides // requested them, so a slide is matched by its job id rather than by position. function attachFigures(deck, files, figureIds) { var byJob = figuresByJob(files, figureIds); var slides = (deck.slides || []).map(function (slide) { var copy = Object.assign({}, slide); // A custom slide carries its figures on image shapes, under the same rule. if (Array.isArray(copy.shapes)) { copy.shapes = copy.shapes.map(function (shape) { var s = Object.assign({}, shape); delete s.image; if (s.image_job && byJob[s.image_job]) s.image = byJob[s.image_job]; return s; }); } // The renderer reads whatever path this field holds and embeds that file in // the download. Nothing today can put a path in a stored deck — normalise() // never copies one and the edit endpoint writes markdown only — but the // field is cleared before it is set, so the only paths that can reach the // renderer are the figures just fetched for this export. delete copy.image; if (copy.image_job && byJob[copy.image_job]) copy.image = byJob[copy.image_job]; // A figure slide whose picture never arrived is still a slide of text. if (!copy.image && (copy.type === 'image' || copy.type === 'figure')) { copy.type = copy.type === 'image' ? 'section' : 'bullets'; } return copy; }); var out = { title: deck.title, subtitle: deck.subtitle, date: deck.date, slides: slides }; if (deck.title && !slides.some(function (s) { return s.type === 'title'; })) { out.slides = [{ type: 'title', heading: deck.title, subtitle: deck.subtitle, date: deck.date }] .concat(slides); } return out; } // A filename someone can find again, without letting a title choose the path. function filename(title, format) { var safe = String(title || 'resource') .replace(/[^a-zA-Z0-9-_\s]/g, '').replace(/\s+/g, '-').toLowerCase().slice(0, 60) || 'resource'; return safe + '.' + (FORMATS[format] || FORMATS.pdf).ext; } // The deck as bytes, without writing a download. The reviewer renders the deck // it is about to judge, and that is the same pipeline an export runs. async function renderDeck(deck, images, figureIds) { var workdir = await fsp.mkdtemp(pathMod.join(os.tmpdir(), 'deck-')); try { await buildDeck('', workdir, images || [], { deck: deck, figureIds: figureIds || [] }); return await fsp.readFile(pathMod.join(workdir, 'doc.pptx')); } finally { await fsp.rm(workdir, { recursive: true, force: true }).catch(function () {}); } } module.exports = { render, renderDeck, filename, mimeFor, isSupported, FORMATS, GOTENBERG };