import * as cheerio from 'cheerio' import sanitizeHtml from 'sanitize-html' import { eq, inArray, sql } from 'drizzle-orm' import { flipFromString, rotateFromString } from '@iconify/utils' import { jobs as jobsTable, pageRenderQueue as renderQueueTable } from '../db/schema.ts' import { CustomError } from '../helpers/common.ts' import { hrefsFrom } from '../helpers/pageLinks.ts' import type { IconifyIcon } from '@iconify/types' import type { RenderableBlocks } from './blocks.ts' import type { IconifyIconCustomisations } from '@iconify/utils' /** * Rendering model * * Markdown becomes HTML in the browser, not here: the editor renders as you type, and what it shows * in its preview is what gets sent up and stored. One renderer, one result — the preview cannot drift * from the saved page because they are the same render. * * What this model does is everything that has to happen *after* that, and cannot be left to the * client: * * - **Sanitizing.** The HTML arrived from a browser, so it is a user input like any other. What * survives depends on what the author is allowed to do — scripts and styles are permissions. * - **Normalizing.** The editor leaves scaffolding in its output (line markers for preview scroll * sync) that has no business being stored, and headings arrive without the anchors a table of * contents needs. * - **Resolving.** An icon is a reference when it is written and a picture when it is read, and this * is where it stops being the former — drawn into the page once, at save time, rather than fetched * by every reader's browser on every view. * - **Extracting.** The table of contents and the plain text the search index is built from are both * derived from the final HTML, once it is settled. * * Re-rendering an existing page from its source — which the server needs when the content is there * but the render is stale — goes back through the very same frontend pipeline, driven in a headless * browser. That is a job rather than part of a request: see `queuePage` and `drainQueue`. */ /** * The editors whose source the renderer bundle knows how to turn into HTML. * * Both pipelines live in the frontend and both are reached through the same `__wikiRender`, which * picks one by the editor it is handed -- so this list is the server's copy of what that function * will accept, and the two have to agree. An editor missing from it is refused before anything is * queued rather than after a browser has been started for it. */ const RENDERABLE_EDITORS = new Set(['markdown', 'asciidoc', 'visual']) /** * Editors whose pages are markdown under another name, and whose CONFIG therefore lives elsewhere. * * The visual editor writes markdown — `contentType: markdown` — and previews it through the very * same renderer, with `editors.markdown`'s config rather than its own empty one (see * `EditorVisual.vue`). A server-side render has to read the same config or it produces a page that * differs from what the author was looking at, over settings like `allowHTML` and `linkify`. */ const EDITOR_CONFIG_ALIAS: Record = { visual: 'markdown' } /** How long the renderer bundle gets to load itself in the headless browser, in milliseconds. */ const RENDER_READY_TIMEOUT = 30000 /** How long a single render gets once the bundle is up, in milliseconds. */ const RENDER_TIMEOUT = 30000 /** The task that drains the render queue. One browser, one page at a time. */ const DRAIN_TASK = 'renderPages' /** A heading in the table of contents, shaped for the Quasar tree the page sidebar draws. */ export interface TocNode { key: string label: string /** * The heading's own level, 1 to 6. * * Kept alongside the nesting because the two say different things: a contents list is asked to show * "H1 to H2", which is about the tag an author reached for, and an `h3` written under an `h1` is * still an `h3` however few levels sit above it. */ level: number children: TocNode[] } export interface PostProcessResult { /** The HTML to store and serve. */ render: string /** The table of contents, derived from the headings. */ toc: TocNode[] /** Plain text, for the search index. */ text: string /** * Every href the page carries, exactly as written and with repeats left in. * * Raw on purpose: what an href ADDRESSES depends on where the page holding it sits, on the site's * locale prefixes and on its page extensions, none of which is rendering's business. See * `helpers/pageLinks.ts`, which reads them, and `models/pageLinks.ts`, which stores what it makes * of them. */ links: string[] } /** * A headless browser standing by on the renderer bundle, good for any number of pages. * * Opening one is the expensive part of rendering, so it is handed out as a handle to be reused and * closed by whoever asked for it rather than opened per page. */ interface PageRenderer { /** * A page's source in, the editor's own HTML out — before `postProcess` gets to it. * * `context` carries what the source cannot say about itself: the page's own path, since a relative * image in a page resolves against the folder it sits in as it would in a repository, and the * `editor` it was written with, which is what picks the pipeline on the other side. A source is not * self-describing — `= Title` is a heading in one syntax and an attribute list in neither — so * nothing can be guessed from the text. */ render( content: string, config: Record, context: Record ): Promise close(): Promise } /** What the author is allowed to put in a page, beyond ordinary content. */ export interface RenderPermissions { /** `write:scripts` — may embed `