From 3b28ac18e5100c59d967c3d1a3000998234dcafd Mon Sep 17 00:00:00 2001 From: NGPixel Date: Mon, 21 Sep 2026 06:40:52 -0400 Subject: [PATCH] feat: add export for wiki.js 3.x utility --- .../admin/admin-utilities-backup.vue | 312 +++++++++ client/components/admin/admin-utilities.vue | 13 +- server/controllers/common.js | 20 + server/core/backup.js | 614 ++++++++++++++++++ server/graph/resolvers/system.js | 39 ++ server/graph/schemas/system.graphql | 15 + server/helpers/zip.js | 389 +++++++++++ server/master.js | 1 + 8 files changed, 1401 insertions(+), 2 deletions(-) create mode 100644 client/components/admin/admin-utilities-backup.vue create mode 100644 server/core/backup.js create mode 100644 server/helpers/zip.js diff --git a/client/components/admin/admin-utilities-backup.vue b/client/components/admin/admin-utilities-backup.vue new file mode 100644 index 000000000..6e4c46b1b --- /dev/null +++ b/client/components/admin/admin-utilities-backup.vue @@ -0,0 +1,312 @@ + + + + + diff --git a/client/components/admin/admin-utilities.vue b/client/components/admin/admin-utilities.vue index 43cb4db15..683cf8332 100644 --- a/client/components/admin/admin-utilities.vue +++ b/client/components/admin/admin-utilities.vue @@ -18,8 +18,8 @@ v-list-item-avatar v-icon(:color='!tool.isAvailable ? `grey lighten-1` : (selectedTool === tool.key ? `blue ` : `grey darken-1`)') {{ tool.icon }} v-list-item-content - v-list-item-title.body-2(:class='!tool.isAvailable ? `grey--text` : (selectedTool === tool.key ? `primary--text` : ``)') {{ $t('admin:utilities.' + tool.i18nKey + 'Title') }} - v-list-item-subtitle: .caption(:class='!tool.isAvailable ? `grey--text text--lighten-1` : (selectedTool === tool.key ? `blue--text ` : ``)') {{ $t('admin:utilities.' + tool.i18nKey + 'Subtitle') }} + v-list-item-title.body-2(:class='!tool.isAvailable ? `grey--text` : (selectedTool === tool.key ? `primary--text` : ``)') {{ tool.title || $t('admin:utilities.' + tool.i18nKey + 'Title') }} + v-list-item-subtitle: .caption(:class='!tool.isAvailable ? `grey--text text--lighten-1` : (selectedTool === tool.key ? `blue--text ` : ``)') {{ tool.subtitle || $t('admin:utilities.' + tool.i18nKey + 'Subtitle') }} v-list-item-avatar(v-if='selectedTool === tool.key') v-icon.animated.fadeInLeft(color='primary', large) mdi-chevron-right v-divider(v-if='idx < tools.length - 1') @@ -35,6 +35,7 @@ export default { components: { UtilityAuth: () => import(/* webpackChunkName: "admin" */ './admin-utilities-auth.vue'), + UtilityBackup: () => import(/* webpackChunkName: "admin" */ './admin-utilities-backup.vue'), UtilityContent: () => import(/* webpackChunkName: "admin" */ './admin-utilities-content.vue'), UtilityCache: () => import(/* webpackChunkName: "admin" */ './admin-utilities-cache.vue'), UtilityExport: () => import(/* webpackChunkName: "admin" */ './admin-utilities-export.vue'), @@ -63,6 +64,14 @@ export default { i18nKey: 'export', isAvailable: true }, + { + key: 'UtilityBackup', + icon: 'mdi-package-variant-closed', + // No localized strings exist for this utility yet, so it carries its own labels. + title: 'Export for Wiki.js 3.x', + subtitle: 'Create a .wkbackup migration package', + isAvailable: true + }, { key: 'UtilityCache', icon: 'mdi-database-refresh', diff --git a/server/controllers/common.js b/server/controllers/common.js index dec6fa3aa..750862088 100644 --- a/server/controllers/common.js +++ b/server/controllers/common.js @@ -5,6 +5,7 @@ const _ = require('lodash') const CleanCSS = require('clean-css') const moment = require('moment') const qs = require('querystring') +const fs = require('fs-extra') /* global WIKI */ @@ -395,6 +396,25 @@ router.get(['/t', '/t/*'], (req, res, next) => { res.render('tags') }) +/** + * Download Backup Package + * + * The .wkbackup package holds password hashes and TOTP secrets, so it is served + * only to administrators, and only from the backup directory. + */ +router.get('/_backup/:filename', async (req, res, next) => { + if (!WIKI.auth.checkAccess(req.user, ['manage:system'])) { + return res.sendStatus(403) + } + + const filePath = WIKI.backup.resolveFile(req.params.filename) + if (!filePath || !await fs.pathExists(filePath)) { + return res.sendStatus(404) + } + + res.download(filePath, req.params.filename) +}) + /** * User Avatar */ diff --git a/server/core/backup.js b/server/core/backup.js new file mode 100644 index 000000000..b7e690b28 --- /dev/null +++ b/server/core/backup.js @@ -0,0 +1,614 @@ +const _ = require('lodash') +const fs = require('fs-extra') +const path = require('path') +const crypto = require('crypto') +const { Readable } = require('node:stream') +const { ZipWriter } = require('../helpers/zip') + +/* global WIKI */ + +/** + * .wkbackup package writer, for migrating a 2.x instance to 3.x. + * + * The package is a ZIP holding a manifest, a set of NDJSON record streams and + * one content-addressed entry per distinct file. See dev/specs/wkbackup.md in + * the 3.x repository for the format itself. + * + * Records are written in 2.x's own shape, with 2.x's integer ids left intact: + * `source.kind: wikijs2` tells the 3.x importer to run the translation, and the + * importer derives every UUID from those source ids. This exporter judges + * nothing it cannot judge from here, and reports what it knows will not carry + * over in `manifest.warnings`. + */ + +const FORMAT_VERSION = 1 +const EXPORTER_VERSION = '1.0.0' + +// 2.x has no sites, so the single site it does have is named `default`. +const SITE_ID = 'default' + +// A content or render field over this size spills to a blob, so that no single +// NDJSON line forces a multi-megabyte string through the reader's parser. +const BLOB_SPILL_THRESHOLD = 1024 * 1024 + +const BATCH_SIZE = { + assets: 25, + comments: 50, + history: 10, + pages: 10, + tree: 100, + users: 50 +} + +/** + * Permissions Wiki.js 3.x knows about, as listed by its group editor. + * + * A 2.x permission missing from this set has no 3.x equivalent. It is carried + * in the package as-is and reported in `manifest.warnings`, never remapped: an + * unrecognised permission string silently never matches, so a wrong guess would + * hide controls with no error anywhere. + */ +const WIKI3_PERMISSIONS = [ + 'access:admin', 'delete:pages', 'manage:assets', 'manage:comments', + 'manage:groups', 'manage:navigation', 'manage:pages', 'manage:scim', + 'manage:sites', 'manage:storage', 'manage:system', 'manage:theme', + 'manage:users', 'manage:webhooks', 'read:assets', 'read:audit', + 'read:comments', 'read:groups', 'read:history', 'read:metrics', + 'read:pages', 'read:source', 'read:users', 'read:webhooks', + 'review:pages', 'write:assets', 'write:comments', 'write:groups', + 'write:pages', 'write:scripts', 'write:styles', 'write:tags', + 'write:users' +] + +/** + * Yield every row of a table in batches, so no table is ever fully resident. + */ +async function * batched (fetch, batchSize, onBatch) { + let offset = 0 + while (true) { + const rows = await fetch(offset, batchSize) + for (const row of rows) { + yield row + } + if (onBatch) { onBatch(rows.length) } + if (rows.length < batchSize) { break } + offset += batchSize + } +} + +/** + * Turn an async iterable of records into a stream of NDJSON lines. + */ +function ndjson (records) { + return Readable.from((async function * () { + for await (const record of records) { + yield Buffer.from(JSON.stringify(record) + '\n') + } + })()) +} + +function sha256 (buf) { + return crypto.createHash('sha256').update(buf).digest('hex') +} + +module.exports = { + status: { + status: 'notrunning', + progress: 0, + message: '', + startedAt: null, + filename: null, + filePath: null, + fileSize: null + }, + + /** + * Directory holding generated packages. The operator downloads the file from + * here; nothing ever expires it, because no server holds a package for the + * import's sake. + */ + get outputDir () { + return path.resolve(WIKI.ROOTPATH, WIKI.config.dataPath, 'backup') + }, + + /** + * Resolve a package filename to a path inside the backup directory, or null + * if it names anything else. + */ + resolveFile (filename) { + if (!/^[A-Za-z0-9._-]+\.wkbackup$/.test(filename)) { + return null + } + const target = path.resolve(this.outputDir, filename) + if (path.dirname(target) !== this.outputDir) { + return null + } + return target + }, + + /** + * Generate a .wkbackup package. + * + * @param {Object} opts + * @param {string[]} opts.entities Entities to include. + */ + async create (opts) { + const status = this.status + const startedAt = new Date() + + status.status = 'running' + status.progress = 0 + status.message = '' + status.startedAt = startedAt + status.filename = null + status.filePath = null + status.fileSize = null + + const entities = opts.entities + const has = entity => entities.includes(entity) + + // Millisecond precision, so two runs never name the same file and abort + // the earlier package's cleanup over a previous one. + const filename = `wikijs-${startedAt.toISOString().replace(/[-:.]/g, '')}.wkbackup` + const filePath = path.join(this.outputDir, filename) + const tmpPath = path.join(this.outputDir, `.tmp-${startedAt.getTime()}`) + + WIKI.logger.info(`Backup started to ${filePath}`) + WIKI.logger.info(`Entities to include: ${entities.join(', ')}`) + + const zip = new ZipWriter(filePath, { tmpPath }) + + // Blobs referenced by the streams, written last. Content addressing makes + // "have I already written this one?" answerable without any bookkeeping. + const blobs = new Map() + + try { + await fs.ensureDir(this.outputDir) + await zip.open() + + // ----------------------------------------- + // COUNTS + // ----------------------------------------- + // Counting first is what lets the manifest be written as the first entry + // with real counts, while the streams are produced afterwards. + const counts = { + assets: has('assets') ? await this.countOf(WIKI.models.assets) : 0, + comments: has('comments') ? await this.countOf(WIKI.models.comments) : 0, + groups: has('groups') ? await this.countOf(WIKI.models.groups) : 0, + history: has('history') ? await this.countOf(WIKI.models.pageHistory) : 0, + locales: await this.countOf(WIKI.models.locales), + pages: has('pages') ? await this.countOf(WIKI.models.pages) : 0, + tree: has('pages') ? await this.countOfFolders() : 0, + users: has('users') ? await this.countOf(WIKI.models.users) : 0 + } + const blobBytes = has('assets') ? await this.sumAssetBytes() : 0 + + const warnings = await this.collectWarnings(entities) + + // -> Steps, for progress reporting + const steps = ['manifest'] + if (has('settings')) { steps.push('settings') } + steps.push('locales') + if (has('groups')) { steps.push('groups') } + if (has('users')) { steps.push('users') } + steps.push('site') + if (has('navigation')) { steps.push('navigation') } + if (has('pages')) { steps.push('tree', 'pages') } + if (has('history')) { steps.push('history') } + if (has('comments')) { steps.push('comments') } + if (has('assets')) { steps.push('assets') } + steps.push('blobs') + + let doneSteps = 0 + const stepDone = () => { + doneSteps++ + status.progress = Math.min(100, (doneSteps / steps.length) * 100) + } + // Report partial progress through the step currently being written. + const stepPartial = fraction => { + status.progress = Math.min(100, ((doneSteps + Math.min(1, fraction)) / steps.length) * 100) + } + // Track a streamed step of `total` records, batch by batch. + const batchTracker = total => { + let done = 0 + return rows => { + done += rows + stepPartial(total > 0 ? done / total : 1) + } + } + + // ----------------------------------------- + // MANIFEST — always the first entry + // ----------------------------------------- + const streams = {} + const siteStreams = {} + + if (has('settings')) { streams.settings = { schema: 1, path: 'streams/settings.json' } } + streams.locales = { schema: 1, path: 'streams/locales.ndjson', count: counts.locales } + if (has('groups')) { streams.groups = { schema: 1, path: 'streams/groups.ndjson', count: counts.groups } } + if (has('users')) { streams.users = { schema: 1, path: 'streams/users.ndjson', count: counts.users } } + + const sitePath = `sites/${SITE_ID}` + if (has('navigation')) { siteStreams.navigation = { schema: 1, path: `${sitePath}/navigation.json` } } + if (has('pages')) { + siteStreams.tree = { schema: 1, path: `${sitePath}/tree.ndjson`, count: counts.tree } + siteStreams.pages = { schema: 1, path: `${sitePath}/pages.ndjson`, count: counts.pages } + } + if (has('history')) { siteStreams['page-history'] = { schema: 1, path: `${sitePath}/page-history.ndjson`, count: counts.history } } + if (has('comments')) { siteStreams.comments = { schema: 1, path: `${sitePath}/comments.ndjson`, count: counts.comments } } + if (has('assets')) { siteStreams.assets = { schema: 1, path: `${sitePath}/assets.ndjson`, count: counts.assets, blobBytes } } + + const manifest = { + format: 'wkbackup', + formatVersion: FORMAT_VERSION, + createdAt: startedAt.toISOString(), + generator: { + product: 'wiki.js', + version: WIKI.version, + exporter: EXPORTER_VERSION + }, + source: { + kind: 'wikijs2', + instanceId: _.get(WIKI.config, 'telemetry.clientId', null), + baseUrl: WIKI.config.host + }, + streams, + sites: [ + { + id: SITE_ID, + title: WIKI.config.title, + hostname: WIKI.config.host, + locales: _.union([WIKI.config.lang.code], _.get(WIKI.config, 'lang.namespaces', [])), + streams: siteStreams + } + ], + options: { includes: entities }, + warnings + } + await zip.addBuffer('manifest.json', Buffer.from(JSON.stringify(manifest, null, 2))) + stepDone() + + // ----------------------------------------- + // SETTINGS + // ----------------------------------------- + // Written whole, keyed as 2.x keys it. The exporter cannot know which + // version will read the package, so the importer owns the mapping. + if (has('settings')) { + WIKI.logger.info('Backup: writing settings...') + const settings = { + ...WIKI.config, + modules: { + analytics: await WIKI.models.analytics.query(), + authentication: (await WIKI.models.authentication.query()).map(a => ({ + ...a, + domainWhitelist: _.get(a, 'domainWhitelist.v', []), + autoEnrollGroups: _.get(a, 'autoEnrollGroups.v', []) + })), + commentProviders: await WIKI.models.commentProviders.query(), + renderers: await WIKI.models.renderers.query(), + searchEngines: await WIKI.models.searchEngines.query(), + storage: await WIKI.models.storage.query() + }, + apiKeys: await WIKI.models.apiKeys.query().where('isRevoked', false) + } + await zip.addBuffer('streams/settings.json', Buffer.from(JSON.stringify(settings, null, 2))) + stepDone() + } + + // ----------------------------------------- + // LOCALES + // ----------------------------------------- + // Codes only; the string sets themselves are 3.x's own. + WIKI.logger.info('Backup: writing locales...') + const primaryLocale = WIKI.config.lang.code + const activeLocales = _.union([primaryLocale], _.get(WIKI.config, 'lang.namespaces', [])) + const locales = await WIKI.models.locales.query() + .select('code', 'name', 'nativeName', 'isRTL', 'availability', 'createdAt', 'updatedAt') + .orderBy('code') + await zip.addStream('streams/locales.ndjson', ndjson(locales.map(lc => ({ + ...lc, + isPrimary: lc.code === primaryLocale, + isActive: activeLocales.includes(lc.code) + })))) + stepDone() + + // ----------------------------------------- + // GROUPS + // ----------------------------------------- + if (has('groups')) { + WIKI.logger.info('Backup: writing groups...') + const groups = await WIKI.models.groups.query().orderBy('id') + await zip.addStream('streams/groups.ndjson', ndjson(groups)) + stepDone() + } + + // ----------------------------------------- + // USERS + // ----------------------------------------- + // Password hashes and TOTP secrets travel as-is: both versions use + // bcryptjs at cost 12 and plain RFC 6238 TOTP, so nobody has to reset a + // password or re-enrol an authenticator after the migration. + if (has('users')) { + WIKI.logger.info(`Backup: writing ${counts.users} users...`) + const onBatch = batchTracker(counts.users) + await zip.addStream('streams/users.ndjson', ndjson(batched(async (offset, limit) => { + const users = await WIKI.models.users.query() + .orderBy('id').offset(offset).limit(limit) + .withGraphJoined({ groups: true }) + .modifyGraph('groups', builder => builder.select('groups.id', 'groups.name')) + return users.map(usr => ({ + ..._.omit(usr, ['groups']), + groups: usr.groups.map(g => g.id) + })) + }, BATCH_SIZE.users, onBatch))) + stepDone() + } + + // ----------------------------------------- + // SITE + // ----------------------------------------- + WIKI.logger.info('Backup: writing site settings...') + await zip.addBuffer(`${sitePath}/site.json`, Buffer.from(JSON.stringify({ + id: SITE_ID, + title: WIKI.config.title, + hostname: WIKI.config.host, + company: WIKI.config.company, + contentLicense: WIKI.config.contentLicense, + footerOverride: WIKI.config.footerOverride, + logoUrl: WIKI.config.logoUrl, + pageExtensions: WIKI.config.pageExtensions, + locales: { + primary: primaryLocale, + active: activeLocales, + namespacing: _.get(WIKI.config, 'lang.namespacing', false) + }, + theme: WIKI.config.theming, + features: WIKI.config.features, + security: WIKI.config.security, + seo: WIKI.config.seo, + editShortcuts: WIKI.config.editShortcuts, + uploads: WIKI.config.uploads + }, null, 2))) + stepDone() + + // ----------------------------------------- + // NAVIGATION + // ----------------------------------------- + // 2.x keeps one tree per locale; the importer turns each into the + // site-wide menu for its locale. + if (has('navigation')) { + WIKI.logger.info('Backup: writing navigation...') + const navigationRaw = await WIKI.models.navigation.query() + const navigation = navigationRaw.reduce((obj, cur) => { + obj[cur.key] = cur.config + return obj + }, {}) + await zip.addBuffer(`${sitePath}/navigation.json`, Buffer.from(JSON.stringify({ + mode: _.get(WIKI.config, 'nav.mode', 'MIXED'), + trees: navigation + }, null, 2))) + stepDone() + } + + // ----------------------------------------- + // TREE (folders) + // ----------------------------------------- + if (has('pages')) { + WIKI.logger.info(`Backup: writing ${counts.tree} folders...`) + const onBatch = batchTracker(counts.tree) + await zip.addStream(`${sitePath}/tree.ndjson`, ndjson(batched(async (offset, limit) => { + return WIKI.models.knex('pageTree') + .select('id', 'path', 'depth', 'title', 'isPrivate', 'privateNS', 'parent', 'localeCode') + .where('isFolder', true) + .orderBy('id').offset(offset).limit(limit) + }, BATCH_SIZE.tree, onBatch))) + stepDone() + + // ----------------------------------------- + // PAGES + // ----------------------------------------- + // The render is deliberately left behind: 3.x renders differently + // enough that a 2.x render would be silently wrong under it, and a + // stored render says nothing about which pipeline produced it. 3.x + // produces the HTML itself. + WIKI.logger.info(`Backup: writing ${counts.pages} pages...`) + const onPageBatch = batchTracker(counts.pages) + await zip.addStream(`${sitePath}/pages.ndjson`, ndjson(batched(async (offset, limit) => { + const pages = await WIKI.models.pages.query() + .orderBy('id').offset(offset).limit(limit) + .withGraphJoined({ tags: true }) + .modifyGraph('tags', builder => builder.select('tags.tag', 'tags.title')) + return Promise.all(pages.map(page => this.spillContent({ + ..._.omit(page, ['render', 'toc']), + tags: page.tags.map(t => t.tag) + }, tmpPath, blobs))) + }, BATCH_SIZE.pages, onPageBatch))) + stepDone() + } + + // ----------------------------------------- + // PAGE HISTORY + // ----------------------------------------- + if (has('history')) { + WIKI.logger.info(`Backup: writing ${counts.history} page history entries...`) + const onBatch = batchTracker(counts.history) + await zip.addStream(`${sitePath}/page-history.ndjson`, ndjson(batched(async (offset, limit) => { + const versions = await WIKI.models.pageHistory.query() + .orderBy('id').offset(offset).limit(limit) + .withGraphJoined({ tags: true }) + .modifyGraph('tags', builder => builder.select('tags.tag', 'tags.title')) + return Promise.all(versions.map(version => this.spillContent({ + ...version, + tags: version.tags.map(t => t.tag) + }, tmpPath, blobs))) + }, BATCH_SIZE.history, onBatch))) + stepDone() + } + + // ----------------------------------------- + // COMMENTS + // ----------------------------------------- + if (has('comments')) { + WIKI.logger.info(`Backup: writing ${counts.comments} comments...`) + const onBatch = batchTracker(counts.comments) + await zip.addStream(`${sitePath}/comments.ndjson`, ndjson(batched(async (offset, limit) => { + return WIKI.models.comments.query().orderBy('id').offset(offset).limit(limit) + }, BATCH_SIZE.comments, onBatch))) + stepDone() + } + + // ----------------------------------------- + // ASSETS + // ----------------------------------------- + // Metadata only; the bytes go to blobs/, keyed by their own digest. + if (has('assets')) { + WIKI.logger.info(`Backup: writing ${counts.assets} assets...`) + const assetFolders = await WIKI.models.assetFolders.getAllPaths() + const onBatch = batchTracker(counts.assets) + await zip.addStream(`${sitePath}/assets.ndjson`, ndjson(batched(async (offset, limit) => { + const assets = await WIKI.models.knex + .select('assets.*', 'assetData.data') + .from('assets') + .join('assetData', 'assets.id', '=', 'assetData.id') + .orderBy('assets.id').offset(offset).limit(limit) + return assets.map(asset => { + const digest = sha256(asset.data) + if (!blobs.has(digest)) { + blobs.set(digest, { kind: 'asset', id: asset.id }) + } + return { + ..._.omit(asset, ['data']), + // 2.x's `hash` is a digest of the path, not of the contents. + folderPath: (asset.folderId && asset.folderId > 0) ? _.get(assetFolders, asset.folderId, '') : '', + blob: digest + } + }) + }, BATCH_SIZE.assets, onBatch))) + stepDone() + } + + // ----------------------------------------- + // BLOBS + // ----------------------------------------- + // One entry per distinct file, named after its own digest, stored rather + // than deflated: deflating a JPEG costs CPU on both ends to make it very + // slightly larger. Oversized page content spilled here too. + WIKI.logger.info(`Backup: writing ${blobs.size} blobs...`) + let doneBlobs = 0 + for (const [digest, ref] of blobs) { + const data = ref.kind === 'asset' ? _.get( + await WIKI.models.knex('assetData').select('data').where('id', ref.id).first(), + 'data', + null + ) : await fs.readFile(ref.path) + if (!data) { + WIKI.logger.warn(`Backup: blob ${digest} has no data, skipping...`) + continue + } + await zip.addBuffer(`blobs/${digest}`, data, { compress: false }) + doneBlobs++ + stepPartial(doneBlobs / blobs.size) + } + stepDone() + + await zip.close() + + const { size } = await fs.stat(filePath) + status.status = 'success' + status.progress = 100 + status.filename = filename + status.filePath = filePath + status.fileSize = size + WIKI.logger.info(`Backup completed: ${filePath} (${size} bytes)`) + } catch (err) { + WIKI.logger.warn(err) + status.status = 'error' + status.message = err.message + await zip.abort().catch(() => {}) + } finally { + await fs.remove(tmpPath).catch(() => {}) + } + }, + + async countOf (model) { + const result = await model.query().count('* as total').first() + return parseInt(result.total) + }, + + async countOfFolders () { + const result = await WIKI.models.knex('pageTree').where('isFolder', true).count('* as total').first() + return parseInt(result.total) + }, + + async sumAssetBytes () { + const result = await WIKI.models.knex('assets').sum('fileSize as total').first() + return parseInt(_.get(result, 'total', 0)) || 0 + }, + + /** + * Move an oversized `content` field out to a blob, so that no NDJSON line + * forces a multi-megabyte string through the reader's parser. + */ + async spillContent (record, tmpPath, blobs) { + if (!record.content || Buffer.byteLength(record.content, 'utf8') <= BLOB_SPILL_THRESHOLD) { + return record + } + const data = Buffer.from(record.content, 'utf8') + const digest = sha256(data) + if (!blobs.has(digest)) { + const spillPath = path.join(tmpPath, `spill-${digest}`) + await fs.outputFile(spillPath, data) + blobs.set(digest, { kind: 'file', path: spillPath }) + } + return { ...record, content: null, contentBlob: digest } + }, + + /** + * Things this exporter knows will not carry over, written into the manifest + * and shown in the import log before the import starts. + */ + async collectWarnings (entities) { + const warnings = [] + + if (entities.includes('groups')) { + const groups = await WIKI.models.groups.query().select('id', 'name', 'permissions') + const unmappable = new Set() + for (const group of groups) { + for (const permission of (group.permissions || [])) { + if (!WIKI3_PERMISSIONS.includes(permission)) { + unmappable.add(permission) + } + } + } + if (unmappable.size > 0) { + warnings.push(`Permissions with no Wiki.js 3.x equivalent will be dropped on import: ${[...unmappable].sort().join(', ')}.`) + } + } + + if (entities.includes('pages') || entities.includes('history')) { + warnings.push('Page renders are not carried over. Wiki.js 3.x renders pages with its own pipeline, so imported pages are re-rendered in the background.') + } + + if (entities.includes('comments')) { + const provider = await WIKI.models.commentProviders.query().where('isEnabled', true).first() + if (provider && provider.key !== 'default') { + warnings.push(`Comments are handled by the '${provider.key}' provider on this instance; only comments stored by the default provider are included.`) + } + } + + if (entities.includes('assets')) { + warnings.push('Asset folders are carried on each asset as `folderPath`; Wiki.js 2.x has no shared tree for them.') + } + + if (entities.includes('settings')) { + warnings.push('Settings are exported in full, keyed as Wiki.js 2.x keys them. The importer applies the keys that have a 3.x equivalent and reports the rest.') + } + + const missing = ['assets', 'comments', 'groups', 'history', 'navigation', 'pages', 'settings', 'users'].filter(e => !entities.includes(e)) + if (missing.length > 0) { + warnings.push(`Not included in this package: ${missing.join(', ')}.`) + } + + return warnings + } +} diff --git a/server/graph/resolvers/system.js b/server/graph/resolvers/system.js index d28ab04cf..fb640f8da 100644 --- a/server/graph/resolvers/system.js +++ b/server/graph/resolvers/system.js @@ -50,6 +50,17 @@ module.exports = { message: WIKI.system.exportStatus.message, startedAt: WIKI.system.exportStatus.startedAt } + }, + async backupStatus () { + return { + status: WIKI.backup.status.status, + progress: Math.ceil(WIKI.backup.status.progress), + message: WIKI.backup.status.message, + startedAt: WIKI.backup.status.startedAt, + filename: WIKI.backup.status.filename, + filePath: WIKI.backup.status.filePath, + fileSize: WIKI.backup.status.fileSize + } } }, SystemMutation: { @@ -305,6 +316,34 @@ module.exports = { } catch (err) { return graphHelper.generateError(err) } + }, + + /** + * Create a .wkbackup package, for migrating to Wiki.js 3.x + */ + async createBackup (obj, args, context) { + try { + // -> Only one export-type job at a time, as both stream the whole wiki + if (WIKI.backup.status.status === 'running') { + throw new Error('Another backup is already running.') + } + if (WIKI.system.exportStatus.status === 'running') { + throw new Error('An export is already running.') + } + // -> Validate entities + if (args.entities.length < 1) { + throw new Error('Must specify at least 1 entity to include.') + } + // -> Start backup + WIKI.backup.create({ + entities: args.entities + }) + return { + responseResult: graphHelper.generateSuccess('Backup started successfully.') + } + } catch (err) { + return graphHelper.generateError(err) + } } }, SystemInfo: { diff --git a/server/graph/schemas/system.graphql b/server/graph/schemas/system.graphql index a0385a8af..d69f3e975 100644 --- a/server/graph/schemas/system.graphql +++ b/server/graph/schemas/system.graphql @@ -19,6 +19,7 @@ type SystemQuery { info: SystemInfo extensions: [SystemExtension] @auth(requires: ["manage:system"]) exportStatus: SystemExportStatus @auth(requires: ["manage:system"]) + backupStatus: SystemBackupStatus @auth(requires: ["manage:system"]) } # ----------------------------------------------- @@ -53,6 +54,10 @@ type SystemMutation { entities: [String]! path: String! ): DefaultResponse @auth(requires: ["manage:system"]) + + createBackup( + entities: [String]! + ): DefaultResponse @auth(requires: ["manage:system"]) } # ----------------------------------------------- @@ -134,3 +139,13 @@ type SystemExportStatus { message: String startedAt: Date } + +type SystemBackupStatus { + status: String + progress: Int + message: String + startedAt: Date + filename: String + filePath: String + fileSize: Float +} diff --git a/server/helpers/zip.js b/server/helpers/zip.js new file mode 100644 index 000000000..d93a7e3fc --- /dev/null +++ b/server/helpers/zip.js @@ -0,0 +1,389 @@ +const fs = require('fs-extra') +const path = require('path') +const zlib = require('zlib') +const { pipeline } = require('node:stream/promises') +const { Transform } = require('node:stream') + +/* global BigInt */ + +/** + * Minimal ZIP writer for the .wkbackup package format. + * + * The wkbackup spec constrains the writer so that readers can stay small: + * - No data descriptors. CRC32 and both sizes always go in the local header, + * and general purpose bit 3 is never set. A deflated entry therefore has to + * be compressed before its local header can be written, which is why + * addStream() stages the compressed bytes in a temp file first. + * - General purpose bit 11 (UTF-8 filenames) is always set. + * - ZIP64 whenever an entry, the archive, or the entry count requires it. + * - Entry order is preserved exactly as added; it is part of the format. + * - Compression method is chosen per entry: deflate for text, store for + * anything already compressed. + */ + +const LOCAL_SIG = 0x04034b50 +const CENTRAL_SIG = 0x02014b50 +const EOCD_SIG = 0x06054b50 +const ZIP64_EOCD_SIG = 0x06064b50 +const ZIP64_LOCATOR_SIG = 0x07064b50 + +const ZIP64_LIMIT = 0xfffffffe +const ZIP64_COUNT_LIMIT = 0xfffe + +const METHOD_STORE = 0 +const METHOD_DEFLATE = 8 + +const FLAG_UTF8 = 0x0800 + +const VERSION_BASE = 20 +const VERSION_ZIP64 = 45 + +// Version made by: UNIX (3) in the upper byte, spec version in the lower. +const VERSION_MADE_BY = (3 << 8) | VERSION_ZIP64 + +const crcTable = (() => { + const table = new Int32Array(256) + for (let n = 0; n < 256; n++) { + let c = n + for (let k = 0; k < 8; k++) { + c = (c & 1) ? (0xedb88320 ^ (c >>> 1)) : (c >>> 1) + } + table[n] = c + } + return table +})() + +/** + * Incremental CRC32, matching the polynomial ZIP uses. + */ +class CRC32 { + constructor () { + this.crc = -1 + } + + update (buf) { + let crc = this.crc + for (let i = 0; i < buf.length; i++) { + crc = crcTable[(crc ^ buf[i]) & 0xff] ^ (crc >>> 8) + } + this.crc = crc + return this + } + + get value () { + return (this.crc ^ -1) >>> 0 + } +} + +function crc32 (buf) { + return new CRC32().update(buf).value +} + +/** + * MS-DOS date / time pair, as stored in ZIP headers. + */ +function dosDateTime (date) { + const year = date.getFullYear() + if (year < 1980) { + return { date: 0x21, time: 0 } + } + return { + date: ((year - 1980) << 9) | ((date.getMonth() + 1) << 5) | date.getDate(), + time: (date.getHours() << 11) | (date.getMinutes() << 5) | Math.floor(date.getSeconds() / 2) + } +} + +class ZipWriter { + /** + * @param {string} outputPath Path of the archive to create. + * @param {Object} opts + * @param {string} opts.tmpPath Directory used to stage compressed entries. + */ + constructor (outputPath, { tmpPath }) { + this.outputPath = outputPath + this.tmpPath = tmpPath + this.entries = [] + this.offset = 0 + this.stream = null + this.tmpSeq = 0 + } + + async open () { + await fs.ensureDir(path.dirname(this.outputPath)) + await fs.ensureDir(this.tmpPath) + this.stream = fs.createWriteStream(this.outputPath) + await new Promise((resolve, reject) => { + this.stream.once('open', resolve) + this.stream.once('error', reject) + }) + } + + /** + * Write raw bytes to the archive, honouring backpressure. + */ + async _write (buf) { + if (!this.stream.write(buf)) { + await new Promise((resolve, reject) => { + const onDrain = () => { + this.stream.removeListener('error', onError) + resolve() + } + const onError = (err) => { + this.stream.removeListener('drain', onDrain) + reject(err) + } + this.stream.once('drain', onDrain) + this.stream.once('error', onError) + }) + } + this.offset += buf.length + } + + /** + * Add an entry whose CRC32 and both sizes are already known, writing the + * local header followed by `writeBody()`. + */ + async _addEntry (name, { method, crc, compressedSize, uncompressedSize, modifiedAt }, writeBody) { + const nameBuf = Buffer.from(name, 'utf8') + const needsZip64 = compressedSize > ZIP64_LIMIT || uncompressedSize > ZIP64_LIMIT + const { date, time } = dosDateTime(modifiedAt || new Date()) + + // -> Local file header + const extraLen = needsZip64 ? 20 : 0 + const header = Buffer.alloc(30 + extraLen) + header.writeUInt32LE(LOCAL_SIG, 0) + header.writeUInt16LE(needsZip64 ? VERSION_ZIP64 : VERSION_BASE, 4) + header.writeUInt16LE(FLAG_UTF8, 6) + header.writeUInt16LE(method, 8) + header.writeUInt16LE(time, 10) + header.writeUInt16LE(date, 12) + header.writeUInt32LE(crc, 14) + header.writeUInt32LE(needsZip64 ? 0xffffffff : compressedSize, 18) + header.writeUInt32LE(needsZip64 ? 0xffffffff : uncompressedSize, 22) + header.writeUInt16LE(nameBuf.length, 26) + header.writeUInt16LE(extraLen, 28) + if (needsZip64) { + // Both sizes are mandatory in a local header's ZIP64 extra field. + header.writeUInt16LE(0x0001, 30) + header.writeUInt16LE(16, 32) + header.writeBigUInt64LE(BigInt(uncompressedSize), 34) + header.writeBigUInt64LE(BigInt(compressedSize), 42) + } + + const localHeaderOffset = this.offset + await this._write(header) + await this._write(nameBuf) + await writeBody() + + this.entries.push({ + nameBuf, + method, + crc, + compressedSize, + uncompressedSize, + localHeaderOffset, + date, + time, + needsZip64 + }) + } + + /** + * Add an entry from a buffer already held in memory. Used for the manifest, + * the small JSON streams and for blobs, whose bytes come out of the database + * as a single buffer anyway. + * + * @param {string} name Entry path within the archive. + * @param {Buffer} buf Uncompressed contents. + * @param {Object} opts + * @param {boolean} opts.compress Deflate the entry (false stores it as-is). + */ + async addBuffer (name, buf, { compress = true, modifiedAt } = {}) { + const uncompressedSize = buf.length + const crc = crc32(buf) + let method = METHOD_STORE + let body = buf + + if (compress && uncompressedSize > 0) { + const deflated = await new Promise((resolve, reject) => { + zlib.deflateRaw(buf, (err, result) => err ? reject(err) : resolve(result)) + }) + // Storing is better than a deflate that made the entry bigger. + if (deflated.length < uncompressedSize) { + method = METHOD_DEFLATE + body = deflated + } + } + + await this._addEntry(name, { + method, + crc, + compressedSize: body.length, + uncompressedSize, + modifiedAt + }, () => this._write(body)) + } + + /** + * Add an entry produced by a stream of unknown length. + * + * Because the local header may not be followed by a data descriptor, the + * contents are staged to a temp file first: that pass computes the CRC32 and + * both sizes, and the staged bytes are then copied into the archive. + * + * @param {string} name Entry path within the archive. + * @param {Readable} source Stream of uncompressed bytes. + * @param {Object} opts + * @param {boolean} opts.compress Deflate the entry (false stores it as-is). + */ + async addStream (name, source, { compress = true, modifiedAt } = {}) { + const stagePath = path.join(this.tmpPath, `entry-${this.tmpSeq++}.part`) + const hasher = new CRC32() + let uncompressedSize = 0 + + const counter = new Transform({ + transform (chunk, enc, cb) { + hasher.update(chunk) + uncompressedSize += chunk.length + cb(null, chunk) + } + }) + + try { + const stages = [source, counter] + if (compress) { + stages.push(zlib.createDeflateRaw()) + } + stages.push(fs.createWriteStream(stagePath)) + await pipeline(...stages) + + const { size: compressedSize } = await fs.stat(stagePath) + + await this._addEntry(name, { + method: compress ? METHOD_DEFLATE : METHOD_STORE, + crc: hasher.value, + compressedSize, + uncompressedSize, + modifiedAt + }, async () => { + for await (const chunk of fs.createReadStream(stagePath)) { + await this._write(chunk) + } + }) + + return { uncompressedSize, compressedSize } + } finally { + await fs.remove(stagePath) + } + } + + /** + * Write the central directory and close the archive. + */ + async close () { + const centralOffset = this.offset + + for (const entry of this.entries) { + const zip64Fields = [] + if (entry.uncompressedSize > ZIP64_LIMIT) { zip64Fields.push(entry.uncompressedSize) } + if (entry.compressedSize > ZIP64_LIMIT) { zip64Fields.push(entry.compressedSize) } + if (entry.localHeaderOffset > ZIP64_LIMIT) { zip64Fields.push(entry.localHeaderOffset) } + + // The ZIP64 extra field carries only the fields that overflowed, in the + // fixed order: uncompressed size, compressed size, local header offset. + const extraLen = zip64Fields.length > 0 ? 4 + (zip64Fields.length * 8) : 0 + const header = Buffer.alloc(46 + extraLen) + header.writeUInt32LE(CENTRAL_SIG, 0) + header.writeUInt16LE(VERSION_MADE_BY, 4) + header.writeUInt16LE(extraLen > 0 ? VERSION_ZIP64 : VERSION_BASE, 6) + header.writeUInt16LE(FLAG_UTF8, 8) + header.writeUInt16LE(entry.method, 10) + header.writeUInt16LE(entry.time, 12) + header.writeUInt16LE(entry.date, 14) + header.writeUInt32LE(entry.crc, 16) + header.writeUInt32LE(entry.compressedSize > ZIP64_LIMIT ? 0xffffffff : entry.compressedSize, 20) + header.writeUInt32LE(entry.uncompressedSize > ZIP64_LIMIT ? 0xffffffff : entry.uncompressedSize, 24) + header.writeUInt16LE(entry.nameBuf.length, 28) + header.writeUInt16LE(extraLen, 30) + header.writeUInt16LE(0, 32) // file comment length + header.writeUInt16LE(0, 34) // disk number start + header.writeUInt16LE(0, 36) // internal attributes + header.writeUInt32LE(0o644 << 16, 38) // external attributes + header.writeUInt32LE(entry.localHeaderOffset > ZIP64_LIMIT ? 0xffffffff : entry.localHeaderOffset, 42) + if (extraLen > 0) { + header.writeUInt16LE(0x0001, 46) + header.writeUInt16LE(zip64Fields.length * 8, 48) + zip64Fields.forEach((value, idx) => { + header.writeBigUInt64LE(BigInt(value), 50 + (idx * 8)) + }) + } + + await this._write(header) + await this._write(entry.nameBuf) + } + + const centralSize = this.offset - centralOffset + const entryCount = this.entries.length + const needsZip64 = entryCount > ZIP64_COUNT_LIMIT || + centralSize > ZIP64_LIMIT || + centralOffset > ZIP64_LIMIT + + if (needsZip64) { + const zip64Offset = this.offset + + const record = Buffer.alloc(56) + record.writeUInt32LE(ZIP64_EOCD_SIG, 0) + record.writeBigUInt64LE(BigInt(44), 4) // size of this record, minus 12 + record.writeUInt16LE(VERSION_MADE_BY, 12) + record.writeUInt16LE(VERSION_ZIP64, 14) + record.writeUInt32LE(0, 16) // this disk + record.writeUInt32LE(0, 20) // disk with start of central directory + record.writeBigUInt64LE(BigInt(entryCount), 24) + record.writeBigUInt64LE(BigInt(entryCount), 32) + record.writeBigUInt64LE(BigInt(centralSize), 40) + record.writeBigUInt64LE(BigInt(centralOffset), 48) + await this._write(record) + + const locator = Buffer.alloc(20) + locator.writeUInt32LE(ZIP64_LOCATOR_SIG, 0) + locator.writeUInt32LE(0, 4) // disk with the ZIP64 end of central directory + locator.writeBigUInt64LE(BigInt(zip64Offset), 8) + locator.writeUInt32LE(1, 16) // total number of disks + await this._write(locator) + } + + const eocd = Buffer.alloc(22) + eocd.writeUInt32LE(EOCD_SIG, 0) + eocd.writeUInt16LE(0, 4) // this disk + eocd.writeUInt16LE(0, 6) // disk with start of central directory + eocd.writeUInt16LE(entryCount > ZIP64_COUNT_LIMIT ? 0xffff : entryCount, 8) + eocd.writeUInt16LE(entryCount > ZIP64_COUNT_LIMIT ? 0xffff : entryCount, 10) + eocd.writeUInt32LE(centralSize > ZIP64_LIMIT ? 0xffffffff : centralSize, 12) + eocd.writeUInt32LE(centralOffset > ZIP64_LIMIT ? 0xffffffff : centralOffset, 16) + eocd.writeUInt16LE(0, 20) // archive comment length + await this._write(eocd) + + await new Promise((resolve, reject) => { + this.stream.once('error', reject) + this.stream.end(resolve) + }) + this.stream = null + } + + /** + * Abort the archive, discarding whatever was written so far. + */ + async abort () { + if (this.stream) { + await new Promise(resolve => this.stream.end(resolve)) + this.stream = null + } + await fs.remove(this.outputPath) + } +} + +module.exports = { + ZipWriter, + CRC32, + crc32 +} diff --git a/server/master.js b/server/master.js index 0679777ef..007f059df 100644 --- a/server/master.js +++ b/server/master.js @@ -18,6 +18,7 @@ module.exports = async () => { // ---------------------------------------- WIKI.auth = require('./core/auth').init() + WIKI.backup = require('./core/backup') WIKI.lang = require('./core/localization').init() WIKI.mail = require('./core/mail').init() WIKI.system = require('./core/system').init()