feat: import from wiki.js v2 utility

pull/8104/head
NGPixel 5 days ago
parent 0f58846bc2
commit 7c6a3bb590
No known key found for this signature in database

@ -0,0 +1,358 @@
import type { FastifyInstance } from 'fastify'
import { audit } from '../helpers/audit.ts'
import {
IMPORT_CONTENT_KINDS,
MAX_BATCH_BYTES,
MAX_BATCH_RECORDS,
MAX_BLOB_BYTES,
type ImportSessionSite
} from '../models/import.ts'
/**
* Import API Routes
*
* `dev/specs/wkbackup.md` §7 is the design. The package is never uploaded: the browser holds it,
* walks it, and drives these routes a batch at a time.
*
* **Every route takes `manage:system`**, and that is the spec's §5 rather than laziness. An import
* writes groups and puts accounts into them, which is precisely the escalation `elevatedGroupGuard`
* exists to stop; anything narrower would have to re-implement those guards per record and would
* still add up to a route to every permission on the wiki. A dedicated `manage:import` would be a new
* global permission, which is the maintainer's call and not something to introduce alongside a
* feature.
*
* The session is instance-level rather than under `/sites/:siteId`, because half of what it writes
* belongs to no site.
*/
async function routes(app: FastifyInstance) {
/*
A blob is the raw bytes rather than a multipart form: one file per request, named by its own
SHA-256 in the path. The catch-all only claims content types nothing else parses, so the JSON
routes below are unaffected.
*/
app.addContentTypeParser(
'*',
{ parseAs: 'buffer', bodyLimit: MAX_BLOB_BYTES },
(req, body, done) => {
done(null, body)
}
)
/**
* PREFLIGHT A PACKAGE
*/
app.post<{ Body: { manifest: any; siteId: string } }>(
'/import/preflight',
{
config: { permissions: ['manage:system'] },
schema: {
summary: 'Check a backup package before importing it',
description:
'Everything answerable from `manifest.json` alone — the container version, the source kind, that every stream it promises has a path and a schema this wiki can read, and that the target site exists. The browser has the manifest in hand within a second of opening the file, whatever the rest of it weighs, so this is what fails an unreadable eight-gigabyte package immediately rather than forty minutes in.\n\n`errors` stops the import; `warnings` are things it will carry on past and the operator should see first — including anything the exporter itself could not represent.',
tags: ['Import'],
body: {
type: 'object',
properties: {
manifest: { type: 'object', additionalProperties: true },
siteId: { type: 'string', format: 'uuid' }
},
required: ['manifest', 'siteId']
},
response: {
200: { $ref: 'ImportPreflight#' }
}
}
},
async (req) => {
return WIKI.models.importer.preflight(req.body.manifest, req.body.siteId)
}
)
/**
* CREATE AN IMPORT SESSION
*/
app.post<{
Body: {
source: string
sourceInstanceId?: string
sites: ImportSessionSite[]
includes: string[]
overwrite?: boolean
}
}>(
'/import/sessions',
{
config: { permissions: ['manage:system'] },
schema: {
summary: 'Open an import session',
description:
'Settles once what every later batch depends on: which target site each package site lands in, what the operator chose to bring over, and whether an existing record is replaced.\n\nThe session is a row rather than server memory, so a batch may be answered by any instance of an HA set and a browser that was interrupted can ask where it got to.\n\nIt also carries the UUID namespace this import derives record ids in, and that namespace is a function of the source wiki and the target site rather than of the session — so an import that fell over can simply be run again from the top without laying down a second copy of every comment and every history entry.',
tags: ['Import'],
body: {
type: 'object',
properties: {
source: {
type: 'string',
description: '`manifest.source.kind`. Only `wikijs2` is implemented.'
},
sourceInstanceId: {
type: 'string',
description:
'`manifest.source.instanceId`. Part of the namespace this import derives record ids in, so that two different source wikis imported into one site cannot derive the same id for two different comments.'
},
sites: {
type: 'array',
items: {
type: 'object',
properties: {
sourceId: { type: 'string' },
siteId: { type: 'string', format: 'uuid' }
},
required: ['sourceId', 'siteId']
}
},
includes: {
type: 'array',
items: { type: 'string', enum: [...IMPORT_CONTENT_KINDS] },
description:
'Settings are not importable — which 2.x key means what in 3.x is being settled separately.'
},
overwrite: { type: 'boolean', default: false }
},
required: ['source', 'sites', 'includes']
},
response: {
200: { $ref: 'ImportSession#' }
}
}
},
async (req) => {
const session = await WIKI.models.importer.createSession({
source: req.body.source,
sourceInstanceId: req.body.sourceInstanceId,
sites: req.body.sites,
includes: req.body.includes,
overwrite: req.body.overwrite === true,
actorId: req.session?.user?.id ?? null
})
await audit(req, 'admin', 'startImport', {
sessionId: session.id,
source: session.source,
includes: session.includes,
overwrite: session.overwrite,
siteIds: session.sites.map((entry) => entry.siteId)
})
return session
}
)
/**
* READ AN IMPORT SESSION
*/
app.get<{ Params: { sessionId: string } }>(
'/import/sessions/:sessionId',
{
config: { permissions: ['manage:system'] },
schema: {
summary: 'Read an import session',
description:
'What has been written so far, per stream, and everything the import could not carry. This is what a browser that was interrupted reads to find its place: records are keyed so that replaying one is an upsert, so resuming is replaying from the last counted batch.',
tags: ['Import'],
params: {
type: 'object',
properties: { sessionId: { type: 'string', format: 'uuid' } },
required: ['sessionId']
},
response: {
200: { $ref: 'ImportSession#' }
}
}
},
async (req, reply) => {
const session = await WIKI.models.importer.getSession(req.params.sessionId)
if (!session) {
return reply.notFound('No such import session.')
}
return session
}
)
/**
* UPLOAD A BLOB
*/
app.post<{ Params: { sessionId: string; digest: string } }>(
'/import/sessions/:sessionId/blobs/:digest',
{
config: { permissions: ['manage:system'] },
schema: {
summary: 'Upload one of a package’s blobs',
description: `The body is the file itself, not a multipart form — send the bytes with \`Content-Type: application/octet-stream\`. At most ${MAX_BLOB_BYTES / 1024 / 1024 / 1024} GB.\n\nThe path segment is the file's SHA-256, which is also what names it inside the package, and it is **verified** rather than trusted: the name being the checksum is the whole of what content addressing establishes.\n\nBlobs are uploaded before the metadata that references them, and once each however many records point at the same bytes. They wait in a staging directory that \`finish\` removes.`,
tags: ['Import'],
consumes: ['application/octet-stream'],
params: {
type: 'object',
properties: {
sessionId: { type: 'string', format: 'uuid' },
digest: { type: 'string', pattern: '^[0-9a-f]{64}$' }
},
required: ['sessionId', 'digest']
},
response: {
200: {
description: 'Blob staged',
type: 'object',
properties: {
ok: { type: 'boolean' },
bytes: { type: 'integer' }
}
}
}
}
},
async (req, reply) => {
await WIKI.models.importer.requireOpenSession(req.params.sessionId)
const data = req.body
if (!Buffer.isBuffer(data) || data.length < 1) {
return reply.badRequest('No file was sent.')
}
const { bytes } = await WIKI.models.importer.putBlob(
req.params.sessionId,
req.params.digest,
data
)
return { ok: true, bytes }
}
)
/**
* INGEST AN INSTANCE-WIDE STREAM
*/
app.post<{ Params: { sessionId: string; stream: string }; Body: { records: any[] } }>(
'/import/sessions/:sessionId/streams/:stream',
{
config: { permissions: ['manage:system'] },
// -> Above the instance-wide JSON limit; see `MAX_BATCH_BYTES`
bodyLimit: MAX_BATCH_BYTES,
schema: {
summary: 'Write a batch of instance-wide records',
description: `Users, groups and locales belong to the wiki rather than to any one site, which is why the session is not itself under a site.\n\nAt most ${MAX_BATCH_RECORDS} records and ${MAX_BATCH_BYTES / 1024 / 1024} MB per request. A stream whose content kind the session was not opened for is refused rather than ignored, and a stream name this wiki has no reader for is refused too — a batch that was quietly dropped is a migration that quietly lost something.`,
tags: ['Import'],
params: {
type: 'object',
properties: {
sessionId: { type: 'string', format: 'uuid' },
stream: { type: 'string', enum: ['locales', 'groups', 'users'] }
},
required: ['sessionId', 'stream']
},
body: {
type: 'object',
properties: {
records: { type: 'array', items: { type: 'object', additionalProperties: true } }
},
required: ['records']
},
response: {
200: { $ref: 'ImportBatch#' }
}
}
},
async (req) => {
return WIKI.models.importer.ingest({
sessionId: req.params.sessionId,
stream: req.params.stream,
records: req.body.records
})
}
)
/**
* INGEST A SITE STREAM
*/
app.post<{
Params: { sessionId: string; siteId: string; stream: string }
Body: { records: any[] }
}>(
'/import/sessions/:sessionId/sites/:siteId/streams/:stream',
{
config: { permissions: ['manage:system'] },
/*
Above the instance-wide JSON limit, and this is the route that needs it: a batch of pages or
of page history is a batch of whole documents, where a batch of anything else is a batch of
rows. See `MAX_BATCH_BYTES`.
*/
bodyLimit: MAX_BATCH_BYTES,
schema: {
summary: 'Write a batch of records belonging to one site',
description: `The site is a target site on THIS instance, one the session was opened for — a package site the operator did not map is never read.\n\nAt most ${MAX_BATCH_RECORDS} records and ${MAX_BATCH_BYTES / 1024 / 1024} MB per request; a batch of pages or of page history is a batch of whole documents, so the byte ceiling is usually what a caller meets first.\n\nOrder matters between these streams and is the caller's to keep: folders, then blobs, then pages, then the history and comments that hang off them, then the assets those blobs belong to, then navigation. A record whose page was not imported is skipped rather than failing its batch.`,
tags: ['Import'],
params: {
type: 'object',
properties: {
sessionId: { type: 'string', format: 'uuid' },
siteId: { type: 'string', format: 'uuid' },
stream: {
type: 'string',
enum: ['tree', 'pages', 'page-history', 'assets', 'comments', 'navigation']
}
},
required: ['sessionId', 'siteId', 'stream']
},
body: {
type: 'object',
properties: {
records: { type: 'array', items: { type: 'object', additionalProperties: true } }
},
required: ['records']
},
response: {
200: { $ref: 'ImportBatch#' }
}
}
},
async (req) => {
return WIKI.models.importer.ingest({
sessionId: req.params.sessionId,
stream: req.params.stream,
siteId: req.params.siteId,
records: req.body.records
})
}
)
/**
* FINISH AN IMPORT SESSION
*/
app.post<{ Params: { sessionId: string } }>(
'/import/sessions/:sessionId/finish',
{
config: { permissions: ['manage:system'] },
schema: {
summary: 'Close an import session',
description:
'Drops the staged blobs and reports what happened. Nothing is rebuilt: every model the records went through did its own bookkeeping as it wrote — the tree entry, the storage-target copies, the search index, the queued render.\n\nThe two counts at the end are what an operator has to know about a 2.x import. A page arrives with no HTML, because a 2.x render is 2.x’s output and would be wrong here in ways nothing could later detect, so it carries a placeholder until the render queue reaches it — one headless browser, one page at a time. `unrenderable` is the pages this wiki has no server-side renderer for at all; those keep their placeholder until somebody opens and saves them, which is a to-do list rather than a failure.',
tags: ['Import'],
params: {
type: 'object',
properties: { sessionId: { type: 'string', format: 'uuid' } },
required: ['sessionId']
},
response: {
200: { $ref: 'ImportSummary#' }
}
}
},
async (req) => {
const summary = await WIKI.models.importer.finishSession(req.params.sessionId)
await audit(req, 'admin', 'finishImport', {
sessionId: req.params.sessionId,
progress: summary.progress,
pendingRenders: summary.pendingRenders,
unrenderable: summary.unrenderable
})
return summary
}
)
}
export default routes

@ -18,6 +18,7 @@ async function routes(app: FastifyInstance) {
await import('./schemas/group.ts').then((m) => m.registerSchemas(app)) await import('./schemas/group.ts').then((m) => m.registerSchemas(app))
await import('./schemas/hook.ts').then((m) => m.registerSchemas(app)) await import('./schemas/hook.ts').then((m) => m.registerSchemas(app))
await import('./schemas/icon.ts').then((m) => m.registerSchemas(app)) await import('./schemas/icon.ts').then((m) => m.registerSchemas(app))
await import('./schemas/import.ts').then((m) => m.registerSchemas(app))
await import('./schemas/locale.ts').then((m) => m.registerSchemas(app)) await import('./schemas/locale.ts').then((m) => m.registerSchemas(app))
await import('./schemas/mail.ts').then((m) => m.registerSchemas(app)) await import('./schemas/mail.ts').then((m) => m.registerSchemas(app))
await import('./schemas/metrics.ts').then((m) => m.registerSchemas(app)) await import('./schemas/metrics.ts').then((m) => m.registerSchemas(app))
@ -44,6 +45,7 @@ async function routes(app: FastifyInstance) {
app.register(import('./groups.ts'), { prefix: '/groups' }) app.register(import('./groups.ts'), { prefix: '/groups' })
app.register(import('./hooks.ts'), { prefix: '/hooks' }) app.register(import('./hooks.ts'), { prefix: '/hooks' })
app.register(import('./icons.ts'), { prefix: '/icons' }) app.register(import('./icons.ts'), { prefix: '/icons' })
app.register(import('./import.ts'))
app.register(import('./locales.ts'), { prefix: '/locales' }) app.register(import('./locales.ts'), { prefix: '/locales' })
app.register(import('./mail.ts'), { prefix: '/mail' }) app.register(import('./mail.ts'), { prefix: '/mail' })
app.register(import('./navigation.ts')) app.register(import('./navigation.ts'))

@ -2,6 +2,7 @@ import { validate as uuidValidate } from 'uuid'
import type { FastifyInstance, FastifyRequest } from 'fastify' import type { FastifyInstance, FastifyRequest } from 'fastify'
import type { PageActor, PageInput } from '../models/pages.ts' import type { PageActor, PageInput } from '../models/pages.ts'
import type { RulePageRef } from '../helpers/pageRules.ts' import type { RulePageRef } from '../helpers/pageRules.ts'
import { PAGE_PERMISSIONS } from '../models/groups.ts'
import { import {
SEARCH_ORDER_BY, SEARCH_ORDER_BY,
SEARCH_TAGS_MATCH, SEARCH_TAGS_MATCH,
@ -90,30 +91,6 @@ export function actorFrom(req: FastifyRequest): PageActor | null {
*/ */
const PASSWORD_BYPASS = ['write:pages', 'manage:pages', 'manage:system'] const PASSWORD_BYPASS = ['write:pages', 'manage:pages', 'manage:system']
/**
* Every page permission a rule can grant, i.e. the whole set `manage:system` amounts to. Mirrors the
* page rules offered in the group editor, and is what the interface asks about per path.
*/
const PAGE_PERMISSIONS = [
'read:pages',
'write:pages',
'review:pages',
'manage:pages',
'delete:pages',
'write:tags',
'write:styles',
'write:scripts',
'read:source',
'read:history',
'read:assets',
'write:assets',
'manage:assets',
'read:comments',
'write:comments',
'manage:comments',
'manage:navigation'
]
export function mayBypassPassword(req: FastifyRequest): boolean { export function mayBypassPassword(req: FastifyRequest): boolean {
const permissions = req.apiKey?.permissions ?? req.session?.permissions ?? [] const permissions = req.apiKey?.permissions ?? req.session?.permissions ?? []
return PASSWORD_BYPASS.some((permission) => permissions.includes(permission)) return PASSWORD_BYPASS.some((permission) => permissions.includes(permission))

@ -0,0 +1,97 @@
import type { FastifyInstance } from 'fastify'
/**
* Shared schemas for the import API.
*
* The record streams themselves are deliberately NOT schema'd here. What a `.wkbackup` line looks
* like is `dev/specs/wkbackup.md` §12's business, it varies per stream and per `schema` version, and
* the models are what read it — validating it twice, in a shape that has to be kept in step with a
* document in another repository, would buy a worse error message than the importer's own.
*/
export function registerSchemas(app: FastifyInstance) {
app.addSchema({
$id: 'ImportPreflight',
type: 'object',
properties: {
ok: { type: 'boolean', description: 'False when `errors` is non-empty.' },
errors: {
type: 'array',
items: { type: 'string' },
description: 'Reasons the package cannot be imported at all.'
},
warnings: {
type: 'array',
items: { type: 'string' },
description:
'Things the import will carry on past, including anything the exporter itself reported it could not represent.'
}
}
})
app.addSchema({
$id: 'ImportSession',
type: 'object',
properties: {
id: { type: 'string', format: 'uuid' },
namespace: {
type: 'string',
format: 'uuid',
description: 'The UUIDv5 namespace this import derives its record ids in.'
},
source: { type: 'string' },
sites: {
type: 'array',
items: {
type: 'object',
properties: {
sourceId: { type: 'string' },
siteId: { type: 'string', format: 'uuid' }
}
}
},
includes: { type: 'array', items: { type: 'string' } },
overwrite: { type: 'boolean' },
state: { type: 'string', enum: ['open', 'finished', 'failed'] },
progress: {
type: 'object',
additionalProperties: { type: 'integer' },
description: 'Records written so far, per stream.'
},
warnings: { type: 'array', items: { type: 'string' } },
createdAt: { type: 'string' },
updatedAt: { type: 'string' }
}
})
app.addSchema({
$id: 'ImportBatch',
type: 'object',
properties: {
imported: { type: 'integer' },
skipped: {
type: 'integer',
description:
'Records the import passed over — already present with overwrite off, missing a page to hang off, or incomplete. Never a failure.'
},
warnings: { type: 'array', items: { type: 'string' } }
}
})
app.addSchema({
$id: 'ImportSummary',
type: 'object',
properties: {
progress: { type: 'object', additionalProperties: { type: 'integer' } },
warnings: { type: 'array', items: { type: 'string' } },
pendingRenders: {
type: 'integer',
description: 'Imported pages waiting on the render queue.'
},
unrenderable: {
type: 'integer',
description:
'Imported pages this wiki has no server-side renderer for. They keep their placeholder until somebody opens and saves them.'
}
}
})
}

@ -0,0 +1,27 @@
CREATE TYPE "importSessionState" AS ENUM('open', 'finished', 'failed');--> statement-breakpoint
CREATE TABLE "importIdMap" (
"sessionId" uuid,
"entity" varchar(32),
"sourceId" varchar(255),
"targetId" uuid NOT NULL,
CONSTRAINT "importIdMap_pkey" PRIMARY KEY("sessionId","entity","sourceId")
);
--> statement-breakpoint
CREATE TABLE "importSessions" (
"id" uuid PRIMARY KEY DEFAULT gen_random_uuid(),
"namespace" uuid NOT NULL,
"source" varchar(32) NOT NULL,
"sites" jsonb DEFAULT '[]' NOT NULL,
"includes" jsonb DEFAULT '[]' NOT NULL,
"overwrite" boolean DEFAULT false NOT NULL,
"state" "importSessionState" DEFAULT 'open'::"importSessionState" NOT NULL,
"progress" jsonb DEFAULT '{}' NOT NULL,
"warnings" jsonb DEFAULT '[]' NOT NULL,
"actorId" uuid,
"createdAt" timestamp DEFAULT now() NOT NULL,
"updatedAt" timestamp DEFAULT now() NOT NULL
);
--> statement-breakpoint
CREATE INDEX "importSessions_createdAt_idx" ON "importSessions" ("createdAt");--> statement-breakpoint
ALTER TABLE "importIdMap" ADD CONSTRAINT "importIdMap_sessionId_importSessions_id_fkey" FOREIGN KEY ("sessionId") REFERENCES "importSessions"("id") ON DELETE CASCADE;--> statement-breakpoint
ALTER TABLE "importSessions" ADD CONSTRAINT "importSessions_actorId_users_id_fkey" FOREIGN KEY ("actorId") REFERENCES "users"("id") ON DELETE SET NULL;

File diff suppressed because it is too large Load Diff

@ -411,6 +411,92 @@ export const icons = pgTable(
(table) => [primaryKey({ columns: [table.prefix, table.name] })] (table) => [primaryKey({ columns: [table.prefix, table.name] })]
) )
// IMPORT SESSIONS ---------------------
/**
* One run of **Administration → Utilities → Import from Wiki.js 2.x**, from the moment the operator
* presses Start until the package has been walked.
*
* A row rather than memory, for two reasons that both come from the import living in a browser tab:
* in an HA set the next batch is answered by a different instance, and a tab that closed has to be
* able to say where it got to. See `dev/specs/wkbackup.md` §7.
*
* Everything the handlers need to agree about across thousands of requests is here — which target
* site each package site lands in, what the operator ticked, and the group mapping the user records
* resolve through — so a batch carries only its own records.
*/
export const importSessionStateEnum = pgEnum('importSessionState', ['open', 'finished', 'failed'])
export const importSessions = pgTable(
'importSessions',
{
id: uuid().primaryKey().defaultRandom(),
/**
* The UUIDv5 namespace the derived ids of this import are built in.
*
* **Derived, not random** — from the source wiki and the site being imported into, so that the
* same package run into the same site a second time derives the same ids and upserts, while two
* operators importing two different packages cannot collide. It has to survive the SESSION and
* not just outlive a batch: an import that fell over is re-run from the top, which would
* otherwise mean a second copy of every comment and every history entry — the two record kinds
* with no natural key to match on. See `dev/specs/wkbackup.md` §4.
*/
namespace: uuid().notNull(),
/** `manifest.source.kind`. Only `wikijs2` is implemented. */
source: varchar({ length: 32 }).notNull(),
/** `[{ sourceId, siteId }]` — each package site paired with the target site it lands in. */
sites: jsonb().notNull().default([]),
/** Which content kinds the operator ticked. A stream for anything absent is refused. */
includes: jsonb().notNull().default([]),
overwrite: boolean().notNull().default(false),
state: importSessionStateEnum().notNull().default('open'),
/** `{ <stream>: <records written> }`, which is what a resumed tab reads to find its place. */
progress: jsonb().notNull().default({}),
/** Everything the import could not carry, in the order it was found. Shown in the log. */
warnings: jsonb().notNull().default([]),
/**
* Who is running it. Null once that account is gone, which costs nothing: a finished session is
* a receipt, and the audit log is where the act itself is recorded.
*
* Also what `usersStream` compares each record against — the account running the import is never
* written to, whatever `overwrite` says.
*/
actorId: uuid().references(() => users.id, { onDelete: 'set null' }),
createdAt: timestamp().notNull().defaultNow(),
updatedAt: timestamp().notNull().defaultNow()
},
(table) => [index('importSessions_createdAt_idx').on(table.createdAt)]
)
/**
* What a record from the source wiki became here — its 2.x integer id paired with the row it is now.
*
* A table rather than a blob on the session, because the things that have to be looked up this way
* are unbounded: a page's author is a 2.x user id, a comment names its page by 2.x page id, and a
* wiki has as many of those as it has users and pages. A jsonb column rewritten once per batch would
* be megabytes of write amplification by the end of a large import, where this is an insert per
* record and one indexed read per batch.
*
* It exists because the package speaks 2.x's ids and this wiki matches on natural keys — a user by
* email, a page by path. Those two answers have to be joined up somewhere, and only for the entities
* something actually references: users, pages and groups.
*
* Rows go with the session, which is what stops this becoming a permanent record of somebody's old
* instance.
*/
export const importIdMap = pgTable(
'importIdMap',
{
sessionId: uuid()
.notNull()
.references(() => importSessions.id, { onDelete: 'cascade' }),
/** `user`, `page` or `group`. */
entity: varchar({ length: 32 }).notNull(),
/** The id the source wiki knew it by, as text — 2.x numbers them, a 3.x package would not. */
sourceId: varchar({ length: 255 }).notNull(),
targetId: uuid().notNull()
},
(table) => [primaryKey({ columns: [table.sessionId, table.entity, table.sourceId] })]
)
// JOB HISTORY ------------------------- // JOB HISTORY -------------------------
export const jobHistoryStateEnum = pgEnum('jobHistoryState', [ export const jobHistoryStateEnum = pgEnum('jobHistoryState', [
'active', 'active',

@ -1517,7 +1517,9 @@
"admin.utilities.wikijs2Import.comments": "Comments", "admin.utilities.wikijs2Import.comments": "Comments",
"admin.utilities.wikijs2Import.groups": "Groups", "admin.utilities.wikijs2Import.groups": "Groups",
"admin.utilities.wikijs2Import.history": "History", "admin.utilities.wikijs2Import.history": "History",
"admin.utilities.wikijs2Import.inProgress": "Import in progress...",
"admin.utilities.wikijs2Import.navigation": "Navigation", "admin.utilities.wikijs2Import.navigation": "Navigation",
"admin.utilities.wikijs2Import.logStarted": "Reading {file}...",
"admin.utilities.wikijs2Import.noArchive": "No archive selected", "admin.utilities.wikijs2Import.noArchive": "No archive selected",
"admin.utilities.wikijs2Import.options": "Options", "admin.utilities.wikijs2Import.options": "Options",
"admin.utilities.wikijs2Import.overwrite": "Overwrite on conflict", "admin.utilities.wikijs2Import.overwrite": "Overwrite on conflict",
@ -1527,7 +1529,6 @@
"admin.utilities.wikijs2Import.progress": "Progress Log", "admin.utilities.wikijs2Import.progress": "Progress Log",
"admin.utilities.wikijs2Import.progressEmpty": "The import has not been started. Progress will be reported here.", "admin.utilities.wikijs2Import.progressEmpty": "The import has not been started. Progress will be reported here.",
"admin.utilities.wikijs2Import.selectArchive": "Select archive...", "admin.utilities.wikijs2Import.selectArchive": "Select archive...",
"admin.utilities.wikijs2Import.settings": "Settings",
"admin.utilities.wikijs2Import.source": "Destination", "admin.utilities.wikijs2Import.source": "Destination",
"admin.utilities.wikijs2Import.start": "Start Import", "admin.utilities.wikijs2Import.start": "Start Import",
"admin.utilities.wikijs2Import.subtitle": "Migrate content from a Wiki.js 2.x backup", "admin.utilities.wikijs2Import.subtitle": "Migrate content from a Wiki.js 2.x backup",

@ -868,13 +868,21 @@ class Assets {
} }
/** /**
* Take a file a storage target already holds into the wiki, without writing it anywhere. * Take a file into the wiki that the wiki has no record of.
* *
* The other direction from an upload: the bytes are already in place — restored from a backup, * The other direction from an upload, and it serves two callers that differ in one thing: whether
* dropped into the folder by another tool — and what is missing is the wiki's record of them. A file * anybody is already holding the bytes.
* the wiki has no entry for is adopted where it lies rather than written out again, which is why *
* nothing is dispatched to the storage layer for it. Any *other* target configured to hold that kind * - **A storage target's import** walks a folder the bytes are already in — restored from a backup,
* will not have a copy until its own export action runs. * dropped there by another tool — and what is missing is only the wiki's record of them. Adopted
* where they lie, with nothing dispatched to the storage layer, which is the default.
* - **A package import** (`.wkbackup`) reads them out of a file in the operator's browser. Nobody
* holds them, so `dispatch` writes them to every target the site stores that kind on, exactly as
* an upload would. Without it the asset gets a row and a thumbnail and no bytes anywhere — a file
* the file manager lists, previews, and cannot open.
*
* Either way, any *other* target configured to hold that kind and not written to here will not have
* a copy until its own export action runs.
* *
* `overwrite` turns the case the wiki DOES have an entry for from a skip into a replacement, for a * `overwrite` turns the case the wiki DOES have an entry for from a skip into a replacement, for a
* restore where the folder is meant to be the authority. That one is dispatched, and has to be: the * restore where the folder is meant to be the authority. That one is dispatched, and has to be: the
@ -887,6 +895,8 @@ class Assets {
* loose file in a folder may take over. * loose file in a folder may take over.
* *
* @param overwrite Replace an asset already at this path instead of leaving it alone * @param overwrite Replace an asset already at this path instead of leaving it alone
* @param dispatch Write the bytes to the site's storage targets, for a caller holding bytes nobody
* else has. The replacement path dispatches regardless, since it has to.
* @returns The asset, or null for a file this passed over * @returns The asset, or null for a file this passed over
*/ */
async adoptStoredFile({ async adoptStoredFile({
@ -896,7 +906,8 @@ class Assets {
fileName, fileName,
data, data,
authorId, authorId,
overwrite overwrite,
dispatch
}: { }: {
siteId: string siteId: string
locale: string locale: string
@ -905,6 +916,7 @@ class Assets {
data: Buffer data: Buffer
authorId: string authorId: string
overwrite?: boolean overwrite?: boolean
dispatch?: boolean
}): Promise<Asset | null> { }): Promise<Asset | null> {
const safeName = sanitizeFileName(fileName) const safeName = sanitizeFileName(fileName)
if (!safeName) { if (!safeName) {
@ -973,8 +985,12 @@ class Assets {
siteId, siteId,
meta: { fileSize: data.length, fileExt, mimeType, ...dimensionMeta(dimensions) } meta: { fileSize: data.length, fileExt, mimeType, ...dimensionMeta(dimensions) }
}) })
// -> Read off the row rather than from the caller: the folder may have just been created, and a
// name that was taken took the next free one
const importedFolderPath = decodeTreePath(entry.folderPath ?? '') ?? ''
try { try {
// -> The metadata row goes in before the bytes, since the database target writes them into it
await WIKI.db.insert(assetsTable).values({ await WIKI.db.insert(assetsTable).values({
id: entry.id, id: entry.id,
fileName: entry.fileName, fileName: entry.fileName,
@ -987,12 +1003,28 @@ class Assets {
authorId, authorId,
siteId siteId
}) })
if (dispatch) {
await WIKI.models.storage.putAsset(
{
id: entry.id,
siteId,
actorId: authorId,
locale,
folderPath: importedFolderPath,
fileName: entry.fileName,
kind,
fileSize: data.length
},
data
)
}
} catch (err) { } catch (err) {
// -> Nothing points at these now, and leaving them would show a file the site cannot serve
await WIKI.db.delete(assetsTable).where(eq(assetsTable.id, entry.id))
await WIKI.db.delete(treeTable).where(eq(treeTable.id, entry.id)) await WIKI.db.delete(treeTable).where(eq(treeTable.id, entry.id))
throw err throw err
} }
const importedFolderPath = decodeTreePath(entry.folderPath ?? '') ?? ''
WIKI.models.hooks.emit('asset:upload', { WIKI.models.hooks.emit('asset:upload', {
id: entry.id, id: entry.id,
fileName: entry.fileName, fileName: entry.fileName,

@ -121,6 +121,8 @@ export const AUDIT_ACTIONS = {
'rebuildSearchIndex', 'rebuildSearchIndex',
'rebuildPageLinks', 'rebuildPageLinks',
'installExtension', 'installExtension',
'startImport',
'finishImport',
'updateApiState', 'updateApiState',
'updateMetricsState', 'updateMetricsState',
'updateScimState', 'updateScimState',

@ -39,6 +39,61 @@ export const ELEVATED_PERMISSIONS = [
SYSTEM_PERMISSION SYSTEM_PERMISSION
] as const ] as const
/**
* Every GLOBAL permission a group may hold — the site-wide list, bound to no path.
*
* Here rather than in the admin screen that offers them or the route hook that checks them, because
* it is a vocabulary rather than a UI concern: nothing validates a permission string on the way in,
* so a name that is not on this list simply never matches anything, and whoever needs to ask "is
* this a real permission" needs one list to ask. `helpers/pageRules.ts` documents the other kind.
*
* Adding a name is the maintainer's call — see CLAUDE.md.
*/
export const GLOBAL_PERMISSIONS = [
'access:admin',
'read:users',
'write:users',
'manage:users',
'read:groups',
'write:groups',
'manage:groups',
'read:audit',
'read:metrics',
'manage:theme',
'manage:storage',
'manage:sites',
'read:webhooks',
'manage:webhooks',
'manage:scim',
SYSTEM_PERMISSION
] as const
/**
* Every PAGE permission a rule may grant — bound to a path, a locale and a site.
*
* The other half of the vocabulary above, and deliberately disjoint from it: `config.permissions` on
* a route reads the global list only, so naming one of these there refuses everybody.
*/
export const PAGE_PERMISSIONS = [
'read:pages',
'write:pages',
'review:pages',
'manage:pages',
'delete:pages',
'write:tags',
'write:styles',
'write:scripts',
'read:source',
'read:history',
'read:assets',
'write:assets',
'manage:assets',
'read:comments',
'write:comments',
'manage:comments',
'manage:navigation'
]
/** Whether a permission list carries any of `ELEVATED_PERMISSIONS`. */ /** Whether a permission list carries any of `ELEVATED_PERMISSIONS`. */
export function isElevated(permissions: readonly string[]): boolean { export function isElevated(permissions: readonly string[]): boolean {
return permissions.some((permission) => return permissions.some((permission) =>

File diff suppressed because it is too large Load Diff

@ -12,6 +12,7 @@ import { flags } from './flags.ts'
import { groups } from './groups.ts' import { groups } from './groups.ts'
import { hooks } from './hooks.ts' import { hooks } from './hooks.ts'
import { icons } from './icons.ts' import { icons } from './icons.ts'
import { importer } from './import.ts'
import { jobs } from './jobs.ts' import { jobs } from './jobs.ts'
import { locales } from './locales.ts' import { locales } from './locales.ts'
import { mail } from './mail.ts' import { mail } from './mail.ts'
@ -50,6 +51,7 @@ export default {
groups, groups,
hooks, hooks,
icons, icons,
importer,
jobs, jobs,
locales, locales,
mail, mail,

@ -40,6 +40,8 @@ export const SYSTEM_SCHEDULE: SystemScheduleEntry[] = [
// -> Daily, off the hour: the retention is expressed in days, so nothing is gained by looking // -> Daily, off the hour: the retention is expressed in days, so nothing is gained by looking
// more often than once a day // more often than once a day
{ task: 'purgeAuditLog', cron: '20 0 * * *' }, { task: 'purgeAuditLog', cron: '20 0 * * *' },
// -> Same reasoning, and the cutoff is in hours but measured in days
{ task: 'purgeImportSessions', cron: '25 0 * * *' },
{ task: 'purgeRateLimits', cron: '10 * * * *' }, { task: 'purgeRateLimits', cron: '10 * * * *' },
{ task: 'updateLocales', cron: '0 0 * * *' }, { task: 'updateLocales', cron: '0 0 * * *' },
// -> Every minute, and the task decides which sites are actually due: the interval is a per-site // -> Every minute, and the task decides which sites are actually due: the interval is a per-site

@ -2663,13 +2663,14 @@ class Pages {
// -> A restore should not report every page as written today. Applied after the fact because the // -> A restore should not report every page as written today. Applied after the fact because the
// two dates are not something an API client may set, only something a file can carry back. // two dates are not something an API client may set, only something a file can carry back.
if (createdAt || updatedAt) { if (createdAt || updatedAt) {
await WIKI.db const dates = {
.update(pagesTable)
.set({
...(createdAt ? { createdAt } : {}), ...(createdAt ? { createdAt } : {}),
...(updatedAt ? { updatedAt } : {}) ...(updatedAt ? { updatedAt } : {})
}) }
.where(eq(pagesTable.id, page.id)) await WIKI.db.update(pagesTable).set(dates).where(eq(pagesTable.id, page.id))
// -> The tree row keeps its own pair, and those are the ones a folder listing and the file
// manager show. Left alone they said the whole of an imported wiki was created today.
await WIKI.db.update(treeTable).set(dates).where(eq(treeTable.id, page.id))
// -> Writing the page already put a copy on every target holding pages, stamped with the dates // -> Writing the page already put a copy on every target holding pages, stamped with the dates
// it had for the moment it existed with the wrong ones // it had for the moment it existed with the wrong ones
const restored = this.toStoragePage( const restored = this.toStoragePage(
@ -2748,9 +2749,17 @@ class Pages {
permissions permissions
) )
/*
`updatedAt` is deliberately not touched. It says when the page last CHANGED, and a render does
not change it: the same source went through the pipeline again, because the markdown config
moved or because the page arrived without HTML. Bumping it here re-dated every page an import
brought in — the render lands seconds later and overwrites the date the package carried — and
told the sitemap a page had been modified when nothing about it had. For a save it made no
difference either way, since `updatePage` has already set it a moment earlier.
*/
const updated = await WIKI.db const updated = await WIKI.db
.update(pagesTable) .update(pagesTable)
.set({ render, toc, searchContent: text, updatedAt: sql`now()` }) .set({ render, toc, searchContent: text })
.where(and(eq(pagesTable.id, id), eq(pagesTable.siteId, siteId))) .where(and(eq(pagesTable.id, id), eq(pagesTable.siteId, siteId)))
.returning({ locale: pagesTable.locale }) .returning({ locale: pagesTable.locale })

@ -0,0 +1,24 @@
/**
* Drop the import sessions nobody came back for, and the blobs they staged.
*
* An import lives in a browser tab (`dev/specs/wkbackup.md` §9 is honest about what that costs), so a
* tab that was closed halfway leaves an open session behind with however many gigabytes of uploaded
* assets waiting in its staging directory for metadata that is never coming. Nothing else will ever
* look at them.
*
* Daily and off the hour, alongside the other retention tasks: the cutoff is expressed in hours but
* measured in days, so there is nothing to gain by looking more often.
*/
export async function task(): Promise<void> {
WIKI.logger.info('Purging abandoned import sessions...')
try {
const purged = await WIKI.models.importer.sweepSessions()
WIKI.logger.info(`Purged ${purged} abandoned import sessions: [ COMPLETED ]`)
} catch (err: any) {
WIKI.logger.error('Purging abandoned import sessions: [ FAILED ]')
WIKI.logger.error(err.message)
throw err
}
}

@ -1,7 +1,12 @@
# `.wkbackup` — backup and migration package format # `.wkbackup` — backup and migration package format
**Status:** proposed. Nothing in this document is implemented. The open questions it carried are **Status:** the v3 importer is implemented for `source.kind: "wikijs2"`; the v2 exporter that writes
answered — [§10](#10-decisions-taken). the packages it reads is not, and neither is the 3.x export or a `wikijs3` restore. The open questions
this document carried are answered — [§10](#10-decisions-taken) — and what implementing the importer
changed is recorded in [§11](#11-amendments-made-while-implementing-the-v3-importer).
**The record shapes are [§12](#12-the-wikijs2-stream-records)**, and that section is written FROM the
2.x exporter (`server/core/backup.js` in the 2.x repository) rather than the other way round: it ships
already, so it is the authority and this reader is held to it.
**Covers:** Wiki.js 2.x → 3.x migration, and 3.x → 3.x backup/restore. **Covers:** Wiki.js 2.x → 3.x migration, and 3.x → 3.x backup/restore.
A `.wkbackup` is one file holding the whole content of a wiki — pages, history, assets, users, groups, A `.wkbackup` is one file holding the whole content of a wiki — pages, history, assets, users, groups,
@ -494,3 +499,362 @@ the section that acts on it.
6. **There is no retention question.** The package is a download that the browser reads from the 6. **There is no retention question.** The package is a download that the browser reads from the
operator's own disk; no server ever holds it for the import — [§6](#6-the-v2-exporter), operator's own disk; no server ever holds it for the import — [§6](#6-the-v2-exporter),
[§7](#7-the-v3-importer). [§7](#7-the-v3-importer).
---
## 11. Amendments made while implementing the v3 importer
The importer described in [§7](#7-the-v3-importer) is implemented for `source.kind: "wikijs2"`
(`backend/models/import.ts`, `backend/api/import.ts`, `frontend/src/helpers/wkbackup/`). Three things
in this document did not survive contact with the code, and the reasons are here rather than in a
commit message.
### Blobs are uploaded before the metadata that references them
[§7](#order-of-operations) ordered a site's streams `pages → page-history → assets → comments`, with
assets as "metadata, then blobs". That order cannot be run. An asset is created by
`assets.adoptStoredFile`, which takes the bytes — there is no half-created asset for a later upload
to fill in — and the same section requires a blob to be uploaded **once** however many records
reference it, which a blob-per-asset request cannot express either.
So a blob is staged before the batch that references it is posted:
```
locales → groups → users
→ per site: tree → pages → page-history → assets → comments → navigation
→ finish
```
and for each batch of any stream, the blobs its records name (`blob`, `contentBlob`) are uploaded
first, once each across the whole import. Demand-driven rather than a phase of its own, because
`blobs/` holds every file in the wiki and an import of pages alone has no business moving eight
gigabytes of images nobody asked for. Content addressing is what makes that safe to decide a batch at
a time — "have I already sent this?" is a question about the digest and nothing else.
Staging is `<dataPath>/cache/import/<sessionId>/<sha256>`, which is a cache in the sense
`<dataPath>/cache/blocks` is: derived, disposable, and removed by `finish` (and by a sweep of
sessions older than `SESSION_MAX_AGE_HOURS`, for the import that never finished). Every blob's
SHA-256 is verified on arrival — the name *is* the checksum, so a reader that did not check it would
be trusting the uploader about the one thing the naming scheme exists to establish.
### The reading is not in a Web Worker of the importer's own
[§7](#frontend) called for one. `@zip.js/zip.js` already runs inflation in a pool of its own workers
by default, which is the part that would jank the main thread; what is left on it is a line split and
a `JSON.parse` per batch of 500 records, between `await`ed uploads that dominate the wall clock by
orders of magnitude. A second worker layer would have to re-export the session's cookie-authenticated
uploads across a message port to buy nothing measurable.
### Identity is natural keys where 2.x has one, derived UUIDs where it does not
[§4](#4-identity-derive-uuids-do-not-map-them) derives every id. That is right for a 3.x restore,
where the package and the target speak the same schema; for a 2.x migration most records have a
natural key in 3.x and using it is what makes the import *converge* with a wiki that is already
running rather than shadowing it:
| Record | Matched on |
| --- | --- |
| user | `email`, lowercased — see below |
| group | `name` |
| page | `siteId` + `locale` + `path` |
| folder | `siteId` + `locale` + `path` |
| asset | `siteId` + `locale` + folder path + file name |
| navigation | `siteId` + `locale` — the site-wide menu |
| comment, page history | derived: `uuidv5(session.namespace, "<site>:<entity>:<sourceId>")` |
The last row is what [§4](#4-identity-derive-uuids-do-not-map-them) is for and keeps its property:
those two are the records with nothing in them that identifies a row, so replaying a batch upserts
rather than duplicates.
**The namespace is derived, not issued.** §4 has the server mint one per session, which makes a batch
safe to *retry* and nothing more. There is no resume button: an import that fell over is re-run from
the top, and with a fresh namespace each run that means a second copy of every comment and every
history entry — precisely the two kinds with no natural key to save them. So it is
`uuidv5(source-kind | source-instance-id | each package-site > target-site, fixed root)`, which is
the same value next week and on a rebuilt instance, and a different one for another wiki's package or
another target site.
**A user is matched on the email address and nothing else.** Not the display name, which is not
unique and which people change; not the 2.x row id, which means nothing here. An address that already
has an account IS that person, so the import fills in what the account is missing rather than making
a second one beside it.
**The account running the import is never written to**, whatever `overwrite` says. It is in practice
the root administrator, it is the session the rest of the import is authenticated by, and the package
was produced by a wiki whose copy of that address may carry a different password hash, a different
2FA secret, or `isActive: false`. Overwriting it mid-import is how an operator locks themselves out
of the instance they are migrating into, with thousands of records still to write. It is reported in
the log rather than passed over quietly.
**2.x system groups are mapped, not created.** `Administrators` (id 1) and `Guests` (id 2) exist in
3.x already, with rules of their own that are not 2.x's to overwrite; their *memberships* are what
carries, onto `systemIds.administratorsGroupId` and `systemIds.guestsGroupId`.
**And the ids have to be joined up somewhere.** A record from 2.x references other records by 2.x's
integers — a page's `authorId`, a comment's `pageId`, a user's `groups` — while this wiki matches on
natural keys, so neither end can resolve the other alone. `importIdMap` is a row per user, page and
group, written as the owning stream runs and read in bulk by the streams after it. A table rather
than a blob on the session because users and pages are unbounded: a jsonb column rewritten once per
batch is megabytes of write amplification by the end of a large import. Rows go with the session,
which is what stops it becoming a permanent record of somebody's old instance.
### Settings are not imported
[§10](#10-decisions-taken) item 5 stands as the design, and `streams/settings.json` is read far
enough to be reported in the log, but nothing is applied: which 2.x key means what in 3.x is a table
of its own and is being settled separately. The overlay offers no Settings tick.
---
## 12. The `wikijs2` stream records
**This section is written from the exporter, not the other way round.** It is
`server/core/backup.js` in the 2.x repository, and the importer here is read against it. An earlier
revision of this section guessed at the field names and every one of the guesses was wrong, which
cost a migration that reported success and imported nothing — so the rule is that a change to this
section follows a change to that file, never precedes it.
### A record is a 2.x database row
The exporter selects a table and writes it out. It renames nothing and resolves nothing, which is the
right division of labour — it cannot know which version will read the package — and it means three
things hold everywhere below:
- **The field names are 2.x's own column names.** `localeCode`, not `locale`. `editorKey`, not
`editor`. `filename`, not `fileName`.
- **The ids are 2.x's own integers**, and every cross-reference is one: a page's `authorId`, a
comment's `pageId`, a user's `groups`. They mean nothing in a 3.x database, so the importer keeps an
`importIdMap` row per user, page and group as the owning stream runs, and the streams after it
resolve through that — see [§11](#identity-is-natural-keys-where-2x-has-one-derived-uuids-where-it-does-not).
- **Columns that mean nothing here come along anyway** — `hash`, `privateNS`, `isPrivate`,
`contentType`, `render` on a comment. They are ignored. Nothing is refused for carrying more than
this reader knows about, which is what lets a 2.x point release add a column without breaking
every import.
Every stream is `schema: 1`. A record missing a field this reader actually needs is skipped with a
warning that **names the missing field and lists the keys the record did carry** — the single most
useful thing a log can say when a package and a reader disagree, and the thing whose absence made the
first failure unreadable.
### `streams/locales.ndjson`
Rows of 2.x's `locales`, plus two flags the exporter adds:
```json
{ "code": "en", "name": "English", "nativeName": "English", "isRTL": false,
"availability": 100, "isPrimary": true, "isActive": true }
```
Only `code` is read. The importer checks each against the locales installed here and reports the ones
that are not; it installs nothing, because that is a download from a third party in the middle of
somebody's migration.
### `streams/groups.ndjson`
Rows of 2.x's `groups`:
```json
{ "id": 5, "name": "Editors", "isSystem": false,
"permissions": ["read:pages", "write:pages", "manage:api"],
"pageRules": [
{ "id": "rul3", "path": "docs", "roles": ["read:pages"], "match": "START",
"deny": false, "locales": ["en"] }
],
"redirectOnLogin": "/", "createdAt": "…", "updatedAt": "…" }
```
2.x keeps one permission list where 3.x keeps two, so `permissions` carries page permissions alongside
global ones. A page permission is dropped from the group's global list **in silence** — it is not a
meaningless name, it is a name that lives in `rules` here, and `pageRules` is where it arrives. A name
in neither vocabulary is dropped **loudly**, which in practice means `manage:api`. The exporter warns
about those in `manifest.warnings` too, having checked against the same list.
`deny` becomes `mode: "DENY"`, otherwise `"ALLOW"`; 2.x has no `FORCEALLOW`. `match` carries across
unchanged for the kinds both versions have (`START`, `EXACT`, `END`, `REGEX`, `TAG`) and a rule using
anything else is dropped, named. Every imported rule is **scoped to the target site** — a rule from a
wiki that had one site must not start speaking for the others here.
`Administrators` (id 1) and `Guests` (id 2) are mapped onto 3.x's own rather than created; only their
memberships carry.
### `streams/users.ndjson`
Rows of 2.x's `users`, with `groups` flattened to a list of group ids:
```json
{ "id": 3, "email": "ana@example.com", "name": "Ana",
"providerKey": "local",
"password": "$2a$12$…",
"tfaIsActive": true, "tfaSecret": "JBSWY3DPEHPK3PXP", "mustChangePwd": false,
"isSystem": false, "isActive": true, "isVerified": true,
"localeCode": "en", "jobTitle": "", "location": "", "timezone": "",
"groups": [5], "createdAt": "…", "updatedAt": "…" }
```
`password` and `tfaSecret` carry over verbatim into `users.auth[<localAuthId>]`
([§10](#10-decisions-taken) item 2) — but only for `providerKey: "local"`. An account that
authenticated somewhere else arrives with no credentials at all and the log says which strategy has
to be configured before they can sign in.
`localeCode` becomes `prefs.locale`; `jobTitle`, `location` and `timezone` become `meta`. A record with
`isSystem` is skipped — 3.x seeds its own guest.
Matching is on **`email`, lowercased**, and the account running the import is never written to. Both
are [§11](#identity-is-natural-keys-where-2x-has-one-derived-uuids-where-it-does-not).
### `sites/<site>/tree.ndjson`
Rows of 2.x's `pageTree` where `isFolder`:
```json
{ "id": 10, "path": "docs/guides", "depth": 2, "title": "Guides",
"isPrivate": false, "privateNS": null, "parent": 10, "localeCode": "en" }
```
Pages and assets create the folders they need on the way in, so this stream is for the two things
they cannot supply: the title somebody gave a folder, and a folder with nothing in it.
### `sites/<site>/pages.ndjson`
Rows of 2.x's `pages` minus `render` and `toc`, with `tags` flattened to a list of strings:
```json
{ "id": 42, "path": "docs/intro", "localeCode": "en",
"title": "Introduction", "description": "The first page",
"editorKey": "markdown", "contentType": "markdown",
"isPublished": true, "isPrivate": false,
"tags": ["onboarding"],
"content": "# Hello…", "contentBlob": null,
"authorId": 3, "creatorId": 3,
"createdAt": "…", "updatedAt": "…" }
```
**No render**, which is the exporter's own decision and the right one
([§8](#8-the-bottleneck-that-shapes-the-schedule)). `contentBlob` is a `blobs/<sha256>` digest and
replaces `content` (set to `null`) when the source is over 1 MiB.
`editorKey` maps `markdown → markdown`, `asciidoc → asciidoc`, `redirect → redirect`, and 2.x's two
HTML editors (`ckeditor`, `code`) onto **markdown** rather than 3.x's `visual`: markdown is configured
with `allowHTML`, so a body of HTML renders as it stood, where `visual` is a WYSIWYG over markdown and
would be handed a document it does not parse. Every page that takes that road is named in the log.
A 2.x redirection's `content` is the target path; 3.x holds a redirection as JSON, so it is rewritten.
### `sites/<site>/page-history.ndjson`
Rows of 2.x's `pageHistory`, with `tags`:
```json
{ "id": 900, "pageId": 42, "path": "docs/intro", "localeCode": "en",
"action": "updated", "title": "Introduction", "description": "",
"editorKey": "markdown", "isPublished": true,
"content": "…", "contentBlob": null,
"authorId": 3, "versionDate": "…", "createdAt": "…" }
```
The page is resolved by **`pageId` through the id map**, not by the path on the record. A history row
records the path the page had AT THE TIME — that is what the column is for, here as much as in 2.x —
so every version written before a page was renamed carries its old path, and matching on that throws
away the history of exactly the pages whose history is most worth having. `pageId` survives a rename.
The path is still what gets **stored**, since "where was this page when this version was written" is
what a history list shows and what finding a deleted page again depends on. Falling back to
`localeCode` + `path` covers a package whose history is imported over pages that are already here.
History must therefore be written after the pages, which the stream order already requires. A row with
nothing to attach to is skipped and **said so**, with a count and examples rather than a line each:
either the pages were not imported, or the page was deleted in the source wiki, which keeps its
history deliberately so that a deleted page can be recovered.
The row's own id is derived, since history has no natural key, so a re-run updates rather than
duplicates.
### `sites/<site>/assets.ndjson`
Rows of 2.x's `assets` minus the bytes, plus two fields the exporter adds:
```json
{ "id": 77, "filename": "logo.png", "ext": ".png", "kind": "image",
"mime": "image/png", "fileSize": 4211, "metadata": {},
"folderId": 2, "folderPath": "docs/img", "blob": "e3b0c442…",
"authorId": 3, "createdAt": "…", "updatedAt": "…" }
```
**There is no locale on an asset.** 2.x files assets in a tree of their own that has no locales in it,
so there is nothing to read and nothing to guess: they go in the target site's **primary locale**,
which is where a one-language 3.x site keeps everything anyway.
`folderPath` is resolved by the exporter from `folderId`, because 2.x's asset folders are a table the
package does not otherwise carry. `blob` names an entry under `blobs/` that must be uploaded first
([§11](#blobs-are-uploaded-before-the-metadata-that-references-them)). `mime` and `fileSize` are read
for the log and nothing else — 3.x derives both from the name and the bytes it actually received.
### `sites/<site>/comments.ndjson`
Rows of 2.x's `comments`:
```json
{ "id": 5, "pageId": 42,
"content": "Looks good to me.", "render": "<p>…</p>",
"name": "Ana", "email": "ana@example.com", "ip": "10.0.0.1",
"authorId": 3, "createdAt": "…", "updatedAt": "…" }
```
**The page is named by 2.x page id and nothing else** — there is no path on the row — so the pages
stream has to have run and recorded what each one became. A comment whose page was not imported is
skipped and named.
`name`, `email` and `ip` are the commenter's own, kept by 2.x for guests and signed-in authors alike.
`authorId` resolves to an account here; when it does not, the comment stays a guest comment under that
name and address, which is what it already was for anybody who commented without signing in. `render`
is ignored — 3.x renders a comment in the browser at display time.
2.x comments are flat, so nothing carries a `parentId`. The row's own id is derived, as page history's
is and for the same reason.
### `sites/<site>/navigation.json`
2.x's `navigation` table, keyed by its `key` column:
```json
{ "mode": "MIXED",
"trees": {
"site": [
{ "locale": "en",
"items": [
{ "id": "n1", "kind": "link", "label": "Home", "icon": "mdi-home",
"targetType": "home", "target": "/",
"visibilityMode": "all", "visibilityGroups": [] }
] },
{ "locale": "fr", "items": [ … ] }
]
} }
```
**Neither half of that key/value pair is what it looks like**, and reading either one wrong produces a
sidebar that imports, reports success and then draws nothing.
- **The key is not a locale.** 2.x keeps its one site-wide tree under the literal `site` — its
navigation model is `idColumn = 'key'` and fetches with `findOne('key', 'site')`. Nothing about the
key says anything about language.
- **The value is not a list of items.** It is a list of **per-locale trees**, `[{ locale, items }]`,
which is what that same model iterates to fill its `nav:sidebar:<locale>` cache. The locale comes
from inside each entry, so the key carries no information the reader needs at all.
One exception, and it is 2.x's own: a config whose **first entry has a `kind`** is the pre-2.3 flat
format, a bare array of items, which 2.x reads as the `en` tree. The exporter writes the column raw,
so a wiki that has not re-saved its navigation since 2.2 still exports that shape.
Each tree becomes **the 3.x site-wide menu for its locale** ([§6](#mapping-notes)) — the sidebar every
page in that locale inherits. The importer logs the item count per locale, because this is a mapping
whose failure is silent.
`kind` maps `link → link`, `header → header`, `divider → separator`, **with no default**: an item that
cannot say what it is is dropped and named, rather than becoming a blank link. `targetType` maps
`home` and `page` onto a path and `external`/`externalblank` onto the URL, the latter with
`openInNewWindow`. A `mdi-home` or `las la-cog` icon is rewritten to its Iconify spelling rather than
stored as a webfont name 3.x no longer ships. `visibilityMode: "restricted"` carries its groups
through the id map.
### `sites/<site>/site.json` and `streams/settings.json`
Read far enough to be reported, not applied — [§11](#settings-are-not-imported).

@ -15,6 +15,7 @@
"@twemoji/api": "17.0.2", "@twemoji/api": "17.0.2",
"@xterm/addon-fit": "0.11.0", "@xterm/addon-fit": "0.11.0",
"@xterm/xterm": "6.0.0", "@xterm/xterm": "6.0.0",
"@zip.js/zip.js": "2.17.0",
"browser-fs-access": "0.38.0", "browser-fs-access": "0.38.0",
"clipboard": "2.0.11", "clipboard": "2.0.11",
"es-toolkit": "1.50.0", "es-toolkit": "1.50.0",
@ -3745,6 +3746,17 @@
"addons/*" "addons/*"
] ]
}, },
"node_modules/@zip.js/zip.js": {
"version": "2.17.0",
"resolved": "https://registry.npmjs.org/@zip.js/zip.js/-/zip.js-2.17.0.tgz",
"integrity": "sha512-WxvSsJvlI+5v4f1gWjGfqvzOB8XkAIughO+iJUPOJR6OSpUkFauBzi1G4xS2MmK1fxHGuDAqq9Lnpq/ri7BWuw==",
"license": "BSD-3-Clause",
"engines": {
"bun": ">=0.7.0",
"deno": ">=1.0.0",
"node": ">=18.0.0"
}
},
"node_modules/acorn": { "node_modules/acorn": {
"version": "8.18.0", "version": "8.18.0",
"license": "MIT", "license": "MIT",

@ -24,6 +24,7 @@
"@twemoji/api": "17.0.2", "@twemoji/api": "17.0.2",
"@xterm/addon-fit": "0.11.0", "@xterm/addon-fit": "0.11.0",
"@xterm/xterm": "6.0.0", "@xterm/xterm": "6.0.0",
"@zip.js/zip.js": "2.17.0",
"browser-fs-access": "0.38.0", "browser-fs-access": "0.38.0",
"clipboard": "2.0.11", "clipboard": "2.0.11",
"es-toolkit": "1.50.0", "es-toolkit": "1.50.0",

@ -5,7 +5,7 @@
never waits on (or depends on) the icon service. Regenerate with `npm run icons` after adding or never waits on (or depends on) the icon service. Regenerate with `npm run icons` after adding or
removing an icon; `check-icons.mjs` fails the build if this drifts. removing an icon; `check-icons.mjs` fails the build if this drifts.
276 icons. 277 icons.
*/ */
export const BUNDLED_ICONS = { export const BUNDLED_ICONS = {
"la:angle-down": {"body":"<path fill=\"currentColor\" d=\"M4.219 10.781L2.78 12.22l12.5 12.5l.719.687l.719-.687l12.5-12.5l-1.438-1.438L16 22.562z\"/>","width":32,"height":32}, "la:angle-down": {"body":"<path fill=\"currentColor\" d=\"M4.219 10.781L2.78 12.22l12.5 12.5l.719.687l.719-.687l12.5-12.5l-1.438-1.438L16 22.562z\"/>","width":32,"height":32},
@ -113,6 +113,7 @@ export const BUNDLED_ICONS = {
"la:pen-fancy": {"body":"<path fill=\"currentColor\" d=\"M23.813 4.031c-1.09 0-2.2.418-3.032 1.25L11.5 14.563l-5.469 2.093l-.531.219l-.094.563L4 26.843l-.188 1.343L5.157 28l9.407-1.406l.562-.094l.219-.531l1.969-5.188l.5-.468l9-9c1.613-1.614 1.644-4.204.125-5.876l-.125-.156a4.22 4.22 0 0 0-3-1.25zm0 1.969c.562 0 1.125.25 1.593.719c.938.937.938 2.25 0 3.187l-5.031 5.031l-3.188-3.187l5.032-5.031C22.687 6.25 23.25 6 23.812 6zm-8.063 7.188l3.188 3.187l-1.813 1.813L13.937 15zm-3.344 3.156h.031l3.22 3.218l-1.97 5.157l-5.843.843l2.687-2.687c.055.004.102.031.156.031c.883 0 1.626-.71 1.626-1.593s-.743-1.625-1.626-1.625s-1.593.742-1.593 1.625c0 .054.027.101.031.156l-2.688 2.687l.844-5.843z\"/>","width":32,"height":32}, "la:pen-fancy": {"body":"<path fill=\"currentColor\" d=\"M23.813 4.031c-1.09 0-2.2.418-3.032 1.25L11.5 14.563l-5.469 2.093l-.531.219l-.094.563L4 26.843l-.188 1.343L5.157 28l9.407-1.406l.562-.094l.219-.531l1.969-5.188l.5-.468l9-9c1.613-1.614 1.644-4.204.125-5.876l-.125-.156a4.22 4.22 0 0 0-3-1.25zm0 1.969c.562 0 1.125.25 1.593.719c.938.937.938 2.25 0 3.187l-5.031 5.031l-3.188-3.187l5.032-5.031C22.687 6.25 23.25 6 23.812 6zm-8.063 7.188l3.188 3.187l-1.813 1.813L13.937 15zm-3.344 3.156h.031l3.22 3.218l-1.97 5.157l-5.843.843l2.687-2.687c.055.004.102.031.156.031c.883 0 1.626-.71 1.626-1.593s-.743-1.625-1.626-1.625s-1.593.742-1.593 1.625c0 .054.027.101.031.156l-2.688 2.687l.844-5.843z\"/>","width":32,"height":32},
"la:pen-nib": {"body":"<path fill=\"currentColor\" d=\"m22 3.586l-4.521 4.521l-6.74 1.928a4.06 4.06 0 0 0-2.802 2.65L3.86 25.274l1.434 1.434l1.434 1.434l12.593-4.08a4.06 4.06 0 0 0 2.64-2.788l1.929-6.748L28.414 10zm0 2.828L25.586 10L23 12.586L19.414 9zm-4.29 3.711l4.165 4.164l-1.842 6.45a2.06 2.06 0 0 1-1.336 1.421L7.69 25.725l5.795-5.795A2 2 0 0 0 14 20a2 2 0 0 0 0-4a2 2 0 0 0-1.93 2.516L6.275 24.31l3.563-11a2.07 2.07 0 0 1 1.437-1.348l6.436-1.838z\"/>","width":32,"height":32}, "la:pen-nib": {"body":"<path fill=\"currentColor\" d=\"m22 3.586l-4.521 4.521l-6.74 1.928a4.06 4.06 0 0 0-2.802 2.65L3.86 25.274l1.434 1.434l1.434 1.434l12.593-4.08a4.06 4.06 0 0 0 2.64-2.788l1.929-6.748L28.414 10zm0 2.828L25.586 10L23 12.586L19.414 9zm-4.29 3.711l4.165 4.164l-1.842 6.45a2.06 2.06 0 0 1-1.336 1.421L7.69 25.725l5.795-5.795A2 2 0 0 0 14 20a2 2 0 0 0 0-4a2 2 0 0 0-1.93 2.516L6.275 24.31l3.563-11a2.07 2.07 0 0 1 1.437-1.348l6.436-1.838z\"/>","width":32,"height":32},
"la:play": {"body":"<path fill=\"currentColor\" d=\"M9 5.156v21.688l1.531-1L25.844 16L10.53 6.156zm2 3.657L22.156 16L11 23.188z\"/>","width":32,"height":32}, "la:play": {"body":"<path fill=\"currentColor\" d=\"M9 5.156v21.688l1.531-1L25.844 16L10.53 6.156zm2 3.657L22.156 16L11 23.188z\"/>","width":32,"height":32},
"la:play-circle": {"body":"<path fill=\"currentColor\" d=\"M16 4C9.383 4 4 9.383 4 16s5.383 12 12 12s12-5.383 12-12S22.617 4 16 4m0 2c5.535 0 10 4.465 10 10s-4.465 10-10 10S6 21.535 6 16S10.465 6 16 6m-4 3.125v13.75L13.5 22l9-5.125L24 16l-1.5-.875l-9-5.125zm2 3.438L19.969 16L14 19.438z\"/>","width":32,"height":32},
"la:plug": {"body":"<path fill=\"currentColor\" d=\"m22 3.594l-4 3.969l-2.281-2.282L14.28 6.72l.75.75l-5.125 5.125a3.126 3.126 0 0 0 0 4.406l1.844 1.844l-7.469 7.437l1.44 1.438l7.437-7.469L15 22.094a3.126 3.126 0 0 0 4.406 0l5.125-5.125l.75.75l1.438-1.438L24.437 14l3.97-4L27 8.594l-4 3.969L19.437 9l3.97-4zm-5.563 5.281l6.688 6.688L18 20.686c-.387.387-1.207.387-1.594 0l-5.093-5.093c-.387-.387-.387-1.207 0-1.594z\"/>","width":32,"height":32}, "la:plug": {"body":"<path fill=\"currentColor\" d=\"m22 3.594l-4 3.969l-2.281-2.282L14.28 6.72l.75.75l-5.125 5.125a3.126 3.126 0 0 0 0 4.406l1.844 1.844l-7.469 7.437l1.44 1.438l7.437-7.469L15 22.094a3.126 3.126 0 0 0 4.406 0l5.125-5.125l.75.75l1.438-1.438L24.437 14l3.97-4L27 8.594l-4 3.969L19.437 9l3.97-4zm-5.563 5.281l6.688 6.688L18 20.686c-.387.387-1.207.387-1.594 0l-5.093-5.093c-.387-.387-.387-1.207 0-1.594z\"/>","width":32,"height":32},
"la:plus": {"body":"<path fill=\"currentColor\" d=\"M15 5v10H5v2h10v10h2V17h10v-2H17V5z\"/>","width":32,"height":32}, "la:plus": {"body":"<path fill=\"currentColor\" d=\"M15 5v10H5v2h10v10h2V17h10v-2H17V5z\"/>","width":32,"height":32},
"la:plus-circle": {"body":"<path fill=\"currentColor\" d=\"M16 3C8.832 3 3 8.832 3 16s5.832 13 13 13s13-5.832 13-13S23.168 3 16 3m0 2c6.086 0 11 4.914 11 11s-4.914 11-11 11S5 22.086 5 16S9.914 5 16 5m-1 5v5h-5v2h5v5h2v-5h5v-2h-5v-5z\"/>","width":32,"height":32}, "la:plus-circle": {"body":"<path fill=\"currentColor\" d=\"M16 3C8.832 3 3 8.832 3 16s5.832 13 13 13s13-5.832 13-13S23.168 3 16 3m0 2c6.086 0 11 4.914 11 11s-4.914 11-11 11S5 22.086 5 16S9.914 5 16 5m-1 5v5h-5v2h5v5h2v-5h5v-2h-5v-5z\"/>","width":32,"height":32},

@ -1,6 +1,13 @@
<template> <template>
<w-layout view="hHh lpR fFf" container> <w-layout view="hHh lpR fFf" container>
<w-header class="card-header px-4 py-2"> <!--
The bar and the progress strip under it are two rows of the ONE header element, rather than the
strip being a sibling below it: `WLayout` is a grid of named areas and an unassigned child
would be auto-placed into the empty left drawer cell, a column wide. Stacking them here is also
what makes the strip flush -- it spans the header's own width, inside its border.
-->
<w-header class="card-header import-header">
<div class="flex items-center px-4 py-2">
<w-icon name="img:/_assets/icons/ultraviolet-database-restore.svg" left size="md" /> <w-icon name="img:/_assets/icons/ultraviolet-database-restore.svg" left size="md" />
<div> <div>
<span>{{ t('admin.utilities.importWikijs2') }}</span> <span>{{ t('admin.utilities.importWikijs2') }}</span>
@ -27,6 +34,14 @@
icon="la:times" icon="la:times"
@click="close" /> @click="close" />
</w-btn-group> </w-btn-group>
</div>
<w-linear-progress
v-if="showProgress"
size="10px"
:striped="isRunning"
:color="state.phase === 'failed' ? 'negative' : 'positive'"
:value="state.progress"
:aria-label="t('admin.utilities.wikijs2Import.progress')" />
</w-header> </w-header>
<w-page-container> <w-page-container>
<w-page class="p-4"> <w-page class="p-4">
@ -54,6 +69,7 @@
style="min-width: 200px" style="min-width: 200px"
v-model="state.siteId" v-model="state.siteId"
:options="siteOptions" :options="siteOptions"
:disabled="isRunning"
:aria-label="t('admin.utilities.wikijs2Import.targetSite')" /> :aria-label="t('admin.utilities.wikijs2Import.targetSite')" />
</w-item-section> </w-item-section>
</w-item> </w-item>
@ -87,7 +103,9 @@
v-model="state.content[entry.key]" v-model="state.content[entry.key]"
color="primary" color="primary"
:label="entry.label" :label="entry.label"
:disabled="entry.parent ? !state.content[entry.parent] : false" /> :disabled="
isRunning || (entry.parent ? !state.content[entry.parent] : false)
" />
</div> </div>
</div> </div>
</div> </div>
@ -114,6 +132,7 @@
<w-toggle <w-toggle
v-model="state.overwrite" v-model="state.overwrite"
color="negative" color="negative"
:disabled="isRunning"
checked-icon="la:check" checked-icon="la:check"
unchecked-icon="la:times" unchecked-icon="la:times"
:aria-label="t('admin.utilities.wikijs2Import.overwrite')" /> :aria-label="t('admin.utilities.wikijs2Import.overwrite')" />
@ -144,22 +163,35 @@
color="primary" color="primary"
icon="la:file-upload" icon="la:file-upload"
:label="t('admin.utilities.wikijs2Import.selectArchive')" :label="t('admin.utilities.wikijs2Import.selectArchive')"
:disabled="isRunning"
@click="pickArchive" /> @click="pickArchive" />
</w-item-section> </w-item-section>
</w-item> </w-item>
</w-card> </w-card>
<!-- -> Below the cards rather than inside the last one: it acts on all four --> <!-- -> Below the cards rather than inside the last one: it acts on all four -->
<!--
Spelt out in the slot rather than through `icon`/`label`, because `WBtn`'s own
`loading` prop hides the content behind a centred spinner -- which is the right
treatment for a button whose label would be a lie mid-request, and the wrong one here:
the label is what says what is happening.
-->
<w-btn <w-btn
unelevated unelevated
size="lg" size="lg"
class="w-full" class="w-full"
color="positive" color="positive"
text-color="white" text-color="white"
icon="la:play-circle" :disabled="!canStart || isRunning"
:label="t('admin.utilities.wikijs2Import.start')" @click="startImport">
:disabled="!canStart" <w-spinner v-if="isRunning" size="1.2em" />
@click="startImport" /> <w-icon v-else name="la:play-circle" class="shrink-0" />
<span>{{
isRunning
? t('admin.utilities.wikijs2Import.inProgress')
: t('admin.utilities.wikijs2Import.start')
}}</span>
</w-btn>
</div> </div>
<!-- ----------------------- --> <!-- ----------------------- -->
@ -169,7 +201,7 @@
<w-card> <w-card>
<w-card-header>{{ t('admin.utilities.wikijs2Import.progress') }}</w-card-header> <w-card-header>{{ t('admin.utilities.wikijs2Import.progress') }}</w-card-header>
<div class="p-4 pt-0"> <div class="p-4 pt-0">
<div class="import-log" :class="dark.isActive ? `import-log--dark` : ``"> <div ref="logPanel" class="import-log">
<div v-if="state.log.length < 1" class="import-log__empty"> <div v-if="state.log.length < 1" class="import-log__empty">
{{ t('admin.utilities.wikijs2Import.progressEmpty') }} {{ t('admin.utilities.wikijs2Import.progressEmpty') }}
</div> </div>
@ -200,20 +232,23 @@
</template> </template>
<script setup> <script setup>
import { computed, reactive, ref, watch } from 'vue' import { computed, nextTick, reactive, ref, watch } from 'vue'
import { useI18n } from 'vue-i18n' import { useI18n } from 'vue-i18n'
import { useDark } from '@/composables/dark'
import { useAdminStore } from '@/stores/admin' import { useAdminStore } from '@/stores/admin'
import { useSiteStore } from '@/stores/site' import { useSiteStore } from '@/stores/site'
/** /**
* Import a 2.x backup into a site on this instance. * Import a 2.x backup into a site on this instance.
* *
* UI only for now: nothing is uploaded and `startImport` does not exist yet on the backend. The two * The package is never uploaded as a file: this component opens it, walks it, and drives the import
* columns are the shape the finished thing needs -- what to bring in on the left, what happened on * API a batch at a time. `helpers/wkbackup.js` is the reader and `helpers/wkbackupImport.js` is what
* the right -- so the log panel is here and empty rather than added later. * knows which stream goes where; both are lazily imported, so an instance that never opens this
* screen never fetches the ZIP library. `dev/specs/wkbackup.md` is the format and the design.
*
* Two columns, because there are two things to watch: what is being brought in, on the left, and what
* happened to it, on the right. The left half locks for the length of the run -- those four cards
* describe the import that is already going, so a control that still moved would be lying.
*/ */
// STORES // STORES
@ -221,10 +256,6 @@ import { useSiteStore } from '@/stores/site'
const adminStore = useAdminStore() const adminStore = useAdminStore()
const siteStore = useSiteStore() const siteStore = useSiteStore()
// COMPOSABLES
const dark = useDark()
// I18N // I18N
const { t } = useI18n() const { t } = useI18n()
@ -232,6 +263,7 @@ const { t } = useI18n()
// DATA // DATA
const archiveIpt = ref(null) const archiveIpt = ref(null)
const logPanel = ref(null)
const state = reactive({ const state = reactive({
/** Defaults to whichever site the admin area is already pointed at. */ /** Defaults to whichever site the admin area is already pointed at. */
@ -248,18 +280,39 @@ const state = reactive({
history: true, history: true,
groups: true, groups: true,
users: true, users: true,
navigation: true, navigation: true
settings: true
}, },
overwrite: false, overwrite: false,
/** The `File` the picker handed back, kept whole so its name and size can be shown. */ /** The `File` the picker handed back, kept whole so its name and size can be shown. */
archive: null, archive: null,
/** `{ ts, level, message }`, appended as the import reports its progress. */ /** `{ ts, level, message }`, appended as the import reports its progress. */
log: [] log: [],
/**
* `idle` before the first run, then `running`, then `done` or `failed`.
*
* One value rather than a pair of booleans, because the three things that read it want different
* cuts of it: the form locks on `running` alone, the progress bar is drawn for anything but `idle`
* -- a finished import should still show what it reached -- and its colour is the difference
* between `done` and `failed`.
*/
phase: 'idle',
/** 0..1, as the importer reports it against the record counts in the package manifest. */
progress: 0
}) })
// COMPUTED // COMPUTED
const isRunning = computed(() => state.phase === 'running')
/**
* The strip under the header, while there is something to watch.
*
* Gone once the import finishes: a full bar says nothing the log's closing line does not, and leaving
* it up reads as though something were still happening. A FAILED import keeps it, because there the
* bar carries the one thing the log cannot say at a glance — how far it got before it stopped.
*/
const showProgress = computed(() => state.phase === 'running' || state.phase === 'failed')
const siteOptions = computed(() => const siteOptions = computed(() =>
adminStore.sites.map((site) => ({ value: site.id, label: site.title })) adminStore.sites.map((site) => ({ value: site.id, label: site.title }))
) )
@ -276,10 +329,21 @@ const contentEntries = computed(() => [
{ key: 'history', col: 1, parent: 'pages', label: t('admin.utilities.wikijs2Import.history') }, { key: 'history', col: 1, parent: 'pages', label: t('admin.utilities.wikijs2Import.history') },
{ key: 'groups', col: 2, label: t('admin.utilities.wikijs2Import.groups') }, { key: 'groups', col: 2, label: t('admin.utilities.wikijs2Import.groups') },
{ key: 'users', col: 2, label: t('admin.utilities.wikijs2Import.users') }, { key: 'users', col: 2, label: t('admin.utilities.wikijs2Import.users') },
{ key: 'navigation', col: 2, label: t('admin.utilities.wikijs2Import.navigation') }, { key: 'navigation', col: 2, label: t('admin.utilities.wikijs2Import.navigation') }
{ key: 'settings', col: 2, label: t('admin.utilities.wikijs2Import.settings') }
]) ])
/**
* What is actually going to be imported: what is ticked, minus anything whose parent is not.
*
* The same reading `canStart` makes, and what the session is opened with -- a child of an unticked
* `pages` is not imported, so sending it would have the server reading a stream nobody asked for.
*/
const selectedContent = computed(() =>
contentEntries.value
.filter((entry) => state.content[entry.key] && (!entry.parent || state.content[entry.parent]))
.map((entry) => entry.key)
)
/** /**
* The same rows dealt into the two columns they are drawn in. Which column an entry belongs to is a * The same rows dealt into the two columns they are drawn in. Which column an entry belongs to is a
* field on the entry rather than a slice of the list, so that a row added later lands where it is * field on the entry rather than a slice of the list, so that a row added later lands where it is
@ -308,9 +372,7 @@ const canStart = computed(() => {
if (!state.archive || !state.siteId) { if (!state.archive || !state.siteId) {
return false return false
} }
return contentEntries.value.some( return selectedContent.value.length > 0
(entry) => state.content[entry.key] && (!entry.parent || state.content[entry.parent])
)
}) })
// WATCHERS // WATCHERS
@ -358,15 +420,114 @@ function formatBytes(bytes) {
} }
/** /**
* Not implemented yet — there is no endpoint behind this. The button is wired so that the form's * Run the import.
* enabled state is exercised, and so that there is one place for the upload to go when it exists. *
* Everything heavy is behind the dynamic import: the ZIP reader and the stream logic are a lazy chunk
* that an instance which never migrates from 2.x never fetches.
*
* A failure unlocks the form rather than leaving the screen dead. Every record is keyed so that
* writing it twice is an upsert, so running it again after fixing whatever went wrong carries on
* instead of duplicating what already landed.
*/
async function startImport() {
state.phase = 'running'
state.progress = 0
state.log = []
addLog('info', t('admin.utilities.wikijs2Import.logStarted', { file: state.archive.name }))
try {
const { runImport } = await import('@/helpers/wkbackupImport')
await runImport({
file: state.archive,
siteId: state.siteId,
includes: selectedContent.value,
overwrite: state.overwrite,
log: addLog,
onProgress: (value) => {
state.progress = value
}
})
state.phase = 'done'
} catch (err) {
addLog('error', err.message)
state.phase = 'failed'
}
}
/**
* Add a line to the log, and keep the panel at the bottom.
*
* Pinned only when the reader is already there, so that scrolling back to read a warning is not
* undone by the next of the several thousand lines a large import writes.
*/ */
function startImport() { function addLog(level, message) {
window.alert('Not yet implemented') const panel = logPanel.value
const wasAtBottom = panel ? panel.scrollHeight - panel.scrollTop - panel.clientHeight < 40 : true
state.log.push({
ts: Temporal.Now.plainTimeISO().toString({ smallestUnit: 'second' }),
level,
message
})
if (wasAtBottom) {
nextTick(() => {
if (logPanel.value) {
logPanel.value.scrollTop = logPanel.value.scrollHeight
}
})
}
} }
</script> </script>
<style scoped lang="scss"> <style scoped lang="scss">
/*
`.card-header` is a centred flex ROW, which is what the bar itself wants; the strip beneath it
belongs to the header too, so the element becomes a column of the two. Scoped styles carry an
attribute selector, so this outweighs `.card-header` regardless of stylesheet order.
*/
.import-header {
flex-direction: column;
align-items: stretch;
position: relative;
}
/*
The line that closes the header, under the progress strip.
`.card-header` draws TWO of them -- a 1px border plus a 1px `box-shadow` sitting under it -- which
is the bevel every dialog header in the app wears, and under a progress bar reads as one thick
rule. On top of that a 1px CSS border lands on a fractional device row under display scaling and
paints 1.5 device pixels at 150%, so it looks heavier still.
So: neither, and one line drawn by a pseudo-element scaled to `1 / dpr` instead -- the same
technique as the `.w-hairline` utility in `css/tailwind.css`, with `--w-dpr` published by
`helpers/hairline.js` and a fallback of 1 for an unscaled display.
Selectors are written from `body` because `.card-header`'s own rules are, and a scoped style is
not guaranteed to be injected after the global stylesheet -- matching their specificity would
leave which one wins up to Vite's ordering.
*/
body.body--light .import-header,
body.body--dark .import-header {
border-bottom: 0;
box-shadow: none;
}
.import-header::after {
content: '';
position: absolute;
left: 0;
right: 0;
bottom: 0;
height: 1px;
background-color: $dark-6;
transform: scaleY(calc(1 / var(--w-dpr, 1)));
transform-origin: bottom;
}
body.body--dark .import-header::after {
background-color: #000;
}
/* /*
The elbow that ties Comments and History to the Pages checkbox above them. The elbow that ties Comments and History to the Pages checkbox above them.
@ -415,9 +576,18 @@ body.body--dark .content-child {
} }
/* /*
The log is a terminal, so it is monospaced, scrolls on its own and keeps a fixed height whether it The log is a terminal, so it looks like one: monospaced, dark, scrolling on its own and keeping a
holds two lines or two hundred -- a panel that grows with the import would move the form beside it fixed height whether it holds two lines or two hundred -- a panel that grew with the import would
on every message. `--font-mono` is the app's own, as `UtilCodeEditor` uses. move the form beside it on every message. `--font-mono` is the app's own, as `UtilCodeEditor` uses.
Dark in both themes, rather than following the app. This is machine output scrolling past, and every
other place anybody reads that -- a terminal, a CI log, the browser console -- is dark; a light one
reads as a document. It also keeps the colours below meaning one thing: a warning and an error have
to be legible against exactly one background instead of two, which is why they can be the brighter
end of the ramp and stay readable.
The palette is the app's own dark surfaces (`_theme.scss`), so the panel sits in an admin overlay
rather than looking like a component from somewhere else.
*/ */
.import-log { .import-log {
height: calc(100vh - 260px); height: calc(100vh - 260px);
@ -425,17 +595,19 @@ body.body--dark .content-child {
overflow-y: auto; overflow-y: auto;
border-radius: 4px; border-radius: 4px;
padding: 12px; padding: 12px;
background-color: $grey-1; background-color: $dark-6;
border: 1px solid $grey-4; border: 1px solid $dark-3;
color: $grey-4;
font-family: var(--font-mono, monospace); font-family: var(--font-mono, monospace);
font-size: 12px; font-size: 12px;
line-height: 1.6; line-height: 1.6;
&--dark { /*
background-color: $dark-6; A step lighter than the timestamps, because it is not secondary the way they are: it is the only
border-color: $dark-3; thing on the panel before an import starts, so it has to read as an instruction rather than as
} something greyed out. `$grey-7` was 4.3:1 against this background and `$grey-6` is 7.4:1, while
still sitting a step below the message colour so it does not pass for output.
*/
&__empty { &__empty {
color: $grey-6; color: $grey-6;
font-style: italic; font-style: italic;
@ -450,29 +622,24 @@ body.body--dark .content-child {
&__ts { &__ts {
flex: none; flex: none;
color: $grey-6; color: $grey-7;
} }
/* /*
The full Material ramp is in `tailwind.css` as custom properties; `_palette.scss` carries only The full Material ramp is in `tailwind.css` as custom properties; `_palette.scss` carries only the
the steps its own stylesheets happened to need, and the two mid tones a log wants on a dark steps its own stylesheets happened to need, and the mid tones a log wants are not among them.
background are not among them. These are the light end of each hue, which is what stays legible on the dark panel above.
*/ */
&__line--warn .import-log__msg { &__line--success .import-log__msg {
color: var(--color-orange-9); color: var(--color-green-4);
} font-weight: 600;
&__line--error .import-log__msg {
color: var(--color-red-9);
}
} }
.import-log--dark { &__line--warn .import-log__msg {
.import-log__line--warn .import-log__msg {
color: var(--color-orange-4); color: var(--color-orange-4);
} }
.import-log__line--error .import-log__msg { &__line--error .import-log__msg {
color: var(--color-red-4); color: var(--color-red-4);
} }
} }

@ -2,14 +2,17 @@
<div <div
class="w-linear-progress relative w-full overflow-hidden" class="w-linear-progress relative w-full overflow-hidden"
:class="rounded ? 'rounded-full' : ''" :class="rounded ? 'rounded-full' : ''"
:style="{ height: resolvedSize }" :style="{ height: resolvedSize, '--w-linear-progress-color': trackColor }"
role="progressbar" role="progressbar"
:aria-valuenow="indeterminate ? undefined : Math.round(value * 100)" :aria-valuenow="indeterminate ? undefined : Math.round(value * 100)"
:aria-valuemin="indeterminate ? undefined : 0" :aria-valuemin="indeterminate ? undefined : 0"
:aria-valuemax="indeterminate ? undefined : 100"> :aria-valuemax="indeterminate ? undefined : 100">
<div class="absolute inset-0 opacity-25" :style="{ backgroundColor: trackColor }" />
<div <div
class="absolute inset-y-0 left-0" class="absolute inset-0"
:class="striped ? 'w-linear-progress-striped' : 'opacity-25'"
:style="striped ? undefined : { backgroundColor: trackColor }" />
<div
class="w-linear-progress-bar absolute inset-y-0 left-0"
:class="indeterminate ? 'w-linear-progress-indeterminate' : ''" :class="indeterminate ? 'w-linear-progress-indeterminate' : ''"
:style="barStyle" /> :style="barStyle" />
</div> </div>
@ -50,6 +53,14 @@ const props = defineProps({
rounded: { rounded: {
type: Boolean, type: Boolean,
default: false default: false
},
/**
* Draw the track as diagonal stripes travelling right to left, rather than as a flat tint. For
* work that is running: the bar says how far along it is, the stripes say it is still moving.
*/
striped: {
type: Boolean,
default: false
} }
}) })
@ -74,6 +85,101 @@ const barStyle = computed(() => ({
</script> </script>
<style scoped> <style scoped>
/*
The fill: a band of the colour and a lighter tone of it, tiled across the bar, plus a glow.
The tile has the SAME colour at both ends (`c -> lighter -> c`), which is what makes it repeat
seamlessly -- a two-stop ramp would show a hard step at every tile boundary. It is sized in px
rather than as a percentage of the bar, because a percentage is a percentage of the FILL: the band
would stretch as the bar grew, and the drift below would change speed with the progress.
The glow is in the bar's own colour, so it reads as the bar being lit rather than as a drop shadow
under it. The track clips it (`overflow: hidden`, which the indeterminate bar and the rounded ends
both need), so what survives is the spill ahead of the leading edge -- which is the half worth
having: a soft edge where the bar meets the track and nothing anywhere else. Two shadows rather
than one, a tight near-opaque core that carries over a striped track and a wider soft halo around
it; turning a single blur up far enough to show against the stripes makes a hard collar at the
bar's edge instead of a fade.
Both paint over the track because the bar is the later of the two children -- a box-shadow is drawn
behind its OWN element but above everything under it in the stacking order.
*/
.w-linear-progress-bar {
background-image: linear-gradient(
to right,
var(--w-linear-progress-color),
color-mix(in srgb, var(--w-linear-progress-color) 64%, white),
var(--w-linear-progress-color)
);
background-size: 72px 100%;
box-shadow:
0 0 5px 1px color-mix(in srgb, var(--w-linear-progress-color) 90%, transparent),
0 0 14px 4px color-mix(in srgb, var(--w-linear-progress-color) 55%, transparent);
}
/*
The band drifts left to right, one whole period per cycle so the loop has no seam, which is what
makes a determinate bar read as running even while its width holds still.
Not on the indeterminate bar: that one already moves, and its own keyframes are the same `animation`
shorthand -- declaring both here would leave which survives to the order of the two rules.
*/
.w-linear-progress-bar:not(.w-linear-progress-indeterminate) {
animation: w-linear-progress-sheen 1.8s linear infinite;
}
/*
A square tile crossed by one 45deg band, repeated -- the arrangement that tiles seamlessly in both
axes, which a `repeating-linear-gradient` at an angle does not without matching its period to the
tile by hand. Both tones are the bar's own colour, so a striped track reads as the unfilled part of
the same bar on any surface rather than as a grey the background may or may not suit.
It animates towards x = 0, which moves the pattern leftward -- against the direction the solid bar
grows, so the two do not read as one drifting shape.
The two tones are far enough apart to read at a glance rather than as a texture: the light stripe
is most of the colour and the dark one is nearly nothing, so the track is legible on a light and a
dark surface alike, where a pair of close mid tones is only ever legible on one of them.
Both are mixed off a DARKENED copy of the bar's colour, which is what keeps the track sitting
behind the fill rather than competing with it. Black rather than a lower alpha: alpha is darker
only on a dark surface and lighter on a light one, so it would settle the two tones against the
header here and undo them anywhere else.
*/
.w-linear-progress-striped {
--w-linear-progress-track: color-mix(in srgb, var(--w-linear-progress-color) 81%, black);
background-image: linear-gradient(
45deg,
color-mix(in srgb, var(--w-linear-progress-track) 70%, transparent) 25%,
color-mix(in srgb, var(--w-linear-progress-track) 8%, transparent) 25%,
color-mix(in srgb, var(--w-linear-progress-track) 8%, transparent) 50%,
color-mix(in srgb, var(--w-linear-progress-track) 70%, transparent) 50%,
color-mix(in srgb, var(--w-linear-progress-track) 70%, transparent) 75%,
color-mix(in srgb, var(--w-linear-progress-track) 8%, transparent) 75%
);
background-size: 12px 12px;
animation: w-linear-progress-stripes 0.35s linear infinite;
}
@keyframes w-linear-progress-sheen {
from {
background-position-x: 0;
}
to {
background-position-x: 72px;
}
}
@keyframes w-linear-progress-stripes {
from {
background-position-x: 12px;
}
to {
background-position-x: 0;
}
}
.w-linear-progress-indeterminate { .w-linear-progress-indeterminate {
width: 50%; width: 50%;
animation: w-linear-progress 1.4s ease-in-out infinite; animation: w-linear-progress 1.4s ease-in-out infinite;
@ -92,5 +198,15 @@ const barStyle = computed(() => ({
.w-linear-progress-indeterminate { .w-linear-progress-indeterminate {
animation-duration: 3s; animation-duration: 3s;
} }
/*
-> The stripes and the drift carry no information the bar does not; motion is all they are. The
indeterminate bar's own animation is all it has to say anything with, so it is slowed above
rather than stopped here.
*/
.w-linear-progress-striped,
.w-linear-progress-bar:not(.w-linear-progress-indeterminate) {
animation: none;
}
} }
</style> </style>

@ -0,0 +1,188 @@
import { BlobReader, TextWriter, Uint8ArrayWriter, ZipReader } from '@zip.js/zip.js'
/**
* Reading a `.wkbackup` package in the browser.
*
* The format is `dev/specs/wkbackup.md`; §12 of it is the contract for the records this yields. The
* package is never uploaded — the browser holds the `File`, walks it, and drives the import API a
* batch at a time — so everything here is about getting at one part of a possibly enormous archive
* without materialising the rest.
*
* ZIP earns its keep for exactly that reason: a local `File` is a `Blob` with random access, so the
* central directory at the end of the file is an asset rather than the liability it is over a socket.
* `manifest.json` is readable in the first second whatever the other eight gigabytes weigh.
*
* `@zip.js/zip.js` runs inflation in a pool of workers of its own, which is the part that would jank
* the log this is drawing. What is left on the main thread is a line split and a `JSON.parse` per
* batch, between uploads that dominate the wall clock by orders of magnitude.
*/
/** Refused by the reader rather than by the server, since the server never sees the file. */
export class PackageError extends Error {}
/**
* Open a package and read its manifest.
*
* @returns `{ manifest, entry(path), has(path), close() }`. `entry` is how every other function here
* is reached — a stream is named by the manifest, never guessed at from a convention.
*/
export async function openPackage(file) {
const reader = new ZipReader(new BlobReader(file))
let entries
try {
entries = await reader.getEntries()
} catch (err) {
await reader.close().catch(() => {})
throw new PackageError(
`This file could not be opened as a Wiki.js backup package. (${err.message})`
)
}
const byName = new Map()
for (const entry of entries) {
if (!entry.directory) {
byName.set(entry.filename, entry)
}
}
const manifestEntry = byName.get('manifest.json')
if (!manifestEntry) {
await reader.close().catch(() => {})
throw new PackageError(
'The package has no manifest.json, so it is not a Wiki.js backup package.'
)
}
let manifest
try {
manifest = JSON.parse(await manifestEntry.getData(new TextWriter()))
} catch (err) {
await reader.close().catch(() => {})
throw new PackageError(`The package manifest could not be read. (${err.message})`)
}
return {
manifest,
entry: (path) => byName.get(path) ?? null,
has: (path) => byName.has(path),
close: () => reader.close().catch(() => {})
}
}
/** One whole entry as parsed JSON, for the small documents (`navigation.json`, `settings.json`). */
export async function readJson(entry) {
return JSON.parse(await entry.getData(new TextWriter()))
}
/** One whole entry as bytes. Used for blobs, which are stored rather than deflated. */
export async function readBytes(entry) {
return entry.getData(new Uint8ArrayWriter())
}
/**
* The records of an NDJSON stream, one at a time.
*
* Streamed rather than read whole: `pages.ndjson` for a large wiki is hundreds of megabytes, and the
* point of one entry per stream instead of one per page is that it can be walked in a single pass.
*
* A line that will not parse is **skipped**, not thrown — one corrupt line in ninety thousand is not
* a reason to abandon a migration — and reported through `onMalformed` so it reaches the log.
*/
export async function* readRecords(entry, { onMalformed } = {}) {
const stream = new TransformStream()
// -> Kicked off, not awaited: it resolves when the last byte has been written, which is after the
// loop below has consumed it. Its rejection is claimed here so it is never unhandled.
const finished = entry.getData(stream.writable)
finished.catch(() => {})
const reader = stream.readable.pipeThrough(new TextDecoderStream()).getReader()
let buffer = ''
try {
for (;;) {
const { value, done } = await reader.read()
if (done) {
break
}
buffer += value
let breakAt = buffer.indexOf('\n')
while (breakAt >= 0) {
const line = buffer.slice(0, breakAt).trim()
buffer = buffer.slice(breakAt + 1)
if (line) {
const record = parseLine(line, onMalformed)
if (record) {
yield record
}
}
breakAt = buffer.indexOf('\n')
}
}
// -> A writer that did not end the file with a newline still wrote a record
const tail = buffer.trim()
if (tail) {
const record = parseLine(tail, onMalformed)
if (record) {
yield record
}
}
} finally {
reader.cancel().catch(() => {})
}
await finished
}
function parseLine(line, onMalformed) {
try {
return JSON.parse(line)
} catch {
onMalformed?.(line)
return null
}
}
/**
* Group an async iterable into batches, bounded by BOTH a record count and a rough payload size.
*
* Batching is what keeps the import to one request per few hundred records instead of one per record,
* and it is also why nothing here ever holds a whole stream: only the batch in flight is resident.
*
* The size bound is the one that matters. A count alone is fine for rows — a locale, a group, a tree
* entry — and hopeless for the two streams that carry whole documents: 500 page-history records off a
* real wiki is tens of megabytes, which is a request no sane body limit accepts. Counting records was
* how this shipped and how it fell over on the first package with a page history in it.
*
* `JSON.stringify(...).length` is UTF-16 code units rather than bytes, so it UNDER-counts by up to 3x
* for text that is not Latin — which is why the budget the caller passes sits well under what the
* server will take, and why the caller also halves a batch that comes back 413 rather than trusting
* this. An item bigger than the budget on its own is yielded alone: a record cannot be split, and one
* enormous page is the server's business to accept or refuse.
*/
export async function* inBatches(iterable, { maxRecords, maxBytes }) {
let batch = []
let bytes = 0
for await (const item of iterable) {
const size = JSON.stringify(item).length
if (batch.length > 0 && (batch.length >= maxRecords || bytes + size > maxBytes)) {
yield batch
batch = []
bytes = 0
}
batch.push(item)
bytes += size
}
if (batch.length > 0) {
yield batch
}
}
/**
* The SHA-256 of some bytes, as lowercase hex.
*
* Used to check a blob against the name it is filed under before it is uploaded: the name being the
* checksum is the whole of what content addressing buys, and a reader that took it on trust would be
* uploading whatever the file happened to hold under a digest the server would then reject.
*/
export async function sha256Hex(bytes) {
const digest = await crypto.subtle.digest('SHA-256', bytes)
return [...new Uint8Array(digest)].map((byte) => byte.toString(16).padStart(2, '0')).join('')
}

@ -0,0 +1,425 @@
import { inBatches, openPackage, readBytes, readJson, readRecords, sha256Hex } from './wkbackup'
/**
* Driving an import of a `.wkbackup` package from the browser.
*
* `dev/specs/wkbackup.md` is the design and §11 has the order of operations this follows. The reader
* is `helpers/wkbackup.js`; this is the half that knows what the records MEAN and which endpoint each
* stream goes to.
*
* It reports as it goes rather than returning at the end: an import of a large wiki runs for hours,
* so `log` and `onProgress` are how it is watched, and the value it resolves to is only the summary.
*/
/** Records per request. The server's own ceiling — see `MAX_BATCH_RECORDS` in `models/import.ts`. */
const BATCH_SIZE = 500
/**
* What a batch aims to weigh, in UTF-16 code units of JSON.
*
* A quarter of the server's `MAX_BATCH_BYTES`, and the gap is deliberate: the measure under-counts
* bytes by up to 3x for text outside Latin-1, so a budget at the ceiling would be a 413 for every
* wiki written in Chinese, Japanese, Korean, Greek, Cyrillic or Arabic. `postBatch` halves anything
* that is refused anyway, so this only has to be close.
*/
const BATCH_BYTES = 4 * 1024 * 1024
/**
* The order the streams are walked in, and it is load-bearing.
*
* Instance-wide first: a page needs an author and a comment needs one too, and both are resolved by
* email against the accounts that exist at the time — so a users stream that ran afterwards would
* leave every record attributed to whoever pressed the button. Groups before users for the same
* reason one rung down, since a user record names the groups it belongs to.
*
* Then, per site: folders (so a folder somebody titled *Guides* is not created as *guides* by the
* first page filed under it), pages, then the history and comments that hang off them, then assets,
* then navigation.
*/
const INSTANCE_STREAMS = [
{ name: 'locales', requires: null },
{ name: 'groups', requires: 'groups' },
{ name: 'users', requires: 'users' }
]
const SITE_STREAMS = [
{ name: 'tree', requires: null },
{ name: 'pages', requires: 'pages' },
{ name: 'page-history', requires: 'history' },
{ name: 'assets', requires: 'assets' },
{ name: 'comments', requires: 'comments' }
]
/**
* Import one package.
*
* @param file The `.wkbackup` the operator picked
* @param siteId The site on this instance everything site-scoped lands in
* @param includes The content kinds ticked in the overlay
* @param overwrite Whether an existing record is replaced
* @param log `(level, message)` — `info` / `success` / `warn` / `error`
* @param onProgress `(fraction)` from 0 to 1
*/
export async function runImport({ file, siteId, includes, overwrite, log, onProgress }) {
const pkg = await openPackage(file)
try {
return await drive({ pkg, siteId, includes, overwrite, log, onProgress })
} finally {
await pkg.close()
}
}
async function drive({ pkg, siteId, includes, overwrite, log, onProgress }) {
const { manifest } = pkg
log(
'info',
`Package written ${manifest.createdAt ?? 'at an unknown date'} by ${describeGenerator(manifest)}.`
)
/*
Everything answerable from the manifest, before a single record is written. The server owns this
question rather than the browser: what this wiki can read is what this wiki knows.
*/
const preflight = await API_CLIENT.post('import/preflight', {
json: { manifest, siteId }
}).json()
for (const warning of preflight.warnings) {
log('warn', warning)
}
if (!preflight.ok) {
for (const error of preflight.errors) {
log('error', error)
}
throw new Error(preflight.errors[0] ?? 'This package cannot be imported.')
}
log('info', 'Package checked.')
// -> A 2.x package always holds exactly one site; a multi-site package is reported by preflight
const packageSite = manifest.sites[0]
const sourceId = String(packageSite.id ?? 'default')
const session = await API_CLIENT.post('import/sessions', {
json: {
source: manifest.source.kind,
// -> Part of what the server derives this import's record ids from, so that running the same
// package again after a failure upserts instead of laying down a second copy of everything
// that has no natural key
sourceInstanceId: manifest.source.instanceId ?? '',
sites: [{ sourceId, siteId }],
includes,
overwrite
}
}).json()
log('info', `Import session opened.`)
const plan = buildPlan({ manifest, packageSite, includes, log })
const progress = { done: 0, total: Math.max(plan.total, 1) }
const advance = (n) => {
progress.done += n
onProgress(Math.min(1, progress.done / progress.total))
}
onProgress(0)
/** Blobs already staged this session, so one shared by forty records is uploaded once. */
const staged = new Set()
for (const step of plan.steps) {
const label = step.label
log('info', `${label}: starting (${step.count === null ? 'unknown' : step.count} records)...`)
const totals = { imported: 0, skipped: 0 }
let read = 0
if (step.kind === 'json') {
const records = await step.read(pkg)
read = records.length
const result = await postBatch(step.url(session, siteId), records, label, log)
tally(totals, result, log)
advance(step.count ?? records.length)
} else {
for await (const batch of inBatches(
readRecords(pkg.entry(step.path), {
onMalformed: () => log('warn', `${label}: a record could not be read and was skipped.`)
}),
{ maxRecords: BATCH_SIZE, maxBytes: BATCH_BYTES }
)) {
read += batch.length
await stageBlobsFor({ pkg, session, batch, staged, log })
const result = await postBatch(step.url(session, siteId), batch, label, log)
tally(totals, result, log)
advance(batch.length)
}
}
/*
The manifest counts the records the exporter meant to write; this is how many were actually
there. When they disagree the package is short, and saying so plainly is the difference between
"the import lost my pages" and "the export only wrote nine of them" — which are fixed in
completely different places, and only one of them here.
A shortfall cascades: everything hanging off the missing records is skipped too, so the log
below this line fills with history and comments that have nowhere to go. This is the line that
explains those, which is why it says what it says.
*/
if (step.count !== null && read !== step.count) {
const verb = read < step.count ? 'only held' : 'held'
log(
'warn',
`${label}: the package says it contains ${step.count} records but the stream ${verb} ${read}. The package itself is incomplete \u2014 nothing below can import what is not in it.`
)
}
log('info', `${label}: ${totals.imported} imported, ${totals.skipped} skipped.`)
}
const summary = await API_CLIENT.post(`import/sessions/${session.id}/finish`).json()
reportSummary(summary, log)
onProgress(1)
return summary
}
/**
* What will be read, and how many records that is.
*
* Built up front from the manifest's counts, which is what makes the progress bar a real fraction
* rather than a spinner — the counts are in the central directory's first entry, so they are known
* before anything has been inflated.
*/
function buildPlan({ manifest, packageSite, includes, log }) {
const steps = []
let total = 0
const add = (step) => {
steps.push(step)
total += step.count ?? 0
}
for (const { name, requires } of INSTANCE_STREAMS) {
const declared = manifest.streams?.[name]
if (!declared?.path || (requires && !includes.includes(requires))) {
continue
}
add({
kind: 'ndjson',
label: labelFor(name),
path: declared.path,
count: declared.count ?? null,
url: (session) => `import/sessions/${session.id}/streams/${name}`
})
}
for (const { name, requires } of SITE_STREAMS) {
const declared = packageSite.streams?.[name]
if (!declared?.path || (requires && !includes.includes(requires))) {
continue
}
add({
kind: 'ndjson',
label: labelFor(name),
path: declared.path,
count: declared.count ?? null,
url: (session, siteId) => `import/sessions/${session.id}/sites/${siteId}/streams/${name}`
})
}
/*
Navigation is one document rather than a stream — a whole menu is a single column here, so a batch
of halves would be a sidebar that flickered between them.
`{ mode, trees: { <key>: <config> } }` is what the 2.x exporter writes, keyed by 2.x's
`navigation.key`. **Neither half of that key/value pair is what it looks like.**
The key is not a locale: 2.x keeps its one site-wide tree under the literal `site`
(`idColumn = 'key'`, `findOne('key', 'site')` in its navigation model). And the value is not a
list of menu items — it is a list of PER-LOCALE TREES, `[{ locale, items }]`, which is what that
model iterates to fill `nav:sidebar:<locale>`. So the locale comes from inside the config and the
key carries nothing at all.
Reading the config as items produces one blank link per locale, which imports, reports success and
draws an empty sidebar. That is what this looked like the first two times.
The one exception is 2.x's own: a config whose first entry has a `kind` is the pre-2.3 flat format,
and 2.x reads that as locale `en`. Handled here because the exporter writes the column raw, so a
wiki that has not re-saved its navigation since 2.2 still exports the old shape.
*/
const nav = packageSite.streams?.navigation
if (nav?.path && includes.includes('navigation')) {
add({
kind: 'json',
label: labelFor('navigation'),
count: null,
url: (session, siteId) => `import/sessions/${session.id}/sites/${siteId}/streams/navigation`,
read: async (pkg) => {
const entry = pkg.entry(nav.path)
if (!entry) {
return []
}
const document = await readJson(entry)
const primary = packageSite.locales?.length > 0 ? packageSite.locales[0] : 'en'
const trees = []
for (const config of Object.values(document?.trees ?? {})) {
if (!Array.isArray(config) || config.length < 1) {
continue
}
// -> Pre-2.3: a flat list of items, which 2.x itself reads as the `en` tree
if (config[0]?.kind) {
trees.push({ localeCode: 'en', items: config })
continue
}
for (const tree of config) {
if (Array.isArray(tree?.items)) {
trees.push({ localeCode: tree.locale || primary, items: tree.items })
}
}
}
/*
Said out loud per tree. A sidebar that lands in the wrong locale, or that is built out of
the wrong level of the document, is invisible rather than wrong — it reports success and
draws nothing — so the log has to carry enough to tell those apart without a database.
*/
for (const tree of trees) {
log('info', `Navigation: ${tree.items.length} items for the ${tree.localeCode} sidebar.`)
}
if (trees.length < 1) {
log('warn', 'Navigation: the package holds no menu this wiki could read.')
}
return trees
}
})
}
return { steps, total }
}
/**
* Upload the blobs a batch is about to reference, once each.
*
* Demand-driven rather than a phase of its own: a package's `blobs/` holds every asset in the wiki,
* and an import of pages alone has no business uploading eight gigabytes of images nobody asked for.
* Content addressing is what makes this safe to do per batch — "have I already sent this?" is a
* question about the digest and nothing else.
*/
async function stageBlobsFor({ pkg, session, batch, staged, log }) {
const wanted = new Set()
for (const record of batch) {
for (const digest of [record?.blob, record?.contentBlob]) {
if (typeof digest === 'string' && digest && !staged.has(digest)) {
wanted.add(digest)
}
}
}
for (const digest of wanted) {
const entry = pkg.entry(`blobs/${digest}`)
if (!entry) {
log('warn', `A file the package refers to (${digest.slice(0, 12)}…) is missing from it.`)
// -> Marked as seen either way, so a blob forty records share is reported once rather than forty
// times; the records themselves are then skipped by the server, which says which they were
staged.add(digest)
continue
}
const bytes = await readBytes(entry)
const actual = await sha256Hex(bytes)
if (actual !== digest) {
log(
'warn',
`A file in the package does not match its checksum (${digest.slice(0, 12)}…) and was not imported.`
)
staged.add(digest)
continue
}
await API_CLIENT.post(`import/sessions/${session.id}/blobs/${digest}`, {
body: bytes,
headers: { 'content-type': 'application/octet-stream' }
})
staged.add(digest)
}
}
/**
* Post a batch, halving it if the server says it is too large.
*
* The size budget above is an estimate over a format this code does not control, so it will sometimes
* be wrong — a wiki written in a script where one character is three bytes, a page carrying a base64
* image inline. Halving converges in a handful of requests and needs no tuning, which is a better
* answer than a constant somebody has to guess right for every wiki in the world.
*
* A single record the server still refuses is skipped rather than fatal: one page too large to send
* is a page to go and look at, not a reason to abandon the other eleven thousand. It is named in the
* log, which is the only place anybody could act on it.
*/
async function postBatch(url, records, label, log) {
try {
return await API_CLIENT.post(url, { json: { records } }).json()
} catch (err) {
if (err.response?.status !== 413) {
throw err
}
if (records.length < 2) {
log('warn', `${label}: one record is too large to send and was skipped.`)
return { imported: 0, skipped: 1, warnings: [] }
}
const half = Math.ceil(records.length / 2)
const first = await postBatch(url, records.slice(0, half), label, log)
const second = await postBatch(url, records.slice(half), label, log)
return {
imported: first.imported + second.imported,
skipped: first.skipped + second.skipped,
warnings: [...(first.warnings ?? []), ...(second.warnings ?? [])]
}
}
}
function tally(totals, result, log) {
totals.imported += result.imported ?? 0
totals.skipped += result.skipped ?? 0
for (const warning of result.warnings ?? []) {
log('warn', warning)
}
}
function reportSummary(summary, log) {
const written = Object.entries(summary.progress ?? {})
.map(([stream, count]) => `${count} ${labelFor(stream).toLowerCase()}`)
.join(', ')
// -> The one line reporting the whole run rather than a step of it, and the one somebody watching a
// long import is waiting for. `success` is what the log panel draws bold and green.
log('success', `Import finished. ${written || 'Nothing was written.'}`)
/*
The one thing an operator has to be told about a 2.x import: a page arrives with no HTML, because
a 2.x render is 2.x's output and would be wrong here in ways nothing could later detect. The queue
that produces the real ones is a single headless browser doing a single page at a time.
*/
if (summary.pendingRenders > 0) {
log(
'info',
`${summary.pendingRenders} pages are waiting to be rendered. They will fill in as the renderer works through them.`
)
}
if (summary.unrenderable > 0) {
log(
'warn',
`${summary.unrenderable} pages cannot be rendered by the server and will stay blank until somebody opens and saves them.`
)
}
}
function describeGenerator(manifest) {
const product = manifest.generator?.product ?? 'an unknown tool'
const version = manifest.generator?.version
return version ? `${product} ${version}` : product
}
const LABELS = {
locales: 'Locales',
groups: 'Groups',
users: 'Users',
tree: 'Folders',
pages: 'Pages',
'page-history': 'Page history',
assets: 'Assets',
comments: 'Comments',
navigation: 'Navigation'
}
function labelFor(stream) {
return LABELS[stream] ?? stream
}
Loading…
Cancel
Save