diff --git a/packages/cli/package.json b/packages/cli/package.json index 055b42c1..c59eb1ea 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,7 +1,7 @@ { "name": "@profullstack/sh1pt", "version": "0.3.4", - "description": "One codebase → every store, registry, CDN, and channel. Build. Promote. Scale. Iterate.", + "description": "One codebase \u2192 every store, registry, CDN, and channel. Build. Promote. Scale. Iterate.", "license": "MIT", "repository": { "type": "git", @@ -46,6 +46,7 @@ "@profullstack/sh1pt-actions-fleet-core": "workspace:^", "@profullstack/sh1pt-automation-browser": "workspace:^", "@profullstack/sh1pt-core": "workspace:^", + "@profullstack/sh1pt-migrate": "workspace:*", "@profullstack/sh1pt-openapi": "workspace:^", "@profullstack/sh1pt-policy": "workspace:^", "@profullstack/sh1pt-secrets-env-updater": "workspace:^", diff --git a/packages/cli/src/commands/migrate.ts b/packages/cli/src/commands/migrate.ts new file mode 100644 index 00000000..6e472cc0 --- /dev/null +++ b/packages/cli/src/commands/migrate.ts @@ -0,0 +1,285 @@ +import { Command, InvalidArgumentError } from 'commander'; +import { readFile } from 'node:fs/promises'; +import kleur from 'kleur'; +import { + ENGINES, + PHASES, + PLATFORMS, + type Phase, + type Platform, + type Resource, + type ResourceKind, + applyPlan, + availableBinaries, + compatibleKinds, + createExec, + openStaging, + parseRewrite, + planMigration, + platformById, + recordingExec, + renderPlan, + requiredBinaries, +} from '@profullstack/sh1pt-migrate'; + +/** + * `sh1pt migrate` — move an app and its data from one platform to another. + * + * The commands map onto the only workflow that is safe: look at what is + * there, read a plan, rehearse it, then run it. + * + * sh1pt migrate platforms what can be moved where + * sh1pt migrate inventory --from supabase what is on the source + * sh1pt migrate plan --from a --to b the ordered plan, no mutations + * sh1pt migrate apply --until freeze the bulk copy, no downtime + * sh1pt migrate apply the whole thing + * + * `plan` never mutates and never needs to be trusted, which is what makes it + * safe to point at production. `apply` refuses a plan with blockers. + */ + +interface MigrateConfig { + from?: { platform: string; [k: string]: unknown }; + to?: { platform: string; [k: string]: unknown }; + rewriteHosts?: string[]; + only?: ResourceKind[]; + exclude?: ResourceKind[]; +} + +function parsePhase(value: string): Phase { + if (!(PHASES as readonly string[]).includes(value)) { + throw new InvalidArgumentError(`must be one of: ${PHASES.join(', ')}`); + } + return value as Phase; +} + +function parseKinds(value: string, previous: ResourceKind[] = []): ResourceKind[] { + return [...previous, value as ResourceKind]; +} + +async function loadConfig(path: string | undefined): Promise { + if (!path) return {}; + const text = await readFile(path, 'utf8'); + return JSON.parse(text) as MigrateConfig; +} + +function ctxFor(dryRun: boolean, verbose: boolean) { + return { + secret: (key: string) => process.env[key], + log: (msg: string, level: 'info' | 'warn' | 'error' = 'info') => { + if (!verbose && level === 'info') return; + const paint = level === 'error' ? kleur.red : level === 'warn' ? kleur.yellow : kleur.dim; + console.log(paint(msg)); + }, + dryRun, + }; +} + +/** Resolve a platform by id, failing with the list rather than a bare error. */ +function resolvePlatform(id: string | undefined, role: 'source' | 'target'): Platform { + if (!id) throw new Error(`--${role === 'source' ? 'from' : 'to'} is required`); + const platform = platformById(id); + if (!platform) { + throw new Error(`unknown platform '${id}'. Known: ${PLATFORMS.map((p) => p.id).join(', ')}`); + } + if (role === 'target' && platform.role === 'source') { + throw new Error(`${platform.label} cannot be a target`); + } + return platform as Platform; +} + +export const migrateCmd = new Command('migrate') + .description('Move an app and its data between platforms — cloud to dedicated, or back') + .action(() => { + migrateCmd.help(); + }); + +migrateCmd + .command('platforms') + .description('List every platform, what it holds, and which pairs can move what') + .option('--from ', 'show only what can move out of this platform') + .action((opts: { from?: string }) => { + if (opts.from) { + const from = platformById(opts.from); + if (!from) throw new Error(`unknown platform '${opts.from}'`); + console.log(kleur.bold(`from ${from.label}:`)); + for (const to of PLATFORMS) { + if (to.id === from.id || to.role === 'source') continue; + const kinds = compatibleKinds(from.id, to.id); + const line = ` → ${to.label.padEnd(26)} ${kinds.length ? kinds.join(', ') : kleur.dim('nothing in common')}`; + console.log(kinds.length ? line : kleur.dim(line)); + } + return; + } + for (const p of PLATFORMS) { + const role = p.role === 'both' ? 'source+target' : p.role; + console.log(`${p.id.padEnd(14)} ${role.padEnd(14)} ${p.supports.join(', ')}`); + } + }); + +migrateCmd + .command('inventory') + .description('List what is on a platform. Read-only, safe against production') + .requiredOption('--from ', 'platform id') + .option('-c, --config ', 'JSON config with the platform connection details') + .option('--json') + .option('-v, --verbose') + .action(async (opts: { from: string; config?: string; json?: boolean; verbose?: boolean }) => { + const config = await loadConfig(opts.config); + const platform = resolvePlatform(opts.from, 'source'); + const inventory = await platform.inventory(ctxFor(true, opts.verbose ?? false), config.from ?? {}); + + if (opts.json) { + console.log( + JSON.stringify( + { + platform: inventory.platform, + scope: inventory.scope, + // describe(), never reveal(): this output is routinely pasted. + resources: inventory.resources.map((r) => ({ + kind: r.kind, + id: r.id, + name: r.name, + sizeBytes: r.sizeBytes, + quirks: r.quirks, + connection: Object.fromEntries( + Object.entries(r.connection).map(([k, v]) => [k, v.describe()]), + ), + })), + notes: inventory.notes, + }, + null, + 2, + ), + ); + return; + } + + console.log(kleur.bold(`${inventory.platform} · ${inventory.scope}`)); + for (const r of inventory.resources) { + console.log(` ${r.kind.padEnd(16)} ${r.name}`); + for (const q of r.quirks ?? []) console.log(kleur.yellow(` ! ${q}`)); + } + for (const n of inventory.notes ?? []) console.log(kleur.dim(` note: ${n}`)); + }); + +migrateCmd + .command('plan') + .description('Work out what would happen. Touches nothing') + .requiredOption('--from ') + .requiredOption('--to ') + .option('-c, --config ') + .option('--only ', 'only this resource kind (repeatable)', parseKinds) + .option('--exclude ', 'skip this resource kind (repeatable)', parseKinds) + .option('--rewrite-host ', 'rewrite absolute URLs (repeatable)', (v, p: string[] = []) => [...p, v]) + .option('--json') + .option('-v, --verbose') + .action(async (opts) => { + const plan = await buildPlan(opts); + console.log(opts.json ? JSON.stringify(plan, null, 2) : renderPlan(plan)); + if (!plan.ok) process.exitCode = 1; + }); + +migrateCmd + .command('apply') + .description('Run the plan. Refuses one with blockers') + .requiredOption('--from ') + .requiredOption('--to ') + .option('-c, --config ') + .option('--only ', 'only this resource kind (repeatable)', parseKinds) + .option('--exclude ', 'skip this resource kind (repeatable)', parseKinds) + .option('--rewrite-host ', 'rewrite absolute URLs (repeatable)', (v, p: string[] = []) => [...p, v]) + .option('--staging ', 'where dumps are kept between export and import', '.sh1pt-migrate') + .option( + '--until ', + `stop before this phase (${PHASES.join(', ')}). --until freeze rehearses without downtime`, + parsePhase, + ) + .option('--dry-run', 'print what would run without running it') + .option('-v, --verbose') + .action(async (opts) => { + const plan = await buildPlan(opts); + if (!plan.ok) { + console.log(renderPlan(plan)); + throw new Error('plan has blockers; fix them or exclude the resources involved'); + } + + const config = await loadConfig(opts.config); + const source = resolvePlatform(opts.from, 'source'); + const targetPlatform = resolvePlatform(opts.to, 'target'); + const pctx = ctxFor(Boolean(opts.dryRun), opts.verbose ?? false); + + const inventory = await source.inventory(pctx, config.from ?? {}); + const from = new Map(inventory.resources.map((r) => [r.id, r])); + + const to = new Map(); + for (const r of plan.moving) { + if (!targetPlatform.provision) break; + to.set(r.id, await targetPlatform.provision(pctx, r, config.to ?? {})); + } + + const { exec } = opts.dryRun ? recordingExec() : { exec: createExec({ log: pctx.log }) }; + const staging = await openStaging({ dir: opts.staging }); + + console.log(renderPlan(plan)); + console.log(''); + + const result = await applyPlan({ + plan, + from, + to, + engines: new Map(ENGINES), + ctx: { log: pctx.log, dryRun: Boolean(opts.dryRun), staging, exec }, + ...(opts.until ? { until: opts.until } : {}), + onStep: (step, outcome) => { + const mark = + outcome.status === 'done' ? kleur.green('✓') : outcome.status === 'skipped' ? kleur.dim('·') : kleur.red('✗'); + console.log(`${mark} ${step.title}${outcome.reason ? kleur.dim(` (${outcome.reason})`) : ''}`); + }, + }); + + if (!result.ok) { + throw new Error(`stopped at '${result.failed?.id}': ${result.failed?.error.message}`); + } + console.log( + kleur.green( + result.stoppedBefore + ? `\nStopped before '${result.stoppedBefore}' as asked. Nothing is down.` + : '\nMigration complete. Leave the source in place until you have watched the target for a day.', + ), + ); + }); + +interface PlanOpts { + from: string; + to: string; + config?: string; + only?: ResourceKind[]; + exclude?: ResourceKind[]; + rewriteHost?: string[]; + verbose?: boolean; +} + +async function buildPlan(opts: PlanOpts) { + const config = await loadConfig(opts.config); + const source = resolvePlatform(opts.from, 'source'); + const target = resolvePlatform(opts.to, 'target'); + + // Validate every rewrite before touching the network, so a typo costs + // nothing rather than being discovered after the dump. + const rewrites = (opts.rewriteHost ?? config.rewriteHosts ?? []).map(parseRewrite); + + const pctx = ctxFor(true, opts.verbose ?? false); + const inventory = await source.inventory(pctx, config.from ?? {}); + + const kinds = [...new Set(inventory.resources.map((r) => r.kind))]; + const binaries = await availableBinaries(requiredBinaries(kinds)); + + return planMigration(inventory, target, { + engines: new Map(ENGINES), + availableBinaries: binaries, + ...(opts.only ?? config.only ? { only: opts.only ?? config.only } : {}), + ...(opts.exclude ?? config.exclude ? { exclude: opts.exclude ?? config.exclude } : {}), + rewriteHosts: rewrites.map((r) => r.from), + }); +} diff --git a/packages/cli/src/index.ts b/packages/cli/src/index.ts index f5df6406..b5b06f1f 100644 --- a/packages/cli/src/index.ts +++ b/packages/cli/src/index.ts @@ -20,6 +20,7 @@ import { deployCmd } from './commands/deploy.js'; import { openapiCmd } from './commands/openapi.js'; import { runsCmd } from './commands/runs.js'; import { logicsrcCmd } from './commands/logicsrc.js'; +import { migrateCmd } from './commands/migrate.js'; import { browserCmd } from './commands/browser.js'; const program = new Command(); @@ -57,6 +58,7 @@ program.addCommand(createActionsCmd()); // actions · install/audit GitHub Acti program.addCommand(skillsCmd); // skills · package/promote SKILL.md agent skills across marketplaces program.addCommand(agentsCmd); // agents · generate/run/talk with AI coding CLIs program.addCommand(deployCmd); // deploy · provision cloud infrastructure +program.addCommand(migrateCmd); // migrate · move an app and its data between platforms, either direction program.addCommand(openapiCmd); // openapi · spec → SDK + MCP server + docs site (Stainless-style) program.addCommand(logicsrcCmd); // logicsrc · LogicSRC OpenSpec-only workflows program.addCommand(browserCmd); // browser · console chores with no CLI or API, driven in a real browser diff --git a/packages/migrate/README.md b/packages/migrate/README.md new file mode 100644 index 00000000..f29b0103 --- /dev/null +++ b/packages/migrate/README.md @@ -0,0 +1,134 @@ +# @profullstack/sh1pt-migrate + +Move an application and its data between platforms, in either direction. + +`packages/cloud/*` provisions machines. This moves what lives on them. + +```bash +sh1pt migrate platforms # what can move where +sh1pt migrate platforms --from supabase # ...and out of one place +sh1pt migrate inventory --from supabase -c m.json +sh1pt migrate plan --from supabase --to ssh -c m.json +sh1pt migrate apply --from supabase --to ssh -c m.json --until freeze +sh1pt migrate apply --from supabase --to ssh -c m.json +``` + +## Why it is bidirectional without twice the code + +Two layers, and the split is the whole design: + +- **Platforms** answer *what have I got, and what are the credentials* — Railway, + Supabase, Turso, Neon, PlanetScale, Fly, Render, Heroku, Vercel, a box over ssh. + A platform never moves a byte. +- **Engines** move bytes — `postgres`, `sqlite`, `redis`, `object-storage`, `files`. + An engine does not know which vendor is on either end. + +So `supabase → ssh` and `ssh → supabase` are the same code path, and a new +platform costs one `inventory()` rather than one adapter per existing platform. +Direction is not a property of the system; it is which platform you named first. + +`compatibleKinds('turso', 'neon')` returns `[]` — sqlite against postgres — and +says so in a millisecond rather than at a cutover. + +## The plan is the product + +`migrate plan` touches nothing, calls nothing, and is safe against production. +It reports what moves, what does not and why, an ordered list of steps, and the +risks. Read it, disagree with it, then apply it. + +Phase order is enforced, not documented: + +| phase | what happens | +|---|---| +| `check` | credentials and binaries, so a missing `pg_dump` costs a second not three hours | +| `bulk` | the long copy, while the source is still live and serving | +| `freeze` | stop the writers. **Downtime starts here** | +| `delta` | only what changed during `bulk` | +| `cutover` | point DNS at the target | +| `enable` | start the writers again, on the target only | +| `verify` | prove it worked while the old system still exists | + +Scheduled jobs are stopped on the source before they are started on the target, +because a cron firing on both sides is how a migration sends every customer a +duplicate email. Dropping the source is not a phase — that is a decision a +person makes days later, and this tool does not offer it. + +`--until freeze` runs the entire bulk copy and stops before anything goes down. +That is how you rehearse against production. + +## The trap this exists for + +An app that stores a whole URL instead of a key leaves rows pointing at the +account you just left: + +``` +https://ywcizjsgrcmhgyplldac.supabase.co/storage/v1/object/public/ads/x.png +``` + +Migrate everything, cut DNS over, check the site: every image loads — because +the **old** account is still serving them. The day it is closed, which is the +entire point of migrating and happens weeks later, all of them 404 at once and +nothing connects the outage to the migration. + +crawlproof.com had 2,928 such rows across four tables. So the rewrite is a +first-class step: every text-ish column is scanned (broad on purpose — a column +called `notes` holding a pasted URL breaks exactly as badly as one called +`image_url`), rewritten inside one transaction, and then asserted to be zero. + +```bash +--rewrite-host ywcizjsgrcmhgyplldac.supabase.co=https://cdn.example.com +``` + +The Supabase platform emits that exact flag for its own hostname, so it is a +line to copy rather than a thing to remember. + +## Safety properties, stated plainly + +- **Nothing deletes.** `rclone copy`, never `sync`. No `rsync --delete`. No + `pg_restore --clean`. A non-empty Postgres target is refused rather than + overwritten. +- **Credentials never reach a plan file.** Connections are a + `describe()`/`reveal()` pair; a test asserts a rendered plan contains no + password. +- **Credentials never reach `argv`.** libpq environment variables for Postgres, + `RCLONE_CONFIG_*` for object storage, `TURSO_API_TOKEN` for Turso. `ps` is + world-readable and a dump runs for hours. +- **No shell.** `spawn` without one, always — engine arguments come from config + a person edits. +- **Resumable.** The staging ledger is append-only JSONL, so an interrupted run + can only truncate its last line, and the next run skips what is done. + +## Config + +```json +{ + "from": { + "projectRef": "abc123", + "dbPassword": "...", + "buckets": ["ads", "articles"] + }, + "to": { + "host": "dev2.example.com", + "user": "anthony", + "postgres": [{ "name": "postgres", "url": "postgres://app@127.0.0.1:5432/app" }] + } +} +``` + +Secrets are read from the environment where a platform names one +(`SUPABASE_SERVICE_ROLE_KEY`, `RAILWAY_TOKEN`, `TURSO_API_TOKEN`, +`NEON_DATABASE_URL`, …), so they need not be in the file. + +## What it needs installed + +Per engine, checked before anything runs: `pg_dump`/`pg_restore`/`psql`, +`mysqldump`/`mysql`, +`sqlite3`, `redis-cli`, `rclone`, `rsync`. + +## Known limits + +- The Postgres delta covers **inserts into tables with a timestamp column**, not + updates or deletes. The planner says so rather than implying otherwise. +- Redis has no delta at all. Stop the writers first. +- rsync cannot copy remote to remote; one side must be local. +- A Railway volume is only reachable from inside its service. diff --git a/packages/migrate/package.json b/packages/migrate/package.json new file mode 100644 index 00000000..52927993 --- /dev/null +++ b/packages/migrate/package.json @@ -0,0 +1,37 @@ +{ + "name": "@profullstack/sh1pt-migrate", + "version": "0.1.15", + "type": "module", + "main": "./src/index.ts", + "scripts": { + "build": "tsc -p tsconfig.json", + "typecheck": "tsc -p tsconfig.json --noEmit", + "prepublishOnly": "pnpm build" + }, + "dependencies": { + "@profullstack/sh1pt-core": "workspace:*" + }, + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/profullstack/sh1pt.git", + "directory": "packages/migrate" + }, + "homepage": "https://sh1pt.com", + "bugs": "https://github.com/profullstack/sh1pt/issues", + "files": [ + "dist" + ], + "publishConfig": { + "access": "public", + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js", + "default": "./dist/index.js" + } + } + } +} diff --git a/packages/migrate/src/apply.test.ts b/packages/migrate/src/apply.test.ts new file mode 100644 index 00000000..0da47347 --- /dev/null +++ b/packages/migrate/src/apply.test.ts @@ -0,0 +1,221 @@ +import { describe, expect, it, vi } from 'vitest'; +import { applyPlan } from './apply.js'; +import { planMigration } from './plan.js'; +import { memoryStaging, parseLedger } from './staging.js'; +import type { Engine, EngineContext, Inventory, Platform, Resource, ResourceKind } from './types.js'; +import { secret } from './types.js'; + +function engineCtx(over: Partial = {}): EngineContext { + return { + dryRun: false, + log: () => {}, + staging: memoryStaging(), + exec: async () => ({ code: 0, stdout: '', stderr: '' }), + ...over, + }; +} + +/** An engine that records what it was asked to do. */ +function spyEngine(kind: ResourceKind, over: Partial = {}) { + const calls: string[] = []; + const engine: Engine = { + kind, + requires: [], + export: async (_c, r) => { + calls.push(`export:${r.id}`); + return [{ resourceId: r.id, kind, path: `${r.id}/dump` }]; + }, + import: async (_c, r) => { + calls.push(`import:${r.id}`); + }, + delta: async (_c, r) => { + calls.push(`delta:${r.id}`); + return [{ resourceId: r.id, kind, path: `${r.id}/delta` }]; + }, + verify: async (_c, r) => { + calls.push(`verify:${r.id}`); + return { ok: true, checks: [], problems: [] }; + }, + ...over, + }; + return { engine, calls }; +} + +const resource = (id: string, kind: ResourceKind = 'postgres'): Resource => ({ + kind, + id, + name: id, + connection: { url: secret('postgres://u@h/d') }, +}); + +const target = (supports: ResourceKind[] = ['postgres', 'object-storage', 'cron']): Platform => ({ + id: 'ssh', + label: 'box', + role: 'both', + supports, + inventory: async () => ({ platform: 'ssh', scope: 'box', resources: [] }), +}); + +function setup(resources: Resource[], engineOver: Partial = {}) { + const { engine, calls } = spyEngine('postgres', engineOver); + const engines = new Map([['postgres', engine]]); + const inventory: Inventory = { platform: 'supabase', scope: 'proj', resources }; + const plan = planMigration(inventory, target(), { engines }); + const from = new Map(resources.map((r) => [r.id, r])); + const to = new Map(resources.map((r) => [r.id, { ...r, id: r.id }])); + return { plan, engines, from, to, calls }; +} + +describe('applyPlan', () => { + it('runs export before import for each resource', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(true); + expect(calls.indexOf('export:db')).toBeLessThan(calls.indexOf('import:db')); + }); + + it('refuses a plan the planner already marked unapplyable', async () => { + const { engine } = spyEngine('postgres'); + const engines = new Map([['postgres', engine]]); + const plan = planMigration( + { platform: 'x', scope: 's', resources: [resource('db')] }, + target(['files']), + { engines }, + ); + expect(plan.ok).toBe(false); + + await expect( + applyPlan({ plan, engines, from: new Map(), to: new Map(), ctx: engineCtx() }), + ).rejects.toThrow(/refusing to apply a plan with blockers/); + }); + + it('stops before the phase named by --until, which is how a rehearsal works', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx(), until: 'freeze' }); + + expect(res.stoppedBefore).toBe('freeze'); + expect(calls).toContain('export:db'); + expect(calls).toContain('import:db'); + // Nothing that causes downtime ran. + expect(calls).not.toContain('delta:db'); + }); + + it('skips steps already completed in a previous run', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ + plan, + engines, + from, + to, + ctx: engineCtx(), + completed: new Set(['export:db']), + }); + + expect(calls).not.toContain('export:db'); + expect(res.skipped.some((s) => s.id === 'export:db')).toBe(true); + }); + + it('refuses to run a step whose dependency did not run', async () => { + // An import with no export would restore whatever happens to be in + // staging, possibly from a different migration entirely. + const { plan, engines, from, to, calls } = setup([resource('db')], { + export: async () => { + throw new Error('nope'); + }, + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + expect(res.ok).toBe(false); + expect(calls).not.toContain('import:db'); + }); + + it('stops at the first failure and reports which step', async () => { + const { plan, engines, from, to } = setup([resource('db')], { + import: async () => { + throw new Error('restore blew up'); + }, + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(false); + expect(res.failed?.id).toBe('import:db'); + expect(res.failed?.error.message).toContain('restore blew up'); + }); + + it('fails the migration when verification does not pass', async () => { + const { plan, engines, from, to } = setup([resource('db')], { + verify: async () => ({ ok: false, checks: [], problems: ['public.users: 113 → 9'] }), + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(false); + expect(res.failed?.error.message).toContain('113 → 9'); + }); + + it('reports every step to the callback so progress can be persisted', async () => { + const { plan, engines, from, to } = setup([resource('db')]); + const onStep = vi.fn(); + await applyPlan({ plan, engines, from, to, ctx: engineCtx(), onStep }); + expect(onStep).toHaveBeenCalled(); + }); + + it('runs the delta against the time the bulk copy started, not the freeze', async () => { + const seen: Date[] = []; + const { plan, engines, from, to } = setup([resource('db')], { + delta: async (_c, _r, since) => { + seen.push(since); + return []; + }, + }); + const bulkStartedAt = new Date('2026-09-24T20:00:00Z'); + await applyPlan({ plan, engines, from, to, ctx: engineCtx(), bulkStartedAt }); + expect(seen[0]?.toISOString()).toBe('2026-09-24T20:00:00.000Z'); + }); + + it('never runs a later phase before an earlier one', async () => { + const order: string[] = []; + const { plan, engines, from, to } = setup([resource('db')]); + await applyPlan({ + plan, + engines, + from, + to, + ctx: engineCtx(), + onStep: (step) => { + order.push(step.phase); + }, + }); + const idx = order.map((p) => ['check', 'bulk', 'freeze', 'delta', 'cutover', 'enable', 'verify'].indexOf(p)); + expect(idx).toEqual([...idx].sort((a, b) => a - b)); + }); +}); + +describe('the staging ledger', () => { + it('reads back what was written', () => { + const text = '{"resourceId":"db","kind":"postgres","path":"db/dump.pgc"}\n'; + expect(parseLedger(text)).toHaveLength(1); + }); + + it('skips a truncated final line, which is what an interrupted run leaves', () => { + const text = + '{"resourceId":"db","kind":"postgres","path":"a"}\n{"resourceId":"db","kind":"post'; + const out = parseLedger(text); + expect(out).toHaveLength(1); + expect(out[0]?.path).toBe('a'); + }); + + it('ignores blank lines', () => { + expect(parseLedger('\n\n')).toEqual([]); + }); + + it('drops an entry missing the fields that identify it', () => { + expect(parseLedger('{"nope":true}\n')).toEqual([]); + }); + + it('keeps artifacts per resource', async () => { + const s = memoryStaging(); + await s.record({ resourceId: 'a', kind: 'postgres', path: 'a/1' }); + await s.record({ resourceId: 'b', kind: 'postgres', path: 'b/1' }); + expect(await s.existing('a')).toHaveLength(1); + }); +}); diff --git a/packages/migrate/src/apply.ts b/packages/migrate/src/apply.ts new file mode 100644 index 00000000..81e07aca --- /dev/null +++ b/packages/migrate/src/apply.ts @@ -0,0 +1,210 @@ +import type { MigrationPlan, Phase, Step } from './plan.js'; +import { PHASES } from './plan.js'; +import type { Artifact, Engine, EngineContext, Resource, ResourceKind } from './types.js'; + +/** + * Running a plan. + * + * The planner decided what happens and in what order; this does it, and its + * only real job is refusing to deviate. Two things make a migration + * catastrophic rather than merely failed: doing a later phase before an + * earlier one, and carrying on after a step that was supposed to be a gate. + * Both are prevented here rather than trusted to the caller. + * + * Everything that touches the world is injected — `exec`, the clock, the + * staging — so the whole executor is exercised without a network, a database, + * or a wall-clock wait. + */ + +export interface ApplyOptions { + plan: MigrationPlan; + /** Source resources by id. */ + from: Map; + /** Target resources by id, already resolved by the target platform. */ + to: Map; + engines: Map; + ctx: EngineContext; + /** + * Stop before this phase. `--until freeze` runs the whole bulk copy and + * stops before anything goes down, which is how a migration is rehearsed + * against production without a cutover. + */ + until?: Phase; + /** Steps already completed, from a previous run. */ + completed?: Set; + /** When the bulk copy started, for the delta. Defaults to now at freeze. */ + bulkStartedAt?: Date; + /** Called after each step so a caller can persist progress. */ + onStep?: (step: Step, outcome: StepOutcome) => void | Promise; +} + +export interface StepOutcome { + status: 'done' | 'skipped' | 'failed'; + reason?: string; + artifacts?: Artifact[]; + error?: Error; +} + +export interface ApplyResult { + completed: string[]; + skipped: Array<{ id: string; reason: string }>; + failed?: { id: string; error: Error }; + /** True when everything up to `until` ran. */ + ok: boolean; + stoppedBefore?: Phase; +} + +/** + * Execute a plan. + * + * Refuses a plan with blockers. The planner already said it was not + * applyable, and the one thing worse than a migration that will not start is + * one that starts anyway. + */ +export async function applyPlan(opts: ApplyOptions): Promise { + const { plan, ctx, engines } = opts; + + if (!plan.ok) { + const blockers = plan.risks.filter((r) => r.severity === 'blocker').map((r) => r.message); + throw new Error(`refusing to apply a plan with blockers:\n ${blockers.join('\n ')}`); + } + + const completed = new Set(opts.completed ?? []); + const done: string[] = []; + const skipped: Array<{ id: string; reason: string }> = []; + const stopIndex = opts.until ? PHASES.indexOf(opts.until) : PHASES.length; + const artifactsByResource = new Map(); + let bulkStartedAt = opts.bulkStartedAt; + + for (const step of plan.steps) { + if (PHASES.indexOf(step.phase) >= stopIndex) { + return { + completed: done, + skipped, + ok: true, + stoppedBefore: opts.until!, + }; + } + + if (completed.has(step.id)) { + skipped.push({ id: step.id, reason: 'already done in a previous run' }); + await opts.onStep?.(step, { status: 'skipped', reason: 'already done' }); + continue; + } + + /* + * A step whose dependency did not run must not run either. The planner + * ordered the steps, but a resumed run or a skipped step can leave a gap, + * and "import" running without its "export" would restore whatever was in + * staging from a previous, possibly different, migration. + */ + const missing = step.after.filter( + (dep) => plan.steps.some((s) => s.id === dep) && !completed.has(dep) && !done.includes(dep), + ); + if (missing.length) { + skipped.push({ id: step.id, reason: `depends on ${missing.join(', ')}, which did not run` }); + await opts.onStep?.(step, { status: 'skipped', reason: `unmet dependency: ${missing[0]}` }); + continue; + } + + // The clock for the delta starts when the bulk copy starts, not when the + // freeze does: anything written during the bulk copy is exactly what the + // delta has to catch. + if (step.phase === 'bulk' && !bulkStartedAt) bulkStartedAt = new Date(); + + try { + const outcome = await runStep(step, { + ...opts, + engines, + ctx, + artifactsByResource, + bulkStartedAt: bulkStartedAt ?? new Date(), + }); + if (outcome.status === 'skipped') { + skipped.push({ id: step.id, reason: outcome.reason ?? 'skipped' }); + } else { + done.push(step.id); + completed.add(step.id); + } + await opts.onStep?.(step, outcome); + } catch (err) { + const error = err instanceof Error ? err : new Error(String(err)); + await opts.onStep?.(step, { status: 'failed', error }); + ctx.log(`step '${step.id}' failed: ${error.message}`, 'error'); + return { completed: done, skipped, failed: { id: step.id, error }, ok: false }; + } + } + + return { completed: done, skipped, ok: true }; +} + +interface RunContext extends ApplyOptions { + artifactsByResource: Map; + bulkStartedAt: Date; +} + +async function runStep(step: Step, rc: RunContext): Promise { + const { ctx, engines, from, to, artifactsByResource } = rc; + + // Steps with no resource are gates and instructions: the freeze, the DNS + // cutover, the URL rewrite. They are real work, but not work this executor + // can do unattended — pointing DNS at a new host is not something to do on + // a caller's behalf without being asked very explicitly. + if (!step.resourceId) { + ctx.log(`${step.phase}: ${step.title}`); + return { status: 'done' }; + } + + const source = from.get(step.resourceId); + const target = to.get(step.resourceId); + const kind = step.kind; + const engine = kind ? engines.get(kind) : undefined; + + if (!engine || !source) { + return { status: 'skipped', reason: `no engine or source for ${step.resourceId}` }; + } + + const verb = step.id.split(':')[0]; + + switch (verb) { + case 'export': { + const artifacts = await engine.export(ctx, source); + artifactsByResource.set(step.resourceId, artifacts); + return { status: 'done', artifacts }; + } + case 'import': { + if (!target) return { status: 'skipped', reason: `no target resolved for ${step.resourceId}` }; + const staged = + artifactsByResource.get(step.resourceId) ?? (await ctx.staging.existing(step.resourceId)); + await engine.import(ctx, target, staged, source); + return { status: 'done' }; + } + case 'delta': { + if (!engine.delta) return { status: 'skipped', reason: 'engine has no delta' }; + const artifacts = await engine.delta(ctx, source, rc.bulkStartedAt); + if (target && artifacts.length) await engine.import(ctx, target, artifacts, source); + return { status: 'done', artifacts }; + } + case 'verify': { + if (!engine.verify) return { status: 'skipped', reason: 'engine has no verify' }; + if (!target) return { status: 'skipped', reason: 'no target to compare against' }; + const result = await engine.verify(ctx, source, target); + for (const line of result.checks) ctx.log(` ${line}`); + if (!result.ok) { + throw new Error(`verification failed for ${source.name}:\n ${result.problems.join('\n ')}`); + } + return { status: 'done' }; + } + case 'disable': + case 'enable': + // Scheduled jobs. Surfaced rather than automated, for the same reason as + // the DNS cutover: turning cron on at the wrong moment is the failure + // the phase ordering exists to prevent, and doing it silently would put + // the decision back in the tool. + ctx.log(`${step.phase}: ${step.title}`, 'warn'); + return { status: 'done' }; + default: + ctx.log(`${step.phase}: ${step.title}`); + return { status: 'done' }; + } +} diff --git a/packages/migrate/src/engines/files.ts b/packages/migrate/src/engines/files.ts new file mode 100644 index 00000000..34a07e88 --- /dev/null +++ b/packages/migrate/src/engines/files.ts @@ -0,0 +1,158 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a directory of files — a mounted volume, an uploads directory, a + * docroot. + * + * This is the engine for the thing that is not a database and not a bucket: + * Railway volumes, Fly volumes, and `~/www/` on a dedicated box. rsync + * does it because rsync is what does this, and because its delta algorithm + * makes the second pass — the one during the cutover window — proportional to + * what changed rather than to the size of the tree. + * + * As with object storage, `--delete` is never passed. rsync's delete is the + * single most effective way to destroy a directory by typing the wrong target, + * and a migration has no reason to remove anything. + */ + +/** + * `[user@host:]/path/` as rsync wants it. + * + * The trailing slash is always present and always matters: with it rsync + * copies the CONTENTS of the directory, without it it nests the directory + * inside the destination, producing `~/www/app/app/` — which looks from the + * outside like the copy silently did nothing. + */ +export function endpoint(r: Resource): string { + const path = r.connection.path?.reveal(); + if (!path) throw new Error(`files resource '${r.name}' has no connection.path`); + const withSlash = `${path.replace(/\/*$/, '')}/`; + const host = r.connection.host?.reveal(); + if (!host) return withSlash; + const user = r.connection.user?.reveal(); + return `${user ? `${user}@` : ''}${host}:${withSlash}`; +} + +export function isRemote(r: Resource): boolean { + return Boolean(r.connection.host); +} + +/** + * rsync cannot copy remote to remote. + * + * It is a hard limitation of the protocol, not a flag that was missed: one + * side must be local. A VPS-to-VPS move therefore has to relay through the + * machine running the migration, which is a real cost (the bytes cross the + * wire twice) and needs to be said out loud rather than discovered when rsync + * exits with "The source and destination cannot both be remote." + */ +export function assertCopyable(from: Resource, to: Resource): void { + if (isRemote(from) && isRemote(to)) { + throw new Error( + `rsync cannot copy directly between two remote hosts (${from.name} → ${to.name}). Run the migration from one of them, or stage the directory locally first.`, + ); + } +} + +/** The ssh transport, including a key when one is configured. */ +function rsyncTransport(r: Resource): string[] { + const key = r.connection.sshKeyPath?.reveal(); + const port = r.connection.sshPort?.reveal(); + if (!r.connection.host) return []; + const parts = ['ssh', '-o', 'BatchMode=yes']; + if (key) parts.push('-i', key); + if (port) parts.push('-p', port); + return ['-e', parts.join(' ')]; +} + +export const filesEngine: Engine = { + kind: 'files', + requires: ['rsync'], + + /** + * Nothing is staged: like object storage, files go host to host. The export + * records what is there so `verify` has something to compare and so a plan + * can show a size. + */ + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/files.manifest`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'files', path }]; + + const res = await ctx.exec( + 'rsync', + [...rsyncTransport(from), '--dry-run', '--archive', '--stats', endpoint(from), '/dev/null'], + { check: false }, + ); + + const files = /Number of files: ([\d,]+)/.exec(res.stdout)?.[1]?.replace(/,/g, ''); + const artifact: Artifact = { + resourceId: from.id, + kind: 'files', + path, + metadata: { fileCount: files ? Number(files) : 0 }, + }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, _artifacts: Artifact[], from?: Resource): Promise { + if (!from) throw new Error('files import needs the source resource: files are copied host to host'); + assertCopyable(from, to); + + if (ctx.dryRun) { + ctx.log(`would rsync ${endpoint(from)} → ${endpoint(to)}`); + return; + } + + await ctx.exec( + 'rsync', + [ + ...rsyncTransport(isRemote(from) ? from : to), + '--archive', + '--compress', + '--partial', + '--human-readable', + endpoint(from), + endpoint(to), + ], + { timeoutMs: 12 * 60 * 60 * 1000 }, + ); + }, + + /** The second pass, during the cutover window: only what changed. */ + async delta(ctx: EngineContext, from: Resource, _since: Date): Promise { + const path = `${from.id}/files.delta`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'files', path }]; + // rsync compares by size and mtime on its own, so the delta pass is the + // same command. It is fast because almost nothing has changed. + const artifact: Artifact = { resourceId: from.id, kind: 'files', path, metadata: { mode: 'delta' } }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + // A dry-run rsync from source to target lists exactly what still differs. + // Zero transfers is the pass condition. + const res = await ctx.exec( + 'rsync', + [ + ...rsyncTransport(isRemote(from) ? from : to), + '--dry-run', + '--archive', + '--itemize-changes', + endpoint(from), + endpoint(to), + ], + { check: false }, + ); + + const differing = res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + return { + ok: differing.length === 0, + checks: [`${differing.length} path(s) still differ`], + problems: differing.slice(0, 20), + }; + }, +}; diff --git a/packages/migrate/src/engines/index.test.ts b/packages/migrate/src/engines/index.test.ts new file mode 100644 index 00000000..2cbce881 --- /dev/null +++ b/packages/migrate/src/engines/index.test.ts @@ -0,0 +1,202 @@ +import { describe, expect, it } from 'vitest'; +import { ENGINES, engineFor, requiredBinaries } from './index.js'; +import { assertCopyable, endpoint, filesEngine, isRemote } from './files.js'; +import { redisEngine } from './redis.js'; +import { sqliteEngine } from './sqlite.js'; +import type { EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; +import { RESOURCE_KINDS } from '../types.js'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +describe('the engine registry', () => { + it('registers every engine under its own kind', () => { + for (const [kind, engine] of ENGINES) expect(engine.kind).toBe(kind); + }); + + it('declares its required binaries so the planner can check them', () => { + for (const engine of ENGINES.values()) { + expect(Array.isArray(engine.requires)).toBe(true); + expect(engine.requires.length).toBeGreaterThan(0); + } + }); + + it('collects the binaries needed for a set of kinds, without duplicates', () => { + const bins = requiredBinaries(['postgres', 'files', 'postgres']); + expect(bins).toContain('pg_dump'); + expect(bins).toContain('rsync'); + expect(new Set(bins).size).toBe(bins.length); + }); + + it('returns nothing for a kind no engine handles', () => { + // env, cron and dns are real resource kinds with no byte-moving engine; + // they are handled as plan steps rather than copies. + expect(engineFor('env')).toBeUndefined(); + expect(engineFor('cron')).toBeUndefined(); + }); + + it('covers a documented subset of the resource kinds', () => { + const covered = [...ENGINES.keys()]; + for (const k of covered) expect(RESOURCE_KINDS).toContain(k); + }); +}); + +describe('sqlite / libSQL', () => { + const file = (over: Partial = {}): Resource => ({ + kind: 'sqlite', + id: 'db', + name: 'app.db', + connection: { path: plain('/data/app.db') }, + ...over, + }); + const turso = (): Resource => ({ + kind: 'sqlite', + id: 'db', + name: 'prod', + connection: { tursoDatabase: plain('prod'), authToken: secret('tok') }, + }); + + it('dumps a local file to staging via stdout redirection', async () => { + const c = ctx(); + await sqliteEngine.export(c, file()); + expect(c.calls[0]!.cmd).toBe('sqlite3'); + expect(c.calls[0]!.args).toContain('.dump'); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + }); + + it('dumps a Turso database with the same engine', async () => { + const c = ctx(); + await sqliteEngine.export(c, turso()); + expect(c.calls[0]!.cmd).toBe('turso'); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + }); + + it('keeps the Turso token out of argv', async () => { + const c = ctx(); + await sqliteEngine.export(c, turso()); + expect(c.calls[0]!.args.join(' ')).not.toContain('tok'); + expect(c.calls[0]!.opts?.env?.TURSO_API_TOKEN).toBe('tok'); + }); + + it('loads into Turso from stdin, which is the direction that makes it bidirectional', async () => { + const c = ctx(); + await sqliteEngine.import(c, turso(), [{ resourceId: 'db', kind: 'sqlite', path: 'db/dump.sql' }]); + expect(c.calls[0]!.opts?.stdinFile).toBe('/staging/db/dump.sql'); + }); + + it('refuses a resource with no path', async () => { + const c = ctx(); + await expect(sqliteEngine.export(c, file({ connection: {} }))).rejects.toThrow(/no connection.path/); + }); +}); + +describe('redis', () => { + const r = (over: Partial = {}): Resource => ({ + kind: 'redis', + id: 'r', + name: 'cache', + connection: { url: secret('redis://h:6379') }, + ...over, + }); + + it('uses --rdb, which is consistent, rather than walking keys, which is not', async () => { + const c = ctx(); + await redisEngine.export(c, r()); + expect(c.calls[0]!.args).toContain('--rdb'); + }); + + it('has no delta, so the planner warns that writes during the copy are lost', () => { + expect(redisEngine.delta).toBeUndefined(); + }); + + it('refuses to load over the wire and says what to do instead', async () => { + const c = ctx(); + await expect( + redisEngine.import(c, r(), [{ resourceId: 'r', kind: 'redis', path: 'r/dump.rdb' }]), + ).rejects.toThrow(/data directory/); + }); + + it('places the file when the target declares a data directory', async () => { + const c = ctx(); + const target = r({ connection: { url: secret('redis://h'), dataDir: plain('/var/lib/redis') } }); + await redisEngine.import(c, target, [{ resourceId: 'r', kind: 'redis', path: 'r/dump.rdb' }]); + expect(c.calls[0]!.args[1]).toBe('/var/lib/redis/dump.rdb'); + }); +}); + +describe('files', () => { + const local = (): Resource => ({ + kind: 'files', + id: 'v', + name: 'uploads', + connection: { path: plain('/data/uploads') }, + }); + const remote = (over: Record = {}): Resource => ({ + kind: 'files', + id: 'v2', + name: 'www', + connection: { + path: plain('/home/anthony/www'), + host: plain('dev2.example.com'), + user: plain('anthony'), + ...Object.fromEntries(Object.entries(over).map(([k, v]) => [k, plain(v)])), + }, + }); + + it('always ends an endpoint with a slash, so rsync copies contents not the directory', () => { + expect(endpoint(local())).toBe('/data/uploads/'); + expect(endpoint(remote())).toBe('anthony@dev2.example.com:/home/anthony/www/'); + }); + + it('knows which side is remote', () => { + expect(isRemote(local())).toBe(false); + expect(isRemote(remote())).toBe(true); + }); + + it('refuses remote-to-remote, which rsync cannot do at all', () => { + expect(() => assertCopyable(remote(), remote())).toThrow(/cannot copy directly between two remote/); + }); + + it('allows a copy when one side is local', () => { + expect(() => assertCopyable(remote(), local())).not.toThrow(); + expect(() => assertCopyable(local(), remote())).not.toThrow(); + }); + + it('never passes --delete, which is how a wrong target destroys a directory', async () => { + const c = ctx(); + await filesEngine.import(c, remote(), [], local()); + const args = c.calls[0]!.args; + expect(args).not.toContain('--delete'); + expect(args.some((a) => a.startsWith('--delete'))).toBe(false); + }); + + it('carries an ssh key and port into the transport when configured', async () => { + const c = ctx(); + await filesEngine.import(c, remote({ sshKeyPath: '/k/id', sshPort: '2222' }), [], local()); + const e = c.calls[0]!.args[c.calls[0]!.args.indexOf('-e') + 1]; + expect(e).toContain('-i /k/id'); + expect(e).toContain('-p 2222'); + }); +}); diff --git a/packages/migrate/src/engines/index.ts b/packages/migrate/src/engines/index.ts new file mode 100644 index 00000000..d95bfb0d --- /dev/null +++ b/packages/migrate/src/engines/index.ts @@ -0,0 +1,38 @@ +import type { Engine, ResourceKind } from '../types.js'; +import { filesEngine } from './files.js'; +import { mysqlEngine } from './mysql.js'; +import { objectStorageEngine } from './object-storage.js'; +import { postgresEngine } from './postgres.js'; +import { redisEngine } from './redis.js'; +import { sqliteEngine } from './sqlite.js'; + +/** + * Every engine this build can run, keyed by what it moves. + * + * The planner takes this map and refuses, up front, to plan a migration for a + * kind that is not in it — which is the difference between "we do not support + * that" printed before anything happens and a crash after the freeze. + */ +export const ENGINES: ReadonlyMap = new Map([ + ['postgres', postgresEngine], + ['mysql', mysqlEngine], + ['sqlite', sqliteEngine], + ['redis', redisEngine], + ['object-storage', objectStorageEngine], + ['files', filesEngine], +]); + +export function engineFor(kind: ResourceKind): Engine | undefined { + return ENGINES.get(kind); +} + +/** Every binary any engine needs, for a one-shot preflight check. */ +export function requiredBinaries(kinds: ResourceKind[]): string[] { + const out = new Set(); + for (const kind of kinds) { + for (const bin of ENGINES.get(kind)?.requires ?? []) out.add(bin); + } + return [...out].sort(); +} + +export { filesEngine, mysqlEngine, objectStorageEngine, postgresEngine, redisEngine, sqliteEngine }; diff --git a/packages/migrate/src/engines/mysql.test.ts b/packages/migrate/src/engines/mysql.test.ts new file mode 100644 index 00000000..ba6e154a --- /dev/null +++ b/packages/migrate/src/engines/mysql.test.ts @@ -0,0 +1,152 @@ +import { describe, expect, it } from 'vitest'; +import { databaseName, mysqlConnection, mysqlEngine } from './mysql.js'; +import type { Artifact, EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { secret } from '../types.js'; + +/* Named rather than inlined; see postgres.test.ts. */ +const FAKE_PASSWORD = 'p%40ss'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +const db = (over: Partial = {}): Resource => ({ + kind: 'mysql', + id: 'db', + name: 'app', + connection: { url: secret(`mysql://u:${FAKE_PASSWORD}@db.example.com:3306/appdb`) }, + ...over, +}); + +describe('mysqlConnection', () => { + it('splits host, port and user into flags', () => { + const { args } = mysqlConnection(db()); + expect(args).toContain('--host=db.example.com'); + expect(args).toContain('--port=3306'); + expect(args).toContain('--user=u'); + }); + + it('puts the password in MYSQL_PWD, never in argv where ps can read it', () => { + const { args, env } = mysqlConnection(db()); + expect(env.MYSQL_PWD).toBe('p@ss'); + expect(args.join(' ')).not.toContain('p@ss'); + expect(args.some((a) => a.startsWith('--password'))).toBe(false); + }); + + it('requires TLS by default', () => { + expect(mysqlConnection(db()).args).toContain('--ssl-mode=REQUIRED'); + }); + + it('honours an explicit ssl-mode', () => { + const r = db({ connection: { url: secret('mysql://u@h:3306/d?ssl-mode=DISABLED') } }); + expect(mysqlConnection(r).args).toContain('--ssl-mode=DISABLED'); + }); + + it('refuses a url that is not a URL', () => { + expect(() => mysqlConnection(db({ connection: { url: secret('nope') } }))).toThrow(/not a URL/); + }); +}); + +describe('databaseName', () => { + it('is the path of the DSN', () => { + expect(databaseName(db())).toBe('appdb'); + }); + + it('refuses a DSN naming no database', () => { + expect(() => databaseName(db({ connection: { url: secret('mysql://u@h:3306/') } }))).toThrow( + /no database/, + ); + }); +}); + +describe('export', () => { + it('dumps in one transaction rather than locking every table for hours', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.cmd).toBe('mysqldump'); + expect(c.calls[0]!.args).toContain('--single-transaction'); + }); + + it('turns off GTID state, which otherwise refuses to load elsewhere', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.args).toContain('--set-gtid-purged=OFF'); + }); + + it('skips tablespaces, which need a privilege managed providers do not grant', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.args).toContain('--no-tablespaces'); + }); + + it('writes to staging via stdout redirection', async () => { + const c = ctx(); + const [artifact] = await mysqlEngine.export(c, db()); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + expect(artifact?.path).toBe('db/dump.sql'); + }); + + it('runs nothing on a dry run', async () => { + const c = ctx([], { dryRun: true }); + await mysqlEngine.export(c, db()); + expect(c.calls).toHaveLength(0); + }); +}); + +describe('import', () => { + const dump: Artifact = { resourceId: 'db', kind: 'mysql', path: 'db/dump.sql' }; + + it('loads into an empty database from stdin', async () => { + const c = ctx([{ stdout: '0' }, {}]); + await mysqlEngine.import(c, db(), [dump]); + expect(c.calls[1]!.opts?.stdinFile).toBe('/staging/db/dump.sql'); + }); + + it('refuses to load over a database that already has tables', async () => { + const c = ctx([{ stdout: '17' }]); + await expect(mysqlEngine.import(c, db(), [dump])).rejects.toThrow(/already has 17 table/); + expect(c.calls).toHaveLength(1); + }); + + it('fails when nothing was staged', async () => { + await expect(mysqlEngine.import(ctx(), db(), [])).rejects.toThrow(/no mysql dump staged/); + }); +}); + +describe('verify', () => { + it('reports estimated counts without failing on them', async () => { + // information_schema.table_rows is an estimate on InnoDB. Treating it as + // exact would fail every single verification. + const c = ctx([{ stdout: 'users\t113\n' }, { stdout: 'users\t108\n' }]); + const res = await mysqlEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + expect(res.checks[0]).toContain('estimated'); + }); + + it('fails only when a table is missing entirely', async () => { + const c = ctx([{ stdout: 'users\t1\nposts\t2\n' }, { stdout: 'users\t1\n' }]); + const res = await mysqlEngine.verify!(c, db(), db()); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('posts'); + }); +}); diff --git a/packages/migrate/src/engines/mysql.ts b/packages/migrate/src/engines/mysql.ts new file mode 100644 index 00000000..2737986c --- /dev/null +++ b/packages/migrate/src/engines/mysql.ts @@ -0,0 +1,196 @@ +import { quoteIdent, quoteLiteral } from '../transforms.js'; +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a MySQL or MariaDB database. + * + * PlanetScale, RDS, a MariaDB in a container. Same shape as the Postgres + * engine and for the same reasons; the differences are all in the tooling. + * + * ## The flags that matter + * + * `--single-transaction` takes the dump inside one consistent snapshot on + * InnoDB instead of locking every table for the length of the dump. Without + * it, a multi-gigabyte dump is an outage, which rather defeats the point of + * copying while the source is still live. + * + * `--set-gtid-purged=OFF`, because a dump carrying GTID state refuses to load + * into a server with its own replication history, and the error names neither + * the flag nor the cause. + * + * `--no-tablespaces`, because writing tablespace clauses needs PROCESS + * privilege that a managed provider does not grant, and its absence fails the + * dump rather than degrading it. + * + * PlanetScale specifically does not support foreign key constraints in the + * usual way, so a dump taken there restores without constraints a plain MySQL + * would have had. That is recorded as a platform quirk rather than silently + * handled, because the fix is a schema decision, not a flag. + */ + +const DUMP = 'dump.sql'; + +function dsn(r: Resource): string { + const url = r.connection.url; + if (!url) throw new Error(`mysql resource '${r.name}' has no connection.url`); + return url.reveal(); +} + +/** + * Connection details for the mysql client family. + * + * `MYSQL_PWD` rather than `--password=`, for the same reason Postgres uses + * libpq's variables: a password on the command line is readable by every other + * user on the box via `ps` for as long as the dump runs. The client warns + * about `MYSQL_PWD` being insecure on shared machines, which is true and still + * strictly better than argv. + */ +export function mysqlConnection(r: Resource): { args: string[]; env: Record } { + const raw = dsn(r); + let u: URL; + try { + u = new URL(raw); + } catch { + throw new Error(`mysql resource '${r.name}' has a connection.url that is not a URL`); + } + + const args: string[] = []; + if (u.hostname) args.push(`--host=${decodeURIComponent(u.hostname)}`); + if (u.port) args.push(`--port=${u.port}`); + if (u.username) args.push(`--user=${decodeURIComponent(u.username)}`); + + // Managed MySQL is TLS-only in practice, and the client's default is to fall + // back silently where it is not enforced. + const sslMode = u.searchParams.get('ssl-mode') ?? 'REQUIRED'; + args.push(`--ssl-mode=${sslMode}`); + + const env: Record = {}; + if (u.password) env.MYSQL_PWD = decodeURIComponent(u.password); + + return { args, env }; +} + +export function databaseName(r: Resource): string { + const raw = dsn(r); + const name = new URL(raw).pathname.replace(/^\//, ''); + if (!name) throw new Error(`mysql resource '${r.name}' has no database in its connection.url`); + return decodeURIComponent(name); +} + +export const mysqlEngine: Engine = { + kind: 'mysql', + requires: ['mysqldump', 'mysql'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`mysqldump ${from.name} → ${path}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'mysql', path }]; + + const { args, env } = mysqlConnection(from); + await ctx.exec( + 'mysqldump', + [ + ...args, + // A consistent snapshot instead of locking every table for the length + // of the dump. + '--single-transaction', + '--quick', + '--routines', + '--triggers', + '--events', + '--set-gtid-purged=OFF', + '--no-tablespaces', + databaseName(from), + ], + { env, stdoutFile: `${ctx.staging.dir}/${path}`, timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + const artifact: Artifact = { resourceId: from.id, kind: 'mysql', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'mysql'); + if (!dump) throw new Error(`no mysql dump staged for '${to.name}'`); + if (ctx.dryRun) { + ctx.log(`would load ${dump.path} into ${to.name}`); + return; + } + + const { args, env } = mysqlConnection(to); + + // Same refusal as Postgres: a non-empty target is not written over. A + // mysqldump replays CREATE TABLE and INSERT, so aiming it at a populated + // database is a mess of duplicate-key errors on top of live data. + const existing = await ctx.exec( + 'mysql', + [ + ...args, + '--batch', + '--skip-column-names', + '--execute', + `select count(*) from information_schema.tables where table_schema = ${quoteLiteral(databaseName(to))}`, + ], + { env, check: false }, + ); + const tableCount = Number.parseInt(existing.stdout.trim(), 10); + if (Number.isFinite(tableCount) && tableCount > 0) { + throw new Error( + `target database '${to.name}' already has ${tableCount} table(s). Refusing to load over it — drop and recreate it, or point at an empty one.`, + ); + } + + ctx.log(`mysql < ${dump.path}`); + await ctx.exec('mysql', [...args, databaseName(to)], { + env, + stdinFile: `${ctx.staging.dir}/${dump.path}`, + timeoutMs: 6 * 60 * 60 * 1000, + }); + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const counts = async (r: Resource) => { + const { args, env } = mysqlConnection(r); + const res = await ctx.exec( + 'mysql', + [ + ...args, + '--batch', + '--skip-column-names', + '--execute', + `select table_name, table_rows from information_schema.tables + where table_schema = ${quoteLiteral(databaseName(r))} order by table_name`, + ], + { env, check: false }, + ); + const map = new Map(); + for (const line of res.stdout.split('\n')) { + const [name, n] = line.split('\t'); + if (name && n !== undefined) map.set(name.trim(), Number.parseInt(n, 10) || 0); + } + return map; + }; + + const [a, b] = await Promise.all([counts(from), counts(to)]); + const checks: string[] = []; + const problems: string[] = []; + + for (const [table, sourceRows] of a) { + const targetRows = b.get(table); + if (targetRows === undefined) { + problems.push(`${table}: missing on the target`); + continue; + } + // information_schema.table_rows is an ESTIMATE on InnoDB, not a count. + // Treating it as exact would fail every verification, so a table present + // on both sides is reported rather than judged, and only a missing table + // is a problem. + checks.push(`${quoteIdent(table)}: ~${sourceRows} → ~${targetRows} (estimated)`); + } + + return { ok: problems.length === 0, checks, problems }; + }, +}; diff --git a/packages/migrate/src/engines/object-storage.test.ts b/packages/migrate/src/engines/object-storage.test.ts new file mode 100644 index 00000000..e2c50683 --- /dev/null +++ b/packages/migrate/src/engines/object-storage.test.ts @@ -0,0 +1,171 @@ +import { describe, expect, it } from 'vitest'; +import { objectStorageEngine, rcloneEnv, rclonePath } from './object-storage.js'; +import type { EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +const bucket = (over: Partial = {}): Resource => ({ + kind: 'object-storage', + id: 'b1', + name: 'public', + connection: { + bucket: plain('ads'), + accessKeyId: secret('AKIAEXAMPLE'), + secretAccessKey: secret('supersecret'), + endpoint: plain('https://abc.supabase.co/storage/v1/s3'), + region: plain('us-east-1'), + }, + ...over, +}); + +describe('rcloneEnv', () => { + it('builds a remote entirely from the environment so nothing is written to disk', () => { + const env = rcloneEnv(bucket(), 'src'); + expect(env.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(env.RCLONE_CONFIG_SRC_ACCESS_KEY_ID).toBe('AKIAEXAMPLE'); + expect(env.RCLONE_CONFIG_SRC_SECRET_ACCESS_KEY).toBe('supersecret'); + }); + + it('forces path style for a custom endpoint, which otherwise resolves to a host that does not exist', () => { + expect(rcloneEnv(bucket(), 'src').RCLONE_CONFIG_SRC_FORCE_PATH_STYLE).toBe('true'); + }); + + it('leaves path style alone for real AWS', () => { + const r = bucket({ connection: { bucket: plain('b') } }); + expect(rcloneEnv(r, 'src').RCLONE_CONFIG_SRC_FORCE_PATH_STYLE).toBeUndefined(); + }); + + it('namespaces by alias so source and target can both be configured at once', () => { + const merged = { ...rcloneEnv(bucket(), 'src'), ...rcloneEnv(bucket(), 'dst') }; + expect(merged.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(merged.RCLONE_CONFIG_DST_TYPE).toBe('s3'); + }); + + it('refuses a resource with no bucket', () => { + expect(() => rcloneEnv(bucket({ connection: {} }), 'src')).toThrow(/no connection.bucket/); + }); +}); + +describe('rclonePath', () => { + it('is remote:bucket', () => { + expect(rclonePath(bucket(), 'src')).toBe('src:ads'); + }); + + it('appends a prefix when there is one', () => { + const r = bucket({ connection: { bucket: plain('ads'), prefix: plain('/2026') } }); + expect(rclonePath(r, 'src')).toBe('src:ads/2026'); + }); +}); + +describe('export', () => { + it('lists the bucket and records the object count rather than downloading it', async () => { + const c = ctx([{ stdout: JSON.stringify([{ Path: 'a.png', Size: 10 }, { Path: 'b.png', Size: 20 }]) }]); + const [artifact] = await objectStorageEngine.export(c, bucket()); + + expect(c.calls[0]!.cmd).toBe('rclone'); + expect(c.calls[0]!.args).toContain('lsjson'); + expect(artifact?.sizeBytes).toBe(30); + expect(artifact?.metadata?.objectCount).toBe(2); + }); + + it('never puts a secret in argv', async () => { + const c = ctx([{ stdout: '[]' }]); + await objectStorageEngine.export(c, bucket()); + expect(c.calls[0]!.args.join(' ')).not.toContain('supersecret'); + }); + + it('throws on an unparseable listing rather than silently copying nothing', async () => { + const c = ctx([{ stdout: 'not json' }]); + await expect(objectStorageEngine.export(c, bucket())).rejects.toThrow(/could not parse/); + }); +}); + +describe('import', () => { + it('copies remote to remote with both remotes configured', async () => { + const c = ctx(); + const to = bucket({ id: 'b2', name: 'dest', connection: { bucket: plain('dest-ads') } }); + await objectStorageEngine.import(c, to, [], bucket()); + + const call = c.calls[0]!; + expect(call.args[0]).toBe('copy'); + expect(call.args[1]).toBe('src:ads'); + expect(call.args[2]).toBe('dst:dest-ads'); + expect(call.opts?.env?.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(call.opts?.env?.RCLONE_CONFIG_DST_TYPE).toBe('s3'); + }); + + it('uses copy and never sync, so a mistyped target cannot delete a live bucket', async () => { + const c = ctx(); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls[0]!.args).not.toContain('sync'); + expect(c.calls[0]!.args).not.toContain('--delete'); + expect(c.calls[0]!.args).not.toContain('--delete-during'); + }); + + it('is resume-friendly, so a re-run does not re-send the whole bucket', async () => { + const c = ctx(); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls[0]!.args).toContain('--update'); + }); + + it('refuses without the source, since nothing was staged locally', async () => { + const c = ctx(); + await expect(objectStorageEngine.import(c, bucket(), [])).rejects.toThrow(/needs the source/); + }); + + it('runs nothing on a dry run', async () => { + const c = ctx([], { dryRun: true }); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls).toHaveLength(0); + }); +}); + +describe('delta', () => { + it('asks only for objects written since the bulk copy started', async () => { + const c = ctx([{ stdout: '[{"Path":"new.png"}]' }]); + const since = new Date(Date.now() - 3600_000); + const [artifact] = await objectStorageEngine.delta!(c, bucket(), since); + + const ageArg = c.calls[0]!.args.find((a) => a.startsWith('--max-age=')); + expect(ageArg).toBeDefined(); + expect(artifact?.metadata?.objectCount).toBe(1); + }); +}); + +describe('verify', () => { + it('passes when the object counts match', async () => { + const c = ctx([{ stdout: '{"count":8410,"bytes":9560000000}' }, { stdout: '{"count":8410,"bytes":9560000000}' }]); + const res = await objectStorageEngine.verify!(c, bucket(), bucket({ id: 'b2' })); + expect(res.ok).toBe(true); + }); + + it('fails when objects did not arrive', async () => { + const c = ctx([{ stdout: '{"count":8410,"bytes":1}' }, { stdout: '{"count":8000,"bytes":1}' }]); + const res = await objectStorageEngine.verify!(c, bucket(), bucket({ id: 'b2' })); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('410 object(s) did not arrive'); + }); +}); diff --git a/packages/migrate/src/engines/object-storage.ts b/packages/migrate/src/engines/object-storage.ts new file mode 100644 index 00000000..a59dedfb --- /dev/null +++ b/packages/migrate/src/engines/object-storage.ts @@ -0,0 +1,230 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a bucket of objects. + * + * S3, R2, B2, Spaces, Supabase storage, MinIO on a dedicated box. They all + * speak S3 or something close enough, and the differences are endpoint URLs + * and auth styles rather than anything structural — so this is one engine with + * a remote definition per side, not one engine per vendor. + * + * ## Why rclone rather than an SDK + * + * Copying 8,410 objects across 9.56 GB from a script means reimplementing + * concurrency, retries, resume, multipart thresholds and checksum comparison, + * and getting all five right. rclone has those and is a single static binary. + * The alternative — `aws s3 sync` — only speaks S3 and needs a credentials + * file on disk, which is worse for a tool that has to reach six vendors. + * + * The cost is a dependency the planner checks for up front (`requires`), so a + * missing rclone is a blocker before the freeze rather than a failure during + * it. + * + * ## Staging, or not + * + * Every other engine dumps to staging and loads from it. Object storage is the + * exception: pulling 10 GB down to a laptop and pushing it back up doubles the + * transfer and needs the disk. `export` therefore only writes a manifest, and + * `import` runs a remote-to-remote copy that rclone streams server-side where + * it can. The manifest is not busywork — it is what `verify` compares against, + * and what makes a resumed run able to tell what it already did. + */ + +/** + * An rclone remote, built inline from the resource's connection. + * + * rclone is normally configured from a file; passing the whole remote + * definition through `RCLONE_CONFIG_*` environment variables instead means no + * credential is ever written to disk, and none appears in argv. + */ +export function rcloneEnv(r: Resource, alias: string): Record { + const get = (k: string): string | undefined => r.connection[k]?.reveal(); + const bucket = get('bucket'); + if (!bucket) throw new Error(`object-storage resource '${r.name}' has no connection.bucket`); + + const prefix = `RCLONE_CONFIG_${alias.toUpperCase()}`; + const env: Record = { + [`${prefix}_TYPE`]: 's3', + // 'Other' keeps rclone from applying provider-specific assumptions to an + // endpoint that merely speaks S3, which is the case for R2, Supabase, + // MinIO and Backblaze's S3 gateway. + [`${prefix}_PROVIDER`]: get('provider') ?? 'Other', + }; + + const accessKey = get('accessKeyId'); + const secretKey = get('secretAccessKey'); + if (accessKey) env[`${prefix}_ACCESS_KEY_ID`] = accessKey; + if (secretKey) env[`${prefix}_SECRET_ACCESS_KEY`] = secretKey; + + const endpoint = get('endpoint'); + if (endpoint) env[`${prefix}_ENDPOINT`] = endpoint; + const region = get('region'); + if (region) env[`${prefix}_REGION`] = region; + + // Most non-AWS S3 endpoints are path-style; virtual-host style silently + // resolves to a hostname that does not exist. + if (endpoint && !get('forcePathStyle')) env[`${prefix}_FORCE_PATH_STYLE`] = 'true'; + + return env; +} + +/** `remote:bucket/prefix` as rclone wants it. */ +export function rclonePath(r: Resource, alias: string): string { + const bucket = r.connection.bucket?.reveal() ?? ''; + const prefix = r.connection.prefix?.reveal() ?? ''; + return `${alias}:${bucket}${prefix ? `/${prefix.replace(/^\/+/, '')}` : ''}`; +} + +const MANIFEST = 'objects.json'; + +interface ManifestEntry { + path: string; + size: number; +} + +export const objectStorageEngine: Engine = { + kind: 'object-storage', + requires: ['rclone'], + + /** + * List the bucket. Deliberately does not download it — see the note above. + */ + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${MANIFEST}`; + ctx.log(`listing ${from.name}`); + + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'object-storage', path }]; + + const res = await ctx.exec('rclone', ['lsjson', '--recursive', '--files-only', rclonePath(from, 'src')], { + env: rcloneEnv(from, 'src'), + timeoutMs: 60 * 60 * 1000, + }); + + let entries: ManifestEntry[]; + try { + const parsed = JSON.parse(res.stdout || '[]') as Array<{ Path?: string; Size?: number }>; + entries = parsed.map((e) => ({ path: e.Path ?? '', size: e.Size ?? 0 })).filter((e) => e.path); + } catch { + throw new Error(`could not parse the object listing for '${from.name}'`); + } + + const artifact: Artifact = { + resourceId: from.id, + kind: 'object-storage', + path, + sizeBytes: entries.reduce((n, e) => n + e.size, 0), + metadata: { objectCount: entries.length, manifest: JSON.stringify(entries).length }, + }; + await ctx.staging.record(artifact); + ctx.log(`${entries.length} object(s) to copy`); + return [artifact]; + }, + + /** + * Copy source → target directly. + * + * `copy`, never `sync`: sync deletes anything at the destination that is not + * at the source, which for a mistyped target is indistinguishable from + * wiping a live bucket. Migrations should not be able to delete. + */ + async import( + ctx: EngineContext, + to: Resource, + _artifacts: Artifact[], + from?: Resource, + ): Promise { + const source = from; + if (!source) { + throw new Error( + `object-storage import needs the source resource: its objects are copied remote-to-remote rather than staged locally`, + ); + } + + if (ctx.dryRun) { + ctx.log(`would rclone copy ${rclonePath(source, 'src')} → ${rclonePath(to, 'dst')}`); + return; + } + + ctx.log(`rclone copy → ${to.name}`); + await ctx.exec( + 'rclone', + [ + 'copy', + rclonePath(source, 'src'), + rclonePath(to, 'dst'), + '--transfers=16', + '--checkers=32', + // Resume-friendly: an object already present with the same size and + // modification time is not re-sent, so a re-run after a failure costs + // a listing rather than the whole bucket. + '--update', + '--stats=30s', + ], + { + env: { ...rcloneEnv(source, 'src'), ...rcloneEnv(to, 'dst') }, + timeoutMs: 12 * 60 * 60 * 1000, + }, + ); + }, + + /** + * Objects created during the bulk copy. + * + * rclone's own `--max-age` does this server-side, so the delta is the same + * copy restricted to recent objects rather than a different mechanism. + */ + async delta(ctx: EngineContext, from: Resource, since: Date): Promise { + const path = `${from.id}/delta-${since.toISOString().replace(/[:.]/g, '')}.json`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'object-storage', path }]; + + const ageSeconds = Math.max(1, Math.round((Date.now() - since.getTime()) / 1000)); + const res = await ctx.exec( + 'rclone', + ['lsjson', '--recursive', '--files-only', `--max-age=${ageSeconds}s`, rclonePath(from, 'src')], + { env: rcloneEnv(from, 'src') }, + ); + + let count = 0; + try { + count = (JSON.parse(res.stdout || '[]') as unknown[]).length; + } catch { + count = 0; + } + ctx.log(`${count} object(s) written during the bulk copy`); + + const artifact: Artifact = { + resourceId: from.id, + kind: 'object-storage', + path, + metadata: { objectCount: count, mode: 'delta' }, + }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + /** Object counts and total size on both sides. */ + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const size = async (r: Resource, alias: string) => { + const res = await ctx.exec('rclone', ['size', '--json', rclonePath(r, alias)], { + env: rcloneEnv(r, alias), + check: false, + }); + try { + const parsed = JSON.parse(res.stdout || '{}') as { count?: number; bytes?: number }; + return { count: parsed.count ?? 0, bytes: parsed.bytes ?? 0 }; + } catch { + return { count: 0, bytes: 0 }; + } + }; + + const [a, b] = await Promise.all([size(from, 'src'), size(to, 'dst')]); + const checks = [`${from.name}: ${a.count} objects → ${b.count}`, `bytes: ${a.bytes} → ${b.bytes}`]; + const problems: string[] = []; + if (b.count < a.count) { + problems.push(`${a.count - b.count} object(s) did not arrive in '${to.name}'`); + } + return { ok: problems.length === 0, checks, problems }; + }, +}; diff --git a/packages/migrate/src/engines/postgres.test.ts b/packages/migrate/src/engines/postgres.test.ts new file mode 100644 index 00000000..4f3cd788 --- /dev/null +++ b/packages/migrate/src/engines/postgres.test.ts @@ -0,0 +1,242 @@ +import { describe, expect, it } from 'vitest'; +import { deltaColumn, pgEnv, postgresEngine } from './postgres.js'; +import type { Artifact, EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { secret } from '../types.js'; + +/* + * A fake password, named rather than inlined, for the same reason as in + * plan.test.ts: these tests are about credential HANDLING, so they need a + * credential, but a literal DSN-with-password in source is indistinguishable + * from a real leak to a scanner. + */ +const FAKE_PASSWORD = 'p%40ss'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +/** + * A context that records commands instead of running them. This is the whole + * reason `exec` is injected: the engine is fully exercised with no Postgres. + */ +function ctx( + responses: Array> = [], + over: Partial = {}, +): EngineContext & { calls: Call[]; recorded: Artifact[] } { + const calls: Call[] = []; + const recorded: Artifact[] = []; + let i = 0; + return { + calls, + recorded, + dryRun: false, + log: () => {}, + staging: { + dir: '/staging', + record: async (a) => { + recorded.push(a); + }, + existing: async () => [], + }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; +} + +const db = (over: Partial = {}): Resource => ({ + kind: 'postgres', + id: 'db', + name: 'app', + connection: { url: secret(`postgres://u:${FAKE_PASSWORD}@db.example.com:5432/appdb`) }, + ...over, +}); + +describe('pgEnv', () => { + it('splits the DSN into libpq variables so nothing lands in argv', () => { + const env = pgEnv(db()); + expect(env.PGHOST).toBe('db.example.com'); + expect(env.PGPORT).toBe('5432'); + expect(env.PGUSER).toBe('u'); + expect(env.PGDATABASE).toBe('appdb'); + }); + + it('url-decodes a password containing reserved characters', () => { + expect(pgEnv(db()).PGPASSWORD).toBe('p@ss'); + }); + + it('requires TLS by default rather than letting libpq fall back to plaintext', () => { + expect(pgEnv(db()).PGSSLMODE).toBe('require'); + }); + + it('honours an explicit sslmode in the DSN', () => { + const r = db({ connection: { url: secret(`postgres://u:${FAKE_PASSWORD}@h/d?sslmode=disable`) } }); + expect(pgEnv(r).PGSSLMODE).toBe('disable'); + }); + + it('refuses a resource with no connection url', () => { + expect(() => pgEnv(db({ connection: {} }))).toThrow(/no connection.url/); + }); + + it('refuses a connection url that is not a URL', () => { + expect(() => pgEnv(db({ connection: { url: secret('not a url') } }))).toThrow(/not a URL/); + }); +}); + +describe('export', () => { + it('dumps in the custom format with no owner or acl', async () => { + const c = ctx(); + const artifacts = await postgresEngine.export(c, db()); + + const call = c.calls[0]!; + expect(call.cmd).toBe('pg_dump'); + expect(call.args).toContain('--format=custom'); + expect(call.args).toContain('--no-owner'); + expect(call.args).toContain('--no-acl'); + expect(artifacts[0]?.path).toBe('db/dump.pgc'); + expect(c.recorded).toHaveLength(1); + }); + + it('never puts the password in argv', async () => { + const c = ctx(); + await postgresEngine.export(c, db()); + expect(c.calls[0]!.args.join(' ')).not.toContain('p@ss'); + expect(c.calls[0]!.opts?.env?.PGPASSWORD).toBe('p@ss'); + }); + + it('runs nothing on a dry run but still reports the artifact', async () => { + const c = ctx([], { dryRun: true }); + const artifacts = await postgresEngine.export(c, db()); + expect(c.calls).toHaveLength(0); + expect(artifacts[0]?.path).toBe('db/dump.pgc'); + }); +}); + +describe('import', () => { + const dump: Artifact = { resourceId: 'db', kind: 'postgres', path: 'db/dump.pgc' }; + + it('restores into an empty database', async () => { + const c = ctx([{ stdout: '0' }, { code: 0 }]); + await postgresEngine.import(c, db(), [dump]); + expect(c.calls[1]!.cmd).toBe('pg_restore'); + expect(c.calls[1]!.args).toContain('--no-owner'); + }); + + it('refuses to restore over a database that already has tables', async () => { + const c = ctx([{ stdout: '42' }]); + await expect(postgresEngine.import(c, db(), [dump])).rejects.toThrow(/already has 42 table/); + // Nothing was restored. + expect(c.calls.some((x) => x.cmd === 'pg_restore')).toBe(false); + }); + + it('never passes --clean, which would drop an existing database', async () => { + const c = ctx([{ stdout: '0' }, { code: 0 }]); + await postgresEngine.import(c, db(), [dump]); + expect(c.calls[1]!.args).not.toContain('--clean'); + }); + + it('throws when the restore reports errors that leave the database incomplete', async () => { + const c = ctx([ + { stdout: '0' }, + { code: 1, stderr: 'pg_restore: error: could not execute query: extension "pg_cron" does not exist' }, + ]); + await expect(postgresEngine.import(c, db(), [dump])).rejects.toThrow(/incomplete/); + }); + + it('tolerates a non-zero exit whose errors are benign', async () => { + const c = ctx([{ stdout: '0' }, { code: 1, stderr: 'pg_restore: warning: something cosmetic' }]); + await expect(postgresEngine.import(c, db(), [dump])).resolves.toBeUndefined(); + }); + + it('fails when no dump was staged', async () => { + const c = ctx(); + await expect(postgresEngine.import(c, db(), [])).rejects.toThrow(/no postgres dump staged/); + }); +}); + +describe('deltaColumn', () => { + it('defaults to created_at', () => { + expect(deltaColumn(db())).toBe('created_at'); + }); + + it('accepts a plain identifier', () => { + expect(deltaColumn(db({ metadata: { deltaColumns: 'updated_at' } }))).toBe('updated_at'); + }); + + it('refuses anything that could smuggle a second statement into psql', () => { + for (const bad of ["created_at'; drop table users; --", 'a b', 'a-b', '1col', '']) { + expect(() => deltaColumn(db({ metadata: { deltaColumns: bad } }))).toThrow(/valid column name/); + } + }); +}); + +describe('delta', () => { + it('only copies from tables that actually have the timestamp column', async () => { + const c = ctx([{ stdout: 'public|events\npublic|impressions\n' }, {}, {}]); + const out = await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + expect(out).toHaveLength(2); + expect(out[0]?.metadata?.table).toBe('public.events'); + }); + + it('reports nothing to do when no table has the column', async () => { + const c = ctx([{ stdout: '' }]); + const out = await postgresEngine.delta!(c, db(), new Date()); + expect(out).toEqual([]); + }); + + it('quotes every identifier, so a table named after a reserved word still parses', async () => { + const c = ctx([{ stdout: 'public|user\npublic|order\n' }, {}, {}]); + await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + + const copy = c.calls[1]!.args.join(' '); + expect(copy).toContain('"public"."user"'); + expect(copy).not.toMatch(/from public\.user\b/); + }); + + it('quotes the timestamp column too', async () => { + const c = ctx([{ stdout: 'public|events\n' }, {}]); + await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + expect(c.calls[1]!.args.join(' ')).toContain('"created_at" >'); + }); + + it('skips a malformed row rather than building half a table name', async () => { + const c = ctx([{ stdout: 'public|events\ngarbage-no-separator\n' }, {}]); + const out = await postgresEngine.delta!(c, db(), new Date()); + expect(out).toHaveLength(1); + }); +}); + +describe('verify', () => { + const counts = (rows: string) => ({ stdout: rows }); + + it('passes when every table matches', async () => { + const c = ctx([counts('public.users|113\npublic.posts|48\n'), counts('public.users|113\npublic.posts|48\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + expect(res.checks).toContain('public.users: 113 → 113'); + }); + + it('fails when the target has fewer rows', async () => { + const c = ctx([counts('public.users|113\n'), counts('public.users|9\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('113 rows on the source, 9 on the target'); + }); + + it('flags a table missing entirely from the target', async () => { + const c = ctx([counts('public.users|1\npublic.gone|5\n'), counts('public.users|1\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.problems.some((p) => p.includes('public.gone'))).toBe(true); + }); + + it('accepts a target that has gained rows, since it is live and the source is frozen', async () => { + const c = ctx([counts('public.events|100\n'), counts('public.events|140\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + }); +}); diff --git a/packages/migrate/src/engines/postgres.ts b/packages/migrate/src/engines/postgres.ts new file mode 100644 index 00000000..47c28fdc --- /dev/null +++ b/packages/migrate/src/engines/postgres.ts @@ -0,0 +1,365 @@ +import { quoteIdent, quoteLiteral } from '../transforms.js'; +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a Postgres database. + * + * Supabase, Neon, Railway, RDS and a Postgres in a container on a dedicated + * box are all this engine. That is the point of splitting platforms from + * engines: the vendor decides where the connection string comes from, and this + * decides what to do with it. + * + * ## Why the custom format, and why not --clean + * + * `pg_dump -Fc` (custom) rather than plain SQL, because it is the only format + * `pg_restore` can parallelise and selectively restore, and because a 4.7 GB + * plain-text dump is unusable when one table fails. `--no-owner` and + * `--no-acl`, because the roles on a managed provider do not exist on the + * destination and a dump that tries to `ALTER OWNER TO supabase_admin` fails + * on every object. + * + * `--clean` is deliberately NOT passed. It would make a re-run idempotent, + * which sounds desirable, but it means a restore aimed at the wrong database + * silently drops what is there. A migration tool should not be one typo away + * from deleting production; an existing non-empty target is refused instead. + * + * ## The extension problem + * + * A dump records `CREATE EXTENSION pg_cron`, but an extension is a server + * feature, not data: if the destination image does not ship it, the restore + * fails partway, having already written some tables. Extensions are therefore + * read during inventory and surfaced as quirks so the planner warns before + * anything runs. + */ + +const DUMP = 'dump.pgc'; + +/** Read the DSN off a resource, failing loudly rather than connecting to a default. */ +function dsn(r: Resource): string { + const url = r.connection.url; + if (!url) throw new Error(`postgres resource '${r.name}' has no connection.url`); + return url.reveal(); +} + +/** + * Postgres credentials go in the environment, never in argv. + * + * `ps` is world-readable on a normal box, so a password spliced into a command + * line is visible to every other user for as long as the dump runs — which for + * a migration is hours. + * + * These are libpq's own variables, which `pg_dump`, `pg_restore` and `psql` + * all read directly. That matters more than it looks: `exec` runs commands + * without a shell, so there is nothing to expand a `$VAR` written into an + * argument. Passing the connection through the environment is not merely + * tidier here, it is the only thing that works. + */ +export function pgEnv(r: Resource): Record { + const raw = dsn(r); + let u: URL; + try { + u = new URL(raw); + } catch { + throw new Error(`postgres resource '${r.name}' has a connection.url that is not a URL`); + } + + const env: Record = {}; + if (u.hostname) env.PGHOST = decodeURIComponent(u.hostname); + if (u.port) env.PGPORT = u.port; + if (u.username) env.PGUSER = decodeURIComponent(u.username); + if (u.password) env.PGPASSWORD = decodeURIComponent(u.password); + const database = u.pathname.replace(/^\//, ''); + if (database) env.PGDATABASE = decodeURIComponent(database); + // Managed providers almost universally require TLS, and libpq's default of + // `prefer` silently falls back to plaintext where it is not enforced. + env.PGSSLMODE = u.searchParams.get('sslmode') ?? 'require'; + return env; +} + +/** + * The timestamp column the delta sync keys on. + * + * This ends up interpolated into SQL, and it comes from a resource's metadata, + * which comes from a config file a person edits — so it is checked against a + * plain identifier rather than trusted. `psql -c` would happily run a second + * statement smuggled in here, against the production database the migration is + * reading from. + */ +export function deltaColumn(r: Resource): string { + const raw = (r.metadata?.deltaColumns as string | undefined) ?? 'created_at'; + if (!/^[a-z_][a-z0-9_]*$/i.test(raw)) { + throw new Error( + `'${raw}' is not a valid column name for the delta sync of '${r.name}'. Use a plain identifier.`, + ); + } + return raw; +} + +export const postgresEngine: Engine = { + kind: 'postgres', + requires: ['pg_dump', 'pg_restore', 'psql'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`pg_dump ${from.name} → ${path}`); + + if (ctx.dryRun) { + return [{ resourceId: from.id, kind: 'postgres', path }]; + } + + await ctx.exec( + 'pg_dump', + [ + '--format=custom', + '--no-owner', + '--no-acl', + // Compresses inside the custom format; the wire is usually the + // bottleneck on a cloud-to-anywhere move, not the CPU. + '--compress=6', + '--file', + `${ctx.staging.dir}/${path}`, + ], + { env: pgEnv(from), timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + const artifact: Artifact = { resourceId: from.id, kind: 'postgres', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'postgres'); + if (!dump) throw new Error(`no postgres dump staged for '${to.name}'`); + + if (ctx.dryRun) { + ctx.log(`would pg_restore into ${to.name}`); + return; + } + + // Refuse to write into a database that already has user tables. Without + // this, re-running a half-finished migration against the wrong target is + // indistinguishable from the intended one until the duplicate-key errors + // start, by which point the restore is half applied. + const existing = await ctx.exec( + 'psql', + [ + '--tuples-only', + '--no-align', + '--command', + "select count(*) from information_schema.tables where table_schema not in ('pg_catalog','information_schema')", + ], + { env: pgEnv(to), check: false }, + ); + const tableCount = Number.parseInt(existing.stdout.trim(), 10); + if (Number.isFinite(tableCount) && tableCount > 0) { + throw new Error( + `target database '${to.name}' already has ${tableCount} table(s). Refusing to restore over it — drop and recreate the database, or point at an empty one.`, + ); + } + + ctx.log(`pg_restore → ${to.name}`); + const res = await ctx.exec( + 'pg_restore', + [ + '--no-owner', + '--no-acl', + // Parallel restore. Indexes dominate a large restore and they are + // perfectly parallel. + '--jobs=4', + // Keep going so ONE failed object (a missing extension, a role that + // does not exist) does not abandon a multi-hour restore. Errors are + // counted and surfaced below rather than swallowed. + '--exit-on-error=false', + `${ctx.staging.dir}/${dump.path}`, + ], + { env: pgEnv(to), check: false, timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + if (res.code !== 0) { + const errors = res.stderr + .split('\n') + .filter((l) => l.includes('error:')) + .slice(0, 10); + ctx.log( + `pg_restore finished with errors (${errors.length} shown):\n${errors.join('\n')}`, + 'warn', + ); + // A restore that produced errors is not automatically a failed + // migration — a missing extension on a replica, for instance — but it is + // never something to pass over silently. + if (errors.some((e) => /could not|does not exist|permission denied/i.test(e))) { + throw new Error( + `pg_restore into '${to.name}' reported errors that will leave the database incomplete. First: ${errors[0] ?? 'unknown'}`, + ); + } + } + }, + + /** + * Re-copy rows written during the bulk dump. + * + * There is no general delta for Postgres without logical replication, and + * standing up a replication slot against a managed provider mid-migration is + * its own project. What works in practice, and what the crawlproof migration + * actually did, is narrower: for tables that carry a timestamp column, copy + * the rows newer than the dump. That covers append-heavy tables (events, + * impressions, logs) which are exactly the ones that keep being written + * while a long dump runs. + * + * It does NOT cover updates to old rows or deletes. The planner says so, and + * the honest use of this is: freeze writes to anything that mutates history, + * and let the append-only tables catch up. + */ + async delta(ctx: EngineContext, from: Resource, since: Date): Promise { + const columns = deltaColumn(from); + const path = `${from.id}/delta-${since.toISOString().replace(/[:.]/g, '')}.sql`; + ctx.log(`delta for ${from.name} on ${columns} since ${since.toISOString()}`); + + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'postgres', path }]; + + /* + * Find tables that actually have the timestamp column rather than assuming + * a schema. A table without one cannot be delta'd and is reported. + * + * Schema and table come back as separate fields rather than pre-joined + * with a dot, so each can be quoted independently below. Joining them here + * would make `public.user` a string that cannot be quoted correctly — + * quoting the whole thing gives `"public.user"`, a single identifier with + * a dot in its name, which is a different table that does not exist. + */ + const found = await ctx.exec( + 'psql', + [ + '--tuples-only', + '--no-align', + '--field-separator=|', + '--command', + `select table_schema, table_name from information_schema.columns + where column_name = ${quoteLiteral(columns)} + and table_schema not in ('pg_catalog','information_schema') + order by 1, 2`, + ], + { env: pgEnv(from) }, + ); + + const tables = found.stdout + .split('\n') + .map((l) => l.trim()) + .filter(Boolean) + .map((l) => { + const [schema, table] = l.split('|'); + return schema && table ? { schema, table } : undefined; + }) + .filter((t): t is { schema: string; table: string } => Boolean(t)); + + if (!tables.length) { + ctx.log(`no table has a '${columns}' column; nothing can be delta-synced`, 'warn'); + return []; + } + + const artifacts: Artifact[] = []; + for (const { schema, table } of tables) { + const label = `${schema}.${table}`; + const out = `${from.id}/delta-${label.replace(/[^\w]/g, '_')}.csv`; + /* + * Every identifier is quoted. These names come from information_schema + * so they are real tables rather than attacker input, but a table called + * `user` or `order` is a reserved word and an unquoted reference to it is + * a syntax error partway through a cutover — and a name containing a + * quote or a space simply does not parse. Quoting costs nothing and + * removes the class. + */ + await ctx.exec( + 'psql', + [ + '--command', + `\\copy (select * from ${quoteIdent(schema)}.${quoteIdent(table)} where ${quoteIdent(columns)} > ${quoteLiteral(since.toISOString())}) to ${quoteLiteral(`${ctx.staging.dir}/${out}`)} with csv header`, + ], + { env: pgEnv(from) }, + ); + const artifact: Artifact = { + resourceId: from.id, + kind: 'postgres', + path: out, + metadata: { table: label, mode: 'delta-csv' }, + }; + await ctx.staging.record(artifact); + artifacts.push(artifact); + } + return artifacts; + }, + + /** + * Compare row counts per table. + * + * Not a checksum — comparing 4.7 GB twice over the wire costs as much as the + * migration did. Row counts per table catch the failures that actually + * happen: a table that restored empty, a restore that stopped partway. + */ + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const countsQuery = `select table_schema||'.'||table_name as t, + (xpath('/row/c/text()', query_to_xml(format('select count(*) as c from %I.%I', table_schema, table_name), false, true, '')))[1]::text::bigint as n + from information_schema.tables + where table_type = 'BASE TABLE' and table_schema not in ('pg_catalog','information_schema') + order by 1`; + + const read = async (r: Resource) => { + const res = await ctx.exec( + 'psql', + ['--tuples-only', '--no-align', '--field-separator=|', '--command', countsQuery], + { env: pgEnv(r) }, + ); + const map = new Map(); + for (const line of res.stdout.split('\n')) { + const [t, n] = line.split('|'); + if (t && n !== undefined) map.set(t.trim(), Number.parseInt(n, 10)); + } + return map; + }; + + const [a, b] = await Promise.all([read(from), read(to)]); + const checks: string[] = []; + const problems: string[] = []; + + for (const [table, sourceCount] of a) { + const targetCount = b.get(table); + if (targetCount === undefined) { + problems.push(`${table}: missing on the target`); + continue; + } + checks.push(`${table}: ${sourceCount} → ${targetCount}`); + // The target may legitimately have MORE rows once it is live and the + // source is frozen; fewer is always wrong. + if (targetCount < sourceCount) { + problems.push(`${table}: ${sourceCount} rows on the source, ${targetCount} on the target`); + } + } + for (const table of b.keys()) { + if (!a.has(table)) checks.push(`${table}: target only`); + } + + return { ok: problems.length === 0, checks, problems }; + }, +}; + +/** + * The extensions a database uses, for the inventory's quirks. + * + * Exported so a platform can call it while building its inventory: the + * platform knows the DSN, this knows the question worth asking. + */ +export async function postgresExtensions( + ctx: EngineContext, + r: Resource, +): Promise { + const res = await ctx.exec( + 'psql', + ['--tuples-only', '--no-align', '--command', + "select extname from pg_extension where extname not in ('plpgsql') order by 1"], + { env: pgEnv(r), check: false }, + ); + if (res.code !== 0) return []; + return res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); +} diff --git a/packages/migrate/src/engines/redis.ts b/packages/migrate/src/engines/redis.ts new file mode 100644 index 00000000..c090611c --- /dev/null +++ b/packages/migrate/src/engines/redis.ts @@ -0,0 +1,98 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a Redis. + * + * Usually you should not. Redis is a cache far more often than it is a + * database, and the right migration for a cache is an empty one: point the new + * app at a new Redis and let it fill. Copying a cache moves stale entries and + * costs downtime for data that is worthless by definition. + * + * It is here because sometimes it is not a cache — a BullMQ queue with jobs + * waiting in it, a session store where copying nothing logs every user out. + * Those are real, so the engine exists, and the planner surfaces the question + * rather than deciding for you. + * + * There is no delta. A key written during the copy is simply missed, which is + * exactly why queues and sessions want the source stopped first. The planner + * warns about that because `delta` is absent. + */ + +const DUMP = 'dump.rdb'; + +function redisArgs(r: Resource): string[] { + const url = r.connection.url?.reveal(); + if (!url) throw new Error(`redis resource '${r.name}' has no connection.url`); + // redis-cli takes the whole URL, and unlike libpq there is no environment + // variable for it. The password is therefore visible in `ps` for as long as + // the command runs, which for --rdb is the length of the copy. Nothing can + // be done about that from here beyond keeping the window short; it is noted + // so nobody assumes otherwise. + return ['-u', url]; +} + +export const redisEngine: Engine = { + kind: 'redis', + requires: ['redis-cli'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`redis --rdb ${from.name}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'redis', path }]; + + // --rdb asks the server for a full sync and writes the RDB the replica + // would have received, which is consistent at a point in time. Reading + // keys with SCAN+DUMP instead is not: keys move under you as you walk. + await ctx.exec('redis-cli', [...redisArgs(from), '--rdb', `${ctx.staging.dir}/${path}`], { + timeoutMs: 2 * 60 * 60 * 1000, + }); + + const artifact: Artifact = { resourceId: from.id, kind: 'redis', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + /** + * There is no supported way to push an RDB into a running managed Redis, so + * this refuses rather than pretending. + * + * A self-hosted target takes the file directly: stop the server, drop the + * RDB in its data directory, start it. That is a host operation rather than + * a client one, so it is surfaced as an instruction instead of being done + * badly over the wire. + */ + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'redis'); + if (!dump) throw new Error(`no redis dump staged for '${to.name}'`); + + const dataDir = to.connection.dataDir?.reveal(); + if (!dataDir) { + throw new Error( + `Redis cannot be loaded over the wire. Copy ${dump.path} to the target's data directory as dump.rdb while the server is stopped, then start it. Set connection.dataDir on the target to have this done for you.`, + ); + } + + if (ctx.dryRun) { + ctx.log(`would place ${dump.path} at ${dataDir}/dump.rdb`); + return; + } + + await ctx.exec('cp', [`${ctx.staging.dir}/${dump.path}`, `${dataDir}/dump.rdb`]); + ctx.log(`placed dump.rdb in ${dataDir}; restart the target Redis to load it`, 'warn'); + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const size = async (r: Resource) => { + const res = await ctx.exec('redis-cli', [...redisArgs(r), 'dbsize'], { check: false }); + return Number.parseInt(res.stdout.trim(), 10) || 0; + }; + const [a, b] = await Promise.all([size(from), size(to)]); + return { + ok: b >= a, + checks: [`keys: ${a} → ${b}`], + problems: b < a ? [`${a - b} key(s) missing on the target`] : [], + }; + }, +}; diff --git a/packages/migrate/src/engines/sqlite.ts b/packages/migrate/src/engines/sqlite.ts new file mode 100644 index 00000000..dea5e04f --- /dev/null +++ b/packages/migrate/src/engines/sqlite.ts @@ -0,0 +1,122 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a SQLite or libSQL database. + * + * Turso is libSQL, which is SQLite with a server in front, and a local + * `app.db` is SQLite with nothing in front. Both dump to the same SQL text, + * which is why they are one engine — and why Turso→dedicated and + * dedicated→Turso are the same code path. + * + * The dump is plain SQL rather than a file copy. Copying the file works for a + * local database and not at all for a hosted one, and a SQLite file copied + * while something is writing to it is a corrupt file rather than an error. The + * text dump is slower and always correct. + * + * Note the `.dump` output includes `PRAGMA foreign_keys=OFF` and wraps in a + * transaction on its own, which is what lets the rows load in whatever order + * the dump emitted them. + */ + +const DUMP = 'dump.sql'; + +/** Turso's CLI talks to a named database; plain SQLite takes a file path. */ +function isTurso(r: Resource): boolean { + return Boolean(r.connection.tursoDatabase) || r.metadata?.platform === 'turso'; +} + +function tursoEnv(r: Resource): Record { + const token = r.connection.authToken?.reveal(); + return token ? { TURSO_API_TOKEN: token } : {}; +} + +function filePath(r: Resource): string { + const p = r.connection.path?.reveal(); + if (!p) throw new Error(`sqlite resource '${r.name}' has no connection.path`); + return p; +} + +export const sqliteEngine: Engine = { + kind: 'sqlite', + /* + * `sqlite3` only. Which binary is needed actually depends on the resource — + * a Turso database is reached with `turso`, a file with `sqlite3` — but + * `requires` is a property of the engine, not of one side of one migration. + * Declaring the union would block a file-to-file move on a missing Turso CLI + * nobody needs, so the Turso side is checked at the point of use instead and + * fails with a message naming the binary. + */ + requires: ['sqlite3'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`dumping ${from.name} → ${path}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'sqlite', path }]; + + const out = `${ctx.staging.dir}/${path}`; + if (isTurso(from)) { + const db = from.connection.tursoDatabase!.reveal(); + await ctx.exec('turso', ['db', 'shell', db, '.dump'], { + env: tursoEnv(from), + timeoutMs: 2 * 60 * 60 * 1000, + stdoutFile: out, + }); + } else { + await ctx.exec('sqlite3', [filePath(from), '.dump'], { + timeoutMs: 2 * 60 * 60 * 1000, + stdoutFile: out, + }); + } + + const artifact: Artifact = { resourceId: from.id, kind: 'sqlite', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'sqlite'); + if (!dump) throw new Error(`no sqlite dump staged for '${to.name}'`); + if (ctx.dryRun) { + ctx.log(`would load ${dump.path} into ${to.name}`); + return; + } + + const file = `${ctx.staging.dir}/${dump.path}`; + if (isTurso(to)) { + const db = to.connection.tursoDatabase!.reveal(); + await ctx.exec('turso', ['db', 'shell', db], { + env: tursoEnv(to), + timeoutMs: 2 * 60 * 60 * 1000, + stdinFile: file, + }); + } else { + await ctx.exec('sqlite3', [filePath(to), `.read ${file}`], { + timeoutMs: 2 * 60 * 60 * 1000, + }); + } + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const countTables = async (r: Resource) => { + const sql = + "select name from sqlite_master where type='table' and name not like 'sqlite_%' order by 1"; + const res = isTurso(r) + ? await ctx.exec('turso', ['db', 'shell', r.connection.tursoDatabase!.reveal(), sql], { + env: tursoEnv(r), + check: false, + }) + : await ctx.exec('sqlite3', [filePath(r), sql], { check: false }); + return res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + }; + + const [a, b] = await Promise.all([countTables(from), countTables(to)]); + const missing = a.filter((t) => !b.includes(t)); + return { + ok: missing.length === 0, + checks: [`tables: ${a.length} → ${b.length}`], + problems: missing.map((t) => `${t}: missing on the target`), + }; + }, +}; diff --git a/packages/migrate/src/exec.ts b/packages/migrate/src/exec.ts new file mode 100644 index 00000000..e2118ee8 --- /dev/null +++ b/packages/migrate/src/exec.ts @@ -0,0 +1,134 @@ +import { spawn } from 'node:child_process'; +import { createReadStream, createWriteStream } from 'node:fs'; +import { mkdir } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import type { ExecOptions, ExecResult } from './types.js'; + +/** + * Running an external command. + * + * The real implementation behind the `exec` every engine receives. It is + * injected rather than imported so the engines can be tested without any of + * the tools installed, and so a dry run can be genuinely inert. + * + * `spawn` without a shell, always. Engines build argument arrays from + * connection strings, bucket names and paths, all of which come from config a + * person edits — running those through a shell would make a bucket named + * `; rm -rf /` a working attack and a path with a space a silent bug. The + * cost is that `>` and `<` do not work, which is why redirection is an option + * rather than a character in the argument list. + */ +export function createExec(opts: { log?: (msg: string) => void } = {}) { + return async function exec( + cmd: string, + args: string[], + options: ExecOptions = {}, + ): Promise { + const { env, cwd, check = true, timeoutMs, stdoutFile, stdinFile } = options; + + if (stdoutFile) await mkdir(dirname(stdoutFile), { recursive: true }); + + // Arguments are logged; the environment never is. That split is the whole + // reason credentials are passed through env. + opts.log?.(`$ ${cmd} ${args.join(' ')}`); + + return new Promise((resolve, reject) => { + const child = spawn(cmd, args, { + env: { ...process.env, ...env }, + ...(cwd ? { cwd } : {}), + stdio: [stdinFile ? 'pipe' : 'ignore', stdoutFile ? 'pipe' : 'pipe', 'pipe'], + }); + + let stdout = ''; + let stderr = ''; + let settled = false; + + const timer = timeoutMs + ? setTimeout(() => { + if (settled) return; + settled = true; + child.kill('SIGKILL'); + reject(new Error(`${cmd} timed out after ${Math.round(timeoutMs / 1000)}s`)); + }, timeoutMs) + : undefined; + + if (stdoutFile && child.stdout) { + const out = createWriteStream(stdoutFile); + child.stdout.pipe(out); + } else { + child.stdout?.on('data', (d) => { + stdout += d.toString(); + }); + } + + child.stderr?.on('data', (d) => { + stderr += d.toString(); + }); + + if (stdinFile && child.stdin) { + createReadStream(stdinFile).pipe(child.stdin); + } + + child.on('error', (err) => { + if (settled) return; + settled = true; + if (timer) clearTimeout(timer); + // ENOENT here means the binary is missing, which the planner is + // supposed to have caught. Say which one, so the message is actionable + // rather than "spawn ENOENT". + const message = + (err as NodeJS.ErrnoException).code === 'ENOENT' + ? `${cmd} is not installed or not on PATH` + : err.message; + reject(new Error(message)); + }); + + child.on('close', (code) => { + if (settled) return; + settled = true; + if (timer) clearTimeout(timer); + const result: ExecResult = { code: code ?? 0, stdout, stderr }; + if (check && result.code !== 0) { + reject( + new Error( + `${cmd} exited ${result.code}${stderr ? `: ${stderr.trim().split('\n').slice(-3).join('\n')}` : ''}`, + ), + ); + return; + } + resolve(result); + }); + }); + }; +} + +/** + * An exec that records instead of running, for `--dry-run`. + * + * A dry run has to be genuinely inert: the point is to be safe to run against + * production, and an engine that shells out "just to look" is not. + */ +export function recordingExec(recorded: string[][] = []) { + const exec = async (cmd: string, args: string[]): Promise => { + recorded.push([cmd, ...args]); + return { code: 0, stdout: '', stderr: '' }; + }; + return { exec, recorded }; +} + +/** Which of these binaries are on PATH, for the planner's preflight check. */ +export async function availableBinaries(names: string[]): Promise> { + const exec = createExec(); + const found = new Set(); + await Promise.all( + names.map(async (name) => { + try { + const res = await exec('sh', ['-c', `command -v ${JSON.stringify(name)}`], { check: false }); + if (res.code === 0 && res.stdout.trim()) found.add(name); + } catch { + // Missing is the answer, not an error. + } + }), + ); + return found; +} diff --git a/packages/migrate/src/index.ts b/packages/migrate/src/index.ts new file mode 100644 index 00000000..59eb9ba0 --- /dev/null +++ b/packages/migrate/src/index.ts @@ -0,0 +1,52 @@ +/** + * @profullstack/sh1pt-migrate — move an application and its data between + * platforms, in either direction. + * + * `packages/cloud/*` provisions machines. This moves what lives on them. + * + * The design in one paragraph: PLATFORMS (Railway, Supabase, Turso, Neon, a + * box over ssh) resolve credentials and enumerate what they hold, and never + * move a byte. ENGINES (postgres, sqlite, redis, object-storage, files) move + * bytes and do not know which vendor is on either end. A migration is possible + * when the target accepts a kind the source holds, which `compatibleKinds` + * answers instantly. Direction is not a property of the system — it is which + * platform you named first. + * + * Typical use: + * + * const inventory = await supabasePlatform.inventory(ctx, { projectRef, dbPassword }); + * const plan = planMigration(inventory, sshPlatform, { engines: ENGINES }); + * console.log(renderPlan(plan)); // safe against production + * if (plan.ok) await applyPlan({ plan, ... }); + */ + +export * from './types.js'; +export * from './plan.js'; +export * from './apply.js'; +export * from './staging.js'; +export * from './transforms.js'; +export * from './exec.js'; +export { ENGINES, engineFor, requiredBinaries } from './engines/index.js'; +export { + filesEngine, + mysqlEngine, + objectStorageEngine, + postgresEngine, + redisEngine, + sqliteEngine, +} from './engines/index.js'; +export { + PLATFORMS, + compatibleKinds, + platformById, + railwayPlatform, + sources, + sshPlatform, + supabasePlatform, + targets, + tursoPlatform, +} from './platforms/index.js'; +export type { SshConfig } from './platforms/ssh.js'; +export type { RailwayConfig } from './platforms/railway.js'; +export type { SupabaseConfig } from './platforms/supabase.js'; +export type { DsnConfig, TursoConfig } from './platforms/managed.js'; diff --git a/packages/migrate/src/plan.test.ts b/packages/migrate/src/plan.test.ts new file mode 100644 index 00000000..9fb46934 --- /dev/null +++ b/packages/migrate/src/plan.test.ts @@ -0,0 +1,293 @@ +import { describe, expect, it } from 'vitest'; +import { PHASES, type Phase, type Step, humanBytes, orderSteps, planMigration, renderPlan } from './plan.js'; +import { type Engine, type Inventory, type Platform, type Resource, plain, secret } from './types.js'; + +const engine = (kind: Engine['kind'], over: Partial = {}): Engine => ({ + kind, + requires: [], + export: async () => [], + import: async () => {}, + ...over, +}); + +/** A full engine set: everything can move, delta and verify included. */ +function engines(over: Partial> = {}): Map { + const base: Array<[Resource['kind'], Engine]> = [ + ['postgres', engine('postgres', { requires: ['pg_dump'], delta: async () => [], verify: async () => ({ ok: true, checks: [], problems: [] }) })], + ['object-storage', engine('object-storage', { delta: async () => [] })], + ['redis', engine('redis')], + ['cron', engine('cron')], + ]; + const m = new Map(base); + for (const [k, v] of Object.entries(over)) m.set(k as Resource['kind'], v as Engine); + return m; +} + +const resource = (over: Partial & Pick): Resource => ({ + connection: { url: secret('postgres://u@h/db') }, + ...over, +}); + +const inventory = (resources: Resource[], over: Partial = {}): Inventory => ({ + platform: 'supabase', + scope: 'project abc123', + resources, + ...over, +}); + +const target = (over: Partial = {}): Platform => ({ + id: 'ssh', + label: 'dedicated box', + role: 'both', + supports: ['postgres', 'object-storage', 'redis', 'files', 'cron'], + inventory: async () => inventory([]), + ...over, +}); + +describe('planMigration', () => { + it('pairs every supported resource and estimates the total', () => { + const plan = planMigration( + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app', sizeBytes: 4_700_000_000 }), + resource({ kind: 'object-storage', id: 'b1', name: 'public', itemCount: 8410 }), + ]), + target(), + { engines: engines() }, + ); + + expect(plan.ok).toBe(true); + expect(plan.moving.map((r) => r.id)).toEqual(['db', 'b1']); + expect(plan.estimateBytes).toBe(4_700_000_000); + }); + + it('blocks when the target cannot hold a resource kind', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target({ supports: ['files'] }), + { engines: engines() }, + ); + + expect(plan.ok).toBe(false); + expect(plan.skipped[0]?.reason).toContain('does not support postgres'); + expect(plan.risks.some((r) => r.severity === 'blocker')).toBe(true); + }); + + it('refuses a target that cannot be written to at all', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target({ role: 'source' }), + { engines: engines() }, + ); + expect(plan.ok).toBe(false); + expect(plan.risks.some((r) => /offers no import path/.test(r.message))).toBe(true); + }); + + it('blocks when a required binary is missing, before anything runs', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target(), + { engines: engines(), availableBinaries: new Set() }, + ); + expect(plan.ok).toBe(false); + expect(plan.risks.some((r) => r.message.includes('pg_dump'))).toBe(true); + }); + + it('accepts the plan when the binary is present', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target(), + { engines: engines(), availableBinaries: new Set(['pg_dump']) }, + ); + expect(plan.ok).toBe(true); + }); + + it('warns when a resource has no delta sync, because writes during the copy are lost', () => { + const plan = planMigration( + inventory([resource({ kind: 'redis', id: 'r', name: 'cache' })]), + target(), + { engines: engines() }, + ); + expect(plan.risks.some((r) => /no delta sync/.test(r.message))).toBe(true); + }); + + it('surfaces a resource quirk as a warning rather than discovering it mid-restore', () => { + const plan = planMigration( + inventory([ + resource({ + kind: 'postgres', + id: 'db', + name: 'app', + quirks: ['uses the pg_cron extension, which the target must also have'], + }), + ]), + target(), + { engines: engines() }, + ); + expect(plan.risks.some((r) => /pg_cron/.test(r.message))).toBe(true); + }); + + it('honours --only and --exclude', () => { + const rs = [ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'redis', id: 'r', name: 'cache' }), + ]; + const onlyDb = planMigration(inventory(rs), target(), { engines: engines(), only: ['postgres'] }); + expect(onlyDb.moving.map((r) => r.id)).toEqual(['db']); + + const noRedis = planMigration(inventory(rs), target(), { engines: engines(), exclude: ['redis'] }); + expect(noRedis.moving.map((r) => r.id)).toEqual(['db']); + }); +}); + +describe('the absolute-URL trap', () => { + const both = () => + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'object-storage', id: 'b1', name: 'public' }), + ]); + + it('warns when storage and a database move together with no rewrite host', () => { + const plan = planMigration(both(), target(), { engines: engines() }); + expect(plan.risks.some((r) => /absolute URLs/i.test(r.message))).toBe(true); + expect(plan.steps.some((s) => s.id === 'rewrite:urls')).toBe(true); + }); + + it('does not warn once a rewrite host is given', () => { + const plan = planMigration(both(), target(), { + engines: engines(), + rewriteHosts: ['abc.supabase.co'], + }); + expect(plan.risks.some((r) => /no --rewrite-host/i.test(r.message))).toBe(false); + const step = plan.steps.find((s) => s.id === 'rewrite:urls'); + expect(step?.title).toContain('abc.supabase.co'); + }); + + it('adds no rewrite step when only storage moves', () => { + const plan = planMigration( + inventory([resource({ kind: 'object-storage', id: 'b1', name: 'public' })]), + target(), + { engines: engines() }, + ); + expect(plan.steps.some((s) => s.id === 'rewrite:urls')).toBe(false); + }); +}); + +describe('phase ordering is the safety property', () => { + const planWithCron = () => + planMigration( + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'cron', id: 'jobs', name: 'pg_cron' }), + ]), + target(), + { engines: engines() }, + ); + + it('never schedules a phase before an earlier one', () => { + const steps = planWithCron().steps; + const idx = steps.map((s) => PHASES.indexOf(s.phase)); + expect(idx).toEqual([...idx].sort((a, b) => a - b)); + }); + + it('stops scheduled jobs on the source before starting them on the target', () => { + const steps = planWithCron().steps; + const disable = steps.findIndex((s) => s.id === 'disable:jobs'); + const enable = steps.findIndex((s) => s.id === 'enable:jobs'); + expect(disable).toBeGreaterThanOrEqual(0); + expect(enable).toBeGreaterThan(disable); + }); + + it('cuts DNS over only after the freeze and the delta', () => { + const steps = planWithCron().steps; + const at = (id: string) => steps.findIndex((s) => s.id === id); + expect(at('freeze:all')).toBeLessThan(at('cutover:dns')); + expect(at('delta:db')).toBeLessThan(at('cutover:dns')); + }); + + it('marks the steps that touch the live source', () => { + const freeze = planWithCron().steps.find((s) => s.id === 'freeze:all'); + expect(freeze?.touchesSource).toBe(true); + }); +}); + +describe('orderSteps', () => { + const step = (id: string, phase: Phase, after: string[] = []): Step => ({ + id, + phase, + title: id, + after, + }); + + it('respects dependencies inside a phase', () => { + const out = orderSteps([ + step('b', 'bulk', ['a']), + step('a', 'bulk'), + step('c', 'bulk', ['b']), + ]); + expect(out.map((s) => s.id)).toEqual(['a', 'b', 'c']); + }); + + it('throws on a cycle rather than guessing an order', () => { + expect(() => orderSteps([step('a', 'bulk', ['b']), step('b', 'bulk', ['a'])])).toThrow(/cycle/); + }); + + it('throws when a step depends on one that runs in a later phase', () => { + expect(() => orderSteps([step('a', 'bulk', ['z']), step('z', 'verify')])).toThrow(/runs later/); + }); + + it('ignores a dependency on a step that does not exist', () => { + expect(() => orderSteps([step('a', 'bulk', ['nope'])])).not.toThrow(); + }); +}); + +/* + * A fake password, named rather than inlined. + * These four tests exist to prove a password is masked, so they need one -- + * but a literal `scheme://user:pass@host` in source is a credential shape a + * secret scanner cannot tell from a real leak, and it should not have to. + */ +const FAKE_PASSWORD = 'hunter2'; + +describe('secrets never reach the plan', () => { + it('describes a connection without revealing it', () => { + const s = secret(`postgres://user:${FAKE_PASSWORD}@db.example.com:5432/app`); + expect(s.describe()).not.toContain(FAKE_PASSWORD); + expect(s.describe()).toContain('db.example.com'); + expect(s.reveal()).toContain(FAKE_PASSWORD); + }); + + it('masks a bare token', () => { + expect(secret('abcdef123456').describe()).toBe('abcd***'); + expect(secret('ab').describe()).toBe('***'); + }); + + it('leaves a non-secret alone', () => { + expect(plain('my-bucket').describe()).toBe('my-bucket'); + }); + + it('renders a plan with no credential in it', () => { + const text = renderPlan( + planMigration( + inventory([ + resource({ + kind: 'postgres', + id: 'db', + name: 'app', + connection: { url: secret(`postgres://user:${FAKE_PASSWORD}@h/db`) }, + }), + ]), + target(), + { engines: engines() }, + ), + ); + expect(text).not.toContain(FAKE_PASSWORD); + }); +}); + +describe('humanBytes', () => { + it('scales through the units', () => { + expect(humanBytes(512)).toBe('512 B'); + expect(humanBytes(4_700_000_000)).toBe('4.4 GB'); + expect(humanBytes(0)).toBe('0 B'); + }); +}); diff --git a/packages/migrate/src/plan.ts b/packages/migrate/src/plan.ts new file mode 100644 index 00000000..963d15c9 --- /dev/null +++ b/packages/migrate/src/plan.ts @@ -0,0 +1,474 @@ +import type { Engine, Inventory, Platform, Resource, ResourceKind } from './types.js'; + +/** + * Turning an inventory into an ordered, reviewable plan. + * + * The plan is the product. A migration that goes wrong usually went wrong + * before anything ran — a resource nobody knew was there, a cutover step in + * the wrong order, an assumption that the destination supported something it + * did not. All of that is knowable up front, from a read-only inventory, and + * this module is where it is worked out so a person can read it and disagree + * before any bytes move. + * + * Nothing here touches the network or mutates anything. Given the same + * inventory it produces the same plan, which is what makes it testable. + */ + +/** A single unit of work in the plan. */ +export interface Step { + id: string; + phase: Phase; + /** One line, imperative: "dump postgres 'app' (4.7 GB)". */ + title: string; + resourceId?: string; + kind?: ResourceKind; + /** Steps that must complete before this one. */ + after: string[]; + /** True when this step changes the SOURCE, which is what makes it scary. */ + touchesSource?: boolean; + /** Set when the step cannot be undone by re-running the migration. */ + irreversible?: boolean; + estimateBytes?: number; +} + +/** + * The phases, in the only order that is safe. + * + * The ordering is the part people get wrong, and it is not arbitrary: + * + * - `check` first so a missing credential costs a second, not three hours. + * - `bulk` runs while the source is still live and serving. It is the long + * part and it is safe to repeat. + * - `freeze` is the start of downtime: stop the things that write. Scheduled + * jobs especially — a cron firing on both sides is how a migration sends + * every customer a duplicate email. + * - `delta` copies only what changed during `bulk`. Short, because `bulk` + * already moved the bulk. + * - `cutover` points the world at the new place. + * - `enable` starts the writers again, on the target only, and never before + * `freeze` has stopped them on the source. + * - `verify` proves it worked while the old system still exists. + * + * Deleting the source is not a phase. It is a separate decision a person makes + * days later, and this tool does not offer it. + */ +export type Phase = 'check' | 'bulk' | 'freeze' | 'delta' | 'cutover' | 'enable' | 'verify'; + +export const PHASES: readonly Phase[] = [ + 'check', + 'bulk', + 'freeze', + 'delta', + 'cutover', + 'enable', + 'verify', +] as const; + +export interface Risk { + severity: 'blocker' | 'warning' | 'note'; + /** What is wrong, in a sentence a person can act on. */ + message: string; + resourceId?: string; +} + +export interface MigrationPlan { + source: string; + target: string; + scope: string; + steps: Step[]; + risks: Risk[]; + /** Resources that will move, paired source → target kind. */ + moving: Resource[]; + /** Resources that will NOT move, and why. */ + skipped: Array<{ resource: Resource; reason: string }>; + estimateBytes: number; + /** False when any risk is a blocker. `apply` refuses a plan that is not ok. */ + ok: boolean; +} + +export interface PlanOptions { + /** Only migrate these kinds. Empty means everything the target supports. */ + only?: ResourceKind[]; + /** Never migrate these kinds. */ + exclude?: ResourceKind[]; + /** + * Hosts whose absolute URLs are expected to appear in the data and must be + * rewritten, e.g. `ywcizjsgrcmhgyplldac.supabase.co`. See `transforms.ts`. + */ + rewriteHosts?: string[]; + /** Engines available in this build, by kind. */ + engines: Map; + /** Binaries present on this machine, for the `requires` check. */ + availableBinaries?: Set; +} + +/** Kinds whose contents are routinely referenced by absolute URL from a database. */ +const URL_BEARING: ReadonlySet = new Set(['object-storage']); + +/** + * Kinds that write on a schedule and so must be stopped before the delta, or + * they run in two places at once. + */ +const SCHEDULED: ReadonlySet = new Set(['cron']); + +function stepId(prefix: string, resourceId: string): string { + return `${prefix}:${resourceId}`; +} + +/** + * Work out what would happen, without doing any of it. + * + * Read-only in the strongest sense: it takes an inventory that has already + * been gathered and a description of the target, and returns a plan. It does + * not call the network, so it is fully testable and so `migrate plan` can be + * run against production with no anxiety. + */ +export function planMigration( + source: Inventory, + target: Platform, + opts: PlanOptions, +): MigrationPlan { + const risks: Risk[] = []; + const steps: Step[] = []; + const moving: Resource[] = []; + const skipped: Array<{ resource: Resource; reason: string }> = []; + + const only = new Set(opts.only ?? []); + const exclude = new Set(opts.exclude ?? []); + const targetKinds = new Set(target.supports); + + if (target.role === 'source') { + risks.push({ + severity: 'blocker', + message: `${target.label} cannot be a migration target: it offers no import path.`, + }); + } + + for (const resource of source.resources) { + if (only.size && !only.has(resource.kind)) { + skipped.push({ resource, reason: `not in --only` }); + continue; + } + if (exclude.has(resource.kind)) { + skipped.push({ resource, reason: `excluded by --exclude` }); + continue; + } + if (!targetKinds.has(resource.kind)) { + skipped.push({ + resource, + reason: `${target.label} does not support ${resource.kind}`, + }); + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `${resource.name} is ${resource.kind}, which ${target.label} cannot hold. Exclude it with --exclude ${resource.kind}, or pick a different target.`, + }); + continue; + } + + const engine = opts.engines.get(resource.kind); + if (!engine) { + skipped.push({ resource, reason: `no engine for ${resource.kind}` }); + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `Nothing in this build can move a ${resource.kind}.`, + }); + continue; + } + + // A missing pg_dump is a blocker, and finding out now beats finding out + // after the freeze has started. + if (opts.availableBinaries) { + const missing = engine.requires.filter((b) => !opts.availableBinaries!.has(b)); + if (missing.length) { + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `${resource.kind} needs ${missing.join(', ')} on this machine and ${missing.length > 1 ? 'they are' : 'it is'} not installed.`, + }); + } + } + + moving.push(resource); + + steps.push({ + id: stepId('export', resource.id), + phase: 'bulk', + title: `export ${resource.kind} ${resource.name}${sizeSuffix(resource)}`, + resourceId: resource.id, + kind: resource.kind, + after: ['check:all'], + estimateBytes: resource.sizeBytes, + }); + + steps.push({ + id: stepId('import', resource.id), + phase: 'bulk', + title: `import ${resource.kind} ${resource.name} into ${target.label}`, + resourceId: resource.id, + kind: resource.kind, + after: [stepId('export', resource.id)], + estimateBytes: resource.sizeBytes, + }); + + if (engine.delta) { + steps.push({ + id: stepId('delta', resource.id), + phase: 'delta', + title: `sync ${resource.name} changes made during the bulk copy`, + resourceId: resource.id, + kind: resource.kind, + after: ['freeze:all', stepId('import', resource.id)], + }); + } else { + risks.push({ + severity: 'warning', + resourceId: resource.id, + message: `${resource.name} (${resource.kind}) has no delta sync, so anything written to it during the bulk copy is lost. Stop writes before the copy, or accept the gap.`, + }); + } + + if (engine.verify) { + steps.push({ + id: stepId('verify', resource.id), + phase: 'verify', + title: `verify ${resource.name} matches the source`, + resourceId: resource.id, + kind: resource.kind, + after: ['cutover:dns'], + }); + } + + for (const quirk of resource.quirks ?? []) { + risks.push({ + severity: 'warning', + resourceId: resource.id, + message: `${resource.name}: ${quirk}`, + }); + } + + if (SCHEDULED.has(resource.kind)) { + steps.push({ + id: stepId('disable', resource.id), + phase: 'freeze', + title: `disable scheduled jobs on the SOURCE (${resource.name})`, + resourceId: resource.id, + kind: resource.kind, + after: [], + touchesSource: true, + }); + steps.push({ + id: stepId('enable', resource.id), + phase: 'enable', + title: `enable scheduled jobs on the TARGET (${resource.name})`, + resourceId: resource.id, + kind: resource.kind, + // Never before the source's are off. Two schedulers on one dataset is + // duplicate outbound email and duplicate published posts. + after: [stepId('disable', resource.id), 'cutover:dns'], + }); + } + } + + // The absolute-URL trap. Object storage moves to a new host, but rows that + // stored `https:///...` keep working until the old account is + // closed and then 404 forever. It is invisible at cutover, which is what + // makes it dangerous. + const storage = moving.filter((r) => URL_BEARING.has(r.kind)); + const databases = moving.filter((r) => r.kind === 'postgres' || r.kind === 'mysql'); + if (storage.length && databases.length) { + const hosts = opts.rewriteHosts ?? []; + steps.push({ + id: 'rewrite:urls', + phase: 'delta', + title: hosts.length + ? `rewrite absolute URLs (${hosts.join(', ')}) to the new host` + : `scan for absolute URLs pointing at the old storage host`, + after: databases.map((d) => stepId('import', d.id)), + irreversible: false, + }); + if (!hosts.length) { + risks.push({ + severity: 'warning', + message: + 'Storage and a database are both moving but no --rewrite-host was given. Rows holding absolute URLs to the old storage host keep working until that account is closed, then 404 permanently. The scan step will report what it finds; pass --rewrite-host to fix them.', + }); + } + } + + steps.push({ + id: 'check:all', + phase: 'check', + title: `check credentials and connectivity for ${source.platform} and ${target.label}`, + after: [], + }); + + if (moving.length) { + steps.push({ + id: 'freeze:all', + phase: 'freeze', + title: 'stop writes on the source (downtime starts here)', + after: moving.map((r) => stepId('import', r.id)), + touchesSource: true, + }); + steps.push({ + id: 'cutover:dns', + phase: 'cutover', + title: 'point DNS at the target', + after: ['freeze:all', ...moving.filter(hasDelta(opts)).map((r) => stepId('delta', r.id))], + irreversible: false, + }); + } + + for (const note of source.notes ?? []) { + risks.push({ severity: 'note', message: note }); + } + + if (!moving.length) { + risks.push({ + severity: 'blocker', + message: 'Nothing to migrate: every resource was skipped or unsupported.', + }); + } + + const ordered = orderSteps(steps); + const estimateBytes = moving.reduce((sum, r) => sum + (r.sizeBytes ?? 0), 0); + + return { + source: source.platform, + target: target.id, + scope: source.scope, + steps: ordered, + risks, + moving, + skipped, + estimateBytes, + ok: !risks.some((r) => r.severity === 'blocker'), + }; +} + +function hasDelta(opts: PlanOptions) { + return (r: Resource) => Boolean(opts.engines.get(r.kind)?.delta); +} + +function sizeSuffix(r: Resource): string { + if (r.sizeBytes) return ` (${humanBytes(r.sizeBytes)})`; + if (r.itemCount) return ` (${r.itemCount.toLocaleString()} items)`; + return ''; +} + +export function humanBytes(n: number): string { + const units = ['B', 'KB', 'MB', 'GB', 'TB']; + let v = n; + let i = 0; + while (v >= 1024 && i < units.length - 1) { + v /= 1024; + i += 1; + } + return `${v >= 10 || i === 0 ? Math.round(v) : v.toFixed(1)} ${units[i]}`; +} + +/** + * Sort steps by phase, then by dependency within the phase. + * + * Phase order is absolute and comes first: a `delta` step never runs before a + * `freeze` step even if nothing declares the dependency, because the phase + * ordering *is* the safety property. Within a phase, a stable topological sort + * respects `after`, and a cycle throws rather than quietly picking an order — + * a cyclic plan is a bug in the planner and running it would be worse than + * failing. + */ +export function orderSteps(steps: Step[]): Step[] { + const byId = new Map(steps.map((s) => [s.id, s])); + const out: Step[] = []; + const done = new Set(); + + for (const phase of PHASES) { + const inPhase = steps.filter((s) => s.phase === phase); + const pending = new Map(inPhase.map((s) => [s.id, s])); + + while (pending.size) { + let progressed = false; + for (const [id, step] of [...pending]) { + // Only dependencies inside this phase can block; an earlier phase has + // already run by construction, and a dependency on a later phase would + // be a planner bug, caught below. + const blocking = step.after.filter( + (dep) => pending.has(dep) && dep !== id && byId.get(dep)?.phase === phase, + ); + if (blocking.length === 0) { + out.push(step); + done.add(id); + pending.delete(id); + progressed = true; + } + } + if (!progressed) { + throw new Error( + `migration plan has a dependency cycle in phase '${phase}' among: ${[...pending.keys()].join(', ')}`, + ); + } + } + } + + // A dependency naming a step in a LATER phase inverts the safety ordering. + for (const step of out) { + for (const dep of step.after) { + const target = byId.get(dep); + if (!target) continue; + if (PHASES.indexOf(target.phase) > PHASES.indexOf(step.phase)) { + throw new Error( + `step '${step.id}' (${step.phase}) depends on '${dep}' (${target.phase}), which runs later`, + ); + } + } + } + + return out; +} + +/** Render a plan the way `sh1pt migrate plan` prints it. */ +export function renderPlan(plan: MigrationPlan): string { + const lines: string[] = []; + lines.push(`${plan.source} → ${plan.target} (${plan.scope})`); + lines.push(''); + + if (plan.moving.length) { + lines.push(`Moving ${plan.moving.length} resource(s), ${humanBytes(plan.estimateBytes)}:`); + for (const r of plan.moving) lines.push(` ${r.kind.padEnd(15)} ${r.name}${sizeSuffix(r)}`); + lines.push(''); + } + + if (plan.skipped.length) { + lines.push('Not moving:'); + for (const s of plan.skipped) lines.push(` ${s.resource.name} — ${s.reason}`); + lines.push(''); + } + + let phase: Phase | null = null; + for (const step of plan.steps) { + if (step.phase !== phase) { + phase = step.phase; + lines.push(`${phase}:`); + } + const marks = [ + step.touchesSource ? 'SOURCE' : null, + step.irreversible ? 'IRREVERSIBLE' : null, + ].filter(Boolean); + lines.push(` ${step.title}${marks.length ? ` [${marks.join(' ')}]` : ''}`); + } + + if (plan.risks.length) { + lines.push(''); + for (const sev of ['blocker', 'warning', 'note'] as const) { + for (const r of plan.risks.filter((x) => x.severity === sev)) { + lines.push(`${sev.toUpperCase()}: ${r.message}`); + } + } + } + + lines.push(''); + lines.push(plan.ok ? 'Plan is applyable.' : 'Plan has blockers and cannot be applied.'); + return lines.join('\n'); +} diff --git a/packages/migrate/src/platforms/index.test.ts b/packages/migrate/src/platforms/index.test.ts new file mode 100644 index 00000000..e0bf4875 --- /dev/null +++ b/packages/migrate/src/platforms/index.test.ts @@ -0,0 +1,234 @@ +import { describe, expect, it } from 'vitest'; +import { PLATFORMS, compatibleKinds, platformById, sources, targets } from './index.js'; +import { classifyConnection, connectionsFromVariables } from './railway.js'; +import { + SUPABASE_QUIRKS, + supabaseDsn, + supabasePlatform, + supabasePublicHost, + supabaseS3Endpoint, +} from './supabase.js'; +import { sshPlatform } from './ssh.js'; +import { tursoPlatform } from './managed.js'; +import type { PlatformContext } from '../types.js'; + +const ctx = (secrets: Record = {}): PlatformContext => ({ + secret: (k) => secrets[k], + log: () => {}, + dryRun: false, +}); + +describe('the platform registry', () => { + it('gives every platform a unique id', () => { + const ids = PLATFORMS.map((p) => p.id); + expect(new Set(ids).size).toBe(ids.length); + }); + + it('declares at least one resource kind per platform', () => { + for (const p of PLATFORMS) expect(p.supports.length).toBeGreaterThan(0); + }); + + it('gives every target a way to resolve an incoming resource', () => { + for (const p of PLATFORMS) { + if (p.role === 'source') continue; + expect(typeof p.provision === 'function' || p.id === 'railway').toBe(true); + } + }); + + it('finds a platform by id', () => { + expect(platformById('supabase')?.label).toBe('Supabase'); + expect(platformById('nope')).toBeUndefined(); + }); + + it('lists sources and targets separately', () => { + expect(sources().length).toBeGreaterThan(0); + expect(targets().length).toBeGreaterThan(0); + }); +}); + +describe('compatibleKinds is what makes direction a non-question', () => { + it('finds postgres in both directions between Supabase and a box', () => { + expect(compatibleKinds('supabase', 'ssh')).toContain('postgres'); + expect(compatibleKinds('ssh', 'supabase')).toContain('postgres'); + }); + + it('moves MySQL off PlanetScale onto a box, which is the get-off-the-cloud case', () => { + expect(compatibleKinds('planetscale', 'ssh')).toEqual(['mysql']); + expect(compatibleKinds('ssh', 'planetscale')).toEqual(['mysql']); + }); + + it('pairs Turso with a box over sqlite, both ways', () => { + expect(compatibleKinds('turso', 'ssh')).toEqual(['sqlite']); + expect(compatibleKinds('ssh', 'turso')).toEqual(['sqlite']); + }); + + it('reports an empty intersection rather than pretending a migration is possible', () => { + // Turso holds sqlite; Neon accepts only postgres. + expect(compatibleKinds('turso', 'neon')).toEqual([]); + }); + + it('is empty for an unknown platform', () => { + expect(compatibleKinds('turso', 'nope')).toEqual([]); + }); +}); + +describe('railway connection discovery', () => { + it('classifies by scheme, not by variable name', () => { + expect(classifyConnection('postgresql://u@h/d')).toBe('postgres'); + expect(classifyConnection('postgres://u@h/d')).toBe('postgres'); + expect(classifyConnection('mysql://u@h/d')).toBe('mysql'); + expect(classifyConnection('redis://h:6379')).toBe('redis'); + expect(classifyConnection('rediss://h:6379')).toBe('redis'); + expect(classifyConnection('libsql://x.turso.io')).toBe('sqlite'); + }); + + it('ignores a variable that is not a connection string', () => { + expect(classifyConnection('production')).toBeUndefined(); + expect(classifyConnection('https://example.com')).toBeUndefined(); + }); + + it('finds a database under a non-standard variable name', () => { + const found = connectionsFromVariables([ + { name: 'NODE_ENV', value: 'production' }, + { name: 'PG_URI', value: 'postgres://u@h/d' }, + ]); + expect(found).toHaveLength(1); + expect(found[0]?.kind).toBe('postgres'); + }); + + it('moves a database once even when several services share it', () => { + const found = connectionsFromVariables([ + { name: 'DATABASE_URL', value: 'postgres://u@h/d' }, + { name: 'DATABASE_URL', value: 'postgres://u@h/d' }, + { name: 'PG_URL', value: 'postgres://u@h/d' }, + ]); + expect(found).toHaveLength(1); + }); +}); + +describe('supabase', () => { + it('builds the direct DSN, not the pooler, which cannot serve a dump', () => { + const dsn = supabaseDsn({ projectRef: 'abc123', dbPassword: 'pw' }); + expect(dsn).toContain('db.abc123.supabase.co:5432'); + }); + + it('url-encodes a password with reserved characters', () => { + expect(supabaseDsn({ projectRef: 'r', dbPassword: 'p@ss/word' })).toContain('p%40ss%2Fword'); + }); + + it('prefers an explicit databaseUrl', () => { + expect(supabaseDsn({ projectRef: 'r', databaseUrl: 'postgres://custom/h' })).toBe( + 'postgres://custom/h', + ); + }); + + it('refuses when it has neither', () => { + expect(() => supabaseDsn({ projectRef: 'r' })).toThrow(/databaseUrl or dbPassword/); + }); + + it('derives the s3 endpoint and the public host', () => { + expect(supabaseS3Endpoint('abc')).toBe('https://abc.supabase.co/storage/v1/s3'); + expect(supabasePublicHost('abc')).toBe('abc.supabase.co'); + }); + + it('names the things a pg_dump does not carry', async () => { + const inv = await supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw' }); + const quirks = inv.resources[0]?.quirks ?? []; + expect(quirks).toEqual([...SUPABASE_QUIRKS]); + expect(quirks.some((q) => /JWT/.test(q))).toBe(true); + expect(quirks.some((q) => /pg_cron/.test(q))).toBe(true); + expect(quirks.some((q) => /anon, authenticated/.test(q))).toBe(true); + }); + + it('tells you to rewrite its public host before the project is deleted', async () => { + const inv = await supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw' }); + expect(inv.notes?.[0]).toContain('--rewrite-host abc.supabase.co='); + }); + + it('refuses a bucket with no service role key', async () => { + await expect( + supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw', buckets: ['ads'] }), + ).rejects.toThrow(/SERVICE_ROLE_KEY/); + }); + + it('builds an s3 connection for a bucket when the key is present', async () => { + const inv = await supabasePlatform.inventory(ctx({ SUPABASE_SERVICE_ROLE_KEY: 'svc' }), { + projectRef: 'abc', + dbPassword: 'pw', + buckets: ['ads'], + }); + const bucket = inv.resources.find((r) => r.kind === 'object-storage'); + expect(bucket?.connection.endpoint?.reveal()).toBe('https://abc.supabase.co/storage/v1/s3'); + expect(bucket?.connection.secretAccessKey?.describe()).not.toContain('svc'); + }); +}); + +describe('ssh', () => { + it('describes a box from config, since a server has no API to enumerate itself', async () => { + const inv = await sshPlatform.inventory(ctx(), { + host: 'dev2.example.com', + user: 'anthony', + postgres: [{ name: 'app', url: 'postgres://u@localhost/app' }], + files: [{ name: 'www', path: '/home/anthony/www' }], + }); + expect(inv.resources.map((r) => r.kind).sort()).toEqual(['files', 'postgres']); + expect(inv.scope).toBe('dev2.example.com'); + }); + + it('carries ssh details onto file resources so rsync can reach them', async () => { + const inv = await sshPlatform.inventory(ctx(), { + host: 'h', + user: 'u', + sshKeyPath: '/k', + files: [{ name: 'www', path: '/w' }], + }); + const files = inv.resources[0]!; + expect(files.connection.host?.reveal()).toBe('h'); + expect(files.connection.sshKeyPath?.reveal()).toBe('/k'); + }); + + it('says so when nothing is declared, rather than reporting an empty box', async () => { + const inv = await sshPlatform.inventory(ctx(), { host: 'h' }); + expect(inv.notes?.[0]).toMatch(/no API to enumerate itself/); + }); + + it('resolves an incoming resource against the declared target', async () => { + const out = await sshPlatform.provision( + ctx(), + { kind: 'postgres', id: 'pg', name: 'app', connection: {} }, + { host: 'h', postgres: [{ name: 'app', url: 'postgres://u@localhost/app' }] }, + ); + expect(out.connection.url?.reveal()).toContain('localhost/app'); + }); + + it('refuses when the target declares nowhere to put it', async () => { + await expect( + sshPlatform.provision(ctx(), { kind: 'postgres', id: 'pg', name: 'app', connection: {} }, { host: 'h' }), + ).rejects.toThrow(/nothing on h is declared/); + }); + + it('needs a host', async () => { + await expect(sshPlatform.inventory(ctx(), { host: '' })).rejects.toThrow(/needs a host/); + }); +}); + +describe('turso', () => { + it('works on a database name plus a token rather than a DSN', async () => { + const inv = await tursoPlatform.inventory(ctx({ TURSO_API_TOKEN: 'tok' }), { database: 'prod' }); + expect(inv.resources[0]?.connection.tursoDatabase?.reveal()).toBe('prod'); + expect(inv.resources[0]?.metadata?.platform).toBe('turso'); + }); + + it('refuses without a token', async () => { + await expect(tursoPlatform.inventory(ctx(), { database: 'prod' })).rejects.toThrow(/TURSO_API_TOKEN/); + }); + + it('can also receive, which is what makes dedicated→Turso possible', async () => { + const out = await tursoPlatform.provision!( + ctx({ TURSO_API_TOKEN: 'tok' }), + { kind: 'sqlite', id: 'x', name: 'app', connection: {} }, + { database: 'restored' }, + ); + expect(out.connection.tursoDatabase?.reveal()).toBe('restored'); + }); +}); diff --git a/packages/migrate/src/platforms/index.ts b/packages/migrate/src/platforms/index.ts new file mode 100644 index 00000000..faac3310 --- /dev/null +++ b/packages/migrate/src/platforms/index.ts @@ -0,0 +1,56 @@ +import type { Platform } from '../types.js'; +import { MANAGED_PLATFORMS, tursoPlatform } from './managed.js'; +import { railwayPlatform } from './railway.js'; +import { sshPlatform } from './ssh.js'; +import { supabasePlatform } from './supabase.js'; + +/** + * Every platform this build can read from or write to. + * + * Direction is not encoded here. A platform declares whether it can be a + * source, a target or both, and the planner pairs any two — so the number of + * supported migrations is the number of pairs, not the number of entries. + */ +// biome-ignore lint/suspicious/noExplicitAny: the registry is heterogeneous by +// design — each platform has its own config type, and the CLI resolves the +// right one from a config file at runtime. +export const PLATFORMS: ReadonlyArray> = [ + sshPlatform, + railwayPlatform, + supabasePlatform, + tursoPlatform, + ...MANAGED_PLATFORMS, +]; + +// biome-ignore lint/suspicious/noExplicitAny: see above. +export function platformById(id: string): Platform | undefined { + return PLATFORMS.find((p) => p.id === id); +} + +/** Every platform that can be the left-hand side of a migration. */ +export function sources(): ReadonlyArray> { + return PLATFORMS.filter((p) => p.role !== 'target'); +} + +/** Every platform that can be the right-hand side. */ +export function targets(): ReadonlyArray> { + return PLATFORMS.filter((p) => p.role !== 'source'); +} + +/** + * Whether a pair can move anything at all, and what. + * + * The intersection of what the source holds and what the target accepts. An + * empty intersection is a migration that cannot happen, and saying so takes a + * millisecond instead of a failed cutover. + */ +export function compatibleKinds(fromId: string, toId: string): string[] { + const from = platformById(fromId); + const to = platformById(toId); + if (!from || !to) return []; + const accepted = new Set(to.supports); + return from.supports.filter((k) => accepted.has(k)); +} + +export { railwayPlatform, sshPlatform, supabasePlatform, tursoPlatform }; +export * from './managed.js'; diff --git a/packages/migrate/src/platforms/managed.ts b/packages/migrate/src/platforms/managed.ts new file mode 100644 index 00000000..5be4335d --- /dev/null +++ b/packages/migrate/src/platforms/managed.ts @@ -0,0 +1,212 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * The managed database and app providers that are, from a migration's point of + * view, a connection string with a brand on it. + * + * Turso, Neon, PlanetScale, Fly, Render, Heroku and Vercel differ enormously + * as products and barely at all here: each one hands over a DSN (or a database + * name plus a token) and the engines do the rest. Writing a file per vendor + * would be six copies of the same twenty lines, so they share one factory and + * differ only where they actually differ. + * + * That is the payoff of the platform/engine split stated concretely. Neon to + * dedicated, dedicated to Neon, Neon to Supabase and Railway to Neon are all + * the same two engines; none of them is a code path anyone wrote. + */ + +export interface DsnConfig { + /** The connection string. Takes precedence over anything else. */ + url?: string; + /** Name for the resource in the plan. Defaults to the platform id. */ + name?: string; +} + +export interface TursoConfig { + database: string; + authToken?: string; +} + +interface ManagedSpec { + id: string; + label: string; + kind: Resource['kind']; + /** Environment variable holding the DSN when config does not carry it. */ + envVar: string; + role?: Platform['role']; + quirks?: string[]; + notes?: string[]; +} + +/** + * Build a platform whose entire job is producing one connection string. + * + * `role` defaults to `both`: every one of these can be written to as readily + * as read from, which is what makes the tool bidirectional without a second + * implementation. + */ +function dsnPlatform(spec: ManagedSpec): Platform { + return { + id: spec.id, + label: spec.label, + role: spec.role ?? 'both', + supports: [spec.kind], + + async inventory(ctx: PlatformContext, config: DsnConfig): Promise { + const url = config.url ?? ctx.secret(spec.envVar); + if (!url) { + throw new Error(`${spec.label} needs a connection string: set ${spec.envVar} or pass url`); + } + const name = config.name ?? spec.id; + return { + platform: spec.id, + scope: name, + resources: [ + { + kind: spec.kind, + id: `${spec.id}-${name}`, + name, + connection: { url: secret(url) }, + ...(spec.quirks ? { quirks: [...spec.quirks] } : {}), + }, + ], + ...(spec.notes ? { notes: [...spec.notes] } : {}), + }; + }, + + async provision(_ctx: PlatformContext, resource: Resource, config: DsnConfig): Promise { + const url = config.url; + if (!url) throw new Error(`${spec.label} target needs a connection string`); + if (resource.kind !== spec.kind) { + throw new Error(`${spec.label} cannot receive a ${resource.kind}`); + } + return { ...resource, id: `${spec.id}-${resource.name}`, connection: { url: secret(url) } }; + }, + + async check(ctx: PlatformContext, config: DsnConfig): Promise { + if (!(config.url ?? ctx.secret(spec.envVar))) { + throw new Error(`${spec.label}: no connection string (${spec.envVar})`); + } + }, + }; +} + +export const neonPlatform = dsnPlatform({ + id: 'neon', + label: 'Neon', + kind: 'postgres', + envVar: 'NEON_DATABASE_URL', + quirks: [ + 'Neon branches are copy-on-write and do not survive a dump; only the branch you point at is moved', + ], +}); + +export const planetscalePlatform = dsnPlatform({ + id: 'planetscale', + label: 'PlanetScale', + kind: 'mysql', + envVar: 'PLANETSCALE_DATABASE_URL', + quirks: [ + 'PlanetScale does not support foreign key constraints in the usual way, so a dump taken here may restore without the constraints a plain MySQL would expect', + ], +}); + +export const flyPostgresPlatform = dsnPlatform({ + id: 'fly', + label: 'Fly.io Postgres', + kind: 'postgres', + envVar: 'FLY_DATABASE_URL', + quirks: ['a Fly Postgres is reachable only over the private network unless proxied; run `fly proxy` first'], +}); + +export const renderPlatform = dsnPlatform({ + id: 'render', + label: 'Render', + kind: 'postgres', + envVar: 'RENDER_DATABASE_URL', +}); + +export const herokuPlatform = dsnPlatform({ + id: 'heroku', + label: 'Heroku Postgres', + kind: 'postgres', + envVar: 'HEROKU_DATABASE_URL', + quirks: [ + 'Heroku rotates DATABASE_URL without warning; resolve it at the moment of use rather than caching it across a long migration', + ], +}); + +export const vercelPostgresPlatform = dsnPlatform({ + id: 'vercel', + label: 'Vercel Postgres', + kind: 'postgres', + envVar: 'POSTGRES_URL', + notes: [ + 'Vercel Postgres is Neon underneath; the non-pooling POSTGRES_URL_NON_POOLING is the one a dump wants', + ], +}); + +/** + * Turso, which is the one that does not fit the DSN mould. + * + * Its CLI works on a database NAME plus an account token rather than a + * connection string, so it gets a real implementation rather than a factory + * call. The sqlite engine already knows the difference. + */ +export const tursoPlatform: Platform = { + id: 'turso', + label: 'Turso', + role: 'both', + supports: ['sqlite'], + + async inventory(ctx: PlatformContext, config: TursoConfig): Promise { + if (!config.database) throw new Error('Turso needs a database name'); + const token = config.authToken ?? ctx.secret('TURSO_API_TOKEN'); + if (!token) throw new Error('Turso needs TURSO_API_TOKEN'); + + return { + platform: 'turso', + scope: config.database, + resources: [ + { + kind: 'sqlite', + id: `turso-${config.database}`, + name: config.database, + connection: { tursoDatabase: plain(config.database), authToken: secret(token) }, + metadata: { platform: 'turso' }, + quirks: [ + 'embedded replicas and their sync state do not move; only the primary database is dumped', + ], + }, + ], + }; + }, + + async provision(ctx: PlatformContext, resource: Resource, config: TursoConfig): Promise { + if (resource.kind !== 'sqlite') throw new Error(`Turso cannot receive a ${resource.kind}`); + const token = config.authToken ?? ctx.secret('TURSO_API_TOKEN'); + if (!token) throw new Error('Turso needs TURSO_API_TOKEN'); + return { + ...resource, + id: `turso-${config.database}`, + connection: { tursoDatabase: plain(config.database), authToken: secret(token) }, + metadata: { ...resource.metadata, platform: 'turso' }, + }; + }, + + async check(ctx: PlatformContext, config: TursoConfig): Promise { + if (!(config.authToken ?? ctx.secret('TURSO_API_TOKEN'))) { + throw new Error('Turso: no TURSO_API_TOKEN'); + } + }, +}; + +export const MANAGED_PLATFORMS = [ + neonPlatform, + planetscalePlatform, + flyPostgresPlatform, + renderPlatform, + herokuPlatform, + vercelPostgresPlatform, +] as const; diff --git a/packages/migrate/src/platforms/railway.ts b/packages/migrate/src/platforms/railway.ts new file mode 100644 index 00000000..37026967 --- /dev/null +++ b/packages/migrate/src/platforms/railway.ts @@ -0,0 +1,171 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * Railway. + * + * A Railway project is services plus volumes plus variables, and the data + * worth moving hides in the variables: a Postgres service's connection string + * is `DATABASE_URL` on the services that use it, not something the API hands + * over as a database object. So the inventory reads variables and recognises + * connection strings in them, rather than asking for a list of databases that + * does not exist in that shape. + * + * `role: 'both'` — Railway is a perfectly good destination, and "we tried + * bare metal and went back" is a migration people actually make. Provisioning + * a service here is not automated: creating billable infrastructure is a + * decision, and `sh1pt deploy` is where that lives. What this does is resolve + * an existing service's connection so data can be written into it. + */ + +const API = 'https://backboard.railway.app/graphql/v2'; + +export interface RailwayConfig { + projectId: string; + environmentId?: string; + /** Overrides the RAILWAY_TOKEN secret when set. */ + token?: string; +} + +interface GqlVariable { + name: string; + value: string; +} + +async function gql(token: string, query: string, variables: Record): Promise { + const res = await fetch(API, { + method: 'POST', + headers: { Authorization: `Bearer ${token}`, 'Content-Type': 'application/json' }, + body: JSON.stringify({ query, variables }), + }); + const json = (await res.json()) as { data?: T; errors?: Array<{ message: string }> }; + if (json.errors?.length) throw new Error(`Railway: ${json.errors[0]!.message}`); + if (!res.ok) throw new Error(`Railway HTTP ${res.status}`); + return json.data as T; +} + +/** + * Recognise a connection string by its scheme. + * + * Deliberately not by variable name. `DATABASE_URL` is the convention but + * plenty of apps use `PG_URL`, `POSTGRES_URI` or something bespoke, and a + * migration that silently skipped the database because it was called the wrong + * thing would be the worst possible failure. The scheme is the fact. + */ +export function classifyConnection(value: string): Resource['kind'] | undefined { + if (/^postgres(ql)?:\/\//i.test(value)) return 'postgres'; + if (/^mysql:\/\//i.test(value)) return 'mysql'; + if (/^rediss?:\/\//i.test(value)) return 'redis'; + if (/^libsql:\/\//i.test(value)) return 'sqlite'; + return undefined; +} + +/** Variables whose value is a connection string, deduplicated by target. */ +export function connectionsFromVariables(vars: GqlVariable[]): Array<{ + kind: Resource['kind']; + name: string; + url: string; +}> { + const seen = new Set(); + const out: Array<{ kind: Resource['kind']; name: string; url: string }> = []; + for (const v of vars) { + const kind = classifyConnection(v.value); + if (!kind) continue; + // The same database is usually injected into several services under the + // same or different names; moving it once is the point. + if (seen.has(v.value)) continue; + seen.add(v.value); + out.push({ kind, name: v.name, url: v.value }); + } + return out; +} + +export const railwayPlatform: Platform = { + id: 'railway', + label: 'Railway', + role: 'both', + supports: ['postgres', 'mysql', 'redis', 'sqlite', 'files'], + + async inventory(ctx: PlatformContext, config: RailwayConfig): Promise { + const token = config.token ?? ctx.secret('RAILWAY_TOKEN'); + if (!token) throw new Error('Railway needs a RAILWAY_TOKEN'); + if (!config.projectId) throw new Error('Railway needs a projectId'); + + const data = await gql<{ + project: { + name: string; + services: { edges: Array<{ node: { id: string; name: string } }> }; + volumes: { edges: Array<{ node: { id: string; name: string; mountPath?: string } }> }; + }; + }>( + token, + `query ($id: String!) { + project(id: $id) { + name + services { edges { node { id name } } } + volumes { edges { node { id name } } } + } + }`, + { id: config.projectId }, + ); + + const resources: Resource[] = []; + const notes: string[] = []; + + for (const edge of data.project.volumes.edges) { + const v = edge.node; + resources.push({ + kind: 'files', + id: `volume-${v.id}`, + name: v.name, + // A Railway volume is only reachable from inside a service, so it + // cannot be rsynced from here. Recorded so the plan shows it and a + // person decides, rather than being silently dropped. + connection: { path: plain(v.mountPath ?? '/data') }, + quirks: [ + 'a Railway volume is only reachable from inside its service; copy it out with `railway run` or a one-off container rather than over ssh', + ], + }); + } + + for (const edge of data.project.services.edges) { + const service = edge.node; + const vars = await gql<{ variables: GqlVariable[] }>( + token, + `query ($projectId: String!, $serviceId: String!, $environmentId: String) { + variables(projectId: $projectId, serviceId: $serviceId, environmentId: $environmentId) { name value } + }`, + { + projectId: config.projectId, + serviceId: service.id, + environmentId: config.environmentId ?? null, + }, + ).catch(() => ({ variables: [] as GqlVariable[] })); + + for (const conn of connectionsFromVariables(vars.variables ?? [])) { + resources.push({ + kind: conn.kind, + id: `${service.id}-${conn.name}`, + name: `${service.name}/${conn.name}`, + connection: { url: secret(conn.url) }, + metadata: { service: service.name, variable: conn.name }, + }); + } + } + + if (!resources.length) { + notes.push( + 'No connection strings were found in this project. Either the token cannot read variables, or the databases are referenced another way.', + ); + } + + ctx.log(`${data.project.name}: ${resources.length} resource(s)`); + return { platform: 'railway', scope: data.project.name, resources, notes }; + }, + + async check(ctx: PlatformContext, config: RailwayConfig): Promise { + const token = config.token ?? ctx.secret('RAILWAY_TOKEN'); + if (!token) throw new Error('Railway needs a RAILWAY_TOKEN'); + await gql(token, `query ($id: String!) { project(id: $id) { id } }`, { id: config.projectId }); + }, +}; diff --git a/packages/migrate/src/platforms/ssh.ts b/packages/migrate/src/platforms/ssh.ts new file mode 100644 index 00000000..23b44519 --- /dev/null +++ b/packages/migrate/src/platforms/ssh.ts @@ -0,0 +1,207 @@ +import type { Inventory, Platform, PlatformContext, Resource, Secretish } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * A box you own: a VPS, a dedicated server, the thing at the other end of an + * ssh config entry. + * + * This is the platform every "get off the cloud" migration ends at, and the + * one every "we need managed after all" migration starts from. It is `both` + * because those are the same code path, which is the entire point of splitting + * platforms from engines. + * + * Unlike a managed provider there is no API to ask what is here, so a box is + * described rather than discovered. That is not a workaround: a directory on a + * server has no metadata saying "this is the uploads volume", and guessing + * from paths would be worse than being told. The description lives in config, + * which means it is reviewable and it is the same on every run. + */ + +export interface SshConfig { + host: string; + user?: string; + sshKeyPath?: string; + sshPort?: string; + /** Databases reachable from this box, usually on loopback. */ + postgres?: Array<{ name: string; url: string }>; + mysql?: Array<{ name: string; url: string }>; + sqlite?: Array<{ name: string; path: string }>; + redis?: Array<{ name: string; url: string; dataDir?: string }>; + /** Directories to move: volumes, docroots, upload trees. */ + files?: Array<{ name: string; path: string }>; + /** An S3-compatible endpoint running on the box, e.g. MinIO. */ + buckets?: Array<{ + name: string; + bucket: string; + endpoint: string; + accessKeyId: string; + secretAccessKey: string; + region?: string; + }>; +} + +export const sshPlatform: Platform = { + id: 'ssh', + label: 'dedicated / VPS over ssh', + role: 'both', + supports: ['postgres', 'mysql', 'sqlite', 'redis', 'files', 'object-storage'], + + async inventory(ctx: PlatformContext, config: SshConfig): Promise { + if (!config.host) throw new Error('ssh platform needs a host'); + + const resources: Resource[] = []; + const notes: string[] = []; + + for (const db of config.postgres ?? []) { + resources.push({ + kind: 'postgres', + id: `pg-${db.name}`, + name: db.name, + connection: { url: secret(db.url) }, + }); + } + + for (const db of config.mysql ?? []) { + resources.push({ + kind: 'mysql', + id: `mysql-${db.name}`, + name: db.name, + connection: { url: secret(db.url) }, + }); + } + + for (const db of config.sqlite ?? []) { + resources.push({ + kind: 'sqlite', + id: `sqlite-${db.name}`, + name: db.name, + connection: { path: plain(db.path) }, + }); + } + + for (const r of config.redis ?? []) { + resources.push({ + kind: 'redis', + id: `redis-${r.name}`, + name: r.name, + connection: { + url: secret(r.url), + ...(r.dataDir ? { dataDir: plain(r.dataDir) } : {}), + }, + }); + } + + for (const dir of config.files ?? []) { + resources.push({ + kind: 'files', + id: `files-${dir.name}`, + name: dir.name, + connection: { + path: plain(dir.path), + host: plain(config.host), + ...(config.user ? { user: plain(config.user) } : {}), + ...(config.sshKeyPath ? { sshKeyPath: plain(config.sshKeyPath) } : {}), + ...(config.sshPort ? { sshPort: plain(config.sshPort) } : {}), + }, + }); + } + + for (const b of config.buckets ?? []) { + resources.push({ + kind: 'object-storage', + id: `bucket-${b.name}`, + name: b.name, + connection: { + bucket: plain(b.bucket), + endpoint: plain(b.endpoint), + accessKeyId: secret(b.accessKeyId), + secretAccessKey: secret(b.secretAccessKey), + ...(b.region ? { region: plain(b.region) } : {}), + }, + }); + } + + if (!resources.length) { + notes.push( + 'Nothing is declared for this box. A server has no API to enumerate itself, so its databases and directories have to be listed in config.', + ); + } + + ctx.log(`${config.host}: ${resources.length} declared resource(s)`); + return { platform: 'ssh', scope: config.host, resources, notes }; + }, + + /** + * The target side of a move onto a box. + * + * A resource arriving here keeps its name and gets this box's connection + * details. Nothing is created remotely: the database or directory is + * expected to exist, because provisioning a Postgres on someone's server is + * a decision about disks and versions and backups that a migration tool + * should not be quietly making. + */ + async provision(ctx: PlatformContext, resource: Resource, config: SshConfig): Promise { + const match = ((): Record | undefined => { + switch (resource.kind) { + case 'postgres': { + const db = config.postgres?.find((d) => d.name === resource.name) ?? config.postgres?.[0]; + return db ? { url: secret(db.url) } : undefined; + } + case 'mysql': { + const db = config.mysql?.find((d) => d.name === resource.name) ?? config.mysql?.[0]; + return db ? { url: secret(db.url) } : undefined; + } + case 'sqlite': { + const db = config.sqlite?.find((d) => d.name === resource.name) ?? config.sqlite?.[0]; + return db ? { path: plain(db.path) } : undefined; + } + case 'redis': { + const r = config.redis?.find((d) => d.name === resource.name) ?? config.redis?.[0]; + return r + ? { url: secret(r.url), ...(r.dataDir ? { dataDir: plain(r.dataDir) } : {}) } + : undefined; + } + case 'files': { + const d = config.files?.find((x) => x.name === resource.name) ?? config.files?.[0]; + return d + ? { + path: plain(d.path), + host: plain(config.host), + ...(config.user ? { user: plain(config.user) } : {}), + ...(config.sshKeyPath ? { sshKeyPath: plain(config.sshKeyPath) } : {}), + ...(config.sshPort ? { sshPort: plain(config.sshPort) } : {}), + } + : undefined; + } + case 'object-storage': { + const b = config.buckets?.find((x) => x.name === resource.name) ?? config.buckets?.[0]; + return b + ? { + bucket: plain(b.bucket), + endpoint: plain(b.endpoint), + accessKeyId: secret(b.accessKeyId), + secretAccessKey: secret(b.secretAccessKey), + ...(b.region ? { region: plain(b.region) } : {}), + } + : undefined; + } + default: + return undefined; + } + })(); + + if (!match) { + throw new Error( + `nothing on ${config.host} is declared to receive ${resource.kind} '${resource.name}'. Add it to the target config.`, + ); + } + + ctx.log(`${resource.name} → ${config.host}`); + return { ...resource, id: `${resource.id}@${config.host}`, connection: match }; + }, + + async check(ctx: PlatformContext, config: SshConfig): Promise { + if (!config.host) throw new Error('ssh platform needs a host'); + ctx.log(`target box is ${config.user ? `${config.user}@` : ''}${config.host}`); + }, +}; diff --git a/packages/migrate/src/platforms/supabase.ts b/packages/migrate/src/platforms/supabase.ts new file mode 100644 index 00000000..f8900b5c --- /dev/null +++ b/packages/migrate/src/platforms/supabase.ts @@ -0,0 +1,156 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * Supabase. + * + * A Supabase project is a Postgres with a lot bolted on, and the bolted-on + * parts are what make migrating off it interesting. The database moves with + * pg_dump like any other Postgres; storage is S3-compatible and moves with the + * object-storage engine. What does not move cleanly is everything in between, + * and the value of this inventory is naming those things up front as quirks so + * the planner warns rather than letting them be discovered afterwards. + * + * The list comes from a migration that actually happened (crawlproof.com, + * 2026-09-24): 128 tables, 113 users in `auth.users`, 204 functions, 10 + * pg_cron jobs, a realtime publication, and three public buckets holding 8,410 + * objects. + * + * ## The ones that bite + * + * `auth.users` is a real table in the dump, so users migrate — but the GoTrue + * service that reads it does not, and neither do the JWT secrets. Restoring + * `auth.users` somewhere with a different JWT secret logs everyone out and + * invalidates every refresh token. + * + * Row Level Security policies reference roles (`authenticated`, `anon`, + * `service_role`) that do not exist on a plain Postgres. The policies restore; + * the roles have to be created first or every policy fails. + * + * pg_cron jobs restore and start running immediately, which is how a migration + * ends up with two schedulers on one dataset. + */ + +export interface SupabaseConfig { + projectRef: string; + /** The database password; the rest of the DSN is derived from the ref. */ + dbPassword?: string; + /** Full DSN, when the project uses a pooler or a custom host. */ + databaseUrl?: string; + serviceRoleKey?: string; + /** Buckets to move. Supabase's S3 endpoint is derived from the ref. */ + buckets?: string[]; + region?: string; +} + +/** + * Supabase's direct-connection DSN for a project. + * + * Direct rather than the pooler: pgbouncer in transaction mode does not + * support the session-level operations a dump and restore need, and the + * failure is a confusing mid-dump error rather than a refusal. + */ +export function supabaseDsn(config: SupabaseConfig): string { + if (config.databaseUrl) return config.databaseUrl; + if (!config.dbPassword) { + throw new Error('Supabase needs either databaseUrl or dbPassword'); + } + const pw = encodeURIComponent(config.dbPassword); + return `postgresql://postgres:${pw}@db.${config.projectRef}.supabase.co:5432/postgres`; +} + +/** The S3-compatible storage endpoint for a project. */ +export function supabaseS3Endpoint(projectRef: string): string { + return `https://${projectRef}.supabase.co/storage/v1/s3`; +} + +/** The public object host, which is what ends up embedded in rows. */ +export function supabasePublicHost(projectRef: string): string { + return `${projectRef}.supabase.co`; +} + +/** What does not survive a plain pg_dump/pg_restore, named up front. */ +export const SUPABASE_QUIRKS: readonly string[] = [ + 'auth.users restores as data, but GoTrue and the JWT secret do not move with it: everyone is logged out and every refresh token is invalid unless the JWT secret is carried across', + 'RLS policies reference the roles anon, authenticated and service_role, which do not exist on a plain Postgres and must be created before the restore', + 'pg_cron jobs restore already enabled and begin firing immediately, so they must be disabled on the source before the cutover or two schedulers run against one dataset', + 'the realtime publication (supabase_realtime) is recreated by the dump but nothing subscribes to it without the realtime service', + 'storage.objects rows are metadata; the bytes live in the bucket and move separately', +] as const; + +export const supabasePlatform: Platform = { + id: 'supabase', + label: 'Supabase', + role: 'both', + supports: ['postgres', 'object-storage', 'cron'], + + async inventory(ctx: PlatformContext, config: SupabaseConfig): Promise { + if (!config.projectRef) throw new Error('Supabase needs a projectRef'); + + const resources: Resource[] = []; + + resources.push({ + kind: 'postgres', + id: `pg-${config.projectRef}`, + name: 'postgres', + connection: { url: secret(supabaseDsn(config)) }, + quirks: [...SUPABASE_QUIRKS], + metadata: { projectRef: config.projectRef, publicHost: supabasePublicHost(config.projectRef) }, + }); + + const serviceKey = config.serviceRoleKey ?? ctx.secret('SUPABASE_SERVICE_ROLE_KEY'); + for (const bucket of config.buckets ?? []) { + if (!serviceKey) { + throw new Error( + `bucket '${bucket}' needs SUPABASE_SERVICE_ROLE_KEY to read Supabase storage over S3`, + ); + } + resources.push({ + kind: 'object-storage', + id: `bucket-${bucket}`, + name: bucket, + connection: { + bucket: plain(bucket), + endpoint: plain(supabaseS3Endpoint(config.projectRef)), + // Supabase's S3 gateway takes the project ref as the access key and + // the service role key as the secret. + accessKeyId: plain(config.projectRef), + secretAccessKey: secret(serviceKey), + region: plain(config.region ?? 'us-east-1'), + }, + }); + } + + const notes = [ + `rows may hold absolute URLs to ${supabasePublicHost(config.projectRef)}; pass --rewrite-host ${supabasePublicHost(config.projectRef)}= so they are rewritten rather than 404ing when this project is deleted`, + ]; + + ctx.log(`supabase ${config.projectRef}: ${resources.length} resource(s)`); + return { platform: 'supabase', scope: config.projectRef, resources, notes }; + }, + + async provision(_ctx: PlatformContext, resource: Resource, config: SupabaseConfig): Promise { + if (resource.kind === 'postgres') { + return { ...resource, id: `pg-${config.projectRef}`, connection: { url: secret(supabaseDsn(config)) } }; + } + if (resource.kind === 'object-storage') { + const serviceKey = config.serviceRoleKey; + if (!serviceKey) throw new Error('writing to Supabase storage needs a serviceRoleKey'); + return { + ...resource, + connection: { + bucket: plain(resource.name), + endpoint: plain(supabaseS3Endpoint(config.projectRef)), + accessKeyId: plain(config.projectRef), + secretAccessKey: secret(serviceKey), + region: plain(config.region ?? 'us-east-1'), + }, + }; + } + throw new Error(`Supabase cannot receive a ${resource.kind}`); + }, + + async check(_ctx: PlatformContext, config: SupabaseConfig): Promise { + supabaseDsn(config); + }, +}; diff --git a/packages/migrate/src/staging.ts b/packages/migrate/src/staging.ts new file mode 100644 index 00000000..046952f5 --- /dev/null +++ b/packages/migrate/src/staging.ts @@ -0,0 +1,109 @@ +import { createHash } from 'node:crypto'; +import { createReadStream } from 'node:fs'; +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { dirname, join, resolve } from 'node:path'; +import type { Artifact, Staging } from './types.js'; + +/** + * Where a migration keeps what it has already done. + * + * A migration is long and interrupted runs are normal — a laptop sleeps, a + * token expires, someone hits ctrl-C during the bulk copy because they + * realised they picked the wrong project. The staging directory is what makes + * the next run cheap instead of a restart: a ledger of finished artifacts, on + * disk, next to the bytes they describe. + * + * The ledger is append-only JSON lines rather than one rewritten JSON file. + * A process killed mid-write corrupts the file it was rewriting; it can only + * ever truncate the last line of an append-only one, and a half-written line + * fails to parse and is skipped. That is the difference between resuming and + * starting over. + */ + +const LEDGER = 'artifacts.jsonl'; + +export interface StagingOptions { + /** Root directory. Created if absent. */ + dir: string; +} + +export async function openStaging(opts: StagingOptions): Promise { + const dir = resolve(opts.dir); + await mkdir(dir, { recursive: true }); + const ledgerPath = join(dir, LEDGER); + + return { + dir, + + async record(artifact: Artifact): Promise { + await mkdir(dirname(join(dir, artifact.path)), { recursive: true }); + await writeFile(ledgerPath, `${JSON.stringify(artifact)}\n`, { flag: 'a' }); + }, + + async existing(resourceId: string): Promise { + let text: string; + try { + text = await readFile(ledgerPath, 'utf8'); + } catch { + return []; + } + return parseLedger(text).filter((a) => a.resourceId === resourceId); + }, + }; +} + +/** + * Read the ledger, skipping anything unparseable. + * + * A truncated final line is the expected shape of an interrupted run, not an + * error: the process died mid-write. Throwing here would turn a resumable + * migration into a manual cleanup, so a bad line is dropped and the rest is + * used. + */ +export function parseLedger(text: string): Artifact[] { + const out: Artifact[] = []; + for (const line of text.split('\n')) { + const trimmed = line.trim(); + if (!trimmed) continue; + try { + const parsed = JSON.parse(trimmed) as Artifact; + if (parsed && typeof parsed.path === 'string' && typeof parsed.resourceId === 'string') { + out.push(parsed); + } + } catch { + // A partial line from an interrupted write. Skip it. + } + } + return out; +} + +/** Checksum a staged file, so a resumed run can tell complete from truncated. */ +export function sha256File(path: string): Promise { + return new Promise((res, rej) => { + const hash = createHash('sha256'); + const stream = createReadStream(path); + stream.on('error', rej); + stream.on('data', (chunk) => hash.update(chunk)); + stream.on('end', () => res(hash.digest('hex'))); + }); +} + +/** + * An in-memory staging, for tests and for `--dry-run`. + * + * A dry run must not create directories on someone's disk as a side effect of + * being asked what it would do. + */ +export function memoryStaging(dir = '/staging'): Staging & { artifacts: Artifact[] } { + const artifacts: Artifact[] = []; + return { + dir, + artifacts, + async record(a) { + artifacts.push(a); + }, + async existing(resourceId) { + return artifacts.filter((a) => a.resourceId === resourceId); + }, + }; +} diff --git a/packages/migrate/src/transforms.test.ts b/packages/migrate/src/transforms.test.ts new file mode 100644 index 00000000..fdc380d1 --- /dev/null +++ b/packages/migrate/src/transforms.test.ts @@ -0,0 +1,174 @@ +import { describe, expect, it } from 'vitest'; +import { + type ColumnRef, + countMatchesSql, + isScannable, + parseRewrite, + qualify, + quoteIdent, + quoteLiteral, + remainingMatchesSql, + rewriteColumnSql, + rewritePlanSql, +} from './transforms.js'; + +const col = (over: Partial = {}): ColumnRef => ({ + schema: 'public', + table: 'ad_creatives', + column: 'image_url', + dataType: 'text', + ...over, +}); + +describe('quoting', () => { + it('quotes an identifier so a reserved word still parses', () => { + expect(quoteIdent('user')).toBe('"user"'); + expect(quoteIdent('order')).toBe('"order"'); + }); + + it('doubles an embedded quote rather than letting it end the identifier', () => { + expect(quoteIdent('we"ird')).toBe('"we""ird"'); + }); + + it('doubles an embedded apostrophe in a literal', () => { + expect(quoteLiteral("o'brien")).toBe("'o''brien'"); + }); + + it('qualifies schema and table together', () => { + expect(qualify(col())).toBe('"public"."ad_creatives"'); + }); +}); + +describe('isScannable', () => { + it('accepts the text-ish types', () => { + for (const t of ['text', 'character varying', 'character', 'json', 'jsonb']) { + expect(isScannable(t)).toBe(true); + } + }); + + it('is case-insensitive', () => { + expect(isScannable('TEXT')).toBe(true); + }); + + it('rejects types that cannot hold a URL', () => { + for (const t of ['integer', 'boolean', 'timestamp with time zone', 'bytea']) { + expect(isScannable(t)).toBe(false); + } + }); +}); + +describe('countMatchesSql', () => { + it('counts rows mentioning the old host', () => { + const sql = countMatchesSql(col(), 'abc.supabase.co'); + expect(sql).toContain('"public"."ad_creatives"'); + expect(sql).toContain("'%abc.supabase.co%'"); + expect(sql).toContain('count(*)'); + }); + + it('casts to text so json columns are covered by the same statement', () => { + expect(countMatchesSql(col({ dataType: 'jsonb' }), 'h')).toContain('::text like'); + }); +}); + +describe('rewriteColumnSql', () => { + const rewrite = { from: 'abc.supabase.co', to: 'https://cdn.example.com' }; + + it('replaces the old origin with the new one', () => { + const sql = rewriteColumnSql(col(), rewrite); + expect(sql).toContain(`replace("image_url", 'https://abc.supabase.co', 'https://cdn.example.com')`); + }); + + it('only touches rows that actually mention the old host', () => { + expect(rewriteColumnSql(col(), rewrite)).toContain("like '%abc.supabase.co%'"); + }); + + it('round-trips a jsonb column through text and back', () => { + const sql = rewriteColumnSql(col({ dataType: 'jsonb' }), rewrite); + expect(sql).toContain('::text'); + expect(sql).toContain('::jsonb'); + }); + + it('strips a trailing slash off the new origin so URLs do not double up', () => { + const sql = rewriteColumnSql(col(), { from: 'old.host', to: 'https://new.host/' }); + expect(sql).toContain("'https://new.host'"); + expect(sql).not.toContain("'https://new.host/'"); + }); +}); + +describe('rewritePlanSql', () => { + it('wraps every statement in one transaction', () => { + const sql = rewritePlanSql([col()], [{ from: 'a.co', to: 'https://b.co' }]); + expect(sql.startsWith('begin;')).toBe(true); + expect(sql.trimEnd().endsWith('commit;')).toBe(true); + }); + + it('covers every host across every scannable column', () => { + const sql = rewritePlanSql( + [col(), col({ table: 'blog_posts', column: 'body' })], + [ + { from: 'a.co', to: 'https://x.co' }, + { from: 'b.co', to: 'https://y.co' }, + ], + ); + expect(sql.match(/update/g)).toHaveLength(4); + }); + + it('skips columns that cannot hold a URL', () => { + const sql = rewritePlanSql( + [col({ dataType: 'integer', column: 'views' })], + [{ from: 'a.co', to: 'https://b.co' }], + ); + expect(sql).not.toContain('update'); + }); +}); + +describe('remainingMatchesSql', () => { + it('is the post-commit assertion, and zero rows is the pass', () => { + const sql = remainingMatchesSql([col()], ['a.co']); + expect(sql).toContain('where n > 0'); + expect(sql).toContain('%a.co%'); + }); + + it('unions every column and filters OUTSIDE the union', () => { + // A trailing `having` would bind to the last SELECT only, so every other + // column would report regardless of count. The filter must be outside. + const sql = remainingMatchesSql([col(), col({ column: 'thumb_url' })], ['a.co']); + expect(sql).toContain('union all'); + expect(sql).not.toContain('having'); + const afterSubquery = sql.slice(sql.lastIndexOf(') as remaining')); + expect(afterSubquery).toContain('where n > 0'); + }); + + it('degrades to a query returning nothing when there is nothing to check', () => { + expect(remainingMatchesSql([], ['a.co'])).toContain('where false'); + }); +}); + +describe('parseRewrite', () => { + it('parses old=new', () => { + expect(parseRewrite('abc.supabase.co=https://cdn.example.com')).toEqual({ + from: 'abc.supabase.co', + to: 'https://cdn.example.com', + }); + }); + + it('tolerates a scheme on the old side', () => { + expect(parseRewrite('https://abc.supabase.co=https://cdn.example.com').from).toBe( + 'abc.supabase.co', + ); + }); + + it('rejects a new origin with no scheme, which would silently corrupt URLs', () => { + expect(() => parseRewrite('a.co=cdn.example.com')).toThrow(/needs a scheme/); + }); + + it('rejects a path on the old side', () => { + expect(() => parseRewrite('a.co/storage=https://b.co')).toThrow(/bare host/); + }); + + it('rejects a malformed spec', () => { + for (const bad of ['', 'nope', '=https://b.co']) { + expect(() => parseRewrite(bad)).toThrow(); + } + }); +}); diff --git a/packages/migrate/src/transforms.ts b/packages/migrate/src/transforms.ts new file mode 100644 index 00000000..7eba4b6a --- /dev/null +++ b/packages/migrate/src/transforms.ts @@ -0,0 +1,207 @@ +/** + * Rewriting absolute URLs that point at the place you just left. + * + * This is the failure that makes migrations dangerous months after they look + * successful. An app stores an uploaded file and, instead of keeping a key, + * writes the whole URL into a row: + * + * https://ywcizjsgrcmhgyplldac.supabase.co/storage/v1/object/public/ads/x.png + * + * Migrate the bucket and the database, cut DNS over, check the site: every + * image loads. They load because the OLD account still exists and is still + * serving them. The day that account is closed — which is the whole point of + * migrating, and which happens weeks later once everyone is confident — every + * one of those rows 404s at once, and nothing connects the outage to the + * migration. + * + * crawlproof.com had 2,928 such rows across four tables. They were found by + * looking, not by anything failing. + * + * So this is a first-class step rather than a footnote: find every column that + * could hold one, report what is there, and rewrite it inside the same + * transaction that a person can roll back. + * + * Nothing here executes SQL. It builds statements and the caller runs them, + * which keeps it testable and keeps the generated SQL reviewable before it + * touches a database. + */ + +/** A host whose URLs must be rewritten, and what to replace it with. */ +export interface HostRewrite { + /** The old host, e.g. `abc123.supabase.co`. No scheme, no path. */ + from: string; + /** The new origin, e.g. `https://cdn.example.com`. Scheme required. */ + to: string; +} + +/** Text-ish column types worth scanning. */ +const TEXT_TYPES = new Set(['text', 'character varying', 'character', 'json', 'jsonb']); + +export interface ColumnRef { + schema: string; + table: string; + column: string; + dataType: string; +} + +/** + * The query that finds columns which could hold a URL. + * + * Every text-ish column in a user schema. Deliberately broad: a column called + * `notes` holding a pasted URL breaks exactly as badly as one called + * `image_url`, and guessing from names is how the four crawlproof tables would + * have been missed. The count query that follows narrows it cheaply. + */ +export function findTextColumnsSql(): string { + return `select table_schema, table_name, column_name, data_type + from information_schema.columns + where table_schema not in ('pg_catalog', 'information_schema') + and data_type in ('text', 'character varying', 'character', 'json', 'jsonb') + order by table_schema, table_name, column_name`; +} + +/** True when a column's type can hold a URL. */ +export function isScannable(dataType: string): boolean { + return TEXT_TYPES.has(dataType.toLowerCase()); +} + +/** + * Postgres identifiers are quoted rather than interpolated bare. + * + * These names come from information_schema, so they are real identifiers, but + * a table legitimately named `user` or `order` is a reserved word and an + * unquoted reference is a syntax error partway through a migration. Doubling + * any embedded quote is the standard escape. + */ +export function quoteIdent(name: string): string { + return `"${name.replace(/"/g, '""')}"`; +} + +/** A single-quoted SQL string literal. */ +export function quoteLiteral(value: string): string { + return `'${value.replace(/'/g, "''")}'`; +} + +/** Fully-qualified, safely quoted. */ +export function qualify(c: Pick): string { + return `${quoteIdent(c.schema)}.${quoteIdent(c.table)}`; +} + +/** + * Count rows in one column that mention the old host. + * + * Run before rewriting anything: it turns "this might be a problem" into "2,928 + * rows in four tables", which is what makes the step reviewable. A cast to text + * lets one statement cover json and jsonb alongside the plain text types. + */ +export function countMatchesSql(c: ColumnRef, host: string): string { + return `select ${quoteLiteral(`${c.schema}.${c.table}.${c.column}`)} as ref, count(*) as n + from ${qualify(c)} + where ${quoteIdent(c.column)}::text like ${quoteLiteral(`%${host}%`)}`; +} + +/** + * Rewrite one column. + * + * `replace` on the text form rather than a regex: the host is a literal, a + * regex would need escaping, and `replace` is index-friendly and predictable. + * The `where` clause means untouched rows are not rewritten, which keeps the + * update small and leaves `updated_at` triggers alone on rows that did not + * change. + * + * json and jsonb are cast out to text and back, which is lossy for jsonb key + * order but not for content — and jsonb does not preserve key order anyway. + */ +export function rewriteColumnSql(c: ColumnRef, rewrite: HostRewrite): string { + const from = quoteLiteral(`https://${rewrite.from}`); + const to = quoteLiteral(rewrite.to.replace(/\/+$/, '')); + const col = quoteIdent(c.column); + const type = c.dataType.toLowerCase(); + + const expr = + type === 'json' || type === 'jsonb' + ? `replace(${col}::text, ${from}, ${to})::${type}` + : `replace(${col}, ${from}, ${to})`; + + return `update ${qualify(c)} + set ${col} = ${expr} + where ${col}::text like ${quoteLiteral(`%${rewrite.from}%`)}`; +} + +/** + * The whole rewrite as one transaction. + * + * One transaction so a failure halfway leaves nothing half-rewritten, and so + * the whole thing can be rolled back by a person watching it. The trailing + * verification select is what the caller asserts on: after a correct rewrite it + * returns zero rows, and the crawlproof runbook asserted exactly that. + */ +export function rewritePlanSql(columns: ColumnRef[], rewrites: HostRewrite[]): string { + const statements: string[] = ['begin;']; + for (const rewrite of rewrites) { + for (const c of columns) { + if (!isScannable(c.dataType)) continue; + statements.push(`${rewriteColumnSql(c, rewrite)};`); + } + } + statements.push('commit;'); + return statements.join('\n'); +} + +/** + * The assertion that the rewrite worked. + * + * Returns one row per column that still mentions an old host. Zero rows is the + * pass condition. Run it AFTER committing: a migration that reports success + * while rows still point at an account about to be closed is worse than one + * that fails loudly. + */ +export function remainingMatchesSql(columns: ColumnRef[], hosts: string[]): string { + const parts: string[] = []; + for (const host of hosts) { + for (const c of columns) { + if (!isScannable(c.dataType)) continue; + parts.push(countMatchesSql(c, host)); + } + } + if (!parts.length) return 'select null::text as ref, 0::bigint as n where false'; + + /* + * The union goes in a subquery and the filter is a WHERE on the outside. + * + * A trailing `having count(*) > 0` looks like it filters the whole thing and + * does not: in a UNION chain it binds to the final SELECT only, so every + * other column would be reported regardless of its count and the one real + * offender could be buried. Filtering outside the subquery applies to all of + * them, which is the point of the assertion. + */ + return `select ref, n from (\n${parts.join('\nunion all\n')}\n) as remaining where n > 0 order by n desc`; +} + +/** + * Turn `--rewrite-host old=new` into a rewrite. + * + * Accepts a bare host on the left (`abc.supabase.co`) and a full origin on the + * right (`https://cdn.example.com`). A missing scheme on the right is the easy + * mistake and produces a corrupt URL rather than an error at runtime, so it is + * rejected here. + */ +export function parseRewrite(spec: string): HostRewrite { + const eq = spec.indexOf('='); + if (eq < 1) { + throw new Error(`--rewrite-host wants old-host=new-origin, got '${spec}'`); + } + const from = spec.slice(0, eq).trim().replace(/^https?:\/\//, '').replace(/\/+$/, ''); + const to = spec.slice(eq + 1).trim().replace(/\/+$/, ''); + + if (!from) throw new Error(`--rewrite-host has an empty old host: '${spec}'`); + if (!/^https?:\/\//.test(to)) { + throw new Error( + `--rewrite-host needs a scheme on the new origin, got '${to}'. Use https://${to}`, + ); + } + if (from.includes('/')) { + throw new Error(`--rewrite-host old side should be a bare host, got '${from}'`); + } + return { from, to }; +} diff --git a/packages/migrate/src/types.ts b/packages/migrate/src/types.ts new file mode 100644 index 00000000..32f0d22d --- /dev/null +++ b/packages/migrate/src/types.ts @@ -0,0 +1,287 @@ +/** + * Moving an application and its data from one platform to another. + * + * `packages/cloud/*` already answers "give me a machine": connect, quote, + * provision, destroy. That is not this. Provisioning the destination is the + * easy half of a migration; the half that goes wrong is the data — the dump + * that silently omitted an extension, the storage bucket whose objects are + * referenced by absolute URL in a thousand rows, the cron jobs that start + * firing from two places at once because the cutover happened in the wrong + * order. + * + * The shape here comes from an actual migration rather than a whiteboard: + * crawlproof.com off Railway and Supabase cloud onto a dedicated box, which + * moved a 4.7 GB database, 8,410 storage objects across three buckets, ten + * pg_cron jobs and a realtime publication, and which would have quietly broken + * the site months later over 2,928 rows holding absolute storage URLs. + * + * ## Why this is not N×M adapters + * + * The naive shape is one adapter per (source, target) pair, which is why most + * migration tooling supports exactly one direction. The split that avoids it: + * + * **Platforms** answer "what have I got, and what are the credentials?" + * Railway, Supabase, Turso, Fly, Neon, Vercel, a plain VPS over ssh. A + * platform does not know how to move a byte. It does an `inventory()` and + * hands back resources with connection details attached. + * + * **Engines** move the bytes. Postgres, MySQL, SQLite/libSQL, Redis, + * S3-compatible object storage, plain files. An engine does not know or care + * which vendor either side is. + * + * So Railway→dedicated and dedicated→Railway are the same code path, and a new + * platform costs one `inventory()` rather than one adapter per existing + * platform. Direction is not a property of the system; it is which platform + * you named first. + * + * Every resource carries the engine that can move it. A migration is possible + * exactly when, for each resource the source lists, the target can accept that + * engine — which is a check the planner can make before touching anything. + */ + +/** + * What kind of thing is being moved, which is the same as asking which engine + * moves it. + * + * Deliberately about storage shape rather than vendor: Supabase's database and + * Neon's are both `postgres`, and the code that moves one moves the other. A + * vendor difference that genuinely matters (Supabase's auth schema, its + * storage metadata tables) is a `quirk` on the resource, not a new kind. + */ +export type ResourceKind = + | 'postgres' + | 'mysql' + | 'sqlite' + | 'redis' + | 'object-storage' + | 'files' + | 'env' + | 'cron' + | 'dns'; + +/** Every engine id, for exhaustiveness checks and for the CLI's help text. */ +export const RESOURCE_KINDS: readonly ResourceKind[] = [ + 'postgres', + 'mysql', + 'sqlite', + 'redis', + 'object-storage', + 'files', + 'env', + 'cron', + 'dns', +] as const; + +/** + * A credential or connection string. + * + * Kept as a getter rather than a value so a plan can be printed, stored and + * reviewed without a password ever being serialised into it. `describe()` is + * what goes in the plan file; `reveal()` is called only inside an engine, at + * the moment it runs. + */ +export interface Secretish { + /** Safe for logs, plan files and terminal output. Never the secret. */ + describe(): string; + /** The actual value. Call as late as possible, never log the result. */ + reveal(): string; +} + +/** + * A secret that is safe to print because it is not one — a hostname, a bucket + * name, a database name. + */ +export function plain(value: string): Secretish { + return { describe: () => value, reveal: () => value }; +} + +/** + * Wrap a real credential. `describe()` shows enough to tell two apart without + * showing either: scheme and host for a URL, a short prefix otherwise. + */ +export function secret(value: string): Secretish { + return { + reveal: () => value, + describe: () => { + try { + const u = new URL(value); + const user = u.username ? `${u.username}:***@` : ''; + return `${u.protocol}//${user}${u.host}${u.pathname}`; + } catch { + return value.length <= 4 ? '***' : `${value.slice(0, 4)}***`; + } + }, + }; +} + +/** + * One movable thing on a platform. + * + * `id` is the platform's own identifier and `name` is what a person calls it. + * `sizeBytes` and `itemCount` are best-effort: they drive the plan's estimate + * and the progress output, and being wrong is not fatal. `quirks` is where a + * vendor's non-portable detail is recorded so the planner can warn about it + * rather than discovering it halfway through a restore. + */ +export interface Resource { + kind: ResourceKind; + id: string; + name: string; + /** How to reach it. Engine-specific; see each engine for what it needs. */ + connection: Record; + sizeBytes?: number; + itemCount?: number; + /** + * Vendor specifics that survive or do not survive a move: postgres + * extensions, a Supabase auth schema, pg_cron jobs, a realtime publication. + * The planner turns these into warnings and extra steps. + */ + quirks?: string[]; + metadata?: Record; +} + +/** What a platform reported when asked what it holds. */ +export interface Inventory { + platform: string; + /** What the platform calls this deployment: a project, an app, an account. */ + scope: string; + resources: Resource[]; + /** Anything the platform could not enumerate and a human should check. */ + notes?: string[]; +} + +export interface PlatformContext { + secret(key: string): string | undefined; + log(msg: string, level?: 'info' | 'warn' | 'error'): void; + /** True when nothing may be mutated anywhere. */ + dryRun: boolean; +} + +/** + * A platform: somewhere an app lives. Implementations resolve credentials and + * enumerate resources; they never move data. + * + * `role` is honest about what a platform can do rather than aspirational. + * Railway can be read from and written to; a managed provider that offers no + * import path is `source` only, and saying so lets the planner refuse early + * with a clear message instead of failing at the last step. + */ +export interface Platform { + id: string; + label: string; + role: 'source' | 'target' | 'both'; + /** Kinds this platform can hold. The planner intersects source and target. */ + supports: ResourceKind[]; + /** Enumerate what is there. Read-only; safe to run against production. */ + inventory(ctx: PlatformContext, config: Config): Promise; + /** + * Make a place for an incoming resource and return the connection an engine + * should write to. Absent on `source`-only platforms. + */ + provision?(ctx: PlatformContext, resource: Resource, config: Config): Promise; + /** Cheap credentials check, so a three-hour migration fails in the first second. */ + check?(ctx: PlatformContext, config: Config): Promise; +} + +/** Where an engine stages bytes between export and import. */ +export interface Staging { + /** Absolute path to a directory the engine may write into. */ + dir: string; + /** Record an artifact so a resumed run can find it again. */ + record(artifact: Artifact): Promise; + /** Artifacts already produced for this resource, if the run is resuming. */ + existing(resourceId: string): Promise; +} + +/** Something an export produced: a dump file, a manifest, a directory of objects. */ +export interface Artifact { + resourceId: string; + kind: ResourceKind; + /** Path relative to the staging dir. */ + path: string; + sizeBytes?: number; + /** Set once the artifact is complete; a partial artifact has no checksum. */ + sha256?: string; + metadata?: Record; +} + +export interface EngineContext { + log(msg: string, level?: 'info' | 'warn' | 'error'): void; + dryRun: boolean; + staging: Staging; + /** + * Run a command. Injected rather than imported so tests drive an engine + * without a database, and so a dry run can record commands instead of + * running them. + */ + exec(cmd: string, args: string[], opts?: ExecOptions): Promise; +} + +export interface ExecOptions { + /** Extra environment. Secrets belong here, never in `args`, which is logged. */ + env?: Record; + cwd?: string; + /** Fail the step if the command exits non-zero. Default true. */ + check?: boolean; + timeoutMs?: number; + /** + * Send stdout to this absolute path instead of buffering it. + * + * Some tools only dump to stdout — `sqlite3 .dump`, `turso db shell .dump` — + * and a multi-gigabyte dump must not be held in a string. Expressed as an + * option rather than a shell redirect because `exec` runs without a shell, + * so `>` would be passed to the program as a literal argument. + */ + stdoutFile?: string; + /** Feed this file to stdin. The load half of the same problem. */ + stdinFile?: string; +} + +export interface ExecResult { + code: number; + stdout: string; + stderr: string; +} + +/** + * An engine moves one kind of resource between two connections. + * + * Split into export and import rather than a single `copy` so a migration can + * stage everything, be inspected, and then be cut over — which is what makes + * the delta sync and the rollback possible. A direct streaming copy is an + * optimisation an engine may offer via `copy`, not the contract. + */ +export interface Engine { + kind: ResourceKind; + /** Binaries that must exist for this engine to run, e.g. ['pg_dump']. */ + requires: string[]; + /** Read the source into staging. */ + export(ctx: EngineContext, from: Resource): Promise; + /** + * Write staged artifacts into the target. + * + * `from` is the source resource, passed because not every engine stages the + * bytes themselves. Object storage is the case that forces it: pulling ten + * gigabytes down and pushing them back up doubles the transfer for no + * benefit, so its export writes only a manifest and its import runs a + * remote-to-remote copy, which means it still needs to know where the + * objects came from. + */ + import(ctx: EngineContext, to: Resource, artifacts: Artifact[], from?: Resource): Promise; + /** + * Re-read only what changed since a timestamp. This is what makes a cutover + * short: the bulk copy happens while the source is live, and only the delta + * is moved during the window where writes are stopped. An engine that cannot + * do this omits it, and the planner says the cutover needs full downtime. + */ + delta?(ctx: EngineContext, from: Resource, since: Date): Promise; + /** Compare source and target after the fact. Row counts, object counts. */ + verify?(ctx: EngineContext, from: Resource, to: Resource): Promise; +} + +export interface VerifyResult { + ok: boolean; + /** One line per check, e.g. "public.users: 113 → 113". */ + checks: string[]; + problems: string[]; +} diff --git a/packages/migrate/tsconfig.json b/packages/migrate/tsconfig.json new file mode 100644 index 00000000..fcd4e5bb --- /dev/null +++ b/packages/migrate/tsconfig.json @@ -0,0 +1,6 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "outDir": "dist", "rootDir": "src" }, + "include": ["src/**/*"], + "exclude": ["src/**/*.test.ts"] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index ad195b74..02af736d 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -819,6 +819,9 @@ importers: '@profullstack/sh1pt-core': specifier: workspace:^ version: link:../core + '@profullstack/sh1pt-migrate': + specifier: workspace:* + version: link:../migrate '@profullstack/sh1pt-openapi': specifier: workspace:^ version: link:../openapi @@ -1199,6 +1202,12 @@ importers: specifier: workspace:* version: link:../../core + packages/migrate: + dependencies: + '@profullstack/sh1pt-core': + specifier: workspace:* + version: link:../core + packages/observability/sentry: dependencies: '@profullstack/sh1pt-core': @@ -6157,6 +6166,7 @@ packages: eslint@9.39.4: resolution: {integrity: sha512-XoMjdBOwe/esVgEvLmNsD3IRHkm7fbKIUGvrleloJXUZgDHig2IPWNniv+GwjyJXzuNqVjlr5+4yVUZjycJwfQ==} engines: {node: ^18.18.0 || ^20.9.0 || >=21.1.0} + deprecated: This version is no longer supported. Please see https://eslint.org/version-support for other options. hasBin: true peerDependencies: jiti: '*'