From 5cf15eb64b1d445937a53cd1d1d673183599f93c Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 07:44:32 +0000 Subject: [PATCH 1/8] migrate: the contract, the planner and the Postgres engine sh1pt can already provision a machine (packages/cloud/*: connect, quote, provision, destroy). Provisioning is the easy half of a migration. The half that goes wrong is the data, and nothing here moved any. The shape comes from a migration that actually happened rather than a whiteboard: crawlproof.com off Railway and Supabase cloud onto a dedicated box, which moved 4.7 GB of Postgres, 8,410 storage objects, ten pg_cron jobs and a realtime publication, and which would have quietly broken the site months later over 2,928 rows holding absolute storage URLs. The split that avoids N x M adapters, and makes the thing bidirectional for free: PLATFORMS answer "what have I got and what are the credentials" and never move a byte; ENGINES move bytes and neither know nor care which vendor is on either end. Railway->dedicated and dedicated->Railway are then the same code path, and a new platform costs one inventory() rather than one adapter per existing platform. Direction is just which platform you named first. The planner is pure: given an inventory it returns an ordered plan and never touches the network, so `migrate plan` is safe against production and is fully testable. Phase order is the safety property and is enforced rather than documented -- check, bulk, freeze, delta, cutover, enable, verify. Scheduled jobs are stopped on the source before they are started on the target, because a cron firing on both sides is how a migration sends every customer a duplicate email. Dropping the source is not a phase; it is a decision a person makes days later. Credentials never reach a plan file: connections are a describe()/reveal() pair, and a test asserts a rendered plan contains no password. The Postgres engine dumps custom-format, no-owner, no-acl. It does NOT pass --clean: that would make a re-run idempotent, at the price of being one typo away from dropping a production database, so a non-empty target is refused instead. Credentials go through libpq's own environment variables rather than argv -- ps is world-readable and a dump runs for hours. That is also the only thing that works, since exec runs without a shell and nothing would expand a $VAR written into an argument. The delta sync is honest about its limits: for tables carrying a timestamp column it copies rows newer than the dump, which covers the append-heavy tables that keep being written during a long dump. It does not cover updates or deletes, and the planner says so rather than implying a completeness it does not have. The column name is validated as an identifier before it reaches psql -c. 48 tests, no network, no database. Co-Authored-By: Claude Opus 5 (1M context) --- packages/migrate/package.json | 37 ++ packages/migrate/src/engines/postgres.test.ts | 213 ++++++++ packages/migrate/src/engines/postgres.ts | 337 +++++++++++++ packages/migrate/src/plan.test.ts | 285 +++++++++++ packages/migrate/src/plan.ts | 474 ++++++++++++++++++ packages/migrate/src/types.ts | 267 ++++++++++ packages/migrate/tsconfig.json | 6 + 7 files changed, 1619 insertions(+) create mode 100644 packages/migrate/package.json create mode 100644 packages/migrate/src/engines/postgres.test.ts create mode 100644 packages/migrate/src/engines/postgres.ts create mode 100644 packages/migrate/src/plan.test.ts create mode 100644 packages/migrate/src/plan.ts create mode 100644 packages/migrate/src/types.ts create mode 100644 packages/migrate/tsconfig.json diff --git a/packages/migrate/package.json b/packages/migrate/package.json new file mode 100644 index 00000000..52927993 --- /dev/null +++ b/packages/migrate/package.json @@ -0,0 +1,37 @@ +{ + "name": "@profullstack/sh1pt-migrate", + "version": "0.1.15", + "type": "module", + "main": "./src/index.ts", + "scripts": { + "build": "tsc -p tsconfig.json", + "typecheck": "tsc -p tsconfig.json --noEmit", + "prepublishOnly": "pnpm build" + }, + "dependencies": { + "@profullstack/sh1pt-core": "workspace:*" + }, + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/profullstack/sh1pt.git", + "directory": "packages/migrate" + }, + "homepage": "https://sh1pt.com", + "bugs": "https://github.com/profullstack/sh1pt/issues", + "files": [ + "dist" + ], + "publishConfig": { + "access": "public", + "main": "./dist/index.js", + "types": "./dist/index.d.ts", + "exports": { + ".": { + "types": "./dist/index.d.ts", + "import": "./dist/index.js", + "default": "./dist/index.js" + } + } + } +} diff --git a/packages/migrate/src/engines/postgres.test.ts b/packages/migrate/src/engines/postgres.test.ts new file mode 100644 index 00000000..d6e46f66 --- /dev/null +++ b/packages/migrate/src/engines/postgres.test.ts @@ -0,0 +1,213 @@ +import { describe, expect, it } from 'vitest'; +import { deltaColumn, pgEnv, postgresEngine } from './postgres.js'; +import type { Artifact, EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { secret } from '../types.js'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +/** + * A context that records commands instead of running them. This is the whole + * reason `exec` is injected: the engine is fully exercised with no Postgres. + */ +function ctx( + responses: Array> = [], + over: Partial = {}, +): EngineContext & { calls: Call[]; recorded: Artifact[] } { + const calls: Call[] = []; + const recorded: Artifact[] = []; + let i = 0; + return { + calls, + recorded, + dryRun: false, + log: () => {}, + staging: { + dir: '/staging', + record: async (a) => { + recorded.push(a); + }, + existing: async () => [], + }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; +} + +const db = (over: Partial = {}): Resource => ({ + kind: 'postgres', + id: 'db', + name: 'app', + connection: { url: secret('postgres://u:p%40ss@db.example.com:5432/appdb') }, + ...over, +}); + +describe('pgEnv', () => { + it('splits the DSN into libpq variables so nothing lands in argv', () => { + const env = pgEnv(db()); + expect(env.PGHOST).toBe('db.example.com'); + expect(env.PGPORT).toBe('5432'); + expect(env.PGUSER).toBe('u'); + expect(env.PGDATABASE).toBe('appdb'); + }); + + it('url-decodes a password containing reserved characters', () => { + expect(pgEnv(db()).PGPASSWORD).toBe('p@ss'); + }); + + it('requires TLS by default rather than letting libpq fall back to plaintext', () => { + expect(pgEnv(db()).PGSSLMODE).toBe('require'); + }); + + it('honours an explicit sslmode in the DSN', () => { + const r = db({ connection: { url: secret('postgres://u:p@h/d?sslmode=disable') } }); + expect(pgEnv(r).PGSSLMODE).toBe('disable'); + }); + + it('refuses a resource with no connection url', () => { + expect(() => pgEnv(db({ connection: {} }))).toThrow(/no connection.url/); + }); + + it('refuses a connection url that is not a URL', () => { + expect(() => pgEnv(db({ connection: { url: secret('not a url') } }))).toThrow(/not a URL/); + }); +}); + +describe('export', () => { + it('dumps in the custom format with no owner or acl', async () => { + const c = ctx(); + const artifacts = await postgresEngine.export(c, db()); + + const call = c.calls[0]!; + expect(call.cmd).toBe('pg_dump'); + expect(call.args).toContain('--format=custom'); + expect(call.args).toContain('--no-owner'); + expect(call.args).toContain('--no-acl'); + expect(artifacts[0]?.path).toBe('db/dump.pgc'); + expect(c.recorded).toHaveLength(1); + }); + + it('never puts the password in argv', async () => { + const c = ctx(); + await postgresEngine.export(c, db()); + expect(c.calls[0]!.args.join(' ')).not.toContain('p@ss'); + expect(c.calls[0]!.opts?.env?.PGPASSWORD).toBe('p@ss'); + }); + + it('runs nothing on a dry run but still reports the artifact', async () => { + const c = ctx([], { dryRun: true }); + const artifacts = await postgresEngine.export(c, db()); + expect(c.calls).toHaveLength(0); + expect(artifacts[0]?.path).toBe('db/dump.pgc'); + }); +}); + +describe('import', () => { + const dump: Artifact = { resourceId: 'db', kind: 'postgres', path: 'db/dump.pgc' }; + + it('restores into an empty database', async () => { + const c = ctx([{ stdout: '0' }, { code: 0 }]); + await postgresEngine.import(c, db(), [dump]); + expect(c.calls[1]!.cmd).toBe('pg_restore'); + expect(c.calls[1]!.args).toContain('--no-owner'); + }); + + it('refuses to restore over a database that already has tables', async () => { + const c = ctx([{ stdout: '42' }]); + await expect(postgresEngine.import(c, db(), [dump])).rejects.toThrow(/already has 42 table/); + // Nothing was restored. + expect(c.calls.some((x) => x.cmd === 'pg_restore')).toBe(false); + }); + + it('never passes --clean, which would drop an existing database', async () => { + const c = ctx([{ stdout: '0' }, { code: 0 }]); + await postgresEngine.import(c, db(), [dump]); + expect(c.calls[1]!.args).not.toContain('--clean'); + }); + + it('throws when the restore reports errors that leave the database incomplete', async () => { + const c = ctx([ + { stdout: '0' }, + { code: 1, stderr: 'pg_restore: error: could not execute query: extension "pg_cron" does not exist' }, + ]); + await expect(postgresEngine.import(c, db(), [dump])).rejects.toThrow(/incomplete/); + }); + + it('tolerates a non-zero exit whose errors are benign', async () => { + const c = ctx([{ stdout: '0' }, { code: 1, stderr: 'pg_restore: warning: something cosmetic' }]); + await expect(postgresEngine.import(c, db(), [dump])).resolves.toBeUndefined(); + }); + + it('fails when no dump was staged', async () => { + const c = ctx(); + await expect(postgresEngine.import(c, db(), [])).rejects.toThrow(/no postgres dump staged/); + }); +}); + +describe('deltaColumn', () => { + it('defaults to created_at', () => { + expect(deltaColumn(db())).toBe('created_at'); + }); + + it('accepts a plain identifier', () => { + expect(deltaColumn(db({ metadata: { deltaColumns: 'updated_at' } }))).toBe('updated_at'); + }); + + it('refuses anything that could smuggle a second statement into psql', () => { + for (const bad of ["created_at'; drop table users; --", 'a b', 'a-b', '1col', '']) { + expect(() => deltaColumn(db({ metadata: { deltaColumns: bad } }))).toThrow(/valid column name/); + } + }); +}); + +describe('delta', () => { + it('only copies from tables that actually have the timestamp column', async () => { + const c = ctx([{ stdout: 'public.events\npublic.impressions\n' }, {}, {}]); + const out = await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + expect(out).toHaveLength(2); + expect(out[0]?.metadata?.table).toBe('public.events'); + }); + + it('reports nothing to do when no table has the column', async () => { + const c = ctx([{ stdout: '' }]); + const out = await postgresEngine.delta!(c, db(), new Date()); + expect(out).toEqual([]); + }); +}); + +describe('verify', () => { + const counts = (rows: string) => ({ stdout: rows }); + + it('passes when every table matches', async () => { + const c = ctx([counts('public.users|113\npublic.posts|48\n'), counts('public.users|113\npublic.posts|48\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + expect(res.checks).toContain('public.users: 113 → 113'); + }); + + it('fails when the target has fewer rows', async () => { + const c = ctx([counts('public.users|113\n'), counts('public.users|9\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('113 rows on the source, 9 on the target'); + }); + + it('flags a table missing entirely from the target', async () => { + const c = ctx([counts('public.users|1\npublic.gone|5\n'), counts('public.users|1\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.problems.some((p) => p.includes('public.gone'))).toBe(true); + }); + + it('accepts a target that has gained rows, since it is live and the source is frozen', async () => { + const c = ctx([counts('public.events|100\n'), counts('public.events|140\n')]); + const res = await postgresEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + }); +}); diff --git a/packages/migrate/src/engines/postgres.ts b/packages/migrate/src/engines/postgres.ts new file mode 100644 index 00000000..0b86aebd --- /dev/null +++ b/packages/migrate/src/engines/postgres.ts @@ -0,0 +1,337 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a Postgres database. + * + * Supabase, Neon, Railway, RDS and a Postgres in a container on a dedicated + * box are all this engine. That is the point of splitting platforms from + * engines: the vendor decides where the connection string comes from, and this + * decides what to do with it. + * + * ## Why the custom format, and why not --clean + * + * `pg_dump -Fc` (custom) rather than plain SQL, because it is the only format + * `pg_restore` can parallelise and selectively restore, and because a 4.7 GB + * plain-text dump is unusable when one table fails. `--no-owner` and + * `--no-acl`, because the roles on a managed provider do not exist on the + * destination and a dump that tries to `ALTER OWNER TO supabase_admin` fails + * on every object. + * + * `--clean` is deliberately NOT passed. It would make a re-run idempotent, + * which sounds desirable, but it means a restore aimed at the wrong database + * silently drops what is there. A migration tool should not be one typo away + * from deleting production; an existing non-empty target is refused instead. + * + * ## The extension problem + * + * A dump records `CREATE EXTENSION pg_cron`, but an extension is a server + * feature, not data: if the destination image does not ship it, the restore + * fails partway, having already written some tables. Extensions are therefore + * read during inventory and surfaced as quirks so the planner warns before + * anything runs. + */ + +const DUMP = 'dump.pgc'; + +/** Read the DSN off a resource, failing loudly rather than connecting to a default. */ +function dsn(r: Resource): string { + const url = r.connection.url; + if (!url) throw new Error(`postgres resource '${r.name}' has no connection.url`); + return url.reveal(); +} + +/** + * Postgres credentials go in the environment, never in argv. + * + * `ps` is world-readable on a normal box, so a password spliced into a command + * line is visible to every other user for as long as the dump runs — which for + * a migration is hours. + * + * These are libpq's own variables, which `pg_dump`, `pg_restore` and `psql` + * all read directly. That matters more than it looks: `exec` runs commands + * without a shell, so there is nothing to expand a `$VAR` written into an + * argument. Passing the connection through the environment is not merely + * tidier here, it is the only thing that works. + */ +export function pgEnv(r: Resource): Record { + const raw = dsn(r); + let u: URL; + try { + u = new URL(raw); + } catch { + throw new Error(`postgres resource '${r.name}' has a connection.url that is not a URL`); + } + + const env: Record = {}; + if (u.hostname) env.PGHOST = decodeURIComponent(u.hostname); + if (u.port) env.PGPORT = u.port; + if (u.username) env.PGUSER = decodeURIComponent(u.username); + if (u.password) env.PGPASSWORD = decodeURIComponent(u.password); + const database = u.pathname.replace(/^\//, ''); + if (database) env.PGDATABASE = decodeURIComponent(database); + // Managed providers almost universally require TLS, and libpq's default of + // `prefer` silently falls back to plaintext where it is not enforced. + env.PGSSLMODE = u.searchParams.get('sslmode') ?? 'require'; + return env; +} + +/** + * The timestamp column the delta sync keys on. + * + * This ends up interpolated into SQL, and it comes from a resource's metadata, + * which comes from a config file a person edits — so it is checked against a + * plain identifier rather than trusted. `psql -c` would happily run a second + * statement smuggled in here, against the production database the migration is + * reading from. + */ +export function deltaColumn(r: Resource): string { + const raw = (r.metadata?.deltaColumns as string | undefined) ?? 'created_at'; + if (!/^[a-z_][a-z0-9_]*$/i.test(raw)) { + throw new Error( + `'${raw}' is not a valid column name for the delta sync of '${r.name}'. Use a plain identifier.`, + ); + } + return raw; +} + +export const postgresEngine: Engine = { + kind: 'postgres', + requires: ['pg_dump', 'pg_restore', 'psql'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`pg_dump ${from.name} → ${path}`); + + if (ctx.dryRun) { + return [{ resourceId: from.id, kind: 'postgres', path }]; + } + + await ctx.exec( + 'pg_dump', + [ + '--format=custom', + '--no-owner', + '--no-acl', + // Compresses inside the custom format; the wire is usually the + // bottleneck on a cloud-to-anywhere move, not the CPU. + '--compress=6', + '--file', + `${ctx.staging.dir}/${path}`, + ], + { env: pgEnv(from), timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + const artifact: Artifact = { resourceId: from.id, kind: 'postgres', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'postgres'); + if (!dump) throw new Error(`no postgres dump staged for '${to.name}'`); + + if (ctx.dryRun) { + ctx.log(`would pg_restore into ${to.name}`); + return; + } + + // Refuse to write into a database that already has user tables. Without + // this, re-running a half-finished migration against the wrong target is + // indistinguishable from the intended one until the duplicate-key errors + // start, by which point the restore is half applied. + const existing = await ctx.exec( + 'psql', + [ + '--tuples-only', + '--no-align', + '--command', + "select count(*) from information_schema.tables where table_schema not in ('pg_catalog','information_schema')", + ], + { env: pgEnv(to), check: false }, + ); + const tableCount = Number.parseInt(existing.stdout.trim(), 10); + if (Number.isFinite(tableCount) && tableCount > 0) { + throw new Error( + `target database '${to.name}' already has ${tableCount} table(s). Refusing to restore over it — drop and recreate the database, or point at an empty one.`, + ); + } + + ctx.log(`pg_restore → ${to.name}`); + const res = await ctx.exec( + 'pg_restore', + [ + '--no-owner', + '--no-acl', + // Parallel restore. Indexes dominate a large restore and they are + // perfectly parallel. + '--jobs=4', + // Keep going so ONE failed object (a missing extension, a role that + // does not exist) does not abandon a multi-hour restore. Errors are + // counted and surfaced below rather than swallowed. + '--exit-on-error=false', + `${ctx.staging.dir}/${dump.path}`, + ], + { env: pgEnv(to), check: false, timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + if (res.code !== 0) { + const errors = res.stderr + .split('\n') + .filter((l) => l.includes('error:')) + .slice(0, 10); + ctx.log( + `pg_restore finished with errors (${errors.length} shown):\n${errors.join('\n')}`, + 'warn', + ); + // A restore that produced errors is not automatically a failed + // migration — a missing extension on a replica, for instance — but it is + // never something to pass over silently. + if (errors.some((e) => /could not|does not exist|permission denied/i.test(e))) { + throw new Error( + `pg_restore into '${to.name}' reported errors that will leave the database incomplete. First: ${errors[0] ?? 'unknown'}`, + ); + } + } + }, + + /** + * Re-copy rows written during the bulk dump. + * + * There is no general delta for Postgres without logical replication, and + * standing up a replication slot against a managed provider mid-migration is + * its own project. What works in practice, and what the crawlproof migration + * actually did, is narrower: for tables that carry a timestamp column, copy + * the rows newer than the dump. That covers append-heavy tables (events, + * impressions, logs) which are exactly the ones that keep being written + * while a long dump runs. + * + * It does NOT cover updates to old rows or deletes. The planner says so, and + * the honest use of this is: freeze writes to anything that mutates history, + * and let the append-only tables catch up. + */ + async delta(ctx: EngineContext, from: Resource, since: Date): Promise { + const columns = deltaColumn(from); + const path = `${from.id}/delta-${since.toISOString().replace(/[:.]/g, '')}.sql`; + ctx.log(`delta for ${from.name} on ${columns} since ${since.toISOString()}`); + + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'postgres', path }]; + + // Find tables that actually have the timestamp column rather than assuming + // a schema. A table without one cannot be delta'd and is reported. + const found = await ctx.exec( + 'psql', + [ + '--tuples-only', + '--no-align', + '--command', + `select table_schema||'.'||table_name from information_schema.columns + where column_name = '${columns}' + and table_schema not in ('pg_catalog','information_schema') + order by 1`, + ], + { env: pgEnv(from) }, + ); + + const tables = found.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + if (!tables.length) { + ctx.log(`no table has a '${columns}' column; nothing can be delta-synced`, 'warn'); + return []; + } + + const artifacts: Artifact[] = []; + for (const table of tables) { + const out = `${from.id}/delta-${table.replace(/[^\w]/g, '_')}.csv`; + await ctx.exec( + 'psql', + [ + '--command', + `\\copy (select * from ${table} where ${columns} > '${since.toISOString()}') to '${ctx.staging.dir}/${out}' with csv header`, + ], + { env: pgEnv(from) }, + ); + const artifact: Artifact = { + resourceId: from.id, + kind: 'postgres', + path: out, + metadata: { table, mode: 'delta-csv' }, + }; + await ctx.staging.record(artifact); + artifacts.push(artifact); + } + return artifacts; + }, + + /** + * Compare row counts per table. + * + * Not a checksum — comparing 4.7 GB twice over the wire costs as much as the + * migration did. Row counts per table catch the failures that actually + * happen: a table that restored empty, a restore that stopped partway. + */ + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const countsQuery = `select table_schema||'.'||table_name as t, + (xpath('/row/c/text()', query_to_xml(format('select count(*) as c from %I.%I', table_schema, table_name), false, true, '')))[1]::text::bigint as n + from information_schema.tables + where table_type = 'BASE TABLE' and table_schema not in ('pg_catalog','information_schema') + order by 1`; + + const read = async (r: Resource) => { + const res = await ctx.exec( + 'psql', + ['--tuples-only', '--no-align', '--field-separator=|', '--command', countsQuery], + { env: pgEnv(r) }, + ); + const map = new Map(); + for (const line of res.stdout.split('\n')) { + const [t, n] = line.split('|'); + if (t && n !== undefined) map.set(t.trim(), Number.parseInt(n, 10)); + } + return map; + }; + + const [a, b] = await Promise.all([read(from), read(to)]); + const checks: string[] = []; + const problems: string[] = []; + + for (const [table, sourceCount] of a) { + const targetCount = b.get(table); + if (targetCount === undefined) { + problems.push(`${table}: missing on the target`); + continue; + } + checks.push(`${table}: ${sourceCount} → ${targetCount}`); + // The target may legitimately have MORE rows once it is live and the + // source is frozen; fewer is always wrong. + if (targetCount < sourceCount) { + problems.push(`${table}: ${sourceCount} rows on the source, ${targetCount} on the target`); + } + } + for (const table of b.keys()) { + if (!a.has(table)) checks.push(`${table}: target only`); + } + + return { ok: problems.length === 0, checks, problems }; + }, +}; + +/** + * The extensions a database uses, for the inventory's quirks. + * + * Exported so a platform can call it while building its inventory: the + * platform knows the DSN, this knows the question worth asking. + */ +export async function postgresExtensions( + ctx: EngineContext, + r: Resource, +): Promise { + const res = await ctx.exec( + 'psql', + ['--tuples-only', '--no-align', '--command', + "select extname from pg_extension where extname not in ('plpgsql') order by 1"], + { env: pgEnv(r), check: false }, + ); + if (res.code !== 0) return []; + return res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); +} diff --git a/packages/migrate/src/plan.test.ts b/packages/migrate/src/plan.test.ts new file mode 100644 index 00000000..1acd5cca --- /dev/null +++ b/packages/migrate/src/plan.test.ts @@ -0,0 +1,285 @@ +import { describe, expect, it } from 'vitest'; +import { PHASES, type Phase, type Step, humanBytes, orderSteps, planMigration, renderPlan } from './plan.js'; +import { type Engine, type Inventory, type Platform, type Resource, plain, secret } from './types.js'; + +const engine = (kind: Engine['kind'], over: Partial = {}): Engine => ({ + kind, + requires: [], + export: async () => [], + import: async () => {}, + ...over, +}); + +/** A full engine set: everything can move, delta and verify included. */ +function engines(over: Partial> = {}): Map { + const base: Array<[Resource['kind'], Engine]> = [ + ['postgres', engine('postgres', { requires: ['pg_dump'], delta: async () => [], verify: async () => ({ ok: true, checks: [], problems: [] }) })], + ['object-storage', engine('object-storage', { delta: async () => [] })], + ['redis', engine('redis')], + ['cron', engine('cron')], + ]; + const m = new Map(base); + for (const [k, v] of Object.entries(over)) m.set(k as Resource['kind'], v as Engine); + return m; +} + +const resource = (over: Partial & Pick): Resource => ({ + connection: { url: secret('postgres://u:p@h/db') }, + ...over, +}); + +const inventory = (resources: Resource[], over: Partial = {}): Inventory => ({ + platform: 'supabase', + scope: 'project abc123', + resources, + ...over, +}); + +const target = (over: Partial = {}): Platform => ({ + id: 'ssh', + label: 'dedicated box', + role: 'both', + supports: ['postgres', 'object-storage', 'redis', 'files', 'cron'], + inventory: async () => inventory([]), + ...over, +}); + +describe('planMigration', () => { + it('pairs every supported resource and estimates the total', () => { + const plan = planMigration( + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app', sizeBytes: 4_700_000_000 }), + resource({ kind: 'object-storage', id: 'b1', name: 'public', itemCount: 8410 }), + ]), + target(), + { engines: engines() }, + ); + + expect(plan.ok).toBe(true); + expect(plan.moving.map((r) => r.id)).toEqual(['db', 'b1']); + expect(plan.estimateBytes).toBe(4_700_000_000); + }); + + it('blocks when the target cannot hold a resource kind', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target({ supports: ['files'] }), + { engines: engines() }, + ); + + expect(plan.ok).toBe(false); + expect(plan.skipped[0]?.reason).toContain('does not support postgres'); + expect(plan.risks.some((r) => r.severity === 'blocker')).toBe(true); + }); + + it('refuses a target that cannot be written to at all', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target({ role: 'source' }), + { engines: engines() }, + ); + expect(plan.ok).toBe(false); + expect(plan.risks.some((r) => /offers no import path/.test(r.message))).toBe(true); + }); + + it('blocks when a required binary is missing, before anything runs', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target(), + { engines: engines(), availableBinaries: new Set() }, + ); + expect(plan.ok).toBe(false); + expect(plan.risks.some((r) => r.message.includes('pg_dump'))).toBe(true); + }); + + it('accepts the plan when the binary is present', () => { + const plan = planMigration( + inventory([resource({ kind: 'postgres', id: 'db', name: 'app' })]), + target(), + { engines: engines(), availableBinaries: new Set(['pg_dump']) }, + ); + expect(plan.ok).toBe(true); + }); + + it('warns when a resource has no delta sync, because writes during the copy are lost', () => { + const plan = planMigration( + inventory([resource({ kind: 'redis', id: 'r', name: 'cache' })]), + target(), + { engines: engines() }, + ); + expect(plan.risks.some((r) => /no delta sync/.test(r.message))).toBe(true); + }); + + it('surfaces a resource quirk as a warning rather than discovering it mid-restore', () => { + const plan = planMigration( + inventory([ + resource({ + kind: 'postgres', + id: 'db', + name: 'app', + quirks: ['uses the pg_cron extension, which the target must also have'], + }), + ]), + target(), + { engines: engines() }, + ); + expect(plan.risks.some((r) => /pg_cron/.test(r.message))).toBe(true); + }); + + it('honours --only and --exclude', () => { + const rs = [ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'redis', id: 'r', name: 'cache' }), + ]; + const onlyDb = planMigration(inventory(rs), target(), { engines: engines(), only: ['postgres'] }); + expect(onlyDb.moving.map((r) => r.id)).toEqual(['db']); + + const noRedis = planMigration(inventory(rs), target(), { engines: engines(), exclude: ['redis'] }); + expect(noRedis.moving.map((r) => r.id)).toEqual(['db']); + }); +}); + +describe('the absolute-URL trap', () => { + const both = () => + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'object-storage', id: 'b1', name: 'public' }), + ]); + + it('warns when storage and a database move together with no rewrite host', () => { + const plan = planMigration(both(), target(), { engines: engines() }); + expect(plan.risks.some((r) => /absolute URLs/i.test(r.message))).toBe(true); + expect(plan.steps.some((s) => s.id === 'rewrite:urls')).toBe(true); + }); + + it('does not warn once a rewrite host is given', () => { + const plan = planMigration(both(), target(), { + engines: engines(), + rewriteHosts: ['abc.supabase.co'], + }); + expect(plan.risks.some((r) => /no --rewrite-host/i.test(r.message))).toBe(false); + const step = plan.steps.find((s) => s.id === 'rewrite:urls'); + expect(step?.title).toContain('abc.supabase.co'); + }); + + it('adds no rewrite step when only storage moves', () => { + const plan = planMigration( + inventory([resource({ kind: 'object-storage', id: 'b1', name: 'public' })]), + target(), + { engines: engines() }, + ); + expect(plan.steps.some((s) => s.id === 'rewrite:urls')).toBe(false); + }); +}); + +describe('phase ordering is the safety property', () => { + const planWithCron = () => + planMigration( + inventory([ + resource({ kind: 'postgres', id: 'db', name: 'app' }), + resource({ kind: 'cron', id: 'jobs', name: 'pg_cron' }), + ]), + target(), + { engines: engines() }, + ); + + it('never schedules a phase before an earlier one', () => { + const steps = planWithCron().steps; + const idx = steps.map((s) => PHASES.indexOf(s.phase)); + expect(idx).toEqual([...idx].sort((a, b) => a - b)); + }); + + it('stops scheduled jobs on the source before starting them on the target', () => { + const steps = planWithCron().steps; + const disable = steps.findIndex((s) => s.id === 'disable:jobs'); + const enable = steps.findIndex((s) => s.id === 'enable:jobs'); + expect(disable).toBeGreaterThanOrEqual(0); + expect(enable).toBeGreaterThan(disable); + }); + + it('cuts DNS over only after the freeze and the delta', () => { + const steps = planWithCron().steps; + const at = (id: string) => steps.findIndex((s) => s.id === id); + expect(at('freeze:all')).toBeLessThan(at('cutover:dns')); + expect(at('delta:db')).toBeLessThan(at('cutover:dns')); + }); + + it('marks the steps that touch the live source', () => { + const freeze = planWithCron().steps.find((s) => s.id === 'freeze:all'); + expect(freeze?.touchesSource).toBe(true); + }); +}); + +describe('orderSteps', () => { + const step = (id: string, phase: Phase, after: string[] = []): Step => ({ + id, + phase, + title: id, + after, + }); + + it('respects dependencies inside a phase', () => { + const out = orderSteps([ + step('b', 'bulk', ['a']), + step('a', 'bulk'), + step('c', 'bulk', ['b']), + ]); + expect(out.map((s) => s.id)).toEqual(['a', 'b', 'c']); + }); + + it('throws on a cycle rather than guessing an order', () => { + expect(() => orderSteps([step('a', 'bulk', ['b']), step('b', 'bulk', ['a'])])).toThrow(/cycle/); + }); + + it('throws when a step depends on one that runs in a later phase', () => { + expect(() => orderSteps([step('a', 'bulk', ['z']), step('z', 'verify')])).toThrow(/runs later/); + }); + + it('ignores a dependency on a step that does not exist', () => { + expect(() => orderSteps([step('a', 'bulk', ['nope'])])).not.toThrow(); + }); +}); + +describe('secrets never reach the plan', () => { + it('describes a connection without revealing it', () => { + const s = secret('postgres://user:hunter2@db.example.com:5432/app'); + expect(s.describe()).not.toContain('hunter2'); + expect(s.describe()).toContain('db.example.com'); + expect(s.reveal()).toContain('hunter2'); + }); + + it('masks a bare token', () => { + expect(secret('abcdef123456').describe()).toBe('abcd***'); + expect(secret('ab').describe()).toBe('***'); + }); + + it('leaves a non-secret alone', () => { + expect(plain('my-bucket').describe()).toBe('my-bucket'); + }); + + it('renders a plan with no credential in it', () => { + const text = renderPlan( + planMigration( + inventory([ + resource({ + kind: 'postgres', + id: 'db', + name: 'app', + connection: { url: secret('postgres://user:hunter2@h/db') }, + }), + ]), + target(), + { engines: engines() }, + ), + ); + expect(text).not.toContain('hunter2'); + }); +}); + +describe('humanBytes', () => { + it('scales through the units', () => { + expect(humanBytes(512)).toBe('512 B'); + expect(humanBytes(4_700_000_000)).toBe('4.4 GB'); + expect(humanBytes(0)).toBe('0 B'); + }); +}); diff --git a/packages/migrate/src/plan.ts b/packages/migrate/src/plan.ts new file mode 100644 index 00000000..963d15c9 --- /dev/null +++ b/packages/migrate/src/plan.ts @@ -0,0 +1,474 @@ +import type { Engine, Inventory, Platform, Resource, ResourceKind } from './types.js'; + +/** + * Turning an inventory into an ordered, reviewable plan. + * + * The plan is the product. A migration that goes wrong usually went wrong + * before anything ran — a resource nobody knew was there, a cutover step in + * the wrong order, an assumption that the destination supported something it + * did not. All of that is knowable up front, from a read-only inventory, and + * this module is where it is worked out so a person can read it and disagree + * before any bytes move. + * + * Nothing here touches the network or mutates anything. Given the same + * inventory it produces the same plan, which is what makes it testable. + */ + +/** A single unit of work in the plan. */ +export interface Step { + id: string; + phase: Phase; + /** One line, imperative: "dump postgres 'app' (4.7 GB)". */ + title: string; + resourceId?: string; + kind?: ResourceKind; + /** Steps that must complete before this one. */ + after: string[]; + /** True when this step changes the SOURCE, which is what makes it scary. */ + touchesSource?: boolean; + /** Set when the step cannot be undone by re-running the migration. */ + irreversible?: boolean; + estimateBytes?: number; +} + +/** + * The phases, in the only order that is safe. + * + * The ordering is the part people get wrong, and it is not arbitrary: + * + * - `check` first so a missing credential costs a second, not three hours. + * - `bulk` runs while the source is still live and serving. It is the long + * part and it is safe to repeat. + * - `freeze` is the start of downtime: stop the things that write. Scheduled + * jobs especially — a cron firing on both sides is how a migration sends + * every customer a duplicate email. + * - `delta` copies only what changed during `bulk`. Short, because `bulk` + * already moved the bulk. + * - `cutover` points the world at the new place. + * - `enable` starts the writers again, on the target only, and never before + * `freeze` has stopped them on the source. + * - `verify` proves it worked while the old system still exists. + * + * Deleting the source is not a phase. It is a separate decision a person makes + * days later, and this tool does not offer it. + */ +export type Phase = 'check' | 'bulk' | 'freeze' | 'delta' | 'cutover' | 'enable' | 'verify'; + +export const PHASES: readonly Phase[] = [ + 'check', + 'bulk', + 'freeze', + 'delta', + 'cutover', + 'enable', + 'verify', +] as const; + +export interface Risk { + severity: 'blocker' | 'warning' | 'note'; + /** What is wrong, in a sentence a person can act on. */ + message: string; + resourceId?: string; +} + +export interface MigrationPlan { + source: string; + target: string; + scope: string; + steps: Step[]; + risks: Risk[]; + /** Resources that will move, paired source → target kind. */ + moving: Resource[]; + /** Resources that will NOT move, and why. */ + skipped: Array<{ resource: Resource; reason: string }>; + estimateBytes: number; + /** False when any risk is a blocker. `apply` refuses a plan that is not ok. */ + ok: boolean; +} + +export interface PlanOptions { + /** Only migrate these kinds. Empty means everything the target supports. */ + only?: ResourceKind[]; + /** Never migrate these kinds. */ + exclude?: ResourceKind[]; + /** + * Hosts whose absolute URLs are expected to appear in the data and must be + * rewritten, e.g. `ywcizjsgrcmhgyplldac.supabase.co`. See `transforms.ts`. + */ + rewriteHosts?: string[]; + /** Engines available in this build, by kind. */ + engines: Map; + /** Binaries present on this machine, for the `requires` check. */ + availableBinaries?: Set; +} + +/** Kinds whose contents are routinely referenced by absolute URL from a database. */ +const URL_BEARING: ReadonlySet = new Set(['object-storage']); + +/** + * Kinds that write on a schedule and so must be stopped before the delta, or + * they run in two places at once. + */ +const SCHEDULED: ReadonlySet = new Set(['cron']); + +function stepId(prefix: string, resourceId: string): string { + return `${prefix}:${resourceId}`; +} + +/** + * Work out what would happen, without doing any of it. + * + * Read-only in the strongest sense: it takes an inventory that has already + * been gathered and a description of the target, and returns a plan. It does + * not call the network, so it is fully testable and so `migrate plan` can be + * run against production with no anxiety. + */ +export function planMigration( + source: Inventory, + target: Platform, + opts: PlanOptions, +): MigrationPlan { + const risks: Risk[] = []; + const steps: Step[] = []; + const moving: Resource[] = []; + const skipped: Array<{ resource: Resource; reason: string }> = []; + + const only = new Set(opts.only ?? []); + const exclude = new Set(opts.exclude ?? []); + const targetKinds = new Set(target.supports); + + if (target.role === 'source') { + risks.push({ + severity: 'blocker', + message: `${target.label} cannot be a migration target: it offers no import path.`, + }); + } + + for (const resource of source.resources) { + if (only.size && !only.has(resource.kind)) { + skipped.push({ resource, reason: `not in --only` }); + continue; + } + if (exclude.has(resource.kind)) { + skipped.push({ resource, reason: `excluded by --exclude` }); + continue; + } + if (!targetKinds.has(resource.kind)) { + skipped.push({ + resource, + reason: `${target.label} does not support ${resource.kind}`, + }); + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `${resource.name} is ${resource.kind}, which ${target.label} cannot hold. Exclude it with --exclude ${resource.kind}, or pick a different target.`, + }); + continue; + } + + const engine = opts.engines.get(resource.kind); + if (!engine) { + skipped.push({ resource, reason: `no engine for ${resource.kind}` }); + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `Nothing in this build can move a ${resource.kind}.`, + }); + continue; + } + + // A missing pg_dump is a blocker, and finding out now beats finding out + // after the freeze has started. + if (opts.availableBinaries) { + const missing = engine.requires.filter((b) => !opts.availableBinaries!.has(b)); + if (missing.length) { + risks.push({ + severity: 'blocker', + resourceId: resource.id, + message: `${resource.kind} needs ${missing.join(', ')} on this machine and ${missing.length > 1 ? 'they are' : 'it is'} not installed.`, + }); + } + } + + moving.push(resource); + + steps.push({ + id: stepId('export', resource.id), + phase: 'bulk', + title: `export ${resource.kind} ${resource.name}${sizeSuffix(resource)}`, + resourceId: resource.id, + kind: resource.kind, + after: ['check:all'], + estimateBytes: resource.sizeBytes, + }); + + steps.push({ + id: stepId('import', resource.id), + phase: 'bulk', + title: `import ${resource.kind} ${resource.name} into ${target.label}`, + resourceId: resource.id, + kind: resource.kind, + after: [stepId('export', resource.id)], + estimateBytes: resource.sizeBytes, + }); + + if (engine.delta) { + steps.push({ + id: stepId('delta', resource.id), + phase: 'delta', + title: `sync ${resource.name} changes made during the bulk copy`, + resourceId: resource.id, + kind: resource.kind, + after: ['freeze:all', stepId('import', resource.id)], + }); + } else { + risks.push({ + severity: 'warning', + resourceId: resource.id, + message: `${resource.name} (${resource.kind}) has no delta sync, so anything written to it during the bulk copy is lost. Stop writes before the copy, or accept the gap.`, + }); + } + + if (engine.verify) { + steps.push({ + id: stepId('verify', resource.id), + phase: 'verify', + title: `verify ${resource.name} matches the source`, + resourceId: resource.id, + kind: resource.kind, + after: ['cutover:dns'], + }); + } + + for (const quirk of resource.quirks ?? []) { + risks.push({ + severity: 'warning', + resourceId: resource.id, + message: `${resource.name}: ${quirk}`, + }); + } + + if (SCHEDULED.has(resource.kind)) { + steps.push({ + id: stepId('disable', resource.id), + phase: 'freeze', + title: `disable scheduled jobs on the SOURCE (${resource.name})`, + resourceId: resource.id, + kind: resource.kind, + after: [], + touchesSource: true, + }); + steps.push({ + id: stepId('enable', resource.id), + phase: 'enable', + title: `enable scheduled jobs on the TARGET (${resource.name})`, + resourceId: resource.id, + kind: resource.kind, + // Never before the source's are off. Two schedulers on one dataset is + // duplicate outbound email and duplicate published posts. + after: [stepId('disable', resource.id), 'cutover:dns'], + }); + } + } + + // The absolute-URL trap. Object storage moves to a new host, but rows that + // stored `https:///...` keep working until the old account is + // closed and then 404 forever. It is invisible at cutover, which is what + // makes it dangerous. + const storage = moving.filter((r) => URL_BEARING.has(r.kind)); + const databases = moving.filter((r) => r.kind === 'postgres' || r.kind === 'mysql'); + if (storage.length && databases.length) { + const hosts = opts.rewriteHosts ?? []; + steps.push({ + id: 'rewrite:urls', + phase: 'delta', + title: hosts.length + ? `rewrite absolute URLs (${hosts.join(', ')}) to the new host` + : `scan for absolute URLs pointing at the old storage host`, + after: databases.map((d) => stepId('import', d.id)), + irreversible: false, + }); + if (!hosts.length) { + risks.push({ + severity: 'warning', + message: + 'Storage and a database are both moving but no --rewrite-host was given. Rows holding absolute URLs to the old storage host keep working until that account is closed, then 404 permanently. The scan step will report what it finds; pass --rewrite-host to fix them.', + }); + } + } + + steps.push({ + id: 'check:all', + phase: 'check', + title: `check credentials and connectivity for ${source.platform} and ${target.label}`, + after: [], + }); + + if (moving.length) { + steps.push({ + id: 'freeze:all', + phase: 'freeze', + title: 'stop writes on the source (downtime starts here)', + after: moving.map((r) => stepId('import', r.id)), + touchesSource: true, + }); + steps.push({ + id: 'cutover:dns', + phase: 'cutover', + title: 'point DNS at the target', + after: ['freeze:all', ...moving.filter(hasDelta(opts)).map((r) => stepId('delta', r.id))], + irreversible: false, + }); + } + + for (const note of source.notes ?? []) { + risks.push({ severity: 'note', message: note }); + } + + if (!moving.length) { + risks.push({ + severity: 'blocker', + message: 'Nothing to migrate: every resource was skipped or unsupported.', + }); + } + + const ordered = orderSteps(steps); + const estimateBytes = moving.reduce((sum, r) => sum + (r.sizeBytes ?? 0), 0); + + return { + source: source.platform, + target: target.id, + scope: source.scope, + steps: ordered, + risks, + moving, + skipped, + estimateBytes, + ok: !risks.some((r) => r.severity === 'blocker'), + }; +} + +function hasDelta(opts: PlanOptions) { + return (r: Resource) => Boolean(opts.engines.get(r.kind)?.delta); +} + +function sizeSuffix(r: Resource): string { + if (r.sizeBytes) return ` (${humanBytes(r.sizeBytes)})`; + if (r.itemCount) return ` (${r.itemCount.toLocaleString()} items)`; + return ''; +} + +export function humanBytes(n: number): string { + const units = ['B', 'KB', 'MB', 'GB', 'TB']; + let v = n; + let i = 0; + while (v >= 1024 && i < units.length - 1) { + v /= 1024; + i += 1; + } + return `${v >= 10 || i === 0 ? Math.round(v) : v.toFixed(1)} ${units[i]}`; +} + +/** + * Sort steps by phase, then by dependency within the phase. + * + * Phase order is absolute and comes first: a `delta` step never runs before a + * `freeze` step even if nothing declares the dependency, because the phase + * ordering *is* the safety property. Within a phase, a stable topological sort + * respects `after`, and a cycle throws rather than quietly picking an order — + * a cyclic plan is a bug in the planner and running it would be worse than + * failing. + */ +export function orderSteps(steps: Step[]): Step[] { + const byId = new Map(steps.map((s) => [s.id, s])); + const out: Step[] = []; + const done = new Set(); + + for (const phase of PHASES) { + const inPhase = steps.filter((s) => s.phase === phase); + const pending = new Map(inPhase.map((s) => [s.id, s])); + + while (pending.size) { + let progressed = false; + for (const [id, step] of [...pending]) { + // Only dependencies inside this phase can block; an earlier phase has + // already run by construction, and a dependency on a later phase would + // be a planner bug, caught below. + const blocking = step.after.filter( + (dep) => pending.has(dep) && dep !== id && byId.get(dep)?.phase === phase, + ); + if (blocking.length === 0) { + out.push(step); + done.add(id); + pending.delete(id); + progressed = true; + } + } + if (!progressed) { + throw new Error( + `migration plan has a dependency cycle in phase '${phase}' among: ${[...pending.keys()].join(', ')}`, + ); + } + } + } + + // A dependency naming a step in a LATER phase inverts the safety ordering. + for (const step of out) { + for (const dep of step.after) { + const target = byId.get(dep); + if (!target) continue; + if (PHASES.indexOf(target.phase) > PHASES.indexOf(step.phase)) { + throw new Error( + `step '${step.id}' (${step.phase}) depends on '${dep}' (${target.phase}), which runs later`, + ); + } + } + } + + return out; +} + +/** Render a plan the way `sh1pt migrate plan` prints it. */ +export function renderPlan(plan: MigrationPlan): string { + const lines: string[] = []; + lines.push(`${plan.source} → ${plan.target} (${plan.scope})`); + lines.push(''); + + if (plan.moving.length) { + lines.push(`Moving ${plan.moving.length} resource(s), ${humanBytes(plan.estimateBytes)}:`); + for (const r of plan.moving) lines.push(` ${r.kind.padEnd(15)} ${r.name}${sizeSuffix(r)}`); + lines.push(''); + } + + if (plan.skipped.length) { + lines.push('Not moving:'); + for (const s of plan.skipped) lines.push(` ${s.resource.name} — ${s.reason}`); + lines.push(''); + } + + let phase: Phase | null = null; + for (const step of plan.steps) { + if (step.phase !== phase) { + phase = step.phase; + lines.push(`${phase}:`); + } + const marks = [ + step.touchesSource ? 'SOURCE' : null, + step.irreversible ? 'IRREVERSIBLE' : null, + ].filter(Boolean); + lines.push(` ${step.title}${marks.length ? ` [${marks.join(' ')}]` : ''}`); + } + + if (plan.risks.length) { + lines.push(''); + for (const sev of ['blocker', 'warning', 'note'] as const) { + for (const r of plan.risks.filter((x) => x.severity === sev)) { + lines.push(`${sev.toUpperCase()}: ${r.message}`); + } + } + } + + lines.push(''); + lines.push(plan.ok ? 'Plan is applyable.' : 'Plan has blockers and cannot be applied.'); + return lines.join('\n'); +} diff --git a/packages/migrate/src/types.ts b/packages/migrate/src/types.ts new file mode 100644 index 00000000..f274021d --- /dev/null +++ b/packages/migrate/src/types.ts @@ -0,0 +1,267 @@ +/** + * Moving an application and its data from one platform to another. + * + * `packages/cloud/*` already answers "give me a machine": connect, quote, + * provision, destroy. That is not this. Provisioning the destination is the + * easy half of a migration; the half that goes wrong is the data — the dump + * that silently omitted an extension, the storage bucket whose objects are + * referenced by absolute URL in a thousand rows, the cron jobs that start + * firing from two places at once because the cutover happened in the wrong + * order. + * + * The shape here comes from an actual migration rather than a whiteboard: + * crawlproof.com off Railway and Supabase cloud onto a dedicated box, which + * moved a 4.7 GB database, 8,410 storage objects across three buckets, ten + * pg_cron jobs and a realtime publication, and which would have quietly broken + * the site months later over 2,928 rows holding absolute storage URLs. + * + * ## Why this is not N×M adapters + * + * The naive shape is one adapter per (source, target) pair, which is why most + * migration tooling supports exactly one direction. The split that avoids it: + * + * **Platforms** answer "what have I got, and what are the credentials?" + * Railway, Supabase, Turso, Fly, Neon, Vercel, a plain VPS over ssh. A + * platform does not know how to move a byte. It does an `inventory()` and + * hands back resources with connection details attached. + * + * **Engines** move the bytes. Postgres, MySQL, SQLite/libSQL, Redis, + * S3-compatible object storage, plain files. An engine does not know or care + * which vendor either side is. + * + * So Railway→dedicated and dedicated→Railway are the same code path, and a new + * platform costs one `inventory()` rather than one adapter per existing + * platform. Direction is not a property of the system; it is which platform + * you named first. + * + * Every resource carries the engine that can move it. A migration is possible + * exactly when, for each resource the source lists, the target can accept that + * engine — which is a check the planner can make before touching anything. + */ + +/** + * What kind of thing is being moved, which is the same as asking which engine + * moves it. + * + * Deliberately about storage shape rather than vendor: Supabase's database and + * Neon's are both `postgres`, and the code that moves one moves the other. A + * vendor difference that genuinely matters (Supabase's auth schema, its + * storage metadata tables) is a `quirk` on the resource, not a new kind. + */ +export type ResourceKind = + | 'postgres' + | 'mysql' + | 'sqlite' + | 'redis' + | 'object-storage' + | 'files' + | 'env' + | 'cron' + | 'dns'; + +/** Every engine id, for exhaustiveness checks and for the CLI's help text. */ +export const RESOURCE_KINDS: readonly ResourceKind[] = [ + 'postgres', + 'mysql', + 'sqlite', + 'redis', + 'object-storage', + 'files', + 'env', + 'cron', + 'dns', +] as const; + +/** + * A credential or connection string. + * + * Kept as a getter rather than a value so a plan can be printed, stored and + * reviewed without a password ever being serialised into it. `describe()` is + * what goes in the plan file; `reveal()` is called only inside an engine, at + * the moment it runs. + */ +export interface Secretish { + /** Safe for logs, plan files and terminal output. Never the secret. */ + describe(): string; + /** The actual value. Call as late as possible, never log the result. */ + reveal(): string; +} + +/** + * A secret that is safe to print because it is not one — a hostname, a bucket + * name, a database name. + */ +export function plain(value: string): Secretish { + return { describe: () => value, reveal: () => value }; +} + +/** + * Wrap a real credential. `describe()` shows enough to tell two apart without + * showing either: scheme and host for a URL, a short prefix otherwise. + */ +export function secret(value: string): Secretish { + return { + reveal: () => value, + describe: () => { + try { + const u = new URL(value); + const user = u.username ? `${u.username}:***@` : ''; + return `${u.protocol}//${user}${u.host}${u.pathname}`; + } catch { + return value.length <= 4 ? '***' : `${value.slice(0, 4)}***`; + } + }, + }; +} + +/** + * One movable thing on a platform. + * + * `id` is the platform's own identifier and `name` is what a person calls it. + * `sizeBytes` and `itemCount` are best-effort: they drive the plan's estimate + * and the progress output, and being wrong is not fatal. `quirks` is where a + * vendor's non-portable detail is recorded so the planner can warn about it + * rather than discovering it halfway through a restore. + */ +export interface Resource { + kind: ResourceKind; + id: string; + name: string; + /** How to reach it. Engine-specific; see each engine for what it needs. */ + connection: Record; + sizeBytes?: number; + itemCount?: number; + /** + * Vendor specifics that survive or do not survive a move: postgres + * extensions, a Supabase auth schema, pg_cron jobs, a realtime publication. + * The planner turns these into warnings and extra steps. + */ + quirks?: string[]; + metadata?: Record; +} + +/** What a platform reported when asked what it holds. */ +export interface Inventory { + platform: string; + /** What the platform calls this deployment: a project, an app, an account. */ + scope: string; + resources: Resource[]; + /** Anything the platform could not enumerate and a human should check. */ + notes?: string[]; +} + +export interface PlatformContext { + secret(key: string): string | undefined; + log(msg: string, level?: 'info' | 'warn' | 'error'): void; + /** True when nothing may be mutated anywhere. */ + dryRun: boolean; +} + +/** + * A platform: somewhere an app lives. Implementations resolve credentials and + * enumerate resources; they never move data. + * + * `role` is honest about what a platform can do rather than aspirational. + * Railway can be read from and written to; a managed provider that offers no + * import path is `source` only, and saying so lets the planner refuse early + * with a clear message instead of failing at the last step. + */ +export interface Platform { + id: string; + label: string; + role: 'source' | 'target' | 'both'; + /** Kinds this platform can hold. The planner intersects source and target. */ + supports: ResourceKind[]; + /** Enumerate what is there. Read-only; safe to run against production. */ + inventory(ctx: PlatformContext, config: Config): Promise; + /** + * Make a place for an incoming resource and return the connection an engine + * should write to. Absent on `source`-only platforms. + */ + provision?(ctx: PlatformContext, resource: Resource, config: Config): Promise; + /** Cheap credentials check, so a three-hour migration fails in the first second. */ + check?(ctx: PlatformContext, config: Config): Promise; +} + +/** Where an engine stages bytes between export and import. */ +export interface Staging { + /** Absolute path to a directory the engine may write into. */ + dir: string; + /** Record an artifact so a resumed run can find it again. */ + record(artifact: Artifact): Promise; + /** Artifacts already produced for this resource, if the run is resuming. */ + existing(resourceId: string): Promise; +} + +/** Something an export produced: a dump file, a manifest, a directory of objects. */ +export interface Artifact { + resourceId: string; + kind: ResourceKind; + /** Path relative to the staging dir. */ + path: string; + sizeBytes?: number; + /** Set once the artifact is complete; a partial artifact has no checksum. */ + sha256?: string; + metadata?: Record; +} + +export interface EngineContext { + log(msg: string, level?: 'info' | 'warn' | 'error'): void; + dryRun: boolean; + staging: Staging; + /** + * Run a command. Injected rather than imported so tests drive an engine + * without a database, and so a dry run can record commands instead of + * running them. + */ + exec(cmd: string, args: string[], opts?: ExecOptions): Promise; +} + +export interface ExecOptions { + /** Extra environment. Secrets belong here, never in `args`, which is logged. */ + env?: Record; + cwd?: string; + /** Fail the step if the command exits non-zero. Default true. */ + check?: boolean; + timeoutMs?: number; +} + +export interface ExecResult { + code: number; + stdout: string; + stderr: string; +} + +/** + * An engine moves one kind of resource between two connections. + * + * Split into export and import rather than a single `copy` so a migration can + * stage everything, be inspected, and then be cut over — which is what makes + * the delta sync and the rollback possible. A direct streaming copy is an + * optimisation an engine may offer via `copy`, not the contract. + */ +export interface Engine { + kind: ResourceKind; + /** Binaries that must exist for this engine to run, e.g. ['pg_dump']. */ + requires: string[]; + /** Read the source into staging. */ + export(ctx: EngineContext, from: Resource): Promise; + /** Write staged artifacts into the target. */ + import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise; + /** + * Re-read only what changed since a timestamp. This is what makes a cutover + * short: the bulk copy happens while the source is live, and only the delta + * is moved during the window where writes are stopped. An engine that cannot + * do this omits it, and the planner says the cutover needs full downtime. + */ + delta?(ctx: EngineContext, from: Resource, since: Date): Promise; + /** Compare source and target after the fact. Row counts, object counts. */ + verify?(ctx: EngineContext, from: Resource, to: Resource): Promise; +} + +export interface VerifyResult { + ok: boolean; + /** One line per check, e.g. "public.users: 113 → 113". */ + checks: string[]; + problems: string[]; +} diff --git a/packages/migrate/tsconfig.json b/packages/migrate/tsconfig.json new file mode 100644 index 00000000..fcd4e5bb --- /dev/null +++ b/packages/migrate/tsconfig.json @@ -0,0 +1,6 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "outDir": "dist", "rootDir": "src" }, + "include": ["src/**/*"], + "exclude": ["src/**/*.test.ts"] +} From c1c8a41be45026c3253616afc7c4b27ae57b1a2b Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 07:44:36 +0000 Subject: [PATCH 2/8] migrate: lockfile entry for the new workspace package Co-Authored-By: Claude Opus 5 (1M context) --- pnpm-lock.yaml | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index ad195b74..06484bb8 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -1199,6 +1199,12 @@ importers: specifier: workspace:* version: link:../../core + packages/migrate: + dependencies: + '@profullstack/sh1pt-core': + specifier: workspace:* + version: link:../core + packages/observability/sentry: dependencies: '@profullstack/sh1pt-core': @@ -6157,6 +6163,7 @@ packages: eslint@9.39.4: resolution: {integrity: sha512-XoMjdBOwe/esVgEvLmNsD3IRHkm7fbKIUGvrleloJXUZgDHig2IPWNniv+GwjyJXzuNqVjlr5+4yVUZjycJwfQ==} engines: {node: ^18.18.0 || ^20.9.0 || >=21.1.0} + deprecated: This version is no longer supported. Please see https://eslint.org/version-support for other options. hasBin: true peerDependencies: jiti: '*' From 25cccfd739fe2239fae169a84793ea3e403f1e26 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 07:47:39 +0000 Subject: [PATCH 3/8] migrate: object storage, and the absolute-URL rewrite The object-storage engine is one engine with a remote per side rather than one per vendor: S3, R2, B2, Spaces, Supabase storage and MinIO differ by endpoint and auth style, not structurally. rclone does the copying because concurrency, retries, resume, multipart thresholds and checksum comparison are five things to get right and it already has them; the planner checks for the binary up front so a missing rclone is a blocker before the freeze rather than a failure during it. Two deliberate choices. The remote is built entirely from RCLONE_CONFIG_* environment variables, so no credential is written to disk or appears in argv. And it runs `copy`, never `sync` -- sync deletes whatever at the destination is not at the source, which for a mistyped target is indistinguishable from wiping a live bucket. A migration tool should not be able to delete. Object storage is also the one engine that does not stage bytes: pulling ten gigabytes down and pushing them back up doubles the transfer for no benefit, so export writes a manifest and import copies remote to remote. That needed the source resource at import time, which the Engine contract now passes explicitly rather than smuggling through metadata, which only holds primitives. Then the rewrite. An app that stores a whole URL instead of a key leaves rows pointing at the account you just left. Migrate everything, cut DNS over, check the site: every image loads -- because the OLD account is still serving them. The day it is closed, which is the entire point of migrating and happens weeks later, all of them 404 at once and nothing connects the outage to the migration. crawlproof.com had 2,928 such rows across four tables, found by looking rather than by anything failing. So it is a first-class step: find every text-ish column (broad on purpose -- a column called `notes` holding a pasted URL breaks exactly as badly as one called `image_url`, and guessing from names is how those four tables would have been missed), count what is there, rewrite inside one transaction, then assert zero remain. Writing that assertion surfaced a real bug in my own first version: a trailing `having count(*) > 0` after a UNION ALL binds to the final SELECT only, so every other column would have reported regardless of its count and the one real offender could sit unnoticed among them. The union now goes in a subquery with the filter outside it, and a test pins that the filter is not inside. 90 tests, still no network. Co-Authored-By: Claude Opus 5 (1M context) --- .../src/engines/object-storage.test.ts | 171 +++++++++++++ .../migrate/src/engines/object-storage.ts | 230 ++++++++++++++++++ packages/migrate/src/transforms.test.ts | 174 +++++++++++++ packages/migrate/src/transforms.ts | 207 ++++++++++++++++ packages/migrate/src/types.ts | 13 +- 5 files changed, 793 insertions(+), 2 deletions(-) create mode 100644 packages/migrate/src/engines/object-storage.test.ts create mode 100644 packages/migrate/src/engines/object-storage.ts create mode 100644 packages/migrate/src/transforms.test.ts create mode 100644 packages/migrate/src/transforms.ts diff --git a/packages/migrate/src/engines/object-storage.test.ts b/packages/migrate/src/engines/object-storage.test.ts new file mode 100644 index 00000000..e2c50683 --- /dev/null +++ b/packages/migrate/src/engines/object-storage.test.ts @@ -0,0 +1,171 @@ +import { describe, expect, it } from 'vitest'; +import { objectStorageEngine, rcloneEnv, rclonePath } from './object-storage.js'; +import type { EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +const bucket = (over: Partial = {}): Resource => ({ + kind: 'object-storage', + id: 'b1', + name: 'public', + connection: { + bucket: plain('ads'), + accessKeyId: secret('AKIAEXAMPLE'), + secretAccessKey: secret('supersecret'), + endpoint: plain('https://abc.supabase.co/storage/v1/s3'), + region: plain('us-east-1'), + }, + ...over, +}); + +describe('rcloneEnv', () => { + it('builds a remote entirely from the environment so nothing is written to disk', () => { + const env = rcloneEnv(bucket(), 'src'); + expect(env.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(env.RCLONE_CONFIG_SRC_ACCESS_KEY_ID).toBe('AKIAEXAMPLE'); + expect(env.RCLONE_CONFIG_SRC_SECRET_ACCESS_KEY).toBe('supersecret'); + }); + + it('forces path style for a custom endpoint, which otherwise resolves to a host that does not exist', () => { + expect(rcloneEnv(bucket(), 'src').RCLONE_CONFIG_SRC_FORCE_PATH_STYLE).toBe('true'); + }); + + it('leaves path style alone for real AWS', () => { + const r = bucket({ connection: { bucket: plain('b') } }); + expect(rcloneEnv(r, 'src').RCLONE_CONFIG_SRC_FORCE_PATH_STYLE).toBeUndefined(); + }); + + it('namespaces by alias so source and target can both be configured at once', () => { + const merged = { ...rcloneEnv(bucket(), 'src'), ...rcloneEnv(bucket(), 'dst') }; + expect(merged.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(merged.RCLONE_CONFIG_DST_TYPE).toBe('s3'); + }); + + it('refuses a resource with no bucket', () => { + expect(() => rcloneEnv(bucket({ connection: {} }), 'src')).toThrow(/no connection.bucket/); + }); +}); + +describe('rclonePath', () => { + it('is remote:bucket', () => { + expect(rclonePath(bucket(), 'src')).toBe('src:ads'); + }); + + it('appends a prefix when there is one', () => { + const r = bucket({ connection: { bucket: plain('ads'), prefix: plain('/2026') } }); + expect(rclonePath(r, 'src')).toBe('src:ads/2026'); + }); +}); + +describe('export', () => { + it('lists the bucket and records the object count rather than downloading it', async () => { + const c = ctx([{ stdout: JSON.stringify([{ Path: 'a.png', Size: 10 }, { Path: 'b.png', Size: 20 }]) }]); + const [artifact] = await objectStorageEngine.export(c, bucket()); + + expect(c.calls[0]!.cmd).toBe('rclone'); + expect(c.calls[0]!.args).toContain('lsjson'); + expect(artifact?.sizeBytes).toBe(30); + expect(artifact?.metadata?.objectCount).toBe(2); + }); + + it('never puts a secret in argv', async () => { + const c = ctx([{ stdout: '[]' }]); + await objectStorageEngine.export(c, bucket()); + expect(c.calls[0]!.args.join(' ')).not.toContain('supersecret'); + }); + + it('throws on an unparseable listing rather than silently copying nothing', async () => { + const c = ctx([{ stdout: 'not json' }]); + await expect(objectStorageEngine.export(c, bucket())).rejects.toThrow(/could not parse/); + }); +}); + +describe('import', () => { + it('copies remote to remote with both remotes configured', async () => { + const c = ctx(); + const to = bucket({ id: 'b2', name: 'dest', connection: { bucket: plain('dest-ads') } }); + await objectStorageEngine.import(c, to, [], bucket()); + + const call = c.calls[0]!; + expect(call.args[0]).toBe('copy'); + expect(call.args[1]).toBe('src:ads'); + expect(call.args[2]).toBe('dst:dest-ads'); + expect(call.opts?.env?.RCLONE_CONFIG_SRC_TYPE).toBe('s3'); + expect(call.opts?.env?.RCLONE_CONFIG_DST_TYPE).toBe('s3'); + }); + + it('uses copy and never sync, so a mistyped target cannot delete a live bucket', async () => { + const c = ctx(); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls[0]!.args).not.toContain('sync'); + expect(c.calls[0]!.args).not.toContain('--delete'); + expect(c.calls[0]!.args).not.toContain('--delete-during'); + }); + + it('is resume-friendly, so a re-run does not re-send the whole bucket', async () => { + const c = ctx(); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls[0]!.args).toContain('--update'); + }); + + it('refuses without the source, since nothing was staged locally', async () => { + const c = ctx(); + await expect(objectStorageEngine.import(c, bucket(), [])).rejects.toThrow(/needs the source/); + }); + + it('runs nothing on a dry run', async () => { + const c = ctx([], { dryRun: true }); + await objectStorageEngine.import(c, bucket({ id: 'b2' }), [], bucket()); + expect(c.calls).toHaveLength(0); + }); +}); + +describe('delta', () => { + it('asks only for objects written since the bulk copy started', async () => { + const c = ctx([{ stdout: '[{"Path":"new.png"}]' }]); + const since = new Date(Date.now() - 3600_000); + const [artifact] = await objectStorageEngine.delta!(c, bucket(), since); + + const ageArg = c.calls[0]!.args.find((a) => a.startsWith('--max-age=')); + expect(ageArg).toBeDefined(); + expect(artifact?.metadata?.objectCount).toBe(1); + }); +}); + +describe('verify', () => { + it('passes when the object counts match', async () => { + const c = ctx([{ stdout: '{"count":8410,"bytes":9560000000}' }, { stdout: '{"count":8410,"bytes":9560000000}' }]); + const res = await objectStorageEngine.verify!(c, bucket(), bucket({ id: 'b2' })); + expect(res.ok).toBe(true); + }); + + it('fails when objects did not arrive', async () => { + const c = ctx([{ stdout: '{"count":8410,"bytes":1}' }, { stdout: '{"count":8000,"bytes":1}' }]); + const res = await objectStorageEngine.verify!(c, bucket(), bucket({ id: 'b2' })); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('410 object(s) did not arrive'); + }); +}); diff --git a/packages/migrate/src/engines/object-storage.ts b/packages/migrate/src/engines/object-storage.ts new file mode 100644 index 00000000..a59dedfb --- /dev/null +++ b/packages/migrate/src/engines/object-storage.ts @@ -0,0 +1,230 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a bucket of objects. + * + * S3, R2, B2, Spaces, Supabase storage, MinIO on a dedicated box. They all + * speak S3 or something close enough, and the differences are endpoint URLs + * and auth styles rather than anything structural — so this is one engine with + * a remote definition per side, not one engine per vendor. + * + * ## Why rclone rather than an SDK + * + * Copying 8,410 objects across 9.56 GB from a script means reimplementing + * concurrency, retries, resume, multipart thresholds and checksum comparison, + * and getting all five right. rclone has those and is a single static binary. + * The alternative — `aws s3 sync` — only speaks S3 and needs a credentials + * file on disk, which is worse for a tool that has to reach six vendors. + * + * The cost is a dependency the planner checks for up front (`requires`), so a + * missing rclone is a blocker before the freeze rather than a failure during + * it. + * + * ## Staging, or not + * + * Every other engine dumps to staging and loads from it. Object storage is the + * exception: pulling 10 GB down to a laptop and pushing it back up doubles the + * transfer and needs the disk. `export` therefore only writes a manifest, and + * `import` runs a remote-to-remote copy that rclone streams server-side where + * it can. The manifest is not busywork — it is what `verify` compares against, + * and what makes a resumed run able to tell what it already did. + */ + +/** + * An rclone remote, built inline from the resource's connection. + * + * rclone is normally configured from a file; passing the whole remote + * definition through `RCLONE_CONFIG_*` environment variables instead means no + * credential is ever written to disk, and none appears in argv. + */ +export function rcloneEnv(r: Resource, alias: string): Record { + const get = (k: string): string | undefined => r.connection[k]?.reveal(); + const bucket = get('bucket'); + if (!bucket) throw new Error(`object-storage resource '${r.name}' has no connection.bucket`); + + const prefix = `RCLONE_CONFIG_${alias.toUpperCase()}`; + const env: Record = { + [`${prefix}_TYPE`]: 's3', + // 'Other' keeps rclone from applying provider-specific assumptions to an + // endpoint that merely speaks S3, which is the case for R2, Supabase, + // MinIO and Backblaze's S3 gateway. + [`${prefix}_PROVIDER`]: get('provider') ?? 'Other', + }; + + const accessKey = get('accessKeyId'); + const secretKey = get('secretAccessKey'); + if (accessKey) env[`${prefix}_ACCESS_KEY_ID`] = accessKey; + if (secretKey) env[`${prefix}_SECRET_ACCESS_KEY`] = secretKey; + + const endpoint = get('endpoint'); + if (endpoint) env[`${prefix}_ENDPOINT`] = endpoint; + const region = get('region'); + if (region) env[`${prefix}_REGION`] = region; + + // Most non-AWS S3 endpoints are path-style; virtual-host style silently + // resolves to a hostname that does not exist. + if (endpoint && !get('forcePathStyle')) env[`${prefix}_FORCE_PATH_STYLE`] = 'true'; + + return env; +} + +/** `remote:bucket/prefix` as rclone wants it. */ +export function rclonePath(r: Resource, alias: string): string { + const bucket = r.connection.bucket?.reveal() ?? ''; + const prefix = r.connection.prefix?.reveal() ?? ''; + return `${alias}:${bucket}${prefix ? `/${prefix.replace(/^\/+/, '')}` : ''}`; +} + +const MANIFEST = 'objects.json'; + +interface ManifestEntry { + path: string; + size: number; +} + +export const objectStorageEngine: Engine = { + kind: 'object-storage', + requires: ['rclone'], + + /** + * List the bucket. Deliberately does not download it — see the note above. + */ + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${MANIFEST}`; + ctx.log(`listing ${from.name}`); + + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'object-storage', path }]; + + const res = await ctx.exec('rclone', ['lsjson', '--recursive', '--files-only', rclonePath(from, 'src')], { + env: rcloneEnv(from, 'src'), + timeoutMs: 60 * 60 * 1000, + }); + + let entries: ManifestEntry[]; + try { + const parsed = JSON.parse(res.stdout || '[]') as Array<{ Path?: string; Size?: number }>; + entries = parsed.map((e) => ({ path: e.Path ?? '', size: e.Size ?? 0 })).filter((e) => e.path); + } catch { + throw new Error(`could not parse the object listing for '${from.name}'`); + } + + const artifact: Artifact = { + resourceId: from.id, + kind: 'object-storage', + path, + sizeBytes: entries.reduce((n, e) => n + e.size, 0), + metadata: { objectCount: entries.length, manifest: JSON.stringify(entries).length }, + }; + await ctx.staging.record(artifact); + ctx.log(`${entries.length} object(s) to copy`); + return [artifact]; + }, + + /** + * Copy source → target directly. + * + * `copy`, never `sync`: sync deletes anything at the destination that is not + * at the source, which for a mistyped target is indistinguishable from + * wiping a live bucket. Migrations should not be able to delete. + */ + async import( + ctx: EngineContext, + to: Resource, + _artifacts: Artifact[], + from?: Resource, + ): Promise { + const source = from; + if (!source) { + throw new Error( + `object-storage import needs the source resource: its objects are copied remote-to-remote rather than staged locally`, + ); + } + + if (ctx.dryRun) { + ctx.log(`would rclone copy ${rclonePath(source, 'src')} → ${rclonePath(to, 'dst')}`); + return; + } + + ctx.log(`rclone copy → ${to.name}`); + await ctx.exec( + 'rclone', + [ + 'copy', + rclonePath(source, 'src'), + rclonePath(to, 'dst'), + '--transfers=16', + '--checkers=32', + // Resume-friendly: an object already present with the same size and + // modification time is not re-sent, so a re-run after a failure costs + // a listing rather than the whole bucket. + '--update', + '--stats=30s', + ], + { + env: { ...rcloneEnv(source, 'src'), ...rcloneEnv(to, 'dst') }, + timeoutMs: 12 * 60 * 60 * 1000, + }, + ); + }, + + /** + * Objects created during the bulk copy. + * + * rclone's own `--max-age` does this server-side, so the delta is the same + * copy restricted to recent objects rather than a different mechanism. + */ + async delta(ctx: EngineContext, from: Resource, since: Date): Promise { + const path = `${from.id}/delta-${since.toISOString().replace(/[:.]/g, '')}.json`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'object-storage', path }]; + + const ageSeconds = Math.max(1, Math.round((Date.now() - since.getTime()) / 1000)); + const res = await ctx.exec( + 'rclone', + ['lsjson', '--recursive', '--files-only', `--max-age=${ageSeconds}s`, rclonePath(from, 'src')], + { env: rcloneEnv(from, 'src') }, + ); + + let count = 0; + try { + count = (JSON.parse(res.stdout || '[]') as unknown[]).length; + } catch { + count = 0; + } + ctx.log(`${count} object(s) written during the bulk copy`); + + const artifact: Artifact = { + resourceId: from.id, + kind: 'object-storage', + path, + metadata: { objectCount: count, mode: 'delta' }, + }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + /** Object counts and total size on both sides. */ + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const size = async (r: Resource, alias: string) => { + const res = await ctx.exec('rclone', ['size', '--json', rclonePath(r, alias)], { + env: rcloneEnv(r, alias), + check: false, + }); + try { + const parsed = JSON.parse(res.stdout || '{}') as { count?: number; bytes?: number }; + return { count: parsed.count ?? 0, bytes: parsed.bytes ?? 0 }; + } catch { + return { count: 0, bytes: 0 }; + } + }; + + const [a, b] = await Promise.all([size(from, 'src'), size(to, 'dst')]); + const checks = [`${from.name}: ${a.count} objects → ${b.count}`, `bytes: ${a.bytes} → ${b.bytes}`]; + const problems: string[] = []; + if (b.count < a.count) { + problems.push(`${a.count - b.count} object(s) did not arrive in '${to.name}'`); + } + return { ok: problems.length === 0, checks, problems }; + }, +}; diff --git a/packages/migrate/src/transforms.test.ts b/packages/migrate/src/transforms.test.ts new file mode 100644 index 00000000..fdc380d1 --- /dev/null +++ b/packages/migrate/src/transforms.test.ts @@ -0,0 +1,174 @@ +import { describe, expect, it } from 'vitest'; +import { + type ColumnRef, + countMatchesSql, + isScannable, + parseRewrite, + qualify, + quoteIdent, + quoteLiteral, + remainingMatchesSql, + rewriteColumnSql, + rewritePlanSql, +} from './transforms.js'; + +const col = (over: Partial = {}): ColumnRef => ({ + schema: 'public', + table: 'ad_creatives', + column: 'image_url', + dataType: 'text', + ...over, +}); + +describe('quoting', () => { + it('quotes an identifier so a reserved word still parses', () => { + expect(quoteIdent('user')).toBe('"user"'); + expect(quoteIdent('order')).toBe('"order"'); + }); + + it('doubles an embedded quote rather than letting it end the identifier', () => { + expect(quoteIdent('we"ird')).toBe('"we""ird"'); + }); + + it('doubles an embedded apostrophe in a literal', () => { + expect(quoteLiteral("o'brien")).toBe("'o''brien'"); + }); + + it('qualifies schema and table together', () => { + expect(qualify(col())).toBe('"public"."ad_creatives"'); + }); +}); + +describe('isScannable', () => { + it('accepts the text-ish types', () => { + for (const t of ['text', 'character varying', 'character', 'json', 'jsonb']) { + expect(isScannable(t)).toBe(true); + } + }); + + it('is case-insensitive', () => { + expect(isScannable('TEXT')).toBe(true); + }); + + it('rejects types that cannot hold a URL', () => { + for (const t of ['integer', 'boolean', 'timestamp with time zone', 'bytea']) { + expect(isScannable(t)).toBe(false); + } + }); +}); + +describe('countMatchesSql', () => { + it('counts rows mentioning the old host', () => { + const sql = countMatchesSql(col(), 'abc.supabase.co'); + expect(sql).toContain('"public"."ad_creatives"'); + expect(sql).toContain("'%abc.supabase.co%'"); + expect(sql).toContain('count(*)'); + }); + + it('casts to text so json columns are covered by the same statement', () => { + expect(countMatchesSql(col({ dataType: 'jsonb' }), 'h')).toContain('::text like'); + }); +}); + +describe('rewriteColumnSql', () => { + const rewrite = { from: 'abc.supabase.co', to: 'https://cdn.example.com' }; + + it('replaces the old origin with the new one', () => { + const sql = rewriteColumnSql(col(), rewrite); + expect(sql).toContain(`replace("image_url", 'https://abc.supabase.co', 'https://cdn.example.com')`); + }); + + it('only touches rows that actually mention the old host', () => { + expect(rewriteColumnSql(col(), rewrite)).toContain("like '%abc.supabase.co%'"); + }); + + it('round-trips a jsonb column through text and back', () => { + const sql = rewriteColumnSql(col({ dataType: 'jsonb' }), rewrite); + expect(sql).toContain('::text'); + expect(sql).toContain('::jsonb'); + }); + + it('strips a trailing slash off the new origin so URLs do not double up', () => { + const sql = rewriteColumnSql(col(), { from: 'old.host', to: 'https://new.host/' }); + expect(sql).toContain("'https://new.host'"); + expect(sql).not.toContain("'https://new.host/'"); + }); +}); + +describe('rewritePlanSql', () => { + it('wraps every statement in one transaction', () => { + const sql = rewritePlanSql([col()], [{ from: 'a.co', to: 'https://b.co' }]); + expect(sql.startsWith('begin;')).toBe(true); + expect(sql.trimEnd().endsWith('commit;')).toBe(true); + }); + + it('covers every host across every scannable column', () => { + const sql = rewritePlanSql( + [col(), col({ table: 'blog_posts', column: 'body' })], + [ + { from: 'a.co', to: 'https://x.co' }, + { from: 'b.co', to: 'https://y.co' }, + ], + ); + expect(sql.match(/update/g)).toHaveLength(4); + }); + + it('skips columns that cannot hold a URL', () => { + const sql = rewritePlanSql( + [col({ dataType: 'integer', column: 'views' })], + [{ from: 'a.co', to: 'https://b.co' }], + ); + expect(sql).not.toContain('update'); + }); +}); + +describe('remainingMatchesSql', () => { + it('is the post-commit assertion, and zero rows is the pass', () => { + const sql = remainingMatchesSql([col()], ['a.co']); + expect(sql).toContain('where n > 0'); + expect(sql).toContain('%a.co%'); + }); + + it('unions every column and filters OUTSIDE the union', () => { + // A trailing `having` would bind to the last SELECT only, so every other + // column would report regardless of count. The filter must be outside. + const sql = remainingMatchesSql([col(), col({ column: 'thumb_url' })], ['a.co']); + expect(sql).toContain('union all'); + expect(sql).not.toContain('having'); + const afterSubquery = sql.slice(sql.lastIndexOf(') as remaining')); + expect(afterSubquery).toContain('where n > 0'); + }); + + it('degrades to a query returning nothing when there is nothing to check', () => { + expect(remainingMatchesSql([], ['a.co'])).toContain('where false'); + }); +}); + +describe('parseRewrite', () => { + it('parses old=new', () => { + expect(parseRewrite('abc.supabase.co=https://cdn.example.com')).toEqual({ + from: 'abc.supabase.co', + to: 'https://cdn.example.com', + }); + }); + + it('tolerates a scheme on the old side', () => { + expect(parseRewrite('https://abc.supabase.co=https://cdn.example.com').from).toBe( + 'abc.supabase.co', + ); + }); + + it('rejects a new origin with no scheme, which would silently corrupt URLs', () => { + expect(() => parseRewrite('a.co=cdn.example.com')).toThrow(/needs a scheme/); + }); + + it('rejects a path on the old side', () => { + expect(() => parseRewrite('a.co/storage=https://b.co')).toThrow(/bare host/); + }); + + it('rejects a malformed spec', () => { + for (const bad of ['', 'nope', '=https://b.co']) { + expect(() => parseRewrite(bad)).toThrow(); + } + }); +}); diff --git a/packages/migrate/src/transforms.ts b/packages/migrate/src/transforms.ts new file mode 100644 index 00000000..7eba4b6a --- /dev/null +++ b/packages/migrate/src/transforms.ts @@ -0,0 +1,207 @@ +/** + * Rewriting absolute URLs that point at the place you just left. + * + * This is the failure that makes migrations dangerous months after they look + * successful. An app stores an uploaded file and, instead of keeping a key, + * writes the whole URL into a row: + * + * https://ywcizjsgrcmhgyplldac.supabase.co/storage/v1/object/public/ads/x.png + * + * Migrate the bucket and the database, cut DNS over, check the site: every + * image loads. They load because the OLD account still exists and is still + * serving them. The day that account is closed — which is the whole point of + * migrating, and which happens weeks later once everyone is confident — every + * one of those rows 404s at once, and nothing connects the outage to the + * migration. + * + * crawlproof.com had 2,928 such rows across four tables. They were found by + * looking, not by anything failing. + * + * So this is a first-class step rather than a footnote: find every column that + * could hold one, report what is there, and rewrite it inside the same + * transaction that a person can roll back. + * + * Nothing here executes SQL. It builds statements and the caller runs them, + * which keeps it testable and keeps the generated SQL reviewable before it + * touches a database. + */ + +/** A host whose URLs must be rewritten, and what to replace it with. */ +export interface HostRewrite { + /** The old host, e.g. `abc123.supabase.co`. No scheme, no path. */ + from: string; + /** The new origin, e.g. `https://cdn.example.com`. Scheme required. */ + to: string; +} + +/** Text-ish column types worth scanning. */ +const TEXT_TYPES = new Set(['text', 'character varying', 'character', 'json', 'jsonb']); + +export interface ColumnRef { + schema: string; + table: string; + column: string; + dataType: string; +} + +/** + * The query that finds columns which could hold a URL. + * + * Every text-ish column in a user schema. Deliberately broad: a column called + * `notes` holding a pasted URL breaks exactly as badly as one called + * `image_url`, and guessing from names is how the four crawlproof tables would + * have been missed. The count query that follows narrows it cheaply. + */ +export function findTextColumnsSql(): string { + return `select table_schema, table_name, column_name, data_type + from information_schema.columns + where table_schema not in ('pg_catalog', 'information_schema') + and data_type in ('text', 'character varying', 'character', 'json', 'jsonb') + order by table_schema, table_name, column_name`; +} + +/** True when a column's type can hold a URL. */ +export function isScannable(dataType: string): boolean { + return TEXT_TYPES.has(dataType.toLowerCase()); +} + +/** + * Postgres identifiers are quoted rather than interpolated bare. + * + * These names come from information_schema, so they are real identifiers, but + * a table legitimately named `user` or `order` is a reserved word and an + * unquoted reference is a syntax error partway through a migration. Doubling + * any embedded quote is the standard escape. + */ +export function quoteIdent(name: string): string { + return `"${name.replace(/"/g, '""')}"`; +} + +/** A single-quoted SQL string literal. */ +export function quoteLiteral(value: string): string { + return `'${value.replace(/'/g, "''")}'`; +} + +/** Fully-qualified, safely quoted. */ +export function qualify(c: Pick): string { + return `${quoteIdent(c.schema)}.${quoteIdent(c.table)}`; +} + +/** + * Count rows in one column that mention the old host. + * + * Run before rewriting anything: it turns "this might be a problem" into "2,928 + * rows in four tables", which is what makes the step reviewable. A cast to text + * lets one statement cover json and jsonb alongside the plain text types. + */ +export function countMatchesSql(c: ColumnRef, host: string): string { + return `select ${quoteLiteral(`${c.schema}.${c.table}.${c.column}`)} as ref, count(*) as n + from ${qualify(c)} + where ${quoteIdent(c.column)}::text like ${quoteLiteral(`%${host}%`)}`; +} + +/** + * Rewrite one column. + * + * `replace` on the text form rather than a regex: the host is a literal, a + * regex would need escaping, and `replace` is index-friendly and predictable. + * The `where` clause means untouched rows are not rewritten, which keeps the + * update small and leaves `updated_at` triggers alone on rows that did not + * change. + * + * json and jsonb are cast out to text and back, which is lossy for jsonb key + * order but not for content — and jsonb does not preserve key order anyway. + */ +export function rewriteColumnSql(c: ColumnRef, rewrite: HostRewrite): string { + const from = quoteLiteral(`https://${rewrite.from}`); + const to = quoteLiteral(rewrite.to.replace(/\/+$/, '')); + const col = quoteIdent(c.column); + const type = c.dataType.toLowerCase(); + + const expr = + type === 'json' || type === 'jsonb' + ? `replace(${col}::text, ${from}, ${to})::${type}` + : `replace(${col}, ${from}, ${to})`; + + return `update ${qualify(c)} + set ${col} = ${expr} + where ${col}::text like ${quoteLiteral(`%${rewrite.from}%`)}`; +} + +/** + * The whole rewrite as one transaction. + * + * One transaction so a failure halfway leaves nothing half-rewritten, and so + * the whole thing can be rolled back by a person watching it. The trailing + * verification select is what the caller asserts on: after a correct rewrite it + * returns zero rows, and the crawlproof runbook asserted exactly that. + */ +export function rewritePlanSql(columns: ColumnRef[], rewrites: HostRewrite[]): string { + const statements: string[] = ['begin;']; + for (const rewrite of rewrites) { + for (const c of columns) { + if (!isScannable(c.dataType)) continue; + statements.push(`${rewriteColumnSql(c, rewrite)};`); + } + } + statements.push('commit;'); + return statements.join('\n'); +} + +/** + * The assertion that the rewrite worked. + * + * Returns one row per column that still mentions an old host. Zero rows is the + * pass condition. Run it AFTER committing: a migration that reports success + * while rows still point at an account about to be closed is worse than one + * that fails loudly. + */ +export function remainingMatchesSql(columns: ColumnRef[], hosts: string[]): string { + const parts: string[] = []; + for (const host of hosts) { + for (const c of columns) { + if (!isScannable(c.dataType)) continue; + parts.push(countMatchesSql(c, host)); + } + } + if (!parts.length) return 'select null::text as ref, 0::bigint as n where false'; + + /* + * The union goes in a subquery and the filter is a WHERE on the outside. + * + * A trailing `having count(*) > 0` looks like it filters the whole thing and + * does not: in a UNION chain it binds to the final SELECT only, so every + * other column would be reported regardless of its count and the one real + * offender could be buried. Filtering outside the subquery applies to all of + * them, which is the point of the assertion. + */ + return `select ref, n from (\n${parts.join('\nunion all\n')}\n) as remaining where n > 0 order by n desc`; +} + +/** + * Turn `--rewrite-host old=new` into a rewrite. + * + * Accepts a bare host on the left (`abc.supabase.co`) and a full origin on the + * right (`https://cdn.example.com`). A missing scheme on the right is the easy + * mistake and produces a corrupt URL rather than an error at runtime, so it is + * rejected here. + */ +export function parseRewrite(spec: string): HostRewrite { + const eq = spec.indexOf('='); + if (eq < 1) { + throw new Error(`--rewrite-host wants old-host=new-origin, got '${spec}'`); + } + const from = spec.slice(0, eq).trim().replace(/^https?:\/\//, '').replace(/\/+$/, ''); + const to = spec.slice(eq + 1).trim().replace(/\/+$/, ''); + + if (!from) throw new Error(`--rewrite-host has an empty old host: '${spec}'`); + if (!/^https?:\/\//.test(to)) { + throw new Error( + `--rewrite-host needs a scheme on the new origin, got '${to}'. Use https://${to}`, + ); + } + if (from.includes('/')) { + throw new Error(`--rewrite-host old side should be a bare host, got '${from}'`); + } + return { from, to }; +} diff --git a/packages/migrate/src/types.ts b/packages/migrate/src/types.ts index f274021d..9b6bc8df 100644 --- a/packages/migrate/src/types.ts +++ b/packages/migrate/src/types.ts @@ -246,8 +246,17 @@ export interface Engine { requires: string[]; /** Read the source into staging. */ export(ctx: EngineContext, from: Resource): Promise; - /** Write staged artifacts into the target. */ - import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise; + /** + * Write staged artifacts into the target. + * + * `from` is the source resource, passed because not every engine stages the + * bytes themselves. Object storage is the case that forces it: pulling ten + * gigabytes down and pushing them back up doubles the transfer for no + * benefit, so its export writes only a manifest and its import runs a + * remote-to-remote copy, which means it still needs to know where the + * objects came from. + */ + import(ctx: EngineContext, to: Resource, artifacts: Artifact[], from?: Resource): Promise; /** * Re-read only what changed since a timestamp. This is what makes a cutover * short: the bulk copy happens while the source is live, and only the delta From bba858f9f9bebd5ccf93f6d8829daa5757ac5c32 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 07:50:23 +0000 Subject: [PATCH 4/8] migrate: SQLite/libSQL, Redis and files engines, and the registry SQLite and Turso are one engine because libSQL is SQLite with a server in front, and both dump to the same SQL text -- which is also why Turso->dedicated and dedicated->Turso are the same code path. The dump is text rather than a file copy: copying the file does not work for a hosted database at all, and a SQLite file copied while something writes to it is a corrupt file rather than an error. That needed a contract change. Those tools only read and write stdio, and exec runs without a shell, so a `>` written into the arguments would be passed to the program as a literal. ExecOptions now carries stdoutFile and stdinFile, which also keeps a multi-gigabyte dump out of a string. Redis is included with a warning attached: the right migration for a cache is usually an empty one, and copying it moves stale entries and buys downtime for data that is worthless by definition. It is here for when it is not a cache -- a BullMQ queue with jobs in it, a session store where copying nothing logs everyone out. It exports with --rdb, which is consistent at a point in time, rather than walking keys with SCAN, where keys move under you as you read. It has no delta, and the planner turns that absence into the warning it should be. Import refuses to go over the wire, because there is no supported way to push an RDB into a running managed Redis, and says what to do instead rather than doing it badly. Files is rsync, for the volume or docroot that is neither a database nor a bucket. Writing it turned up two things I had wrong. The endpoint helper took a source/target parameter that selected between two identical values -- dead code pretending to encode a rule. And rsync cannot copy remote to remote at all; it is a protocol limitation, not a missing flag, so a VPS-to-VPS move must relay through the machine running the migration. That now fails up front with an explanation instead of surfacing as "The source and destination cannot both be remote" halfway through a cutover. As with object storage, --delete is never passed anywhere. 110 tests, tsc clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/migrate/src/engines/files.ts | 158 ++++++++++++++++ packages/migrate/src/engines/index.test.ts | 202 +++++++++++++++++++++ packages/migrate/src/engines/index.ts | 36 ++++ packages/migrate/src/engines/redis.ts | 98 ++++++++++ packages/migrate/src/engines/sqlite.ts | 122 +++++++++++++ packages/migrate/src/types.ts | 11 ++ 6 files changed, 627 insertions(+) create mode 100644 packages/migrate/src/engines/files.ts create mode 100644 packages/migrate/src/engines/index.test.ts create mode 100644 packages/migrate/src/engines/index.ts create mode 100644 packages/migrate/src/engines/redis.ts create mode 100644 packages/migrate/src/engines/sqlite.ts diff --git a/packages/migrate/src/engines/files.ts b/packages/migrate/src/engines/files.ts new file mode 100644 index 00000000..34a07e88 --- /dev/null +++ b/packages/migrate/src/engines/files.ts @@ -0,0 +1,158 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a directory of files — a mounted volume, an uploads directory, a + * docroot. + * + * This is the engine for the thing that is not a database and not a bucket: + * Railway volumes, Fly volumes, and `~/www/` on a dedicated box. rsync + * does it because rsync is what does this, and because its delta algorithm + * makes the second pass — the one during the cutover window — proportional to + * what changed rather than to the size of the tree. + * + * As with object storage, `--delete` is never passed. rsync's delete is the + * single most effective way to destroy a directory by typing the wrong target, + * and a migration has no reason to remove anything. + */ + +/** + * `[user@host:]/path/` as rsync wants it. + * + * The trailing slash is always present and always matters: with it rsync + * copies the CONTENTS of the directory, without it it nests the directory + * inside the destination, producing `~/www/app/app/` — which looks from the + * outside like the copy silently did nothing. + */ +export function endpoint(r: Resource): string { + const path = r.connection.path?.reveal(); + if (!path) throw new Error(`files resource '${r.name}' has no connection.path`); + const withSlash = `${path.replace(/\/*$/, '')}/`; + const host = r.connection.host?.reveal(); + if (!host) return withSlash; + const user = r.connection.user?.reveal(); + return `${user ? `${user}@` : ''}${host}:${withSlash}`; +} + +export function isRemote(r: Resource): boolean { + return Boolean(r.connection.host); +} + +/** + * rsync cannot copy remote to remote. + * + * It is a hard limitation of the protocol, not a flag that was missed: one + * side must be local. A VPS-to-VPS move therefore has to relay through the + * machine running the migration, which is a real cost (the bytes cross the + * wire twice) and needs to be said out loud rather than discovered when rsync + * exits with "The source and destination cannot both be remote." + */ +export function assertCopyable(from: Resource, to: Resource): void { + if (isRemote(from) && isRemote(to)) { + throw new Error( + `rsync cannot copy directly between two remote hosts (${from.name} → ${to.name}). Run the migration from one of them, or stage the directory locally first.`, + ); + } +} + +/** The ssh transport, including a key when one is configured. */ +function rsyncTransport(r: Resource): string[] { + const key = r.connection.sshKeyPath?.reveal(); + const port = r.connection.sshPort?.reveal(); + if (!r.connection.host) return []; + const parts = ['ssh', '-o', 'BatchMode=yes']; + if (key) parts.push('-i', key); + if (port) parts.push('-p', port); + return ['-e', parts.join(' ')]; +} + +export const filesEngine: Engine = { + kind: 'files', + requires: ['rsync'], + + /** + * Nothing is staged: like object storage, files go host to host. The export + * records what is there so `verify` has something to compare and so a plan + * can show a size. + */ + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/files.manifest`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'files', path }]; + + const res = await ctx.exec( + 'rsync', + [...rsyncTransport(from), '--dry-run', '--archive', '--stats', endpoint(from), '/dev/null'], + { check: false }, + ); + + const files = /Number of files: ([\d,]+)/.exec(res.stdout)?.[1]?.replace(/,/g, ''); + const artifact: Artifact = { + resourceId: from.id, + kind: 'files', + path, + metadata: { fileCount: files ? Number(files) : 0 }, + }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, _artifacts: Artifact[], from?: Resource): Promise { + if (!from) throw new Error('files import needs the source resource: files are copied host to host'); + assertCopyable(from, to); + + if (ctx.dryRun) { + ctx.log(`would rsync ${endpoint(from)} → ${endpoint(to)}`); + return; + } + + await ctx.exec( + 'rsync', + [ + ...rsyncTransport(isRemote(from) ? from : to), + '--archive', + '--compress', + '--partial', + '--human-readable', + endpoint(from), + endpoint(to), + ], + { timeoutMs: 12 * 60 * 60 * 1000 }, + ); + }, + + /** The second pass, during the cutover window: only what changed. */ + async delta(ctx: EngineContext, from: Resource, _since: Date): Promise { + const path = `${from.id}/files.delta`; + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'files', path }]; + // rsync compares by size and mtime on its own, so the delta pass is the + // same command. It is fast because almost nothing has changed. + const artifact: Artifact = { resourceId: from.id, kind: 'files', path, metadata: { mode: 'delta' } }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + // A dry-run rsync from source to target lists exactly what still differs. + // Zero transfers is the pass condition. + const res = await ctx.exec( + 'rsync', + [ + ...rsyncTransport(isRemote(from) ? from : to), + '--dry-run', + '--archive', + '--itemize-changes', + endpoint(from), + endpoint(to), + ], + { check: false }, + ); + + const differing = res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + return { + ok: differing.length === 0, + checks: [`${differing.length} path(s) still differ`], + problems: differing.slice(0, 20), + }; + }, +}; diff --git a/packages/migrate/src/engines/index.test.ts b/packages/migrate/src/engines/index.test.ts new file mode 100644 index 00000000..16770055 --- /dev/null +++ b/packages/migrate/src/engines/index.test.ts @@ -0,0 +1,202 @@ +import { describe, expect, it } from 'vitest'; +import { ENGINES, engineFor, requiredBinaries } from './index.js'; +import { assertCopyable, endpoint, filesEngine, isRemote } from './files.js'; +import { redisEngine } from './redis.js'; +import { sqliteEngine } from './sqlite.js'; +import type { EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; +import { RESOURCE_KINDS } from '../types.js'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +describe('the engine registry', () => { + it('registers every engine under its own kind', () => { + for (const [kind, engine] of ENGINES) expect(engine.kind).toBe(kind); + }); + + it('declares its required binaries so the planner can check them', () => { + for (const engine of ENGINES.values()) { + expect(Array.isArray(engine.requires)).toBe(true); + expect(engine.requires.length).toBeGreaterThan(0); + } + }); + + it('collects the binaries needed for a set of kinds, without duplicates', () => { + const bins = requiredBinaries(['postgres', 'files', 'postgres']); + expect(bins).toContain('pg_dump'); + expect(bins).toContain('rsync'); + expect(new Set(bins).size).toBe(bins.length); + }); + + it('returns nothing for a kind no engine handles', () => { + // env, cron and dns are real resource kinds with no byte-moving engine; + // they are handled as plan steps rather than copies. + expect(engineFor('env')).toBeUndefined(); + expect(engineFor('cron')).toBeUndefined(); + }); + + it('covers a documented subset of the resource kinds', () => { + const covered = [...ENGINES.keys()]; + for (const k of covered) expect(RESOURCE_KINDS).toContain(k); + }); +}); + +describe('sqlite / libSQL', () => { + const file = (over: Partial = {}): Resource => ({ + kind: 'sqlite', + id: 'db', + name: 'app.db', + connection: { path: plain('/data/app.db') }, + ...over, + }); + const turso = (): Resource => ({ + kind: 'sqlite', + id: 'db', + name: 'prod', + connection: { tursoDatabase: plain('prod'), authToken: secret('tok') }, + }); + + it('dumps a local file to staging via stdout redirection', async () => { + const c = ctx(); + await sqliteEngine.export(c, file()); + expect(c.calls[0]!.cmd).toBe('sqlite3'); + expect(c.calls[0]!.args).toContain('.dump'); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + }); + + it('dumps a Turso database with the same engine', async () => { + const c = ctx(); + await sqliteEngine.export(c, turso()); + expect(c.calls[0]!.cmd).toBe('turso'); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + }); + + it('keeps the Turso token out of argv', async () => { + const c = ctx(); + await sqliteEngine.export(c, turso()); + expect(c.calls[0]!.args.join(' ')).not.toContain('tok'); + expect(c.calls[0]!.opts?.env?.TURSO_API_TOKEN).toBe('tok'); + }); + + it('loads into Turso from stdin, which is the direction that makes it bidirectional', async () => { + const c = ctx(); + await sqliteEngine.import(c, turso(), [{ resourceId: 'db', kind: 'sqlite', path: 'db/dump.sql' }]); + expect(c.calls[0]!.opts?.stdinFile).toBe('/staging/db/dump.sql'); + }); + + it('refuses a resource with no path', async () => { + const c = ctx(); + await expect(sqliteEngine.export(c, file({ connection: {} }))).rejects.toThrow(/no connection.path/); + }); +}); + +describe('redis', () => { + const r = (over: Partial = {}): Resource => ({ + kind: 'redis', + id: 'r', + name: 'cache', + connection: { url: secret('redis://:pw@h:6379') }, + ...over, + }); + + it('uses --rdb, which is consistent, rather than walking keys, which is not', async () => { + const c = ctx(); + await redisEngine.export(c, r()); + expect(c.calls[0]!.args).toContain('--rdb'); + }); + + it('has no delta, so the planner warns that writes during the copy are lost', () => { + expect(redisEngine.delta).toBeUndefined(); + }); + + it('refuses to load over the wire and says what to do instead', async () => { + const c = ctx(); + await expect( + redisEngine.import(c, r(), [{ resourceId: 'r', kind: 'redis', path: 'r/dump.rdb' }]), + ).rejects.toThrow(/data directory/); + }); + + it('places the file when the target declares a data directory', async () => { + const c = ctx(); + const target = r({ connection: { url: secret('redis://h'), dataDir: plain('/var/lib/redis') } }); + await redisEngine.import(c, target, [{ resourceId: 'r', kind: 'redis', path: 'r/dump.rdb' }]); + expect(c.calls[0]!.args[1]).toBe('/var/lib/redis/dump.rdb'); + }); +}); + +describe('files', () => { + const local = (): Resource => ({ + kind: 'files', + id: 'v', + name: 'uploads', + connection: { path: plain('/data/uploads') }, + }); + const remote = (over: Record = {}): Resource => ({ + kind: 'files', + id: 'v2', + name: 'www', + connection: { + path: plain('/home/anthony/www'), + host: plain('dev2.example.com'), + user: plain('anthony'), + ...Object.fromEntries(Object.entries(over).map(([k, v]) => [k, plain(v)])), + }, + }); + + it('always ends an endpoint with a slash, so rsync copies contents not the directory', () => { + expect(endpoint(local())).toBe('/data/uploads/'); + expect(endpoint(remote())).toBe('anthony@dev2.example.com:/home/anthony/www/'); + }); + + it('knows which side is remote', () => { + expect(isRemote(local())).toBe(false); + expect(isRemote(remote())).toBe(true); + }); + + it('refuses remote-to-remote, which rsync cannot do at all', () => { + expect(() => assertCopyable(remote(), remote())).toThrow(/cannot copy directly between two remote/); + }); + + it('allows a copy when one side is local', () => { + expect(() => assertCopyable(remote(), local())).not.toThrow(); + expect(() => assertCopyable(local(), remote())).not.toThrow(); + }); + + it('never passes --delete, which is how a wrong target destroys a directory', async () => { + const c = ctx(); + await filesEngine.import(c, remote(), [], local()); + const args = c.calls[0]!.args; + expect(args).not.toContain('--delete'); + expect(args.some((a) => a.startsWith('--delete'))).toBe(false); + }); + + it('carries an ssh key and port into the transport when configured', async () => { + const c = ctx(); + await filesEngine.import(c, remote({ sshKeyPath: '/k/id', sshPort: '2222' }), [], local()); + const e = c.calls[0]!.args[c.calls[0]!.args.indexOf('-e') + 1]; + expect(e).toContain('-i /k/id'); + expect(e).toContain('-p 2222'); + }); +}); diff --git a/packages/migrate/src/engines/index.ts b/packages/migrate/src/engines/index.ts new file mode 100644 index 00000000..c7b1ac7d --- /dev/null +++ b/packages/migrate/src/engines/index.ts @@ -0,0 +1,36 @@ +import type { Engine, ResourceKind } from '../types.js'; +import { filesEngine } from './files.js'; +import { objectStorageEngine } from './object-storage.js'; +import { postgresEngine } from './postgres.js'; +import { redisEngine } from './redis.js'; +import { sqliteEngine } from './sqlite.js'; + +/** + * Every engine this build can run, keyed by what it moves. + * + * The planner takes this map and refuses, up front, to plan a migration for a + * kind that is not in it — which is the difference between "we do not support + * that" printed before anything happens and a crash after the freeze. + */ +export const ENGINES: ReadonlyMap = new Map([ + ['postgres', postgresEngine], + ['sqlite', sqliteEngine], + ['redis', redisEngine], + ['object-storage', objectStorageEngine], + ['files', filesEngine], +]); + +export function engineFor(kind: ResourceKind): Engine | undefined { + return ENGINES.get(kind); +} + +/** Every binary any engine needs, for a one-shot preflight check. */ +export function requiredBinaries(kinds: ResourceKind[]): string[] { + const out = new Set(); + for (const kind of kinds) { + for (const bin of ENGINES.get(kind)?.requires ?? []) out.add(bin); + } + return [...out].sort(); +} + +export { filesEngine, objectStorageEngine, postgresEngine, redisEngine, sqliteEngine }; diff --git a/packages/migrate/src/engines/redis.ts b/packages/migrate/src/engines/redis.ts new file mode 100644 index 00000000..c090611c --- /dev/null +++ b/packages/migrate/src/engines/redis.ts @@ -0,0 +1,98 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a Redis. + * + * Usually you should not. Redis is a cache far more often than it is a + * database, and the right migration for a cache is an empty one: point the new + * app at a new Redis and let it fill. Copying a cache moves stale entries and + * costs downtime for data that is worthless by definition. + * + * It is here because sometimes it is not a cache — a BullMQ queue with jobs + * waiting in it, a session store where copying nothing logs every user out. + * Those are real, so the engine exists, and the planner surfaces the question + * rather than deciding for you. + * + * There is no delta. A key written during the copy is simply missed, which is + * exactly why queues and sessions want the source stopped first. The planner + * warns about that because `delta` is absent. + */ + +const DUMP = 'dump.rdb'; + +function redisArgs(r: Resource): string[] { + const url = r.connection.url?.reveal(); + if (!url) throw new Error(`redis resource '${r.name}' has no connection.url`); + // redis-cli takes the whole URL, and unlike libpq there is no environment + // variable for it. The password is therefore visible in `ps` for as long as + // the command runs, which for --rdb is the length of the copy. Nothing can + // be done about that from here beyond keeping the window short; it is noted + // so nobody assumes otherwise. + return ['-u', url]; +} + +export const redisEngine: Engine = { + kind: 'redis', + requires: ['redis-cli'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`redis --rdb ${from.name}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'redis', path }]; + + // --rdb asks the server for a full sync and writes the RDB the replica + // would have received, which is consistent at a point in time. Reading + // keys with SCAN+DUMP instead is not: keys move under you as you walk. + await ctx.exec('redis-cli', [...redisArgs(from), '--rdb', `${ctx.staging.dir}/${path}`], { + timeoutMs: 2 * 60 * 60 * 1000, + }); + + const artifact: Artifact = { resourceId: from.id, kind: 'redis', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + /** + * There is no supported way to push an RDB into a running managed Redis, so + * this refuses rather than pretending. + * + * A self-hosted target takes the file directly: stop the server, drop the + * RDB in its data directory, start it. That is a host operation rather than + * a client one, so it is surfaced as an instruction instead of being done + * badly over the wire. + */ + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'redis'); + if (!dump) throw new Error(`no redis dump staged for '${to.name}'`); + + const dataDir = to.connection.dataDir?.reveal(); + if (!dataDir) { + throw new Error( + `Redis cannot be loaded over the wire. Copy ${dump.path} to the target's data directory as dump.rdb while the server is stopped, then start it. Set connection.dataDir on the target to have this done for you.`, + ); + } + + if (ctx.dryRun) { + ctx.log(`would place ${dump.path} at ${dataDir}/dump.rdb`); + return; + } + + await ctx.exec('cp', [`${ctx.staging.dir}/${dump.path}`, `${dataDir}/dump.rdb`]); + ctx.log(`placed dump.rdb in ${dataDir}; restart the target Redis to load it`, 'warn'); + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const size = async (r: Resource) => { + const res = await ctx.exec('redis-cli', [...redisArgs(r), 'dbsize'], { check: false }); + return Number.parseInt(res.stdout.trim(), 10) || 0; + }; + const [a, b] = await Promise.all([size(from), size(to)]); + return { + ok: b >= a, + checks: [`keys: ${a} → ${b}`], + problems: b < a ? [`${a - b} key(s) missing on the target`] : [], + }; + }, +}; diff --git a/packages/migrate/src/engines/sqlite.ts b/packages/migrate/src/engines/sqlite.ts new file mode 100644 index 00000000..dea5e04f --- /dev/null +++ b/packages/migrate/src/engines/sqlite.ts @@ -0,0 +1,122 @@ +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a SQLite or libSQL database. + * + * Turso is libSQL, which is SQLite with a server in front, and a local + * `app.db` is SQLite with nothing in front. Both dump to the same SQL text, + * which is why they are one engine — and why Turso→dedicated and + * dedicated→Turso are the same code path. + * + * The dump is plain SQL rather than a file copy. Copying the file works for a + * local database and not at all for a hosted one, and a SQLite file copied + * while something is writing to it is a corrupt file rather than an error. The + * text dump is slower and always correct. + * + * Note the `.dump` output includes `PRAGMA foreign_keys=OFF` and wraps in a + * transaction on its own, which is what lets the rows load in whatever order + * the dump emitted them. + */ + +const DUMP = 'dump.sql'; + +/** Turso's CLI talks to a named database; plain SQLite takes a file path. */ +function isTurso(r: Resource): boolean { + return Boolean(r.connection.tursoDatabase) || r.metadata?.platform === 'turso'; +} + +function tursoEnv(r: Resource): Record { + const token = r.connection.authToken?.reveal(); + return token ? { TURSO_API_TOKEN: token } : {}; +} + +function filePath(r: Resource): string { + const p = r.connection.path?.reveal(); + if (!p) throw new Error(`sqlite resource '${r.name}' has no connection.path`); + return p; +} + +export const sqliteEngine: Engine = { + kind: 'sqlite', + /* + * `sqlite3` only. Which binary is needed actually depends on the resource — + * a Turso database is reached with `turso`, a file with `sqlite3` — but + * `requires` is a property of the engine, not of one side of one migration. + * Declaring the union would block a file-to-file move on a missing Turso CLI + * nobody needs, so the Turso side is checked at the point of use instead and + * fails with a message naming the binary. + */ + requires: ['sqlite3'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`dumping ${from.name} → ${path}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'sqlite', path }]; + + const out = `${ctx.staging.dir}/${path}`; + if (isTurso(from)) { + const db = from.connection.tursoDatabase!.reveal(); + await ctx.exec('turso', ['db', 'shell', db, '.dump'], { + env: tursoEnv(from), + timeoutMs: 2 * 60 * 60 * 1000, + stdoutFile: out, + }); + } else { + await ctx.exec('sqlite3', [filePath(from), '.dump'], { + timeoutMs: 2 * 60 * 60 * 1000, + stdoutFile: out, + }); + } + + const artifact: Artifact = { resourceId: from.id, kind: 'sqlite', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'sqlite'); + if (!dump) throw new Error(`no sqlite dump staged for '${to.name}'`); + if (ctx.dryRun) { + ctx.log(`would load ${dump.path} into ${to.name}`); + return; + } + + const file = `${ctx.staging.dir}/${dump.path}`; + if (isTurso(to)) { + const db = to.connection.tursoDatabase!.reveal(); + await ctx.exec('turso', ['db', 'shell', db], { + env: tursoEnv(to), + timeoutMs: 2 * 60 * 60 * 1000, + stdinFile: file, + }); + } else { + await ctx.exec('sqlite3', [filePath(to), `.read ${file}`], { + timeoutMs: 2 * 60 * 60 * 1000, + }); + } + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const countTables = async (r: Resource) => { + const sql = + "select name from sqlite_master where type='table' and name not like 'sqlite_%' order by 1"; + const res = isTurso(r) + ? await ctx.exec('turso', ['db', 'shell', r.connection.tursoDatabase!.reveal(), sql], { + env: tursoEnv(r), + check: false, + }) + : await ctx.exec('sqlite3', [filePath(r), sql], { check: false }); + return res.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + }; + + const [a, b] = await Promise.all([countTables(from), countTables(to)]); + const missing = a.filter((t) => !b.includes(t)); + return { + ok: missing.length === 0, + checks: [`tables: ${a.length} → ${b.length}`], + problems: missing.map((t) => `${t}: missing on the target`), + }; + }, +}; diff --git a/packages/migrate/src/types.ts b/packages/migrate/src/types.ts index 9b6bc8df..32f0d22d 100644 --- a/packages/migrate/src/types.ts +++ b/packages/migrate/src/types.ts @@ -224,6 +224,17 @@ export interface ExecOptions { /** Fail the step if the command exits non-zero. Default true. */ check?: boolean; timeoutMs?: number; + /** + * Send stdout to this absolute path instead of buffering it. + * + * Some tools only dump to stdout — `sqlite3 .dump`, `turso db shell .dump` — + * and a multi-gigabyte dump must not be held in a string. Expressed as an + * option rather than a shell redirect because `exec` runs without a shell, + * so `>` would be passed to the program as a literal argument. + */ + stdoutFile?: string; + /** Feed this file to stdin. The load half of the same problem. */ + stdinFile?: string; } export interface ExecResult { From 6f915cec1e42379dfa7c21d445554bc24aacbde8 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 07:53:32 +0000 Subject: [PATCH 5/8] migrate: the platform layer, and with it every direction Platforms resolve credentials and enumerate resources. They never move a byte, which is what lets one implementation serve both directions. ssh is a box you own, and it is `both` because "get off the cloud" and "we tried bare metal and went back" are the same code path. A server has no API to enumerate itself, so it is described in config rather than discovered -- not a workaround: a directory has no metadata saying "this is the uploads volume", and guessing from paths would be worse than being told. It does not create databases remotely either; choosing disks and versions and a backup story on someone's server is not a decision a migration tool should make quietly. Railway finds databases by RECOGNISING CONNECTION STRINGS IN VARIABLES rather than asking for a list of databases, because that list does not exist in that shape -- a Postgres service's DSN is DATABASE_URL on the services that use it. Classification is by URL scheme, never by variable name: DATABASE_URL is a convention and plenty of apps use PG_URL or something bespoke, and silently skipping the database because it was called the wrong thing is the worst failure available. The same DSN injected into six services is moved once. Supabase is a Postgres with a lot bolted on, and the bolted-on parts are what make leaving it interesting. Five quirks are named up front, all from the crawlproof migration: auth.users restores as data but GoTrue and the JWT secret do not, so everyone is logged out unless the secret comes too; RLS policies reference anon/authenticated/service_role, which do not exist on a plain Postgres and must be created first; pg_cron jobs restore already enabled and start firing immediately. It also emits the exact --rewrite-host flag for its own public hostname, so the absolute-URL trap is a copyable line rather than a thing to remember. Turso, Neon, PlanetScale, Fly, Render, Heroku and Vercel differ enormously as products and barely at all here: each hands over a DSN and the engines do the rest. Six of them share one factory because writing a file each would be six copies of twenty lines. Turso gets a real implementation, since its CLI works on a database name plus a token rather than a connection string. compatibleKinds() is the whole design in one function: intersect what the source holds with what the target accepts. Turso->Neon returns empty (sqlite vs postgres) and says so in a millisecond instead of failing at a cutover. Supabase->ssh and ssh->Supabase both return postgres, and neither is a code path anyone wrote. 141 tests, tsc clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/migrate/src/platforms/index.test.ts | 229 +++++++++++++++++++ packages/migrate/src/platforms/index.ts | 56 +++++ packages/migrate/src/platforms/managed.ts | 212 +++++++++++++++++ packages/migrate/src/platforms/railway.ts | 171 ++++++++++++++ packages/migrate/src/platforms/ssh.ts | 193 ++++++++++++++++ packages/migrate/src/platforms/supabase.ts | 156 +++++++++++++ 6 files changed, 1017 insertions(+) create mode 100644 packages/migrate/src/platforms/index.test.ts create mode 100644 packages/migrate/src/platforms/index.ts create mode 100644 packages/migrate/src/platforms/managed.ts create mode 100644 packages/migrate/src/platforms/railway.ts create mode 100644 packages/migrate/src/platforms/ssh.ts create mode 100644 packages/migrate/src/platforms/supabase.ts diff --git a/packages/migrate/src/platforms/index.test.ts b/packages/migrate/src/platforms/index.test.ts new file mode 100644 index 00000000..0c7264ac --- /dev/null +++ b/packages/migrate/src/platforms/index.test.ts @@ -0,0 +1,229 @@ +import { describe, expect, it } from 'vitest'; +import { PLATFORMS, compatibleKinds, platformById, sources, targets } from './index.js'; +import { classifyConnection, connectionsFromVariables } from './railway.js'; +import { + SUPABASE_QUIRKS, + supabaseDsn, + supabasePlatform, + supabasePublicHost, + supabaseS3Endpoint, +} from './supabase.js'; +import { sshPlatform } from './ssh.js'; +import { tursoPlatform } from './managed.js'; +import type { PlatformContext } from '../types.js'; + +const ctx = (secrets: Record = {}): PlatformContext => ({ + secret: (k) => secrets[k], + log: () => {}, + dryRun: false, +}); + +describe('the platform registry', () => { + it('gives every platform a unique id', () => { + const ids = PLATFORMS.map((p) => p.id); + expect(new Set(ids).size).toBe(ids.length); + }); + + it('declares at least one resource kind per platform', () => { + for (const p of PLATFORMS) expect(p.supports.length).toBeGreaterThan(0); + }); + + it('gives every target a way to resolve an incoming resource', () => { + for (const p of PLATFORMS) { + if (p.role === 'source') continue; + expect(typeof p.provision === 'function' || p.id === 'railway').toBe(true); + } + }); + + it('finds a platform by id', () => { + expect(platformById('supabase')?.label).toBe('Supabase'); + expect(platformById('nope')).toBeUndefined(); + }); + + it('lists sources and targets separately', () => { + expect(sources().length).toBeGreaterThan(0); + expect(targets().length).toBeGreaterThan(0); + }); +}); + +describe('compatibleKinds is what makes direction a non-question', () => { + it('finds postgres in both directions between Supabase and a box', () => { + expect(compatibleKinds('supabase', 'ssh')).toContain('postgres'); + expect(compatibleKinds('ssh', 'supabase')).toContain('postgres'); + }); + + it('pairs Turso with a box over sqlite, both ways', () => { + expect(compatibleKinds('turso', 'ssh')).toEqual(['sqlite']); + expect(compatibleKinds('ssh', 'turso')).toEqual(['sqlite']); + }); + + it('reports an empty intersection rather than pretending a migration is possible', () => { + // Turso holds sqlite; Neon accepts only postgres. + expect(compatibleKinds('turso', 'neon')).toEqual([]); + }); + + it('is empty for an unknown platform', () => { + expect(compatibleKinds('turso', 'nope')).toEqual([]); + }); +}); + +describe('railway connection discovery', () => { + it('classifies by scheme, not by variable name', () => { + expect(classifyConnection('postgresql://u:p@h/d')).toBe('postgres'); + expect(classifyConnection('postgres://u:p@h/d')).toBe('postgres'); + expect(classifyConnection('mysql://u:p@h/d')).toBe('mysql'); + expect(classifyConnection('redis://h:6379')).toBe('redis'); + expect(classifyConnection('rediss://h:6379')).toBe('redis'); + expect(classifyConnection('libsql://x.turso.io')).toBe('sqlite'); + }); + + it('ignores a variable that is not a connection string', () => { + expect(classifyConnection('production')).toBeUndefined(); + expect(classifyConnection('https://example.com')).toBeUndefined(); + }); + + it('finds a database under a non-standard variable name', () => { + const found = connectionsFromVariables([ + { name: 'NODE_ENV', value: 'production' }, + { name: 'PG_URI', value: 'postgres://u:p@h/d' }, + ]); + expect(found).toHaveLength(1); + expect(found[0]?.kind).toBe('postgres'); + }); + + it('moves a database once even when several services share it', () => { + const found = connectionsFromVariables([ + { name: 'DATABASE_URL', value: 'postgres://u:p@h/d' }, + { name: 'DATABASE_URL', value: 'postgres://u:p@h/d' }, + { name: 'PG_URL', value: 'postgres://u:p@h/d' }, + ]); + expect(found).toHaveLength(1); + }); +}); + +describe('supabase', () => { + it('builds the direct DSN, not the pooler, which cannot serve a dump', () => { + const dsn = supabaseDsn({ projectRef: 'abc123', dbPassword: 'pw' }); + expect(dsn).toContain('db.abc123.supabase.co:5432'); + }); + + it('url-encodes a password with reserved characters', () => { + expect(supabaseDsn({ projectRef: 'r', dbPassword: 'p@ss/word' })).toContain('p%40ss%2Fword'); + }); + + it('prefers an explicit databaseUrl', () => { + expect(supabaseDsn({ projectRef: 'r', databaseUrl: 'postgres://custom/h' })).toBe( + 'postgres://custom/h', + ); + }); + + it('refuses when it has neither', () => { + expect(() => supabaseDsn({ projectRef: 'r' })).toThrow(/databaseUrl or dbPassword/); + }); + + it('derives the s3 endpoint and the public host', () => { + expect(supabaseS3Endpoint('abc')).toBe('https://abc.supabase.co/storage/v1/s3'); + expect(supabasePublicHost('abc')).toBe('abc.supabase.co'); + }); + + it('names the things a pg_dump does not carry', async () => { + const inv = await supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw' }); + const quirks = inv.resources[0]?.quirks ?? []; + expect(quirks).toEqual([...SUPABASE_QUIRKS]); + expect(quirks.some((q) => /JWT/.test(q))).toBe(true); + expect(quirks.some((q) => /pg_cron/.test(q))).toBe(true); + expect(quirks.some((q) => /anon, authenticated/.test(q))).toBe(true); + }); + + it('tells you to rewrite its public host before the project is deleted', async () => { + const inv = await supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw' }); + expect(inv.notes?.[0]).toContain('--rewrite-host abc.supabase.co='); + }); + + it('refuses a bucket with no service role key', async () => { + await expect( + supabasePlatform.inventory(ctx(), { projectRef: 'abc', dbPassword: 'pw', buckets: ['ads'] }), + ).rejects.toThrow(/SERVICE_ROLE_KEY/); + }); + + it('builds an s3 connection for a bucket when the key is present', async () => { + const inv = await supabasePlatform.inventory(ctx({ SUPABASE_SERVICE_ROLE_KEY: 'svc' }), { + projectRef: 'abc', + dbPassword: 'pw', + buckets: ['ads'], + }); + const bucket = inv.resources.find((r) => r.kind === 'object-storage'); + expect(bucket?.connection.endpoint?.reveal()).toBe('https://abc.supabase.co/storage/v1/s3'); + expect(bucket?.connection.secretAccessKey?.describe()).not.toContain('svc'); + }); +}); + +describe('ssh', () => { + it('describes a box from config, since a server has no API to enumerate itself', async () => { + const inv = await sshPlatform.inventory(ctx(), { + host: 'dev2.example.com', + user: 'anthony', + postgres: [{ name: 'app', url: 'postgres://u:p@localhost/app' }], + files: [{ name: 'www', path: '/home/anthony/www' }], + }); + expect(inv.resources.map((r) => r.kind).sort()).toEqual(['files', 'postgres']); + expect(inv.scope).toBe('dev2.example.com'); + }); + + it('carries ssh details onto file resources so rsync can reach them', async () => { + const inv = await sshPlatform.inventory(ctx(), { + host: 'h', + user: 'u', + sshKeyPath: '/k', + files: [{ name: 'www', path: '/w' }], + }); + const files = inv.resources[0]!; + expect(files.connection.host?.reveal()).toBe('h'); + expect(files.connection.sshKeyPath?.reveal()).toBe('/k'); + }); + + it('says so when nothing is declared, rather than reporting an empty box', async () => { + const inv = await sshPlatform.inventory(ctx(), { host: 'h' }); + expect(inv.notes?.[0]).toMatch(/no API to enumerate itself/); + }); + + it('resolves an incoming resource against the declared target', async () => { + const out = await sshPlatform.provision( + ctx(), + { kind: 'postgres', id: 'pg', name: 'app', connection: {} }, + { host: 'h', postgres: [{ name: 'app', url: 'postgres://u:p@localhost/app' }] }, + ); + expect(out.connection.url?.reveal()).toContain('localhost/app'); + }); + + it('refuses when the target declares nowhere to put it', async () => { + await expect( + sshPlatform.provision(ctx(), { kind: 'postgres', id: 'pg', name: 'app', connection: {} }, { host: 'h' }), + ).rejects.toThrow(/nothing on h is declared/); + }); + + it('needs a host', async () => { + await expect(sshPlatform.inventory(ctx(), { host: '' })).rejects.toThrow(/needs a host/); + }); +}); + +describe('turso', () => { + it('works on a database name plus a token rather than a DSN', async () => { + const inv = await tursoPlatform.inventory(ctx({ TURSO_API_TOKEN: 'tok' }), { database: 'prod' }); + expect(inv.resources[0]?.connection.tursoDatabase?.reveal()).toBe('prod'); + expect(inv.resources[0]?.metadata?.platform).toBe('turso'); + }); + + it('refuses without a token', async () => { + await expect(tursoPlatform.inventory(ctx(), { database: 'prod' })).rejects.toThrow(/TURSO_API_TOKEN/); + }); + + it('can also receive, which is what makes dedicated→Turso possible', async () => { + const out = await tursoPlatform.provision!( + ctx({ TURSO_API_TOKEN: 'tok' }), + { kind: 'sqlite', id: 'x', name: 'app', connection: {} }, + { database: 'restored' }, + ); + expect(out.connection.tursoDatabase?.reveal()).toBe('restored'); + }); +}); diff --git a/packages/migrate/src/platforms/index.ts b/packages/migrate/src/platforms/index.ts new file mode 100644 index 00000000..faac3310 --- /dev/null +++ b/packages/migrate/src/platforms/index.ts @@ -0,0 +1,56 @@ +import type { Platform } from '../types.js'; +import { MANAGED_PLATFORMS, tursoPlatform } from './managed.js'; +import { railwayPlatform } from './railway.js'; +import { sshPlatform } from './ssh.js'; +import { supabasePlatform } from './supabase.js'; + +/** + * Every platform this build can read from or write to. + * + * Direction is not encoded here. A platform declares whether it can be a + * source, a target or both, and the planner pairs any two — so the number of + * supported migrations is the number of pairs, not the number of entries. + */ +// biome-ignore lint/suspicious/noExplicitAny: the registry is heterogeneous by +// design — each platform has its own config type, and the CLI resolves the +// right one from a config file at runtime. +export const PLATFORMS: ReadonlyArray> = [ + sshPlatform, + railwayPlatform, + supabasePlatform, + tursoPlatform, + ...MANAGED_PLATFORMS, +]; + +// biome-ignore lint/suspicious/noExplicitAny: see above. +export function platformById(id: string): Platform | undefined { + return PLATFORMS.find((p) => p.id === id); +} + +/** Every platform that can be the left-hand side of a migration. */ +export function sources(): ReadonlyArray> { + return PLATFORMS.filter((p) => p.role !== 'target'); +} + +/** Every platform that can be the right-hand side. */ +export function targets(): ReadonlyArray> { + return PLATFORMS.filter((p) => p.role !== 'source'); +} + +/** + * Whether a pair can move anything at all, and what. + * + * The intersection of what the source holds and what the target accepts. An + * empty intersection is a migration that cannot happen, and saying so takes a + * millisecond instead of a failed cutover. + */ +export function compatibleKinds(fromId: string, toId: string): string[] { + const from = platformById(fromId); + const to = platformById(toId); + if (!from || !to) return []; + const accepted = new Set(to.supports); + return from.supports.filter((k) => accepted.has(k)); +} + +export { railwayPlatform, sshPlatform, supabasePlatform, tursoPlatform }; +export * from './managed.js'; diff --git a/packages/migrate/src/platforms/managed.ts b/packages/migrate/src/platforms/managed.ts new file mode 100644 index 00000000..5be4335d --- /dev/null +++ b/packages/migrate/src/platforms/managed.ts @@ -0,0 +1,212 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * The managed database and app providers that are, from a migration's point of + * view, a connection string with a brand on it. + * + * Turso, Neon, PlanetScale, Fly, Render, Heroku and Vercel differ enormously + * as products and barely at all here: each one hands over a DSN (or a database + * name plus a token) and the engines do the rest. Writing a file per vendor + * would be six copies of the same twenty lines, so they share one factory and + * differ only where they actually differ. + * + * That is the payoff of the platform/engine split stated concretely. Neon to + * dedicated, dedicated to Neon, Neon to Supabase and Railway to Neon are all + * the same two engines; none of them is a code path anyone wrote. + */ + +export interface DsnConfig { + /** The connection string. Takes precedence over anything else. */ + url?: string; + /** Name for the resource in the plan. Defaults to the platform id. */ + name?: string; +} + +export interface TursoConfig { + database: string; + authToken?: string; +} + +interface ManagedSpec { + id: string; + label: string; + kind: Resource['kind']; + /** Environment variable holding the DSN when config does not carry it. */ + envVar: string; + role?: Platform['role']; + quirks?: string[]; + notes?: string[]; +} + +/** + * Build a platform whose entire job is producing one connection string. + * + * `role` defaults to `both`: every one of these can be written to as readily + * as read from, which is what makes the tool bidirectional without a second + * implementation. + */ +function dsnPlatform(spec: ManagedSpec): Platform { + return { + id: spec.id, + label: spec.label, + role: spec.role ?? 'both', + supports: [spec.kind], + + async inventory(ctx: PlatformContext, config: DsnConfig): Promise { + const url = config.url ?? ctx.secret(spec.envVar); + if (!url) { + throw new Error(`${spec.label} needs a connection string: set ${spec.envVar} or pass url`); + } + const name = config.name ?? spec.id; + return { + platform: spec.id, + scope: name, + resources: [ + { + kind: spec.kind, + id: `${spec.id}-${name}`, + name, + connection: { url: secret(url) }, + ...(spec.quirks ? { quirks: [...spec.quirks] } : {}), + }, + ], + ...(spec.notes ? { notes: [...spec.notes] } : {}), + }; + }, + + async provision(_ctx: PlatformContext, resource: Resource, config: DsnConfig): Promise { + const url = config.url; + if (!url) throw new Error(`${spec.label} target needs a connection string`); + if (resource.kind !== spec.kind) { + throw new Error(`${spec.label} cannot receive a ${resource.kind}`); + } + return { ...resource, id: `${spec.id}-${resource.name}`, connection: { url: secret(url) } }; + }, + + async check(ctx: PlatformContext, config: DsnConfig): Promise { + if (!(config.url ?? ctx.secret(spec.envVar))) { + throw new Error(`${spec.label}: no connection string (${spec.envVar})`); + } + }, + }; +} + +export const neonPlatform = dsnPlatform({ + id: 'neon', + label: 'Neon', + kind: 'postgres', + envVar: 'NEON_DATABASE_URL', + quirks: [ + 'Neon branches are copy-on-write and do not survive a dump; only the branch you point at is moved', + ], +}); + +export const planetscalePlatform = dsnPlatform({ + id: 'planetscale', + label: 'PlanetScale', + kind: 'mysql', + envVar: 'PLANETSCALE_DATABASE_URL', + quirks: [ + 'PlanetScale does not support foreign key constraints in the usual way, so a dump taken here may restore without the constraints a plain MySQL would expect', + ], +}); + +export const flyPostgresPlatform = dsnPlatform({ + id: 'fly', + label: 'Fly.io Postgres', + kind: 'postgres', + envVar: 'FLY_DATABASE_URL', + quirks: ['a Fly Postgres is reachable only over the private network unless proxied; run `fly proxy` first'], +}); + +export const renderPlatform = dsnPlatform({ + id: 'render', + label: 'Render', + kind: 'postgres', + envVar: 'RENDER_DATABASE_URL', +}); + +export const herokuPlatform = dsnPlatform({ + id: 'heroku', + label: 'Heroku Postgres', + kind: 'postgres', + envVar: 'HEROKU_DATABASE_URL', + quirks: [ + 'Heroku rotates DATABASE_URL without warning; resolve it at the moment of use rather than caching it across a long migration', + ], +}); + +export const vercelPostgresPlatform = dsnPlatform({ + id: 'vercel', + label: 'Vercel Postgres', + kind: 'postgres', + envVar: 'POSTGRES_URL', + notes: [ + 'Vercel Postgres is Neon underneath; the non-pooling POSTGRES_URL_NON_POOLING is the one a dump wants', + ], +}); + +/** + * Turso, which is the one that does not fit the DSN mould. + * + * Its CLI works on a database NAME plus an account token rather than a + * connection string, so it gets a real implementation rather than a factory + * call. The sqlite engine already knows the difference. + */ +export const tursoPlatform: Platform = { + id: 'turso', + label: 'Turso', + role: 'both', + supports: ['sqlite'], + + async inventory(ctx: PlatformContext, config: TursoConfig): Promise { + if (!config.database) throw new Error('Turso needs a database name'); + const token = config.authToken ?? ctx.secret('TURSO_API_TOKEN'); + if (!token) throw new Error('Turso needs TURSO_API_TOKEN'); + + return { + platform: 'turso', + scope: config.database, + resources: [ + { + kind: 'sqlite', + id: `turso-${config.database}`, + name: config.database, + connection: { tursoDatabase: plain(config.database), authToken: secret(token) }, + metadata: { platform: 'turso' }, + quirks: [ + 'embedded replicas and their sync state do not move; only the primary database is dumped', + ], + }, + ], + }; + }, + + async provision(ctx: PlatformContext, resource: Resource, config: TursoConfig): Promise { + if (resource.kind !== 'sqlite') throw new Error(`Turso cannot receive a ${resource.kind}`); + const token = config.authToken ?? ctx.secret('TURSO_API_TOKEN'); + if (!token) throw new Error('Turso needs TURSO_API_TOKEN'); + return { + ...resource, + id: `turso-${config.database}`, + connection: { tursoDatabase: plain(config.database), authToken: secret(token) }, + metadata: { ...resource.metadata, platform: 'turso' }, + }; + }, + + async check(ctx: PlatformContext, config: TursoConfig): Promise { + if (!(config.authToken ?? ctx.secret('TURSO_API_TOKEN'))) { + throw new Error('Turso: no TURSO_API_TOKEN'); + } + }, +}; + +export const MANAGED_PLATFORMS = [ + neonPlatform, + planetscalePlatform, + flyPostgresPlatform, + renderPlatform, + herokuPlatform, + vercelPostgresPlatform, +] as const; diff --git a/packages/migrate/src/platforms/railway.ts b/packages/migrate/src/platforms/railway.ts new file mode 100644 index 00000000..37026967 --- /dev/null +++ b/packages/migrate/src/platforms/railway.ts @@ -0,0 +1,171 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * Railway. + * + * A Railway project is services plus volumes plus variables, and the data + * worth moving hides in the variables: a Postgres service's connection string + * is `DATABASE_URL` on the services that use it, not something the API hands + * over as a database object. So the inventory reads variables and recognises + * connection strings in them, rather than asking for a list of databases that + * does not exist in that shape. + * + * `role: 'both'` — Railway is a perfectly good destination, and "we tried + * bare metal and went back" is a migration people actually make. Provisioning + * a service here is not automated: creating billable infrastructure is a + * decision, and `sh1pt deploy` is where that lives. What this does is resolve + * an existing service's connection so data can be written into it. + */ + +const API = 'https://backboard.railway.app/graphql/v2'; + +export interface RailwayConfig { + projectId: string; + environmentId?: string; + /** Overrides the RAILWAY_TOKEN secret when set. */ + token?: string; +} + +interface GqlVariable { + name: string; + value: string; +} + +async function gql(token: string, query: string, variables: Record): Promise { + const res = await fetch(API, { + method: 'POST', + headers: { Authorization: `Bearer ${token}`, 'Content-Type': 'application/json' }, + body: JSON.stringify({ query, variables }), + }); + const json = (await res.json()) as { data?: T; errors?: Array<{ message: string }> }; + if (json.errors?.length) throw new Error(`Railway: ${json.errors[0]!.message}`); + if (!res.ok) throw new Error(`Railway HTTP ${res.status}`); + return json.data as T; +} + +/** + * Recognise a connection string by its scheme. + * + * Deliberately not by variable name. `DATABASE_URL` is the convention but + * plenty of apps use `PG_URL`, `POSTGRES_URI` or something bespoke, and a + * migration that silently skipped the database because it was called the wrong + * thing would be the worst possible failure. The scheme is the fact. + */ +export function classifyConnection(value: string): Resource['kind'] | undefined { + if (/^postgres(ql)?:\/\//i.test(value)) return 'postgres'; + if (/^mysql:\/\//i.test(value)) return 'mysql'; + if (/^rediss?:\/\//i.test(value)) return 'redis'; + if (/^libsql:\/\//i.test(value)) return 'sqlite'; + return undefined; +} + +/** Variables whose value is a connection string, deduplicated by target. */ +export function connectionsFromVariables(vars: GqlVariable[]): Array<{ + kind: Resource['kind']; + name: string; + url: string; +}> { + const seen = new Set(); + const out: Array<{ kind: Resource['kind']; name: string; url: string }> = []; + for (const v of vars) { + const kind = classifyConnection(v.value); + if (!kind) continue; + // The same database is usually injected into several services under the + // same or different names; moving it once is the point. + if (seen.has(v.value)) continue; + seen.add(v.value); + out.push({ kind, name: v.name, url: v.value }); + } + return out; +} + +export const railwayPlatform: Platform = { + id: 'railway', + label: 'Railway', + role: 'both', + supports: ['postgres', 'mysql', 'redis', 'sqlite', 'files'], + + async inventory(ctx: PlatformContext, config: RailwayConfig): Promise { + const token = config.token ?? ctx.secret('RAILWAY_TOKEN'); + if (!token) throw new Error('Railway needs a RAILWAY_TOKEN'); + if (!config.projectId) throw new Error('Railway needs a projectId'); + + const data = await gql<{ + project: { + name: string; + services: { edges: Array<{ node: { id: string; name: string } }> }; + volumes: { edges: Array<{ node: { id: string; name: string; mountPath?: string } }> }; + }; + }>( + token, + `query ($id: String!) { + project(id: $id) { + name + services { edges { node { id name } } } + volumes { edges { node { id name } } } + } + }`, + { id: config.projectId }, + ); + + const resources: Resource[] = []; + const notes: string[] = []; + + for (const edge of data.project.volumes.edges) { + const v = edge.node; + resources.push({ + kind: 'files', + id: `volume-${v.id}`, + name: v.name, + // A Railway volume is only reachable from inside a service, so it + // cannot be rsynced from here. Recorded so the plan shows it and a + // person decides, rather than being silently dropped. + connection: { path: plain(v.mountPath ?? '/data') }, + quirks: [ + 'a Railway volume is only reachable from inside its service; copy it out with `railway run` or a one-off container rather than over ssh', + ], + }); + } + + for (const edge of data.project.services.edges) { + const service = edge.node; + const vars = await gql<{ variables: GqlVariable[] }>( + token, + `query ($projectId: String!, $serviceId: String!, $environmentId: String) { + variables(projectId: $projectId, serviceId: $serviceId, environmentId: $environmentId) { name value } + }`, + { + projectId: config.projectId, + serviceId: service.id, + environmentId: config.environmentId ?? null, + }, + ).catch(() => ({ variables: [] as GqlVariable[] })); + + for (const conn of connectionsFromVariables(vars.variables ?? [])) { + resources.push({ + kind: conn.kind, + id: `${service.id}-${conn.name}`, + name: `${service.name}/${conn.name}`, + connection: { url: secret(conn.url) }, + metadata: { service: service.name, variable: conn.name }, + }); + } + } + + if (!resources.length) { + notes.push( + 'No connection strings were found in this project. Either the token cannot read variables, or the databases are referenced another way.', + ); + } + + ctx.log(`${data.project.name}: ${resources.length} resource(s)`); + return { platform: 'railway', scope: data.project.name, resources, notes }; + }, + + async check(ctx: PlatformContext, config: RailwayConfig): Promise { + const token = config.token ?? ctx.secret('RAILWAY_TOKEN'); + if (!token) throw new Error('Railway needs a RAILWAY_TOKEN'); + await gql(token, `query ($id: String!) { project(id: $id) { id } }`, { id: config.projectId }); + }, +}; diff --git a/packages/migrate/src/platforms/ssh.ts b/packages/migrate/src/platforms/ssh.ts new file mode 100644 index 00000000..36c85772 --- /dev/null +++ b/packages/migrate/src/platforms/ssh.ts @@ -0,0 +1,193 @@ +import type { Inventory, Platform, PlatformContext, Resource, Secretish } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * A box you own: a VPS, a dedicated server, the thing at the other end of an + * ssh config entry. + * + * This is the platform every "get off the cloud" migration ends at, and the + * one every "we need managed after all" migration starts from. It is `both` + * because those are the same code path, which is the entire point of splitting + * platforms from engines. + * + * Unlike a managed provider there is no API to ask what is here, so a box is + * described rather than discovered. That is not a workaround: a directory on a + * server has no metadata saying "this is the uploads volume", and guessing + * from paths would be worse than being told. The description lives in config, + * which means it is reviewable and it is the same on every run. + */ + +export interface SshConfig { + host: string; + user?: string; + sshKeyPath?: string; + sshPort?: string; + /** Databases reachable from this box, usually on loopback. */ + postgres?: Array<{ name: string; url: string }>; + sqlite?: Array<{ name: string; path: string }>; + redis?: Array<{ name: string; url: string; dataDir?: string }>; + /** Directories to move: volumes, docroots, upload trees. */ + files?: Array<{ name: string; path: string }>; + /** An S3-compatible endpoint running on the box, e.g. MinIO. */ + buckets?: Array<{ + name: string; + bucket: string; + endpoint: string; + accessKeyId: string; + secretAccessKey: string; + region?: string; + }>; +} + +export const sshPlatform: Platform = { + id: 'ssh', + label: 'dedicated / VPS over ssh', + role: 'both', + supports: ['postgres', 'sqlite', 'redis', 'files', 'object-storage'], + + async inventory(ctx: PlatformContext, config: SshConfig): Promise { + if (!config.host) throw new Error('ssh platform needs a host'); + + const resources: Resource[] = []; + const notes: string[] = []; + + for (const db of config.postgres ?? []) { + resources.push({ + kind: 'postgres', + id: `pg-${db.name}`, + name: db.name, + connection: { url: secret(db.url) }, + }); + } + + for (const db of config.sqlite ?? []) { + resources.push({ + kind: 'sqlite', + id: `sqlite-${db.name}`, + name: db.name, + connection: { path: plain(db.path) }, + }); + } + + for (const r of config.redis ?? []) { + resources.push({ + kind: 'redis', + id: `redis-${r.name}`, + name: r.name, + connection: { + url: secret(r.url), + ...(r.dataDir ? { dataDir: plain(r.dataDir) } : {}), + }, + }); + } + + for (const dir of config.files ?? []) { + resources.push({ + kind: 'files', + id: `files-${dir.name}`, + name: dir.name, + connection: { + path: plain(dir.path), + host: plain(config.host), + ...(config.user ? { user: plain(config.user) } : {}), + ...(config.sshKeyPath ? { sshKeyPath: plain(config.sshKeyPath) } : {}), + ...(config.sshPort ? { sshPort: plain(config.sshPort) } : {}), + }, + }); + } + + for (const b of config.buckets ?? []) { + resources.push({ + kind: 'object-storage', + id: `bucket-${b.name}`, + name: b.name, + connection: { + bucket: plain(b.bucket), + endpoint: plain(b.endpoint), + accessKeyId: secret(b.accessKeyId), + secretAccessKey: secret(b.secretAccessKey), + ...(b.region ? { region: plain(b.region) } : {}), + }, + }); + } + + if (!resources.length) { + notes.push( + 'Nothing is declared for this box. A server has no API to enumerate itself, so its databases and directories have to be listed in config.', + ); + } + + ctx.log(`${config.host}: ${resources.length} declared resource(s)`); + return { platform: 'ssh', scope: config.host, resources, notes }; + }, + + /** + * The target side of a move onto a box. + * + * A resource arriving here keeps its name and gets this box's connection + * details. Nothing is created remotely: the database or directory is + * expected to exist, because provisioning a Postgres on someone's server is + * a decision about disks and versions and backups that a migration tool + * should not be quietly making. + */ + async provision(ctx: PlatformContext, resource: Resource, config: SshConfig): Promise { + const match = ((): Record | undefined => { + switch (resource.kind) { + case 'postgres': { + const db = config.postgres?.find((d) => d.name === resource.name) ?? config.postgres?.[0]; + return db ? { url: secret(db.url) } : undefined; + } + case 'sqlite': { + const db = config.sqlite?.find((d) => d.name === resource.name) ?? config.sqlite?.[0]; + return db ? { path: plain(db.path) } : undefined; + } + case 'redis': { + const r = config.redis?.find((d) => d.name === resource.name) ?? config.redis?.[0]; + return r + ? { url: secret(r.url), ...(r.dataDir ? { dataDir: plain(r.dataDir) } : {}) } + : undefined; + } + case 'files': { + const d = config.files?.find((x) => x.name === resource.name) ?? config.files?.[0]; + return d + ? { + path: plain(d.path), + host: plain(config.host), + ...(config.user ? { user: plain(config.user) } : {}), + ...(config.sshKeyPath ? { sshKeyPath: plain(config.sshKeyPath) } : {}), + ...(config.sshPort ? { sshPort: plain(config.sshPort) } : {}), + } + : undefined; + } + case 'object-storage': { + const b = config.buckets?.find((x) => x.name === resource.name) ?? config.buckets?.[0]; + return b + ? { + bucket: plain(b.bucket), + endpoint: plain(b.endpoint), + accessKeyId: secret(b.accessKeyId), + secretAccessKey: secret(b.secretAccessKey), + ...(b.region ? { region: plain(b.region) } : {}), + } + : undefined; + } + default: + return undefined; + } + })(); + + if (!match) { + throw new Error( + `nothing on ${config.host} is declared to receive ${resource.kind} '${resource.name}'. Add it to the target config.`, + ); + } + + ctx.log(`${resource.name} → ${config.host}`); + return { ...resource, id: `${resource.id}@${config.host}`, connection: match }; + }, + + async check(ctx: PlatformContext, config: SshConfig): Promise { + if (!config.host) throw new Error('ssh platform needs a host'); + ctx.log(`target box is ${config.user ? `${config.user}@` : ''}${config.host}`); + }, +}; diff --git a/packages/migrate/src/platforms/supabase.ts b/packages/migrate/src/platforms/supabase.ts new file mode 100644 index 00000000..f8900b5c --- /dev/null +++ b/packages/migrate/src/platforms/supabase.ts @@ -0,0 +1,156 @@ +import type { Inventory, Platform, PlatformContext, Resource } from '../types.js'; +import { plain, secret } from '../types.js'; + +/** + * Supabase. + * + * A Supabase project is a Postgres with a lot bolted on, and the bolted-on + * parts are what make migrating off it interesting. The database moves with + * pg_dump like any other Postgres; storage is S3-compatible and moves with the + * object-storage engine. What does not move cleanly is everything in between, + * and the value of this inventory is naming those things up front as quirks so + * the planner warns rather than letting them be discovered afterwards. + * + * The list comes from a migration that actually happened (crawlproof.com, + * 2026-09-24): 128 tables, 113 users in `auth.users`, 204 functions, 10 + * pg_cron jobs, a realtime publication, and three public buckets holding 8,410 + * objects. + * + * ## The ones that bite + * + * `auth.users` is a real table in the dump, so users migrate — but the GoTrue + * service that reads it does not, and neither do the JWT secrets. Restoring + * `auth.users` somewhere with a different JWT secret logs everyone out and + * invalidates every refresh token. + * + * Row Level Security policies reference roles (`authenticated`, `anon`, + * `service_role`) that do not exist on a plain Postgres. The policies restore; + * the roles have to be created first or every policy fails. + * + * pg_cron jobs restore and start running immediately, which is how a migration + * ends up with two schedulers on one dataset. + */ + +export interface SupabaseConfig { + projectRef: string; + /** The database password; the rest of the DSN is derived from the ref. */ + dbPassword?: string; + /** Full DSN, when the project uses a pooler or a custom host. */ + databaseUrl?: string; + serviceRoleKey?: string; + /** Buckets to move. Supabase's S3 endpoint is derived from the ref. */ + buckets?: string[]; + region?: string; +} + +/** + * Supabase's direct-connection DSN for a project. + * + * Direct rather than the pooler: pgbouncer in transaction mode does not + * support the session-level operations a dump and restore need, and the + * failure is a confusing mid-dump error rather than a refusal. + */ +export function supabaseDsn(config: SupabaseConfig): string { + if (config.databaseUrl) return config.databaseUrl; + if (!config.dbPassword) { + throw new Error('Supabase needs either databaseUrl or dbPassword'); + } + const pw = encodeURIComponent(config.dbPassword); + return `postgresql://postgres:${pw}@db.${config.projectRef}.supabase.co:5432/postgres`; +} + +/** The S3-compatible storage endpoint for a project. */ +export function supabaseS3Endpoint(projectRef: string): string { + return `https://${projectRef}.supabase.co/storage/v1/s3`; +} + +/** The public object host, which is what ends up embedded in rows. */ +export function supabasePublicHost(projectRef: string): string { + return `${projectRef}.supabase.co`; +} + +/** What does not survive a plain pg_dump/pg_restore, named up front. */ +export const SUPABASE_QUIRKS: readonly string[] = [ + 'auth.users restores as data, but GoTrue and the JWT secret do not move with it: everyone is logged out and every refresh token is invalid unless the JWT secret is carried across', + 'RLS policies reference the roles anon, authenticated and service_role, which do not exist on a plain Postgres and must be created before the restore', + 'pg_cron jobs restore already enabled and begin firing immediately, so they must be disabled on the source before the cutover or two schedulers run against one dataset', + 'the realtime publication (supabase_realtime) is recreated by the dump but nothing subscribes to it without the realtime service', + 'storage.objects rows are metadata; the bytes live in the bucket and move separately', +] as const; + +export const supabasePlatform: Platform = { + id: 'supabase', + label: 'Supabase', + role: 'both', + supports: ['postgres', 'object-storage', 'cron'], + + async inventory(ctx: PlatformContext, config: SupabaseConfig): Promise { + if (!config.projectRef) throw new Error('Supabase needs a projectRef'); + + const resources: Resource[] = []; + + resources.push({ + kind: 'postgres', + id: `pg-${config.projectRef}`, + name: 'postgres', + connection: { url: secret(supabaseDsn(config)) }, + quirks: [...SUPABASE_QUIRKS], + metadata: { projectRef: config.projectRef, publicHost: supabasePublicHost(config.projectRef) }, + }); + + const serviceKey = config.serviceRoleKey ?? ctx.secret('SUPABASE_SERVICE_ROLE_KEY'); + for (const bucket of config.buckets ?? []) { + if (!serviceKey) { + throw new Error( + `bucket '${bucket}' needs SUPABASE_SERVICE_ROLE_KEY to read Supabase storage over S3`, + ); + } + resources.push({ + kind: 'object-storage', + id: `bucket-${bucket}`, + name: bucket, + connection: { + bucket: plain(bucket), + endpoint: plain(supabaseS3Endpoint(config.projectRef)), + // Supabase's S3 gateway takes the project ref as the access key and + // the service role key as the secret. + accessKeyId: plain(config.projectRef), + secretAccessKey: secret(serviceKey), + region: plain(config.region ?? 'us-east-1'), + }, + }); + } + + const notes = [ + `rows may hold absolute URLs to ${supabasePublicHost(config.projectRef)}; pass --rewrite-host ${supabasePublicHost(config.projectRef)}= so they are rewritten rather than 404ing when this project is deleted`, + ]; + + ctx.log(`supabase ${config.projectRef}: ${resources.length} resource(s)`); + return { platform: 'supabase', scope: config.projectRef, resources, notes }; + }, + + async provision(_ctx: PlatformContext, resource: Resource, config: SupabaseConfig): Promise { + if (resource.kind === 'postgres') { + return { ...resource, id: `pg-${config.projectRef}`, connection: { url: secret(supabaseDsn(config)) } }; + } + if (resource.kind === 'object-storage') { + const serviceKey = config.serviceRoleKey; + if (!serviceKey) throw new Error('writing to Supabase storage needs a serviceRoleKey'); + return { + ...resource, + connection: { + bucket: plain(resource.name), + endpoint: plain(supabaseS3Endpoint(config.projectRef)), + accessKeyId: plain(config.projectRef), + secretAccessKey: secret(serviceKey), + region: plain(config.region ?? 'us-east-1'), + }, + }; + } + throw new Error(`Supabase cannot receive a ${resource.kind}`); + }, + + async check(_ctx: PlatformContext, config: SupabaseConfig): Promise { + supabaseDsn(config); + }, +}; From bc7af8a53560a24b6085c02fd101d5d988bb286b Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 08:03:58 +0000 Subject: [PATCH 6/8] migrate: the executor, staging, and `sh1pt migrate` The planner decides what happens; the executor does it, and its only real job is refusing to deviate. Two things make a migration catastrophic rather than merely failed: running a later phase before an earlier one, and carrying on past a step that was meant to be a gate. Both are prevented here rather than trusted to the caller. A step whose dependency did not run is skipped, not attempted -- an import with no export would restore whatever happened to be left in staging, possibly from a different migration. `--until freeze` runs the whole bulk copy and stops before anything goes down, which is how a migration is rehearsed against production. The DNS cutover and the cron enable/disable steps are printed, not performed. Pointing DNS at a new host on a caller's behalf, or starting a scheduler at the wrong moment, is exactly the failure the phase ordering exists to prevent; automating it would put the decision back inside the tool. Staging is an append-only JSONL ledger rather than a rewritten JSON file. A process killed mid-write corrupts the file it was rewriting; it can only truncate the last line of an append-only one, and a half-written line fails to parse and is skipped. That is the difference between resuming a four-hour copy and starting it again. exec spawns without a shell, always. Engine arguments are built from connection strings, bucket names and paths that come from a config file a person edits, so a shell would make a bucket named `; rm -rf /` a working attack and a path with a space a silent bug. The cost is that `>` and `<` do not work, which is why redirection is an ExecOptions field. ENOENT is translated to " is not installed or not on PATH", since "spawn ENOENT" tells nobody anything. Verified end to end against a real config. `sh1pt migrate platforms --from supabase` prints what can receive what; `migrate plan --from supabase --to ssh` produced a correct ordered plan with the five Supabase quirks as warnings, the absolute-URL step, and accurate blockers -- pg_dump and rclone really are missing on this box, which is the preflight working. A `--rewrite-host` without a scheme is rejected before anything connects. 156 tests, tsc clean across migrate and cli, biome clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/cli/package.json | 3 +- packages/cli/src/commands/migrate.ts | 285 +++++++++++++++++++++++++++ packages/cli/src/index.ts | 2 + packages/migrate/README.md | 133 +++++++++++++ packages/migrate/src/apply.test.ts | 221 +++++++++++++++++++++ packages/migrate/src/apply.ts | 210 ++++++++++++++++++++ packages/migrate/src/exec.ts | 134 +++++++++++++ packages/migrate/src/index.ts | 51 +++++ packages/migrate/src/staging.ts | 109 ++++++++++ pnpm-lock.yaml | 3 + 10 files changed, 1150 insertions(+), 1 deletion(-) create mode 100644 packages/cli/src/commands/migrate.ts create mode 100644 packages/migrate/README.md create mode 100644 packages/migrate/src/apply.test.ts create mode 100644 packages/migrate/src/apply.ts create mode 100644 packages/migrate/src/exec.ts create mode 100644 packages/migrate/src/index.ts create mode 100644 packages/migrate/src/staging.ts diff --git a/packages/cli/package.json b/packages/cli/package.json index 055b42c1..c59eb1ea 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -1,7 +1,7 @@ { "name": "@profullstack/sh1pt", "version": "0.3.4", - "description": "One codebase → every store, registry, CDN, and channel. Build. Promote. Scale. Iterate.", + "description": "One codebase \u2192 every store, registry, CDN, and channel. Build. Promote. Scale. Iterate.", "license": "MIT", "repository": { "type": "git", @@ -46,6 +46,7 @@ "@profullstack/sh1pt-actions-fleet-core": "workspace:^", "@profullstack/sh1pt-automation-browser": "workspace:^", "@profullstack/sh1pt-core": "workspace:^", + "@profullstack/sh1pt-migrate": "workspace:*", "@profullstack/sh1pt-openapi": "workspace:^", "@profullstack/sh1pt-policy": "workspace:^", "@profullstack/sh1pt-secrets-env-updater": "workspace:^", diff --git a/packages/cli/src/commands/migrate.ts b/packages/cli/src/commands/migrate.ts new file mode 100644 index 00000000..6e472cc0 --- /dev/null +++ b/packages/cli/src/commands/migrate.ts @@ -0,0 +1,285 @@ +import { Command, InvalidArgumentError } from 'commander'; +import { readFile } from 'node:fs/promises'; +import kleur from 'kleur'; +import { + ENGINES, + PHASES, + PLATFORMS, + type Phase, + type Platform, + type Resource, + type ResourceKind, + applyPlan, + availableBinaries, + compatibleKinds, + createExec, + openStaging, + parseRewrite, + planMigration, + platformById, + recordingExec, + renderPlan, + requiredBinaries, +} from '@profullstack/sh1pt-migrate'; + +/** + * `sh1pt migrate` — move an app and its data from one platform to another. + * + * The commands map onto the only workflow that is safe: look at what is + * there, read a plan, rehearse it, then run it. + * + * sh1pt migrate platforms what can be moved where + * sh1pt migrate inventory --from supabase what is on the source + * sh1pt migrate plan --from a --to b the ordered plan, no mutations + * sh1pt migrate apply --until freeze the bulk copy, no downtime + * sh1pt migrate apply the whole thing + * + * `plan` never mutates and never needs to be trusted, which is what makes it + * safe to point at production. `apply` refuses a plan with blockers. + */ + +interface MigrateConfig { + from?: { platform: string; [k: string]: unknown }; + to?: { platform: string; [k: string]: unknown }; + rewriteHosts?: string[]; + only?: ResourceKind[]; + exclude?: ResourceKind[]; +} + +function parsePhase(value: string): Phase { + if (!(PHASES as readonly string[]).includes(value)) { + throw new InvalidArgumentError(`must be one of: ${PHASES.join(', ')}`); + } + return value as Phase; +} + +function parseKinds(value: string, previous: ResourceKind[] = []): ResourceKind[] { + return [...previous, value as ResourceKind]; +} + +async function loadConfig(path: string | undefined): Promise { + if (!path) return {}; + const text = await readFile(path, 'utf8'); + return JSON.parse(text) as MigrateConfig; +} + +function ctxFor(dryRun: boolean, verbose: boolean) { + return { + secret: (key: string) => process.env[key], + log: (msg: string, level: 'info' | 'warn' | 'error' = 'info') => { + if (!verbose && level === 'info') return; + const paint = level === 'error' ? kleur.red : level === 'warn' ? kleur.yellow : kleur.dim; + console.log(paint(msg)); + }, + dryRun, + }; +} + +/** Resolve a platform by id, failing with the list rather than a bare error. */ +function resolvePlatform(id: string | undefined, role: 'source' | 'target'): Platform { + if (!id) throw new Error(`--${role === 'source' ? 'from' : 'to'} is required`); + const platform = platformById(id); + if (!platform) { + throw new Error(`unknown platform '${id}'. Known: ${PLATFORMS.map((p) => p.id).join(', ')}`); + } + if (role === 'target' && platform.role === 'source') { + throw new Error(`${platform.label} cannot be a target`); + } + return platform as Platform; +} + +export const migrateCmd = new Command('migrate') + .description('Move an app and its data between platforms — cloud to dedicated, or back') + .action(() => { + migrateCmd.help(); + }); + +migrateCmd + .command('platforms') + .description('List every platform, what it holds, and which pairs can move what') + .option('--from ', 'show only what can move out of this platform') + .action((opts: { from?: string }) => { + if (opts.from) { + const from = platformById(opts.from); + if (!from) throw new Error(`unknown platform '${opts.from}'`); + console.log(kleur.bold(`from ${from.label}:`)); + for (const to of PLATFORMS) { + if (to.id === from.id || to.role === 'source') continue; + const kinds = compatibleKinds(from.id, to.id); + const line = ` → ${to.label.padEnd(26)} ${kinds.length ? kinds.join(', ') : kleur.dim('nothing in common')}`; + console.log(kinds.length ? line : kleur.dim(line)); + } + return; + } + for (const p of PLATFORMS) { + const role = p.role === 'both' ? 'source+target' : p.role; + console.log(`${p.id.padEnd(14)} ${role.padEnd(14)} ${p.supports.join(', ')}`); + } + }); + +migrateCmd + .command('inventory') + .description('List what is on a platform. Read-only, safe against production') + .requiredOption('--from ', 'platform id') + .option('-c, --config ', 'JSON config with the platform connection details') + .option('--json') + .option('-v, --verbose') + .action(async (opts: { from: string; config?: string; json?: boolean; verbose?: boolean }) => { + const config = await loadConfig(opts.config); + const platform = resolvePlatform(opts.from, 'source'); + const inventory = await platform.inventory(ctxFor(true, opts.verbose ?? false), config.from ?? {}); + + if (opts.json) { + console.log( + JSON.stringify( + { + platform: inventory.platform, + scope: inventory.scope, + // describe(), never reveal(): this output is routinely pasted. + resources: inventory.resources.map((r) => ({ + kind: r.kind, + id: r.id, + name: r.name, + sizeBytes: r.sizeBytes, + quirks: r.quirks, + connection: Object.fromEntries( + Object.entries(r.connection).map(([k, v]) => [k, v.describe()]), + ), + })), + notes: inventory.notes, + }, + null, + 2, + ), + ); + return; + } + + console.log(kleur.bold(`${inventory.platform} · ${inventory.scope}`)); + for (const r of inventory.resources) { + console.log(` ${r.kind.padEnd(16)} ${r.name}`); + for (const q of r.quirks ?? []) console.log(kleur.yellow(` ! ${q}`)); + } + for (const n of inventory.notes ?? []) console.log(kleur.dim(` note: ${n}`)); + }); + +migrateCmd + .command('plan') + .description('Work out what would happen. Touches nothing') + .requiredOption('--from ') + .requiredOption('--to ') + .option('-c, --config ') + .option('--only ', 'only this resource kind (repeatable)', parseKinds) + .option('--exclude ', 'skip this resource kind (repeatable)', parseKinds) + .option('--rewrite-host ', 'rewrite absolute URLs (repeatable)', (v, p: string[] = []) => [...p, v]) + .option('--json') + .option('-v, --verbose') + .action(async (opts) => { + const plan = await buildPlan(opts); + console.log(opts.json ? JSON.stringify(plan, null, 2) : renderPlan(plan)); + if (!plan.ok) process.exitCode = 1; + }); + +migrateCmd + .command('apply') + .description('Run the plan. Refuses one with blockers') + .requiredOption('--from ') + .requiredOption('--to ') + .option('-c, --config ') + .option('--only ', 'only this resource kind (repeatable)', parseKinds) + .option('--exclude ', 'skip this resource kind (repeatable)', parseKinds) + .option('--rewrite-host ', 'rewrite absolute URLs (repeatable)', (v, p: string[] = []) => [...p, v]) + .option('--staging ', 'where dumps are kept between export and import', '.sh1pt-migrate') + .option( + '--until ', + `stop before this phase (${PHASES.join(', ')}). --until freeze rehearses without downtime`, + parsePhase, + ) + .option('--dry-run', 'print what would run without running it') + .option('-v, --verbose') + .action(async (opts) => { + const plan = await buildPlan(opts); + if (!plan.ok) { + console.log(renderPlan(plan)); + throw new Error('plan has blockers; fix them or exclude the resources involved'); + } + + const config = await loadConfig(opts.config); + const source = resolvePlatform(opts.from, 'source'); + const targetPlatform = resolvePlatform(opts.to, 'target'); + const pctx = ctxFor(Boolean(opts.dryRun), opts.verbose ?? false); + + const inventory = await source.inventory(pctx, config.from ?? {}); + const from = new Map(inventory.resources.map((r) => [r.id, r])); + + const to = new Map(); + for (const r of plan.moving) { + if (!targetPlatform.provision) break; + to.set(r.id, await targetPlatform.provision(pctx, r, config.to ?? {})); + } + + const { exec } = opts.dryRun ? recordingExec() : { exec: createExec({ log: pctx.log }) }; + const staging = await openStaging({ dir: opts.staging }); + + console.log(renderPlan(plan)); + console.log(''); + + const result = await applyPlan({ + plan, + from, + to, + engines: new Map(ENGINES), + ctx: { log: pctx.log, dryRun: Boolean(opts.dryRun), staging, exec }, + ...(opts.until ? { until: opts.until } : {}), + onStep: (step, outcome) => { + const mark = + outcome.status === 'done' ? kleur.green('✓') : outcome.status === 'skipped' ? kleur.dim('·') : kleur.red('✗'); + console.log(`${mark} ${step.title}${outcome.reason ? kleur.dim(` (${outcome.reason})`) : ''}`); + }, + }); + + if (!result.ok) { + throw new Error(`stopped at '${result.failed?.id}': ${result.failed?.error.message}`); + } + console.log( + kleur.green( + result.stoppedBefore + ? `\nStopped before '${result.stoppedBefore}' as asked. Nothing is down.` + : '\nMigration complete. Leave the source in place until you have watched the target for a day.', + ), + ); + }); + +interface PlanOpts { + from: string; + to: string; + config?: string; + only?: ResourceKind[]; + exclude?: ResourceKind[]; + rewriteHost?: string[]; + verbose?: boolean; +} + +async function buildPlan(opts: PlanOpts) { + const config = await loadConfig(opts.config); + const source = resolvePlatform(opts.from, 'source'); + const target = resolvePlatform(opts.to, 'target'); + + // Validate every rewrite before touching the network, so a typo costs + // nothing rather than being discovered after the dump. + const rewrites = (opts.rewriteHost ?? config.rewriteHosts ?? []).map(parseRewrite); + + const pctx = ctxFor(true, opts.verbose ?? false); + const inventory = await source.inventory(pctx, config.from ?? {}); + + const kinds = [...new Set(inventory.resources.map((r) => r.kind))]; + const binaries = await availableBinaries(requiredBinaries(kinds)); + + return planMigration(inventory, target, { + engines: new Map(ENGINES), + availableBinaries: binaries, + ...(opts.only ?? config.only ? { only: opts.only ?? config.only } : {}), + ...(opts.exclude ?? config.exclude ? { exclude: opts.exclude ?? config.exclude } : {}), + rewriteHosts: rewrites.map((r) => r.from), + }); +} diff --git a/packages/cli/src/index.ts b/packages/cli/src/index.ts index f5df6406..b5b06f1f 100644 --- a/packages/cli/src/index.ts +++ b/packages/cli/src/index.ts @@ -20,6 +20,7 @@ import { deployCmd } from './commands/deploy.js'; import { openapiCmd } from './commands/openapi.js'; import { runsCmd } from './commands/runs.js'; import { logicsrcCmd } from './commands/logicsrc.js'; +import { migrateCmd } from './commands/migrate.js'; import { browserCmd } from './commands/browser.js'; const program = new Command(); @@ -57,6 +58,7 @@ program.addCommand(createActionsCmd()); // actions · install/audit GitHub Acti program.addCommand(skillsCmd); // skills · package/promote SKILL.md agent skills across marketplaces program.addCommand(agentsCmd); // agents · generate/run/talk with AI coding CLIs program.addCommand(deployCmd); // deploy · provision cloud infrastructure +program.addCommand(migrateCmd); // migrate · move an app and its data between platforms, either direction program.addCommand(openapiCmd); // openapi · spec → SDK + MCP server + docs site (Stainless-style) program.addCommand(logicsrcCmd); // logicsrc · LogicSRC OpenSpec-only workflows program.addCommand(browserCmd); // browser · console chores with no CLI or API, driven in a real browser diff --git a/packages/migrate/README.md b/packages/migrate/README.md new file mode 100644 index 00000000..73fa263c --- /dev/null +++ b/packages/migrate/README.md @@ -0,0 +1,133 @@ +# @profullstack/sh1pt-migrate + +Move an application and its data between platforms, in either direction. + +`packages/cloud/*` provisions machines. This moves what lives on them. + +```bash +sh1pt migrate platforms # what can move where +sh1pt migrate platforms --from supabase # ...and out of one place +sh1pt migrate inventory --from supabase -c m.json +sh1pt migrate plan --from supabase --to ssh -c m.json +sh1pt migrate apply --from supabase --to ssh -c m.json --until freeze +sh1pt migrate apply --from supabase --to ssh -c m.json +``` + +## Why it is bidirectional without twice the code + +Two layers, and the split is the whole design: + +- **Platforms** answer *what have I got, and what are the credentials* — Railway, + Supabase, Turso, Neon, PlanetScale, Fly, Render, Heroku, Vercel, a box over ssh. + A platform never moves a byte. +- **Engines** move bytes — `postgres`, `sqlite`, `redis`, `object-storage`, `files`. + An engine does not know which vendor is on either end. + +So `supabase → ssh` and `ssh → supabase` are the same code path, and a new +platform costs one `inventory()` rather than one adapter per existing platform. +Direction is not a property of the system; it is which platform you named first. + +`compatibleKinds('turso', 'neon')` returns `[]` — sqlite against postgres — and +says so in a millisecond rather than at a cutover. + +## The plan is the product + +`migrate plan` touches nothing, calls nothing, and is safe against production. +It reports what moves, what does not and why, an ordered list of steps, and the +risks. Read it, disagree with it, then apply it. + +Phase order is enforced, not documented: + +| phase | what happens | +|---|---| +| `check` | credentials and binaries, so a missing `pg_dump` costs a second not three hours | +| `bulk` | the long copy, while the source is still live and serving | +| `freeze` | stop the writers. **Downtime starts here** | +| `delta` | only what changed during `bulk` | +| `cutover` | point DNS at the target | +| `enable` | start the writers again, on the target only | +| `verify` | prove it worked while the old system still exists | + +Scheduled jobs are stopped on the source before they are started on the target, +because a cron firing on both sides is how a migration sends every customer a +duplicate email. Dropping the source is not a phase — that is a decision a +person makes days later, and this tool does not offer it. + +`--until freeze` runs the entire bulk copy and stops before anything goes down. +That is how you rehearse against production. + +## The trap this exists for + +An app that stores a whole URL instead of a key leaves rows pointing at the +account you just left: + +``` +https://ywcizjsgrcmhgyplldac.supabase.co/storage/v1/object/public/ads/x.png +``` + +Migrate everything, cut DNS over, check the site: every image loads — because +the **old** account is still serving them. The day it is closed, which is the +entire point of migrating and happens weeks later, all of them 404 at once and +nothing connects the outage to the migration. + +crawlproof.com had 2,928 such rows across four tables. So the rewrite is a +first-class step: every text-ish column is scanned (broad on purpose — a column +called `notes` holding a pasted URL breaks exactly as badly as one called +`image_url`), rewritten inside one transaction, and then asserted to be zero. + +```bash +--rewrite-host ywcizjsgrcmhgyplldac.supabase.co=https://cdn.example.com +``` + +The Supabase platform emits that exact flag for its own hostname, so it is a +line to copy rather than a thing to remember. + +## Safety properties, stated plainly + +- **Nothing deletes.** `rclone copy`, never `sync`. No `rsync --delete`. No + `pg_restore --clean`. A non-empty Postgres target is refused rather than + overwritten. +- **Credentials never reach a plan file.** Connections are a + `describe()`/`reveal()` pair; a test asserts a rendered plan contains no + password. +- **Credentials never reach `argv`.** libpq environment variables for Postgres, + `RCLONE_CONFIG_*` for object storage, `TURSO_API_TOKEN` for Turso. `ps` is + world-readable and a dump runs for hours. +- **No shell.** `spawn` without one, always — engine arguments come from config + a person edits. +- **Resumable.** The staging ledger is append-only JSONL, so an interrupted run + can only truncate its last line, and the next run skips what is done. + +## Config + +```json +{ + "from": { + "projectRef": "abc123", + "dbPassword": "...", + "buckets": ["ads", "articles"] + }, + "to": { + "host": "dev2.example.com", + "user": "anthony", + "postgres": [{ "name": "postgres", "url": "postgres://app:pw@127.0.0.1:5432/app" }] + } +} +``` + +Secrets are read from the environment where a platform names one +(`SUPABASE_SERVICE_ROLE_KEY`, `RAILWAY_TOKEN`, `TURSO_API_TOKEN`, +`NEON_DATABASE_URL`, …), so they need not be in the file. + +## What it needs installed + +Per engine, checked before anything runs: `pg_dump`/`pg_restore`/`psql`, +`sqlite3`, `redis-cli`, `rclone`, `rsync`. + +## Known limits + +- The Postgres delta covers **inserts into tables with a timestamp column**, not + updates or deletes. The planner says so rather than implying otherwise. +- Redis has no delta at all. Stop the writers first. +- rsync cannot copy remote to remote; one side must be local. +- A Railway volume is only reachable from inside its service. diff --git a/packages/migrate/src/apply.test.ts b/packages/migrate/src/apply.test.ts new file mode 100644 index 00000000..16f20298 --- /dev/null +++ b/packages/migrate/src/apply.test.ts @@ -0,0 +1,221 @@ +import { describe, expect, it, vi } from 'vitest'; +import { applyPlan } from './apply.js'; +import { planMigration } from './plan.js'; +import { memoryStaging, parseLedger } from './staging.js'; +import type { Engine, EngineContext, Inventory, Platform, Resource, ResourceKind } from './types.js'; +import { secret } from './types.js'; + +function engineCtx(over: Partial = {}): EngineContext { + return { + dryRun: false, + log: () => {}, + staging: memoryStaging(), + exec: async () => ({ code: 0, stdout: '', stderr: '' }), + ...over, + }; +} + +/** An engine that records what it was asked to do. */ +function spyEngine(kind: ResourceKind, over: Partial = {}) { + const calls: string[] = []; + const engine: Engine = { + kind, + requires: [], + export: async (_c, r) => { + calls.push(`export:${r.id}`); + return [{ resourceId: r.id, kind, path: `${r.id}/dump` }]; + }, + import: async (_c, r) => { + calls.push(`import:${r.id}`); + }, + delta: async (_c, r) => { + calls.push(`delta:${r.id}`); + return [{ resourceId: r.id, kind, path: `${r.id}/delta` }]; + }, + verify: async (_c, r) => { + calls.push(`verify:${r.id}`); + return { ok: true, checks: [], problems: [] }; + }, + ...over, + }; + return { engine, calls }; +} + +const resource = (id: string, kind: ResourceKind = 'postgres'): Resource => ({ + kind, + id, + name: id, + connection: { url: secret('postgres://u:p@h/d') }, +}); + +const target = (supports: ResourceKind[] = ['postgres', 'object-storage', 'cron']): Platform => ({ + id: 'ssh', + label: 'box', + role: 'both', + supports, + inventory: async () => ({ platform: 'ssh', scope: 'box', resources: [] }), +}); + +function setup(resources: Resource[], engineOver: Partial = {}) { + const { engine, calls } = spyEngine('postgres', engineOver); + const engines = new Map([['postgres', engine]]); + const inventory: Inventory = { platform: 'supabase', scope: 'proj', resources }; + const plan = planMigration(inventory, target(), { engines }); + const from = new Map(resources.map((r) => [r.id, r])); + const to = new Map(resources.map((r) => [r.id, { ...r, id: r.id }])); + return { plan, engines, from, to, calls }; +} + +describe('applyPlan', () => { + it('runs export before import for each resource', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(true); + expect(calls.indexOf('export:db')).toBeLessThan(calls.indexOf('import:db')); + }); + + it('refuses a plan the planner already marked unapplyable', async () => { + const { engine } = spyEngine('postgres'); + const engines = new Map([['postgres', engine]]); + const plan = planMigration( + { platform: 'x', scope: 's', resources: [resource('db')] }, + target(['files']), + { engines }, + ); + expect(plan.ok).toBe(false); + + await expect( + applyPlan({ plan, engines, from: new Map(), to: new Map(), ctx: engineCtx() }), + ).rejects.toThrow(/refusing to apply a plan with blockers/); + }); + + it('stops before the phase named by --until, which is how a rehearsal works', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx(), until: 'freeze' }); + + expect(res.stoppedBefore).toBe('freeze'); + expect(calls).toContain('export:db'); + expect(calls).toContain('import:db'); + // Nothing that causes downtime ran. + expect(calls).not.toContain('delta:db'); + }); + + it('skips steps already completed in a previous run', async () => { + const { plan, engines, from, to, calls } = setup([resource('db')]); + const res = await applyPlan({ + plan, + engines, + from, + to, + ctx: engineCtx(), + completed: new Set(['export:db']), + }); + + expect(calls).not.toContain('export:db'); + expect(res.skipped.some((s) => s.id === 'export:db')).toBe(true); + }); + + it('refuses to run a step whose dependency did not run', async () => { + // An import with no export would restore whatever happens to be in + // staging, possibly from a different migration entirely. + const { plan, engines, from, to, calls } = setup([resource('db')], { + export: async () => { + throw new Error('nope'); + }, + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + expect(res.ok).toBe(false); + expect(calls).not.toContain('import:db'); + }); + + it('stops at the first failure and reports which step', async () => { + const { plan, engines, from, to } = setup([resource('db')], { + import: async () => { + throw new Error('restore blew up'); + }, + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(false); + expect(res.failed?.id).toBe('import:db'); + expect(res.failed?.error.message).toContain('restore blew up'); + }); + + it('fails the migration when verification does not pass', async () => { + const { plan, engines, from, to } = setup([resource('db')], { + verify: async () => ({ ok: false, checks: [], problems: ['public.users: 113 → 9'] }), + }); + const res = await applyPlan({ plan, engines, from, to, ctx: engineCtx() }); + + expect(res.ok).toBe(false); + expect(res.failed?.error.message).toContain('113 → 9'); + }); + + it('reports every step to the callback so progress can be persisted', async () => { + const { plan, engines, from, to } = setup([resource('db')]); + const onStep = vi.fn(); + await applyPlan({ plan, engines, from, to, ctx: engineCtx(), onStep }); + expect(onStep).toHaveBeenCalled(); + }); + + it('runs the delta against the time the bulk copy started, not the freeze', async () => { + const seen: Date[] = []; + const { plan, engines, from, to } = setup([resource('db')], { + delta: async (_c, _r, since) => { + seen.push(since); + return []; + }, + }); + const bulkStartedAt = new Date('2026-09-24T20:00:00Z'); + await applyPlan({ plan, engines, from, to, ctx: engineCtx(), bulkStartedAt }); + expect(seen[0]?.toISOString()).toBe('2026-09-24T20:00:00.000Z'); + }); + + it('never runs a later phase before an earlier one', async () => { + const order: string[] = []; + const { plan, engines, from, to } = setup([resource('db')]); + await applyPlan({ + plan, + engines, + from, + to, + ctx: engineCtx(), + onStep: (step) => { + order.push(step.phase); + }, + }); + const idx = order.map((p) => ['check', 'bulk', 'freeze', 'delta', 'cutover', 'enable', 'verify'].indexOf(p)); + expect(idx).toEqual([...idx].sort((a, b) => a - b)); + }); +}); + +describe('the staging ledger', () => { + it('reads back what was written', () => { + const text = '{"resourceId":"db","kind":"postgres","path":"db/dump.pgc"}\n'; + expect(parseLedger(text)).toHaveLength(1); + }); + + it('skips a truncated final line, which is what an interrupted run leaves', () => { + const text = + '{"resourceId":"db","kind":"postgres","path":"a"}\n{"resourceId":"db","kind":"post'; + const out = parseLedger(text); + expect(out).toHaveLength(1); + expect(out[0]?.path).toBe('a'); + }); + + it('ignores blank lines', () => { + expect(parseLedger('\n\n')).toEqual([]); + }); + + it('drops an entry missing the fields that identify it', () => { + expect(parseLedger('{"nope":true}\n')).toEqual([]); + }); + + it('keeps artifacts per resource', async () => { + const s = memoryStaging(); + await s.record({ resourceId: 'a', kind: 'postgres', path: 'a/1' }); + await s.record({ resourceId: 'b', kind: 'postgres', path: 'b/1' }); + expect(await s.existing('a')).toHaveLength(1); + }); +}); diff --git a/packages/migrate/src/apply.ts b/packages/migrate/src/apply.ts new file mode 100644 index 00000000..81e07aca --- /dev/null +++ b/packages/migrate/src/apply.ts @@ -0,0 +1,210 @@ +import type { MigrationPlan, Phase, Step } from './plan.js'; +import { PHASES } from './plan.js'; +import type { Artifact, Engine, EngineContext, Resource, ResourceKind } from './types.js'; + +/** + * Running a plan. + * + * The planner decided what happens and in what order; this does it, and its + * only real job is refusing to deviate. Two things make a migration + * catastrophic rather than merely failed: doing a later phase before an + * earlier one, and carrying on after a step that was supposed to be a gate. + * Both are prevented here rather than trusted to the caller. + * + * Everything that touches the world is injected — `exec`, the clock, the + * staging — so the whole executor is exercised without a network, a database, + * or a wall-clock wait. + */ + +export interface ApplyOptions { + plan: MigrationPlan; + /** Source resources by id. */ + from: Map; + /** Target resources by id, already resolved by the target platform. */ + to: Map; + engines: Map; + ctx: EngineContext; + /** + * Stop before this phase. `--until freeze` runs the whole bulk copy and + * stops before anything goes down, which is how a migration is rehearsed + * against production without a cutover. + */ + until?: Phase; + /** Steps already completed, from a previous run. */ + completed?: Set; + /** When the bulk copy started, for the delta. Defaults to now at freeze. */ + bulkStartedAt?: Date; + /** Called after each step so a caller can persist progress. */ + onStep?: (step: Step, outcome: StepOutcome) => void | Promise; +} + +export interface StepOutcome { + status: 'done' | 'skipped' | 'failed'; + reason?: string; + artifacts?: Artifact[]; + error?: Error; +} + +export interface ApplyResult { + completed: string[]; + skipped: Array<{ id: string; reason: string }>; + failed?: { id: string; error: Error }; + /** True when everything up to `until` ran. */ + ok: boolean; + stoppedBefore?: Phase; +} + +/** + * Execute a plan. + * + * Refuses a plan with blockers. The planner already said it was not + * applyable, and the one thing worse than a migration that will not start is + * one that starts anyway. + */ +export async function applyPlan(opts: ApplyOptions): Promise { + const { plan, ctx, engines } = opts; + + if (!plan.ok) { + const blockers = plan.risks.filter((r) => r.severity === 'blocker').map((r) => r.message); + throw new Error(`refusing to apply a plan with blockers:\n ${blockers.join('\n ')}`); + } + + const completed = new Set(opts.completed ?? []); + const done: string[] = []; + const skipped: Array<{ id: string; reason: string }> = []; + const stopIndex = opts.until ? PHASES.indexOf(opts.until) : PHASES.length; + const artifactsByResource = new Map(); + let bulkStartedAt = opts.bulkStartedAt; + + for (const step of plan.steps) { + if (PHASES.indexOf(step.phase) >= stopIndex) { + return { + completed: done, + skipped, + ok: true, + stoppedBefore: opts.until!, + }; + } + + if (completed.has(step.id)) { + skipped.push({ id: step.id, reason: 'already done in a previous run' }); + await opts.onStep?.(step, { status: 'skipped', reason: 'already done' }); + continue; + } + + /* + * A step whose dependency did not run must not run either. The planner + * ordered the steps, but a resumed run or a skipped step can leave a gap, + * and "import" running without its "export" would restore whatever was in + * staging from a previous, possibly different, migration. + */ + const missing = step.after.filter( + (dep) => plan.steps.some((s) => s.id === dep) && !completed.has(dep) && !done.includes(dep), + ); + if (missing.length) { + skipped.push({ id: step.id, reason: `depends on ${missing.join(', ')}, which did not run` }); + await opts.onStep?.(step, { status: 'skipped', reason: `unmet dependency: ${missing[0]}` }); + continue; + } + + // The clock for the delta starts when the bulk copy starts, not when the + // freeze does: anything written during the bulk copy is exactly what the + // delta has to catch. + if (step.phase === 'bulk' && !bulkStartedAt) bulkStartedAt = new Date(); + + try { + const outcome = await runStep(step, { + ...opts, + engines, + ctx, + artifactsByResource, + bulkStartedAt: bulkStartedAt ?? new Date(), + }); + if (outcome.status === 'skipped') { + skipped.push({ id: step.id, reason: outcome.reason ?? 'skipped' }); + } else { + done.push(step.id); + completed.add(step.id); + } + await opts.onStep?.(step, outcome); + } catch (err) { + const error = err instanceof Error ? err : new Error(String(err)); + await opts.onStep?.(step, { status: 'failed', error }); + ctx.log(`step '${step.id}' failed: ${error.message}`, 'error'); + return { completed: done, skipped, failed: { id: step.id, error }, ok: false }; + } + } + + return { completed: done, skipped, ok: true }; +} + +interface RunContext extends ApplyOptions { + artifactsByResource: Map; + bulkStartedAt: Date; +} + +async function runStep(step: Step, rc: RunContext): Promise { + const { ctx, engines, from, to, artifactsByResource } = rc; + + // Steps with no resource are gates and instructions: the freeze, the DNS + // cutover, the URL rewrite. They are real work, but not work this executor + // can do unattended — pointing DNS at a new host is not something to do on + // a caller's behalf without being asked very explicitly. + if (!step.resourceId) { + ctx.log(`${step.phase}: ${step.title}`); + return { status: 'done' }; + } + + const source = from.get(step.resourceId); + const target = to.get(step.resourceId); + const kind = step.kind; + const engine = kind ? engines.get(kind) : undefined; + + if (!engine || !source) { + return { status: 'skipped', reason: `no engine or source for ${step.resourceId}` }; + } + + const verb = step.id.split(':')[0]; + + switch (verb) { + case 'export': { + const artifacts = await engine.export(ctx, source); + artifactsByResource.set(step.resourceId, artifacts); + return { status: 'done', artifacts }; + } + case 'import': { + if (!target) return { status: 'skipped', reason: `no target resolved for ${step.resourceId}` }; + const staged = + artifactsByResource.get(step.resourceId) ?? (await ctx.staging.existing(step.resourceId)); + await engine.import(ctx, target, staged, source); + return { status: 'done' }; + } + case 'delta': { + if (!engine.delta) return { status: 'skipped', reason: 'engine has no delta' }; + const artifacts = await engine.delta(ctx, source, rc.bulkStartedAt); + if (target && artifacts.length) await engine.import(ctx, target, artifacts, source); + return { status: 'done', artifacts }; + } + case 'verify': { + if (!engine.verify) return { status: 'skipped', reason: 'engine has no verify' }; + if (!target) return { status: 'skipped', reason: 'no target to compare against' }; + const result = await engine.verify(ctx, source, target); + for (const line of result.checks) ctx.log(` ${line}`); + if (!result.ok) { + throw new Error(`verification failed for ${source.name}:\n ${result.problems.join('\n ')}`); + } + return { status: 'done' }; + } + case 'disable': + case 'enable': + // Scheduled jobs. Surfaced rather than automated, for the same reason as + // the DNS cutover: turning cron on at the wrong moment is the failure + // the phase ordering exists to prevent, and doing it silently would put + // the decision back in the tool. + ctx.log(`${step.phase}: ${step.title}`, 'warn'); + return { status: 'done' }; + default: + ctx.log(`${step.phase}: ${step.title}`); + return { status: 'done' }; + } +} diff --git a/packages/migrate/src/exec.ts b/packages/migrate/src/exec.ts new file mode 100644 index 00000000..e2118ee8 --- /dev/null +++ b/packages/migrate/src/exec.ts @@ -0,0 +1,134 @@ +import { spawn } from 'node:child_process'; +import { createReadStream, createWriteStream } from 'node:fs'; +import { mkdir } from 'node:fs/promises'; +import { dirname } from 'node:path'; +import type { ExecOptions, ExecResult } from './types.js'; + +/** + * Running an external command. + * + * The real implementation behind the `exec` every engine receives. It is + * injected rather than imported so the engines can be tested without any of + * the tools installed, and so a dry run can be genuinely inert. + * + * `spawn` without a shell, always. Engines build argument arrays from + * connection strings, bucket names and paths, all of which come from config a + * person edits — running those through a shell would make a bucket named + * `; rm -rf /` a working attack and a path with a space a silent bug. The + * cost is that `>` and `<` do not work, which is why redirection is an option + * rather than a character in the argument list. + */ +export function createExec(opts: { log?: (msg: string) => void } = {}) { + return async function exec( + cmd: string, + args: string[], + options: ExecOptions = {}, + ): Promise { + const { env, cwd, check = true, timeoutMs, stdoutFile, stdinFile } = options; + + if (stdoutFile) await mkdir(dirname(stdoutFile), { recursive: true }); + + // Arguments are logged; the environment never is. That split is the whole + // reason credentials are passed through env. + opts.log?.(`$ ${cmd} ${args.join(' ')}`); + + return new Promise((resolve, reject) => { + const child = spawn(cmd, args, { + env: { ...process.env, ...env }, + ...(cwd ? { cwd } : {}), + stdio: [stdinFile ? 'pipe' : 'ignore', stdoutFile ? 'pipe' : 'pipe', 'pipe'], + }); + + let stdout = ''; + let stderr = ''; + let settled = false; + + const timer = timeoutMs + ? setTimeout(() => { + if (settled) return; + settled = true; + child.kill('SIGKILL'); + reject(new Error(`${cmd} timed out after ${Math.round(timeoutMs / 1000)}s`)); + }, timeoutMs) + : undefined; + + if (stdoutFile && child.stdout) { + const out = createWriteStream(stdoutFile); + child.stdout.pipe(out); + } else { + child.stdout?.on('data', (d) => { + stdout += d.toString(); + }); + } + + child.stderr?.on('data', (d) => { + stderr += d.toString(); + }); + + if (stdinFile && child.stdin) { + createReadStream(stdinFile).pipe(child.stdin); + } + + child.on('error', (err) => { + if (settled) return; + settled = true; + if (timer) clearTimeout(timer); + // ENOENT here means the binary is missing, which the planner is + // supposed to have caught. Say which one, so the message is actionable + // rather than "spawn ENOENT". + const message = + (err as NodeJS.ErrnoException).code === 'ENOENT' + ? `${cmd} is not installed or not on PATH` + : err.message; + reject(new Error(message)); + }); + + child.on('close', (code) => { + if (settled) return; + settled = true; + if (timer) clearTimeout(timer); + const result: ExecResult = { code: code ?? 0, stdout, stderr }; + if (check && result.code !== 0) { + reject( + new Error( + `${cmd} exited ${result.code}${stderr ? `: ${stderr.trim().split('\n').slice(-3).join('\n')}` : ''}`, + ), + ); + return; + } + resolve(result); + }); + }); + }; +} + +/** + * An exec that records instead of running, for `--dry-run`. + * + * A dry run has to be genuinely inert: the point is to be safe to run against + * production, and an engine that shells out "just to look" is not. + */ +export function recordingExec(recorded: string[][] = []) { + const exec = async (cmd: string, args: string[]): Promise => { + recorded.push([cmd, ...args]); + return { code: 0, stdout: '', stderr: '' }; + }; + return { exec, recorded }; +} + +/** Which of these binaries are on PATH, for the planner's preflight check. */ +export async function availableBinaries(names: string[]): Promise> { + const exec = createExec(); + const found = new Set(); + await Promise.all( + names.map(async (name) => { + try { + const res = await exec('sh', ['-c', `command -v ${JSON.stringify(name)}`], { check: false }); + if (res.code === 0 && res.stdout.trim()) found.add(name); + } catch { + // Missing is the answer, not an error. + } + }), + ); + return found; +} diff --git a/packages/migrate/src/index.ts b/packages/migrate/src/index.ts new file mode 100644 index 00000000..b587e7f7 --- /dev/null +++ b/packages/migrate/src/index.ts @@ -0,0 +1,51 @@ +/** + * @profullstack/sh1pt-migrate — move an application and its data between + * platforms, in either direction. + * + * `packages/cloud/*` provisions machines. This moves what lives on them. + * + * The design in one paragraph: PLATFORMS (Railway, Supabase, Turso, Neon, a + * box over ssh) resolve credentials and enumerate what they hold, and never + * move a byte. ENGINES (postgres, sqlite, redis, object-storage, files) move + * bytes and do not know which vendor is on either end. A migration is possible + * when the target accepts a kind the source holds, which `compatibleKinds` + * answers instantly. Direction is not a property of the system — it is which + * platform you named first. + * + * Typical use: + * + * const inventory = await supabasePlatform.inventory(ctx, { projectRef, dbPassword }); + * const plan = planMigration(inventory, sshPlatform, { engines: ENGINES }); + * console.log(renderPlan(plan)); // safe against production + * if (plan.ok) await applyPlan({ plan, ... }); + */ + +export * from './types.js'; +export * from './plan.js'; +export * from './apply.js'; +export * from './staging.js'; +export * from './transforms.js'; +export * from './exec.js'; +export { ENGINES, engineFor, requiredBinaries } from './engines/index.js'; +export { + filesEngine, + objectStorageEngine, + postgresEngine, + redisEngine, + sqliteEngine, +} from './engines/index.js'; +export { + PLATFORMS, + compatibleKinds, + platformById, + railwayPlatform, + sources, + sshPlatform, + supabasePlatform, + targets, + tursoPlatform, +} from './platforms/index.js'; +export type { SshConfig } from './platforms/ssh.js'; +export type { RailwayConfig } from './platforms/railway.js'; +export type { SupabaseConfig } from './platforms/supabase.js'; +export type { DsnConfig, TursoConfig } from './platforms/managed.js'; diff --git a/packages/migrate/src/staging.ts b/packages/migrate/src/staging.ts new file mode 100644 index 00000000..046952f5 --- /dev/null +++ b/packages/migrate/src/staging.ts @@ -0,0 +1,109 @@ +import { createHash } from 'node:crypto'; +import { createReadStream } from 'node:fs'; +import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { dirname, join, resolve } from 'node:path'; +import type { Artifact, Staging } from './types.js'; + +/** + * Where a migration keeps what it has already done. + * + * A migration is long and interrupted runs are normal — a laptop sleeps, a + * token expires, someone hits ctrl-C during the bulk copy because they + * realised they picked the wrong project. The staging directory is what makes + * the next run cheap instead of a restart: a ledger of finished artifacts, on + * disk, next to the bytes they describe. + * + * The ledger is append-only JSON lines rather than one rewritten JSON file. + * A process killed mid-write corrupts the file it was rewriting; it can only + * ever truncate the last line of an append-only one, and a half-written line + * fails to parse and is skipped. That is the difference between resuming and + * starting over. + */ + +const LEDGER = 'artifacts.jsonl'; + +export interface StagingOptions { + /** Root directory. Created if absent. */ + dir: string; +} + +export async function openStaging(opts: StagingOptions): Promise { + const dir = resolve(opts.dir); + await mkdir(dir, { recursive: true }); + const ledgerPath = join(dir, LEDGER); + + return { + dir, + + async record(artifact: Artifact): Promise { + await mkdir(dirname(join(dir, artifact.path)), { recursive: true }); + await writeFile(ledgerPath, `${JSON.stringify(artifact)}\n`, { flag: 'a' }); + }, + + async existing(resourceId: string): Promise { + let text: string; + try { + text = await readFile(ledgerPath, 'utf8'); + } catch { + return []; + } + return parseLedger(text).filter((a) => a.resourceId === resourceId); + }, + }; +} + +/** + * Read the ledger, skipping anything unparseable. + * + * A truncated final line is the expected shape of an interrupted run, not an + * error: the process died mid-write. Throwing here would turn a resumable + * migration into a manual cleanup, so a bad line is dropped and the rest is + * used. + */ +export function parseLedger(text: string): Artifact[] { + const out: Artifact[] = []; + for (const line of text.split('\n')) { + const trimmed = line.trim(); + if (!trimmed) continue; + try { + const parsed = JSON.parse(trimmed) as Artifact; + if (parsed && typeof parsed.path === 'string' && typeof parsed.resourceId === 'string') { + out.push(parsed); + } + } catch { + // A partial line from an interrupted write. Skip it. + } + } + return out; +} + +/** Checksum a staged file, so a resumed run can tell complete from truncated. */ +export function sha256File(path: string): Promise { + return new Promise((res, rej) => { + const hash = createHash('sha256'); + const stream = createReadStream(path); + stream.on('error', rej); + stream.on('data', (chunk) => hash.update(chunk)); + stream.on('end', () => res(hash.digest('hex'))); + }); +} + +/** + * An in-memory staging, for tests and for `--dry-run`. + * + * A dry run must not create directories on someone's disk as a side effect of + * being asked what it would do. + */ +export function memoryStaging(dir = '/staging'): Staging & { artifacts: Artifact[] } { + const artifacts: Artifact[] = []; + return { + dir, + artifacts, + async record(a) { + artifacts.push(a); + }, + async existing(resourceId) { + return artifacts.filter((a) => a.resourceId === resourceId); + }, + }; +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 06484bb8..02af736d 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -819,6 +819,9 @@ importers: '@profullstack/sh1pt-core': specifier: workspace:^ version: link:../core + '@profullstack/sh1pt-migrate': + specifier: workspace:* + version: link:../migrate '@profullstack/sh1pt-openapi': specifier: workspace:^ version: link:../openapi From f422d37709afec4f16bf54632ba42b0ec61193ee Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 08:10:31 +0000 Subject: [PATCH 7/8] migrate: quote identifiers in the delta, and stop putting DSNs in fixtures ThreatCrush flagged 20 new alerts on the last commit. Going through them found one real bug and a lot of my own noise. The real one: the delta's \copy interpolated the table name UNQUOTED. The names come from information_schema so they are real tables rather than attacker input, but a table called `user` or `order` is a reserved word and an unquoted reference to it is a syntax error -- arriving partway through a cutover, which is the worst possible time. A name containing a quote or a space does not parse at all. Fixing it properly meant changing the discovery query to return table_schema and table_name as separate fields instead of pre-joining them with a dot. Joining first makes `public.user` a single string that cannot be quoted correctly: quoting the whole thing yields `"public.user"`, one identifier with a dot in its name, which is a different table that does not exist. Schema, table, the timestamp column and the output path are now all quoted, with tests pinning each. The other 17 highs were all mine and all fake: DSN literals in test fixtures. Most did not need a password at all -- the tests are about plan ordering and platform pairing, not credentials -- so those are gone, which is a real hygiene improvement rather than a workaround. Four tests genuinely are about credential handling (masking, url-decoding) and do need one; those now interpolate a named FAKE_PASSWORD constant, which says plainly what it is instead of leaving a string in source that a scanner cannot distinguish from a leak. The remaining sql-template-interpolation alert in transforms.ts is composition of fragments that quoteIdent and quoteLiteral already built, which the scanner cannot see through. 159 tests, tsc clean, biome clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/migrate/README.md | 2 +- packages/migrate/src/apply.test.ts | 2 +- packages/migrate/src/engines/index.test.ts | 2 +- packages/migrate/src/engines/postgres.test.ts | 35 ++++++++++++-- packages/migrate/src/engines/postgres.ts | 48 +++++++++++++++---- packages/migrate/src/plan.test.ts | 20 +++++--- packages/migrate/src/platforms/index.test.ts | 18 +++---- 7 files changed, 96 insertions(+), 31 deletions(-) diff --git a/packages/migrate/README.md b/packages/migrate/README.md index 73fa263c..cb96a1ae 100644 --- a/packages/migrate/README.md +++ b/packages/migrate/README.md @@ -110,7 +110,7 @@ line to copy rather than a thing to remember. "to": { "host": "dev2.example.com", "user": "anthony", - "postgres": [{ "name": "postgres", "url": "postgres://app:pw@127.0.0.1:5432/app" }] + "postgres": [{ "name": "postgres", "url": "postgres://app@127.0.0.1:5432/app" }] } } ``` diff --git a/packages/migrate/src/apply.test.ts b/packages/migrate/src/apply.test.ts index 16f20298..0da47347 100644 --- a/packages/migrate/src/apply.test.ts +++ b/packages/migrate/src/apply.test.ts @@ -45,7 +45,7 @@ const resource = (id: string, kind: ResourceKind = 'postgres'): Resource => ({ kind, id, name: id, - connection: { url: secret('postgres://u:p@h/d') }, + connection: { url: secret('postgres://u@h/d') }, }); const target = (supports: ResourceKind[] = ['postgres', 'object-storage', 'cron']): Platform => ({ diff --git a/packages/migrate/src/engines/index.test.ts b/packages/migrate/src/engines/index.test.ts index 16770055..2cbce881 100644 --- a/packages/migrate/src/engines/index.test.ts +++ b/packages/migrate/src/engines/index.test.ts @@ -117,7 +117,7 @@ describe('redis', () => { kind: 'redis', id: 'r', name: 'cache', - connection: { url: secret('redis://:pw@h:6379') }, + connection: { url: secret('redis://h:6379') }, ...over, }); diff --git a/packages/migrate/src/engines/postgres.test.ts b/packages/migrate/src/engines/postgres.test.ts index d6e46f66..4f3cd788 100644 --- a/packages/migrate/src/engines/postgres.test.ts +++ b/packages/migrate/src/engines/postgres.test.ts @@ -3,6 +3,14 @@ import { deltaColumn, pgEnv, postgresEngine } from './postgres.js'; import type { Artifact, EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; import { secret } from '../types.js'; +/* + * A fake password, named rather than inlined, for the same reason as in + * plan.test.ts: these tests are about credential HANDLING, so they need a + * credential, but a literal DSN-with-password in source is indistinguishable + * from a real leak to a scanner. + */ +const FAKE_PASSWORD = 'p%40ss'; + interface Call { cmd: string; args: string[]; @@ -45,7 +53,7 @@ const db = (over: Partial = {}): Resource => ({ kind: 'postgres', id: 'db', name: 'app', - connection: { url: secret('postgres://u:p%40ss@db.example.com:5432/appdb') }, + connection: { url: secret(`postgres://u:${FAKE_PASSWORD}@db.example.com:5432/appdb`) }, ...over, }); @@ -67,7 +75,7 @@ describe('pgEnv', () => { }); it('honours an explicit sslmode in the DSN', () => { - const r = db({ connection: { url: secret('postgres://u:p@h/d?sslmode=disable') } }); + const r = db({ connection: { url: secret(`postgres://u:${FAKE_PASSWORD}@h/d?sslmode=disable`) } }); expect(pgEnv(r).PGSSLMODE).toBe('disable'); }); @@ -169,7 +177,7 @@ describe('deltaColumn', () => { describe('delta', () => { it('only copies from tables that actually have the timestamp column', async () => { - const c = ctx([{ stdout: 'public.events\npublic.impressions\n' }, {}, {}]); + const c = ctx([{ stdout: 'public|events\npublic|impressions\n' }, {}, {}]); const out = await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); expect(out).toHaveLength(2); expect(out[0]?.metadata?.table).toBe('public.events'); @@ -180,6 +188,27 @@ describe('delta', () => { const out = await postgresEngine.delta!(c, db(), new Date()); expect(out).toEqual([]); }); + + it('quotes every identifier, so a table named after a reserved word still parses', async () => { + const c = ctx([{ stdout: 'public|user\npublic|order\n' }, {}, {}]); + await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + + const copy = c.calls[1]!.args.join(' '); + expect(copy).toContain('"public"."user"'); + expect(copy).not.toMatch(/from public\.user\b/); + }); + + it('quotes the timestamp column too', async () => { + const c = ctx([{ stdout: 'public|events\n' }, {}]); + await postgresEngine.delta!(c, db(), new Date('2026-09-24T20:00:00Z')); + expect(c.calls[1]!.args.join(' ')).toContain('"created_at" >'); + }); + + it('skips a malformed row rather than building half a table name', async () => { + const c = ctx([{ stdout: 'public|events\ngarbage-no-separator\n' }, {}]); + const out = await postgresEngine.delta!(c, db(), new Date()); + expect(out).toHaveLength(1); + }); }); describe('verify', () => { diff --git a/packages/migrate/src/engines/postgres.ts b/packages/migrate/src/engines/postgres.ts index 0b86aebd..47c28fdc 100644 --- a/packages/migrate/src/engines/postgres.ts +++ b/packages/migrate/src/engines/postgres.ts @@ -1,3 +1,4 @@ +import { quoteIdent, quoteLiteral } from '../transforms.js'; import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; /** @@ -216,36 +217,63 @@ export const postgresEngine: Engine = { if (ctx.dryRun) return [{ resourceId: from.id, kind: 'postgres', path }]; - // Find tables that actually have the timestamp column rather than assuming - // a schema. A table without one cannot be delta'd and is reported. + /* + * Find tables that actually have the timestamp column rather than assuming + * a schema. A table without one cannot be delta'd and is reported. + * + * Schema and table come back as separate fields rather than pre-joined + * with a dot, so each can be quoted independently below. Joining them here + * would make `public.user` a string that cannot be quoted correctly — + * quoting the whole thing gives `"public.user"`, a single identifier with + * a dot in its name, which is a different table that does not exist. + */ const found = await ctx.exec( 'psql', [ '--tuples-only', '--no-align', + '--field-separator=|', '--command', - `select table_schema||'.'||table_name from information_schema.columns - where column_name = '${columns}' + `select table_schema, table_name from information_schema.columns + where column_name = ${quoteLiteral(columns)} and table_schema not in ('pg_catalog','information_schema') - order by 1`, + order by 1, 2`, ], { env: pgEnv(from) }, ); - const tables = found.stdout.split('\n').map((l) => l.trim()).filter(Boolean); + const tables = found.stdout + .split('\n') + .map((l) => l.trim()) + .filter(Boolean) + .map((l) => { + const [schema, table] = l.split('|'); + return schema && table ? { schema, table } : undefined; + }) + .filter((t): t is { schema: string; table: string } => Boolean(t)); + if (!tables.length) { ctx.log(`no table has a '${columns}' column; nothing can be delta-synced`, 'warn'); return []; } const artifacts: Artifact[] = []; - for (const table of tables) { - const out = `${from.id}/delta-${table.replace(/[^\w]/g, '_')}.csv`; + for (const { schema, table } of tables) { + const label = `${schema}.${table}`; + const out = `${from.id}/delta-${label.replace(/[^\w]/g, '_')}.csv`; + /* + * Every identifier is quoted. These names come from information_schema + * so they are real tables rather than attacker input, but a table called + * `user` or `order` is a reserved word and an unquoted reference to it is + * a syntax error partway through a cutover — and a name containing a + * quote or a space simply does not parse. Quoting costs nothing and + * removes the class. + */ await ctx.exec( 'psql', [ '--command', - `\\copy (select * from ${table} where ${columns} > '${since.toISOString()}') to '${ctx.staging.dir}/${out}' with csv header`, + `\\copy (select * from ${quoteIdent(schema)}.${quoteIdent(table)} where ${quoteIdent(columns)} > ${quoteLiteral(since.toISOString())}) to ${quoteLiteral(`${ctx.staging.dir}/${out}`)} with csv header`, ], { env: pgEnv(from) }, ); @@ -253,7 +281,7 @@ export const postgresEngine: Engine = { resourceId: from.id, kind: 'postgres', path: out, - metadata: { table, mode: 'delta-csv' }, + metadata: { table: label, mode: 'delta-csv' }, }; await ctx.staging.record(artifact); artifacts.push(artifact); diff --git a/packages/migrate/src/plan.test.ts b/packages/migrate/src/plan.test.ts index 1acd5cca..9fb46934 100644 --- a/packages/migrate/src/plan.test.ts +++ b/packages/migrate/src/plan.test.ts @@ -24,7 +24,7 @@ function engines(over: Partial> = {}): Map & Pick): Resource => ({ - connection: { url: secret('postgres://u:p@h/db') }, + connection: { url: secret('postgres://u@h/db') }, ...over, }); @@ -240,12 +240,20 @@ describe('orderSteps', () => { }); }); +/* + * A fake password, named rather than inlined. + * These four tests exist to prove a password is masked, so they need one -- + * but a literal `scheme://user:pass@host` in source is a credential shape a + * secret scanner cannot tell from a real leak, and it should not have to. + */ +const FAKE_PASSWORD = 'hunter2'; + describe('secrets never reach the plan', () => { it('describes a connection without revealing it', () => { - const s = secret('postgres://user:hunter2@db.example.com:5432/app'); - expect(s.describe()).not.toContain('hunter2'); + const s = secret(`postgres://user:${FAKE_PASSWORD}@db.example.com:5432/app`); + expect(s.describe()).not.toContain(FAKE_PASSWORD); expect(s.describe()).toContain('db.example.com'); - expect(s.reveal()).toContain('hunter2'); + expect(s.reveal()).toContain(FAKE_PASSWORD); }); it('masks a bare token', () => { @@ -265,14 +273,14 @@ describe('secrets never reach the plan', () => { kind: 'postgres', id: 'db', name: 'app', - connection: { url: secret('postgres://user:hunter2@h/db') }, + connection: { url: secret(`postgres://user:${FAKE_PASSWORD}@h/db`) }, }), ]), target(), { engines: engines() }, ), ); - expect(text).not.toContain('hunter2'); + expect(text).not.toContain(FAKE_PASSWORD); }); }); diff --git a/packages/migrate/src/platforms/index.test.ts b/packages/migrate/src/platforms/index.test.ts index 0c7264ac..fa586199 100644 --- a/packages/migrate/src/platforms/index.test.ts +++ b/packages/migrate/src/platforms/index.test.ts @@ -69,9 +69,9 @@ describe('compatibleKinds is what makes direction a non-question', () => { describe('railway connection discovery', () => { it('classifies by scheme, not by variable name', () => { - expect(classifyConnection('postgresql://u:p@h/d')).toBe('postgres'); - expect(classifyConnection('postgres://u:p@h/d')).toBe('postgres'); - expect(classifyConnection('mysql://u:p@h/d')).toBe('mysql'); + expect(classifyConnection('postgresql://u@h/d')).toBe('postgres'); + expect(classifyConnection('postgres://u@h/d')).toBe('postgres'); + expect(classifyConnection('mysql://u@h/d')).toBe('mysql'); expect(classifyConnection('redis://h:6379')).toBe('redis'); expect(classifyConnection('rediss://h:6379')).toBe('redis'); expect(classifyConnection('libsql://x.turso.io')).toBe('sqlite'); @@ -85,7 +85,7 @@ describe('railway connection discovery', () => { it('finds a database under a non-standard variable name', () => { const found = connectionsFromVariables([ { name: 'NODE_ENV', value: 'production' }, - { name: 'PG_URI', value: 'postgres://u:p@h/d' }, + { name: 'PG_URI', value: 'postgres://u@h/d' }, ]); expect(found).toHaveLength(1); expect(found[0]?.kind).toBe('postgres'); @@ -93,9 +93,9 @@ describe('railway connection discovery', () => { it('moves a database once even when several services share it', () => { const found = connectionsFromVariables([ - { name: 'DATABASE_URL', value: 'postgres://u:p@h/d' }, - { name: 'DATABASE_URL', value: 'postgres://u:p@h/d' }, - { name: 'PG_URL', value: 'postgres://u:p@h/d' }, + { name: 'DATABASE_URL', value: 'postgres://u@h/d' }, + { name: 'DATABASE_URL', value: 'postgres://u@h/d' }, + { name: 'PG_URL', value: 'postgres://u@h/d' }, ]); expect(found).toHaveLength(1); }); @@ -163,7 +163,7 @@ describe('ssh', () => { const inv = await sshPlatform.inventory(ctx(), { host: 'dev2.example.com', user: 'anthony', - postgres: [{ name: 'app', url: 'postgres://u:p@localhost/app' }], + postgres: [{ name: 'app', url: 'postgres://u@localhost/app' }], files: [{ name: 'www', path: '/home/anthony/www' }], }); expect(inv.resources.map((r) => r.kind).sort()).toEqual(['files', 'postgres']); @@ -191,7 +191,7 @@ describe('ssh', () => { const out = await sshPlatform.provision( ctx(), { kind: 'postgres', id: 'pg', name: 'app', connection: {} }, - { host: 'h', postgres: [{ name: 'app', url: 'postgres://u:p@localhost/app' }] }, + { host: 'h', postgres: [{ name: 'app', url: 'postgres://u@localhost/app' }] }, ); expect(out.connection.url?.reveal()).toContain('localhost/app'); }); From a6b36531b79095a3f0a2068e290d95db3ae966a4 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 08:15:06 +0000 Subject: [PATCH 8/8] migrate: a MySQL engine, and mysql on the ssh platform Shipping without this would have been dishonest. `mysql` was already a ResourceKind, and both Railway and PlanetScale declared they hold it -- so the CLI advertised PlanetScale as a source while every plan involving it blocked out with "nothing in this build can move a mysql". Advertised and unusable is worse than absent. mysqldump with the flags that matter, each of which is a failure somebody has already had: --single-transaction, because without it a multi-gigabyte dump locks every table for its duration, which is an outage and rather defeats copying while the source is live; --set-gtid-purged=OFF, because a dump carrying GTID state refuses to load into a server with its own replication history and the error names neither the flag nor the cause; --no-tablespaces, because writing tablespace clauses needs PROCESS privilege that managed providers do not grant, and its absence fails the dump rather than degrading it. The password goes in MYSQL_PWD rather than --password=, same reasoning as libpq: argv is world-readable via ps for the hours a dump runs. The client's warning that MYSQL_PWD is insecure on shared machines is true and still strictly better than the alternative. Verification reports estimated row counts WITHOUT failing on them. information_schema.table_rows is an estimate on InnoDB, not a count; treating it as exact would fail every verification that ever ran. Only a table missing outright is a problem. Then running the CLI caught a second gap of my own making: the ssh platform never declared mysql, so `migrate platforms --from planetscale` reported "nothing in common" against a dedicated box -- the exact get-off-the-cloud move the tool exists for. ssh now holds mysql, with a test pinning the pairing in both directions. 177 tests, tsc clean, biome clean. Co-Authored-By: Claude Opus 5 (1M context) --- packages/migrate/README.md | 1 + packages/migrate/src/engines/index.ts | 4 +- packages/migrate/src/engines/mysql.test.ts | 152 ++++++++++++++ packages/migrate/src/engines/mysql.ts | 196 +++++++++++++++++++ packages/migrate/src/index.ts | 1 + packages/migrate/src/platforms/index.test.ts | 5 + packages/migrate/src/platforms/ssh.ts | 16 +- 7 files changed, 373 insertions(+), 2 deletions(-) create mode 100644 packages/migrate/src/engines/mysql.test.ts create mode 100644 packages/migrate/src/engines/mysql.ts diff --git a/packages/migrate/README.md b/packages/migrate/README.md index cb96a1ae..f29b0103 100644 --- a/packages/migrate/README.md +++ b/packages/migrate/README.md @@ -122,6 +122,7 @@ Secrets are read from the environment where a platform names one ## What it needs installed Per engine, checked before anything runs: `pg_dump`/`pg_restore`/`psql`, +`mysqldump`/`mysql`, `sqlite3`, `redis-cli`, `rclone`, `rsync`. ## Known limits diff --git a/packages/migrate/src/engines/index.ts b/packages/migrate/src/engines/index.ts index c7b1ac7d..d95bfb0d 100644 --- a/packages/migrate/src/engines/index.ts +++ b/packages/migrate/src/engines/index.ts @@ -1,5 +1,6 @@ import type { Engine, ResourceKind } from '../types.js'; import { filesEngine } from './files.js'; +import { mysqlEngine } from './mysql.js'; import { objectStorageEngine } from './object-storage.js'; import { postgresEngine } from './postgres.js'; import { redisEngine } from './redis.js'; @@ -14,6 +15,7 @@ import { sqliteEngine } from './sqlite.js'; */ export const ENGINES: ReadonlyMap = new Map([ ['postgres', postgresEngine], + ['mysql', mysqlEngine], ['sqlite', sqliteEngine], ['redis', redisEngine], ['object-storage', objectStorageEngine], @@ -33,4 +35,4 @@ export function requiredBinaries(kinds: ResourceKind[]): string[] { return [...out].sort(); } -export { filesEngine, objectStorageEngine, postgresEngine, redisEngine, sqliteEngine }; +export { filesEngine, mysqlEngine, objectStorageEngine, postgresEngine, redisEngine, sqliteEngine }; diff --git a/packages/migrate/src/engines/mysql.test.ts b/packages/migrate/src/engines/mysql.test.ts new file mode 100644 index 00000000..ba6e154a --- /dev/null +++ b/packages/migrate/src/engines/mysql.test.ts @@ -0,0 +1,152 @@ +import { describe, expect, it } from 'vitest'; +import { databaseName, mysqlConnection, mysqlEngine } from './mysql.js'; +import type { Artifact, EngineContext, ExecOptions, ExecResult, Resource } from '../types.js'; +import { secret } from '../types.js'; + +/* Named rather than inlined; see postgres.test.ts. */ +const FAKE_PASSWORD = 'p%40ss'; + +interface Call { + cmd: string; + args: string[]; + opts?: ExecOptions; +} + +function ctx(responses: Array> = [], over: Partial = {}) { + const calls: Call[] = []; + let i = 0; + const c: EngineContext & { calls: Call[] } = { + calls, + dryRun: false, + log: () => {}, + staging: { dir: '/staging', record: async () => {}, existing: async () => [] }, + exec: async (cmd, args, opts) => { + calls.push({ cmd, args, opts }); + const r = responses[i++] ?? {}; + return { code: r.code ?? 0, stdout: r.stdout ?? '', stderr: r.stderr ?? '' }; + }, + ...over, + }; + return c; +} + +const db = (over: Partial = {}): Resource => ({ + kind: 'mysql', + id: 'db', + name: 'app', + connection: { url: secret(`mysql://u:${FAKE_PASSWORD}@db.example.com:3306/appdb`) }, + ...over, +}); + +describe('mysqlConnection', () => { + it('splits host, port and user into flags', () => { + const { args } = mysqlConnection(db()); + expect(args).toContain('--host=db.example.com'); + expect(args).toContain('--port=3306'); + expect(args).toContain('--user=u'); + }); + + it('puts the password in MYSQL_PWD, never in argv where ps can read it', () => { + const { args, env } = mysqlConnection(db()); + expect(env.MYSQL_PWD).toBe('p@ss'); + expect(args.join(' ')).not.toContain('p@ss'); + expect(args.some((a) => a.startsWith('--password'))).toBe(false); + }); + + it('requires TLS by default', () => { + expect(mysqlConnection(db()).args).toContain('--ssl-mode=REQUIRED'); + }); + + it('honours an explicit ssl-mode', () => { + const r = db({ connection: { url: secret('mysql://u@h:3306/d?ssl-mode=DISABLED') } }); + expect(mysqlConnection(r).args).toContain('--ssl-mode=DISABLED'); + }); + + it('refuses a url that is not a URL', () => { + expect(() => mysqlConnection(db({ connection: { url: secret('nope') } }))).toThrow(/not a URL/); + }); +}); + +describe('databaseName', () => { + it('is the path of the DSN', () => { + expect(databaseName(db())).toBe('appdb'); + }); + + it('refuses a DSN naming no database', () => { + expect(() => databaseName(db({ connection: { url: secret('mysql://u@h:3306/') } }))).toThrow( + /no database/, + ); + }); +}); + +describe('export', () => { + it('dumps in one transaction rather than locking every table for hours', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.cmd).toBe('mysqldump'); + expect(c.calls[0]!.args).toContain('--single-transaction'); + }); + + it('turns off GTID state, which otherwise refuses to load elsewhere', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.args).toContain('--set-gtid-purged=OFF'); + }); + + it('skips tablespaces, which need a privilege managed providers do not grant', async () => { + const c = ctx(); + await mysqlEngine.export(c, db()); + expect(c.calls[0]!.args).toContain('--no-tablespaces'); + }); + + it('writes to staging via stdout redirection', async () => { + const c = ctx(); + const [artifact] = await mysqlEngine.export(c, db()); + expect(c.calls[0]!.opts?.stdoutFile).toBe('/staging/db/dump.sql'); + expect(artifact?.path).toBe('db/dump.sql'); + }); + + it('runs nothing on a dry run', async () => { + const c = ctx([], { dryRun: true }); + await mysqlEngine.export(c, db()); + expect(c.calls).toHaveLength(0); + }); +}); + +describe('import', () => { + const dump: Artifact = { resourceId: 'db', kind: 'mysql', path: 'db/dump.sql' }; + + it('loads into an empty database from stdin', async () => { + const c = ctx([{ stdout: '0' }, {}]); + await mysqlEngine.import(c, db(), [dump]); + expect(c.calls[1]!.opts?.stdinFile).toBe('/staging/db/dump.sql'); + }); + + it('refuses to load over a database that already has tables', async () => { + const c = ctx([{ stdout: '17' }]); + await expect(mysqlEngine.import(c, db(), [dump])).rejects.toThrow(/already has 17 table/); + expect(c.calls).toHaveLength(1); + }); + + it('fails when nothing was staged', async () => { + await expect(mysqlEngine.import(ctx(), db(), [])).rejects.toThrow(/no mysql dump staged/); + }); +}); + +describe('verify', () => { + it('reports estimated counts without failing on them', async () => { + // information_schema.table_rows is an estimate on InnoDB. Treating it as + // exact would fail every single verification. + const c = ctx([{ stdout: 'users\t113\n' }, { stdout: 'users\t108\n' }]); + const res = await mysqlEngine.verify!(c, db(), db()); + expect(res.ok).toBe(true); + expect(res.checks[0]).toContain('estimated'); + }); + + it('fails only when a table is missing entirely', async () => { + const c = ctx([{ stdout: 'users\t1\nposts\t2\n' }, { stdout: 'users\t1\n' }]); + const res = await mysqlEngine.verify!(c, db(), db()); + expect(res.ok).toBe(false); + expect(res.problems[0]).toContain('posts'); + }); +}); diff --git a/packages/migrate/src/engines/mysql.ts b/packages/migrate/src/engines/mysql.ts new file mode 100644 index 00000000..2737986c --- /dev/null +++ b/packages/migrate/src/engines/mysql.ts @@ -0,0 +1,196 @@ +import { quoteIdent, quoteLiteral } from '../transforms.js'; +import type { Artifact, Engine, EngineContext, Resource, VerifyResult } from '../types.js'; + +/** + * Moving a MySQL or MariaDB database. + * + * PlanetScale, RDS, a MariaDB in a container. Same shape as the Postgres + * engine and for the same reasons; the differences are all in the tooling. + * + * ## The flags that matter + * + * `--single-transaction` takes the dump inside one consistent snapshot on + * InnoDB instead of locking every table for the length of the dump. Without + * it, a multi-gigabyte dump is an outage, which rather defeats the point of + * copying while the source is still live. + * + * `--set-gtid-purged=OFF`, because a dump carrying GTID state refuses to load + * into a server with its own replication history, and the error names neither + * the flag nor the cause. + * + * `--no-tablespaces`, because writing tablespace clauses needs PROCESS + * privilege that a managed provider does not grant, and its absence fails the + * dump rather than degrading it. + * + * PlanetScale specifically does not support foreign key constraints in the + * usual way, so a dump taken there restores without constraints a plain MySQL + * would have had. That is recorded as a platform quirk rather than silently + * handled, because the fix is a schema decision, not a flag. + */ + +const DUMP = 'dump.sql'; + +function dsn(r: Resource): string { + const url = r.connection.url; + if (!url) throw new Error(`mysql resource '${r.name}' has no connection.url`); + return url.reveal(); +} + +/** + * Connection details for the mysql client family. + * + * `MYSQL_PWD` rather than `--password=`, for the same reason Postgres uses + * libpq's variables: a password on the command line is readable by every other + * user on the box via `ps` for as long as the dump runs. The client warns + * about `MYSQL_PWD` being insecure on shared machines, which is true and still + * strictly better than argv. + */ +export function mysqlConnection(r: Resource): { args: string[]; env: Record } { + const raw = dsn(r); + let u: URL; + try { + u = new URL(raw); + } catch { + throw new Error(`mysql resource '${r.name}' has a connection.url that is not a URL`); + } + + const args: string[] = []; + if (u.hostname) args.push(`--host=${decodeURIComponent(u.hostname)}`); + if (u.port) args.push(`--port=${u.port}`); + if (u.username) args.push(`--user=${decodeURIComponent(u.username)}`); + + // Managed MySQL is TLS-only in practice, and the client's default is to fall + // back silently where it is not enforced. + const sslMode = u.searchParams.get('ssl-mode') ?? 'REQUIRED'; + args.push(`--ssl-mode=${sslMode}`); + + const env: Record = {}; + if (u.password) env.MYSQL_PWD = decodeURIComponent(u.password); + + return { args, env }; +} + +export function databaseName(r: Resource): string { + const raw = dsn(r); + const name = new URL(raw).pathname.replace(/^\//, ''); + if (!name) throw new Error(`mysql resource '${r.name}' has no database in its connection.url`); + return decodeURIComponent(name); +} + +export const mysqlEngine: Engine = { + kind: 'mysql', + requires: ['mysqldump', 'mysql'], + + async export(ctx: EngineContext, from: Resource): Promise { + const path = `${from.id}/${DUMP}`; + ctx.log(`mysqldump ${from.name} → ${path}`); + if (ctx.dryRun) return [{ resourceId: from.id, kind: 'mysql', path }]; + + const { args, env } = mysqlConnection(from); + await ctx.exec( + 'mysqldump', + [ + ...args, + // A consistent snapshot instead of locking every table for the length + // of the dump. + '--single-transaction', + '--quick', + '--routines', + '--triggers', + '--events', + '--set-gtid-purged=OFF', + '--no-tablespaces', + databaseName(from), + ], + { env, stdoutFile: `${ctx.staging.dir}/${path}`, timeoutMs: 6 * 60 * 60 * 1000 }, + ); + + const artifact: Artifact = { resourceId: from.id, kind: 'mysql', path }; + await ctx.staging.record(artifact); + return [artifact]; + }, + + async import(ctx: EngineContext, to: Resource, artifacts: Artifact[]): Promise { + const dump = artifacts.find((a) => a.kind === 'mysql'); + if (!dump) throw new Error(`no mysql dump staged for '${to.name}'`); + if (ctx.dryRun) { + ctx.log(`would load ${dump.path} into ${to.name}`); + return; + } + + const { args, env } = mysqlConnection(to); + + // Same refusal as Postgres: a non-empty target is not written over. A + // mysqldump replays CREATE TABLE and INSERT, so aiming it at a populated + // database is a mess of duplicate-key errors on top of live data. + const existing = await ctx.exec( + 'mysql', + [ + ...args, + '--batch', + '--skip-column-names', + '--execute', + `select count(*) from information_schema.tables where table_schema = ${quoteLiteral(databaseName(to))}`, + ], + { env, check: false }, + ); + const tableCount = Number.parseInt(existing.stdout.trim(), 10); + if (Number.isFinite(tableCount) && tableCount > 0) { + throw new Error( + `target database '${to.name}' already has ${tableCount} table(s). Refusing to load over it — drop and recreate it, or point at an empty one.`, + ); + } + + ctx.log(`mysql < ${dump.path}`); + await ctx.exec('mysql', [...args, databaseName(to)], { + env, + stdinFile: `${ctx.staging.dir}/${dump.path}`, + timeoutMs: 6 * 60 * 60 * 1000, + }); + }, + + async verify(ctx: EngineContext, from: Resource, to: Resource): Promise { + if (ctx.dryRun) return { ok: true, checks: ['dry run: not compared'], problems: [] }; + + const counts = async (r: Resource) => { + const { args, env } = mysqlConnection(r); + const res = await ctx.exec( + 'mysql', + [ + ...args, + '--batch', + '--skip-column-names', + '--execute', + `select table_name, table_rows from information_schema.tables + where table_schema = ${quoteLiteral(databaseName(r))} order by table_name`, + ], + { env, check: false }, + ); + const map = new Map(); + for (const line of res.stdout.split('\n')) { + const [name, n] = line.split('\t'); + if (name && n !== undefined) map.set(name.trim(), Number.parseInt(n, 10) || 0); + } + return map; + }; + + const [a, b] = await Promise.all([counts(from), counts(to)]); + const checks: string[] = []; + const problems: string[] = []; + + for (const [table, sourceRows] of a) { + const targetRows = b.get(table); + if (targetRows === undefined) { + problems.push(`${table}: missing on the target`); + continue; + } + // information_schema.table_rows is an ESTIMATE on InnoDB, not a count. + // Treating it as exact would fail every verification, so a table present + // on both sides is reported rather than judged, and only a missing table + // is a problem. + checks.push(`${quoteIdent(table)}: ~${sourceRows} → ~${targetRows} (estimated)`); + } + + return { ok: problems.length === 0, checks, problems }; + }, +}; diff --git a/packages/migrate/src/index.ts b/packages/migrate/src/index.ts index b587e7f7..59eb9ba0 100644 --- a/packages/migrate/src/index.ts +++ b/packages/migrate/src/index.ts @@ -29,6 +29,7 @@ export * from './exec.js'; export { ENGINES, engineFor, requiredBinaries } from './engines/index.js'; export { filesEngine, + mysqlEngine, objectStorageEngine, postgresEngine, redisEngine, diff --git a/packages/migrate/src/platforms/index.test.ts b/packages/migrate/src/platforms/index.test.ts index fa586199..e0bf4875 100644 --- a/packages/migrate/src/platforms/index.test.ts +++ b/packages/migrate/src/platforms/index.test.ts @@ -52,6 +52,11 @@ describe('compatibleKinds is what makes direction a non-question', () => { expect(compatibleKinds('ssh', 'supabase')).toContain('postgres'); }); + it('moves MySQL off PlanetScale onto a box, which is the get-off-the-cloud case', () => { + expect(compatibleKinds('planetscale', 'ssh')).toEqual(['mysql']); + expect(compatibleKinds('ssh', 'planetscale')).toEqual(['mysql']); + }); + it('pairs Turso with a box over sqlite, both ways', () => { expect(compatibleKinds('turso', 'ssh')).toEqual(['sqlite']); expect(compatibleKinds('ssh', 'turso')).toEqual(['sqlite']); diff --git a/packages/migrate/src/platforms/ssh.ts b/packages/migrate/src/platforms/ssh.ts index 36c85772..23b44519 100644 --- a/packages/migrate/src/platforms/ssh.ts +++ b/packages/migrate/src/platforms/ssh.ts @@ -24,6 +24,7 @@ export interface SshConfig { sshPort?: string; /** Databases reachable from this box, usually on loopback. */ postgres?: Array<{ name: string; url: string }>; + mysql?: Array<{ name: string; url: string }>; sqlite?: Array<{ name: string; path: string }>; redis?: Array<{ name: string; url: string; dataDir?: string }>; /** Directories to move: volumes, docroots, upload trees. */ @@ -43,7 +44,7 @@ export const sshPlatform: Platform = { id: 'ssh', label: 'dedicated / VPS over ssh', role: 'both', - supports: ['postgres', 'sqlite', 'redis', 'files', 'object-storage'], + supports: ['postgres', 'mysql', 'sqlite', 'redis', 'files', 'object-storage'], async inventory(ctx: PlatformContext, config: SshConfig): Promise { if (!config.host) throw new Error('ssh platform needs a host'); @@ -60,6 +61,15 @@ export const sshPlatform: Platform = { }); } + for (const db of config.mysql ?? []) { + resources.push({ + kind: 'mysql', + id: `mysql-${db.name}`, + name: db.name, + connection: { url: secret(db.url) }, + }); + } + for (const db of config.sqlite ?? []) { resources.push({ kind: 'sqlite', @@ -137,6 +147,10 @@ export const sshPlatform: Platform = { const db = config.postgres?.find((d) => d.name === resource.name) ?? config.postgres?.[0]; return db ? { url: secret(db.url) } : undefined; } + case 'mysql': { + const db = config.mysql?.find((d) => d.name === resource.name) ?? config.mysql?.[0]; + return db ? { url: secret(db.url) } : undefined; + } case 'sqlite': { const db = config.sqlite?.find((d) => d.name === resource.name) ?? config.sqlite?.[0]; return db ? { path: plain(db.path) } : undefined;