From 2c1fcf5803a54bc05e75e8f9d514c1da7bafc5f3 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 28 Aug 2026 15:14:50 +0000 Subject: [PATCH 1/3] fix(hardware): repoint dead Pioneer DJ manual links The downloads.support.alphatheta.com/manuals tree no longer exists, so every link under it 404s. AlphaTheta moved manual content into support articles. EFX-1000 has a dedicated Specifications article holding the same content the dead ?page=28 anchor pointed at, so it keeps its title. The other six get the instruction-manual article and are retitled, since the per-page anchor no longer exists to link. Fixes #654 Co-Authored-By: Claude --- data/hardware/pioneer-dj-efx-1000.yaml | 2 +- data/hardware/pioneer-dj-efx-500-r.yaml | 4 ++-- data/hardware/pioneer-dj-efx-500.yaml | 4 ++-- data/hardware/pioneer-dj-rmx-1000-m.yaml | 4 ++-- data/hardware/pioneer-dj-rmx-1000-w.yaml | 4 ++-- data/hardware/pioneer-dj-rmx-1000.yaml | 4 ++-- data/hardware/pioneer-dj-rmx-500.yaml | 4 ++-- 7 files changed, 13 insertions(+), 13 deletions(-) diff --git a/data/hardware/pioneer-dj-efx-1000.yaml b/data/hardware/pioneer-dj-efx-1000.yaml index b953430c3..1b1861ce4 100644 --- a/data/hardware/pioneer-dj-efx-1000.yaml +++ b/data/hardware/pioneer-dj-efx-1000.yaml @@ -19,6 +19,6 @@ links: - url: https://support.alphatheta.com/products/4416594001049 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/EFX-1000/EFX-1000_DRB1368_manual?page=28 + - url: https://support.alphatheta.com/articles/24199943617177 type: resource title: Specifications diff --git a/data/hardware/pioneer-dj-efx-500-r.yaml b/data/hardware/pioneer-dj-efx-500-r.yaml index 30e8d21c6..53e1fc22d 100644 --- a/data/hardware/pioneer-dj-efx-500-r.yaml +++ b/data/hardware/pioneer-dj-efx-500-r.yaml @@ -16,6 +16,6 @@ links: - url: https://support.pioneerdj.com/hc/sections/4416580053785-EFX-500 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/EFX-500/EFX-500_DRB1236_manual?page=26 + - url: https://support.alphatheta.com/articles/4550917677209 type: resource - title: Specifications + title: Instruction Manual diff --git a/data/hardware/pioneer-dj-efx-500.yaml b/data/hardware/pioneer-dj-efx-500.yaml index 7412b1f44..56caddc94 100644 --- a/data/hardware/pioneer-dj-efx-500.yaml +++ b/data/hardware/pioneer-dj-efx-500.yaml @@ -16,6 +16,6 @@ links: - url: https://support.alphatheta.com/products/4416580053785 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/EFX-500/EFX-500_DRB1236_manual?page=26 + - url: https://support.alphatheta.com/articles/4550917677209 type: resource - title: Specifications + title: Instruction Manual diff --git a/data/hardware/pioneer-dj-rmx-1000-m.yaml b/data/hardware/pioneer-dj-rmx-1000-m.yaml index efdef86bf..8cad94095 100644 --- a/data/hardware/pioneer-dj-rmx-1000-m.yaml +++ b/data/hardware/pioneer-dj-rmx-1000-m.yaml @@ -29,6 +29,6 @@ links: - url: https://support.alphatheta.com/products/4416587889433 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/RMX-1000/RMX-1000_DRB1587D_manual?page=24 + - url: https://support.alphatheta.com/articles/4404659858201 type: resource - title: Specifications + title: Instruction Manual diff --git a/data/hardware/pioneer-dj-rmx-1000-w.yaml b/data/hardware/pioneer-dj-rmx-1000-w.yaml index 643a9a28c..20b4fcd81 100644 --- a/data/hardware/pioneer-dj-rmx-1000-w.yaml +++ b/data/hardware/pioneer-dj-rmx-1000-w.yaml @@ -29,6 +29,6 @@ links: - url: https://support.alphatheta.com/products/4416587889433 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/RMX-1000/RMX-1000_DRB1587D_manual?page=24 + - url: https://support.alphatheta.com/articles/4404659858201 type: resource - title: Specifications + title: Instruction Manual diff --git a/data/hardware/pioneer-dj-rmx-1000.yaml b/data/hardware/pioneer-dj-rmx-1000.yaml index da20313b0..d2f675a21 100644 --- a/data/hardware/pioneer-dj-rmx-1000.yaml +++ b/data/hardware/pioneer-dj-rmx-1000.yaml @@ -32,6 +32,6 @@ links: - url: https://support.alphatheta.com/products/4416587889433 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/RMX-1000/RMX-1000_DRB1587D_manual?page=25 + - url: https://support.alphatheta.com/articles/4404659858201 type: resource - title: Specifications + title: Instruction Manual diff --git a/data/hardware/pioneer-dj-rmx-500.yaml b/data/hardware/pioneer-dj-rmx-500.yaml index d8521bdfb..5f5410ad2 100644 --- a/data/hardware/pioneer-dj-rmx-500.yaml +++ b/data/hardware/pioneer-dj-rmx-500.yaml @@ -19,6 +19,6 @@ links: - url: https://support.alphatheta.com/products/4416580089753 type: support title: Support - - url: https://downloads.support.alphatheta.com/manuals/dj-effectors/RMX-500/RMX-500_DRH1251A_quickstart-manual?page=14 + - url: https://support.alphatheta.com/articles/4404659866137 type: resource - title: Specifications + title: Instruction Manual From 3ee6c7d06f2afc698cf2c4a940e2c27f22029102 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 28 Aug 2026 15:32:48 +0000 Subject: [PATCH 2/3] refactor(scripts): derive patch SQL from the built database generate-patch.ts encoded the YAML a second time and covered 16 of the ~60 tables that build-sqlite.ts writes. A patched database kept stale io, prices, links, versions and variants, held raw markdown in HTML columns, and carried unnormalized category aliases. It now reflects the table graph from the built schema and reads rows back out, so one encoder exists and a new child table is picked up automatically. Root rows are upserted rather than deleted, since another entry's supersedes_id can reference them. Nested rows resolve their parent by natural key, because assigned ids are per-build. The version stamp is written only when every change resolved. Verified against a real range: applying the patch to the older database reproduces all 270 changed entries exactly as a full rebuild does. Fixes #657 Co-Authored-By: Claude --- CLAUDE.md | 32 + scripts/__tests__/generate-patch.test.ts | 229 +++++++ scripts/generate-patch.ts | 834 +++++++++++++---------- 3 files changed, 719 insertions(+), 376 deletions(-) create mode 100644 scripts/__tests__/generate-patch.test.ts diff --git a/CLAUDE.md b/CLAUDE.md index e998b5379..ec44776af 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -55,6 +55,38 @@ cohort working: - Use Prettier for YAML formatting - Data follows strict schemas defined in `scripts/lib/types.ts` +### Incremental Patches + +`pnpm patch` emits SQL that brings a released `catalog.sqlite` up to HEAD. It +reads the rows out of a freshly built database rather than encoding the YAML a +second time, and that is the whole point of its shape. + +- **Never hand-write a table's INSERT into `generate-patch.ts`.** It used to, + and covered sixteen of the sixty-odd tables `build-sqlite.ts` writes: I/O, + prices, links, versions and variants were left stale on a patched database, + markdown reached HTML columns unrendered, and category aliases went in + unnormalized. The table graph is reflected from the built schema, so a new + child table or column is carried for free. Adding a hand-written block + reintroduces the drift. +- **A collection owns its child tables by name prefix.** `content_compatibility` + references both `content` and `software`, but only `content` owns it. A new + child table has to carry its collection's prefix and a foreign key reaching + the root, or reflection fails loudly rather than skipping it. +- **A rewritten entry's root row is upserted, never deleted.** Another entry's + `supersedes_id` can point at it, and dropping the row trips that key even + though it is about to come straight back. +- **A nested row resolves its parent by natural key.** `hardware_variants.id` + is assigned per build, so the target database has its own numbering and + carrying the build's ids across would attach prices to the wrong variant. +- **The version stamp is written only when every change resolved.** A patch + that skips a change and still stamps leaves a database claiming content it + does not have, which is what makes the other failures hard to diagnose after + the fact. + +A patch only rewrites entries whose YAML changed, so renaming a manufacturer +leaves the denormalized `manufacturer_name` stale in the FTS rows of its +unchanged products. Ship a full database when manufacturer names move. + ## Data Entry Format Manufacturers require: name, url diff --git a/scripts/__tests__/generate-patch.test.ts b/scripts/__tests__/generate-patch.test.ts new file mode 100644 index 000000000..80b9ea802 --- /dev/null +++ b/scripts/__tests__/generate-patch.test.ts @@ -0,0 +1,229 @@ +import fs from "node:fs"; +import path from "node:path"; + +import Database from "better-sqlite3"; +import { beforeEach, describe, expect, it } from "vitest"; + +import { + type CollectionSchema, + deleteStatements, + insertStatements, + reflectCollection, +} from "../generate-patch.js"; + +const SCHEMA_SQL = fs.readFileSync(path.join(import.meta.dirname, "..", "schema.sql"), "utf-8"); + +function emptyDatabase(): Database.Database { + const db = new Database(":memory:"); + db.exec(SCHEMA_SQL); + return db; +} + +/** A hardware entry deep enough to exercise every level of the table graph. */ +function seedHardware(db: Database.Database, id: string, name: string): void { + db.prepare(`INSERT INTO manufacturers (id, name) VALUES ('mfg1', 'Maker')`).run(); + db.prepare( + `INSERT INTO hardware (id, name, manufacturer_id, primary_category, description) + VALUES (?, ?, 'mfg1', 'pedal', '

A pedal.

')` + ).run(id, name); + db.prepare(`INSERT INTO hardware_categories (hardware_id, category) VALUES (?, 'effect')`).run( + id + ); + db.prepare( + `INSERT INTO hardware_capabilities (hardware_id, capability) VALUES (?, 'reverb')` + ).run(id); + db.prepare( + `INSERT INTO hardware_io (hardware_id, port_key, name, signal_flow, category, type, connection, max_connections) + VALUES (?, 'in-1', 'Input', 'input', 'audio', 'instrument', '1/4-inch', 1)` + ).run(id); + db.prepare( + `INSERT INTO hardware_prices (hardware_id, amount, currency) VALUES (?, 199, 'USD')` + ).run(id); + db.prepare( + `INSERT INTO hardware_variants (hardware_id, name, slug) VALUES (?, 'Black', 'black')` + ).run(id); + db.prepare( + `INSERT INTO hardware_variant_prices (variant_id, amount, currency) + VALUES ((SELECT id FROM hardware_variants WHERE hardware_id = ? AND slug = 'black'), 219, 'USD')` + ).run(id); + db.prepare( + `INSERT INTO hardware_fts (id, name, manufacturer_name, categories, description, search_terms) + VALUES (?, ?, 'Maker', 'effect', 'A pedal.', '')` + ).run(id, name); +} + +describe("reflectCollection", () => { + let db: Database.Database; + let schema: CollectionSchema; + + beforeEach(() => { + db = emptyDatabase(); + schema = reflectCollection(db, "hardware"); + }); + + it("finds every child table the build writes, not a hand-maintained subset", () => { + const names = schema.children.map((c) => c.name); + // The drift this script existed to cause: io, prices, links and versions + // were among the tables a patch silently left stale. + expect(names).toEqual( + expect.arrayContaining([ + "hardware_io", + "hardware_prices", + "hardware_links", + "hardware_versions", + "hardware_variants", + "hardware_search_terms", + "hardware_translations", + "hardware_capabilities", + ]) + ); + }); + + it("nests a table that hangs off a child deeper than the child", () => { + const variants = schema.children.find((c) => c.name === "hardware_variants")!; + const variantPrices = schema.children.find((c) => c.name === "hardware_variant_prices")!; + expect(variants.depth).toBe(1); + expect(variantPrices.depth).toBe(2); + expect(variantPrices.parent).toBe("hardware_variants"); + }); + + it("omits ids SQLite assigns and keeps composite keys", () => { + const prices = schema.children.find((c) => c.name === "hardware_prices")!; + const categories = schema.children.find((c) => c.name === "hardware_categories")!; + expect(prices.columns).not.toContain("id"); + expect(categories.columns).toEqual(["hardware_id", "category"]); + }); + + it("treats a cross-collection reference as owned by its own collection", () => { + // content_hardware_compatibility references hardware, but content owns it. + expect(schema.children.map((c) => c.name)).not.toContain("content_hardware_compatibility"); + const content = reflectCollection(db, "content"); + expect(content.children.map((c) => c.name)).toContain("content_hardware_compatibility"); + }); + + it("picks up the collection's full-text table without its shadow tables", () => { + expect(schema.fts?.name).toBe("hardware_fts"); + expect(schema.children.map((c) => c.name)).not.toContain("hardware_fts_data"); + }); +}); + +describe("deleteStatements", () => { + it("clears a nested table through its parent, deepest first", () => { + const db = emptyDatabase(); + const schema = reflectCollection(db, "hardware"); + const sql = deleteStatements(schema, "hw1", true); + + const variantPrices = sql.findIndex((s) => s.includes('DELETE FROM "hardware_variant_prices"')); + const variants = sql.findIndex((s) => s.includes('DELETE FROM "hardware_variants"')); + expect(variantPrices).toBeLessThan(variants); + expect(sql[variantPrices]).toContain('SELECT id FROM "hardware_variants"'); + }); + + it("leaves the root row alone when the entry is being rewritten", () => { + const db = emptyDatabase(); + const schema = reflectCollection(db, "hardware"); + // Another entry's supersedes_id may point at it, so it is upserted instead. + expect(deleteStatements(schema, "hw1", false)).not.toContain( + `DELETE FROM "hardware" WHERE id = 'hw1';` + ); + expect(deleteStatements(schema, "hw1", true)).toContain( + `DELETE FROM "hardware" WHERE id = 'hw1';` + ); + }); +}); + +describe("insertStatements", () => { + it("returns null for an entry the built database does not have", () => { + const db = emptyDatabase(); + const schema = reflectCollection(db, "hardware"); + expect(insertStatements(db, schema, "missing")).toBeNull(); + }); + + it("resolves a nested row's parent by natural key, never by assigned id", () => { + const built = emptyDatabase(); + seedHardware(built, "hw1", "Pedal"); + const schema = reflectCollection(built, "hardware"); + + const price = insertStatements(built, schema, "hw1")!.find((s) => + s.includes('INSERT INTO "hardware_variant_prices"') + )!; + expect(price).toContain( + `(SELECT id FROM "hardware_variants" WHERE "hardware_id" = 'hw1' AND "slug" = 'black')` + ); + }); + + it("reproduces the built rows in a database that numbers its ids differently", () => { + const built = emptyDatabase(); + seedHardware(built, "hw1", "Pedal"); + const schema = reflectCollection(built, "hardware"); + + // The target already holds another entry, so its variant ids do not line up + // with the build's. This is the case that makes carrying ids across wrong. + const target = emptyDatabase(); + seedHardware(target, "hw0", "Other"); + target.prepare(`INSERT INTO hardware (id, name) VALUES ('hw1', 'Stale')`).run(); + target + .prepare(`INSERT INTO hardware_categories (hardware_id, category) VALUES ('hw1', 'gone')`) + .run(); + + const sql = [ + ...deleteStatements(schema, "hw1", false), + ...insertStatements(built, schema, "hw1")!, + ].join("\n"); + target.exec(sql); + + const rowsFor = (db: Database.Database, table: string, where: string) => + db.prepare(`SELECT * FROM ${table} WHERE ${where}`).all(); + + expect(rowsFor(target, "hardware", `id = 'hw1'`)).toEqual( + rowsFor(built, "hardware", `id = 'hw1'`) + ); + expect(rowsFor(target, "hardware_categories", `hardware_id = 'hw1'`)).toEqual([ + { hardware_id: "hw1", category: "effect" }, + ]); + + const variantPrice = target + .prepare( + `SELECT p.amount, v.slug FROM hardware_variant_prices p + JOIN hardware_variants v ON v.id = p.variant_id + WHERE v.hardware_id = 'hw1'` + ) + .all(); + expect(variantPrice).toEqual([{ amount: 219, slug: "black" }]); + + // The untouched entry keeps its own rows. + expect( + rowsFor( + target, + "hardware_variant_prices", + `variant_id IN (SELECT id FROM hardware_variants WHERE hardware_id = 'hw0')` + ) + ).toHaveLength(1); + }); + + it("upserts the root row rather than replacing it, so inbound references survive", () => { + const built = emptyDatabase(); + seedHardware(built, "hw1", "Pedal"); + const schema = reflectCollection(built, "hardware"); + + const target = emptyDatabase(); + seedHardware(target, "hw1", "Stale"); + target.pragma("foreign_keys = ON"); + target + .prepare(`INSERT INTO hardware (id, name, supersedes_id) VALUES ('hw2', 'Successor', 'hw1')`) + .run(); + + target.exec( + [...deleteStatements(schema, "hw1", false), ...insertStatements(built, schema, "hw1")!].join( + "\n" + ) + ); + + expect(target.prepare(`SELECT name FROM hardware WHERE id = 'hw1'`).get()).toEqual({ + name: "Pedal", + }); + expect(target.prepare(`SELECT supersedes_id FROM hardware WHERE id = 'hw2'`).get()).toEqual({ + supersedes_id: "hw1", + }); + }); +}); diff --git a/scripts/generate-patch.ts b/scripts/generate-patch.ts index 8972d925a..24431d908 100644 --- a/scripts/generate-patch.ts +++ b/scripts/generate-patch.ts @@ -2,52 +2,240 @@ * Patch Generation Script * * Compares YAML changes since the last release and generates SQL patch files. - * Run with: pnpm patch [from-version] [to-version] + * Run with: pnpm patch [from-version] [to-version] [--db ] [--skip-build] + * + * The statements are derived from a freshly built `catalog.sqlite` rather than + * re-implemented from the YAML. `build-sqlite.ts` normalizes categories, renders + * markdown to HTML, expands search-term synonyms, resolves `supersedes` and + * writes some sixty tables; a second hand-written encoder of all that drifts + * table by table, which is what this script used to be (it covered sixteen of + * those tables, wrote raw markdown into HTML columns, and inserted unnormalized + * category aliases). Reading rows back out of the built database keeps one + * encoder, so a new child table or column is picked up here for free. + * + * Known limitation: a patch only rewrites the entries whose YAML changed. + * Renaming a manufacturer leaves the denormalized `manufacturer_name` in the FTS + * rows of its unchanged products stale, where a full rebuild would refresh them. + * Ship a full database when manufacturer names move. */ import { execFileSync } from "node:child_process"; import fs from "node:fs"; import path from "node:path"; +import { pathToFileURL } from "node:url"; + +import Database from "better-sqlite3"; import { parse as parseYaml } from "yaml"; -import type { Accessory, Change, Content, Hardware, Manufacturer, Software } from "./lib/types.js"; -import { DATA_DIR, escapeSQL, normalizeMarkdown, OUTPUT_DIR } from "./lib/utils.js"; +import type { Change } from "./lib/types.js"; +import { DATA_DIR, escapeSQL, OUTPUT_DIR } from "./lib/utils.js"; const PATCHES_DIR = path.join(OUTPUT_DIR, "patches"); +const DEFAULT_DB = path.join(OUTPUT_DIR, "catalog.sqlite"); +const BUILD_SCRIPT = path.join(import.meta.dirname, "build-sqlite.ts"); + +/** + * Collection root tables, and the prefixes their child tables carry. + * + * Ownership is by name prefix rather than by foreign key: `content_compatibility` + * references both `content` and `software`, but only `content` owns it, and + * `accessories.manufacturer_id` references `manufacturers` without making + * accessories a child of it. Everything else about the shape (which column + * carries the link, which tables nest a level deeper, what columns each has) is + * read from the database. + */ +const COLLECTIONS: Record = { + manufacturers: { table: "manufacturers", prefixes: ["manufacturers_", "manufacturer_"] }, + software: { table: "software", prefixes: ["software_"] }, + content: { table: "content", prefixes: ["content_"] }, + hardware: { table: "hardware", prefixes: ["hardware_"] }, + accessories: { table: "accessories", prefixes: ["accessories_"] }, +}; + +/** Manufacturers must land before the products that reference them. */ +const COLLECTION_ORDER: Change["category"][] = [ + "manufacturers", + "software", + "content", + "hardware", + "accessories", +]; + +/** FTS5 keeps its index in shadow tables that are not ours to write. */ +const FTS_SHADOW_SUFFIX = /_(data|idx|content|docsize|config)$/; // ============================================================================= -// HELPERS +// SQL HELPERS // ============================================================================= -/** Build a map of manufacturer slug -> nanoid from all manufacturer YAML files */ -function buildManufacturerIdMap(): Map { - const map = new Map(); - const dir = path.join(DATA_DIR, "manufacturers"); - if (!fs.existsSync(dir)) return map; - - for (const file of fs.readdirSync(dir).filter((f) => f.endsWith(".yaml") || f.endsWith(".yml"))) { - const content = fs.readFileSync(path.join(dir, file), "utf-8"); - const data = parseYaml(content) as { id?: string }; - const slug = path.basename(file, path.extname(file)); - if (data.id) { - map.set(slug, data.id); +function quoteIdent(name: string): string { + return `"${name.replace(/"/g, '""')}"`; +} + +/** Render a value read back out of SQLite as a SQL literal. */ +function sqlLiteral(value: unknown): string { + if (value === null || value === undefined) return "NULL"; + if (typeof value === "number") return Number.isFinite(value) ? String(value) : "NULL"; + if (typeof value === "bigint") return String(value); + if (Buffer.isBuffer(value)) return `X'${value.toString("hex")}'`; + return escapeSQL(String(value)); +} + +function insertRow(table: string, columns: string[], values: string[]): string { + return ( + `INSERT INTO ${quoteIdent(table)} (${columns.map(quoteIdent).join(", ")}) ` + + `VALUES (${values.join(", ")});` + ); +} + +// ============================================================================= +// SCHEMA REFLECTION +// ============================================================================= + +interface ColumnInfo { + name: string; + pk: number; + type: string; +} + +export interface ChildTable { + /** Table name. */ + name: string; + /** Column on this table carrying the link to `parent`. */ + fkColumn: string; + /** Table this one hangs off: the collection root, or another child. */ + parent: string; + /** Columns to write, excluding any id SQLite assigns. */ + columns: string[]; + /** Distance from the root table (1 = direct child). */ + depth: number; +} + +export interface CollectionSchema { + root: string; + rootColumns: string[]; + /** Root-level full-text table, if the collection has one. */ + fts: { name: string; columns: string[] } | null; + /** Ordered parent-before-child. */ + children: ChildTable[]; + /** + * Natural keys for tables that other tables reference by assigned id. A + * patched database cannot reuse those ids, because a full build numbers them + * globally and the target database has its own numbering; children look their + * parent up by natural key instead. + */ + naturalKeys: Map; +} + +/** True for an id SQLite assigns, which must be left out of an INSERT. */ +function isAssignedPk(column: ColumnInfo, createSql: string): boolean { + if (column.pk !== 1 || column.type.toUpperCase() !== "INTEGER") return false; + // A composite primary key or a TEXT id is real data and has to be written. + return new RegExp(`\\b${column.name}\\s+INTEGER\\s+PRIMARY\\s+KEY`, "i").test(createSql); +} + +export function reflectCollection( + db: Database.Database, + category: Change["category"] +): CollectionSchema { + const { table: root, prefixes } = COLLECTIONS[category]; + + const objects = db + .prepare( + `SELECT name, sql FROM sqlite_master WHERE type = 'table' AND name NOT LIKE 'sqlite_%'` + ) + .all() as { name: string; sql: string | null }[]; + + const createSql = new Map(objects.map((o) => [o.name, o.sql ?? ""])); + const isVirtual = (name: string) => /CREATE\s+VIRTUAL\s+TABLE/i.test(createSql.get(name) ?? ""); + + const columnsOf = (table: string): ColumnInfo[] => + db.prepare(`PRAGMA table_info(${quoteIdent(table)})`).all() as ColumnInfo[]; + + const writableColumns = (table: string): string[] => { + const sql = createSql.get(table) ?? ""; + return columnsOf(table) + .filter((c) => !isAssignedPk(c, sql)) + .map((c) => c.name); + }; + + const owned = objects + .map((o) => o.name) + .filter((name) => name !== root && prefixes.some((p) => name.startsWith(p))) + .filter((name) => !FTS_SHADOW_SUFFIX.test(name)); + + const ftsName = owned.find((name) => isVirtual(name)) ?? null; + const fts = ftsName ? { name: ftsName, columns: columnsOf(ftsName).map((c) => c.name) } : null; + + // Walk outward from the root, so a table hanging off a child (variant links + // off variants) is discovered after the child it depends on. + const children: ChildTable[] = []; + const placed = new Set([root]); + const candidates = owned.filter((name) => !isVirtual(name)); + + for (let depth = 1; depth <= candidates.length; depth++) { + // Only tables placed in earlier passes count as parents, or a table + // discovered mid-pass would let its own child in at the same depth and the + // delete order would depend on the order sqlite_master happens to list them. + const reachable = new Set(placed); + let added = false; + for (const name of candidates) { + if (placed.has(name)) continue; + const fks = db.prepare(`PRAGMA foreign_key_list(${quoteIdent(name)})`).all() as { + table: string; + from: string; + }[]; + const link = fks.find((fk) => reachable.has(fk.table) && fk.table !== name); + if (!link) continue; + children.push({ + name, + fkColumn: link.from, + parent: link.table, + columns: writableColumns(name), + depth, + }); + placed.add(name); + added = true; } + if (!added) break; } - return map; -} -/** Get the nanoid of a deleted entry by reading it from git history */ -function getDeletedEntryId(since: string, filePath: string): string | null { - try { - const content = execFileSync("git", ["show", `${since}:${filePath}`], { - encoding: "utf-8", - stdio: ["pipe", "pipe", "pipe"], - }); - const data = parseYaml(content) as { id?: string }; - return data.id ?? null; - } catch { - return null; + const unreachable = candidates.filter((name) => !placed.has(name)); + if (unreachable.length > 0) { + throw new Error( + `Cannot patch ${root}: no foreign key path from ${unreachable.join(", ")}. ` + + `Add one, or move the table out of the ${root} name prefix.` + ); } + + // A child keyed on something other than the root's text id needs its parent + // resolved at apply time. + const naturalKeys = new Map(); + for (const child of children) { + if (child.parent === root || naturalKeys.has(child.parent)) continue; + const indexes = db.prepare(`PRAGMA index_list(${quoteIdent(child.parent)})`).all() as { + name: string; + unique: number; + }[]; + const unique = indexes.find((i) => i.unique === 1); + if (!unique) { + throw new Error( + `Cannot patch ${root}: ${child.name} references ${child.parent} by assigned id, ` + + `but ${child.parent} has no unique index to resolve it by.` + ); + } + const cols = ( + db.prepare(`PRAGMA index_info(${quoteIdent(unique.name)})`).all() as { name: string }[] + ).map((c) => c.name); + naturalKeys.set(child.parent, cols); + } + + return { root, rootColumns: writableColumns(root), fts, children, naturalKeys }; +} + +/** The column on `parent` that links it back toward the collection root. */ +function linkToRoot(schema: CollectionSchema, parent: string): string { + return schema.children.find((c) => c.name === parent)?.fkColumn ?? `${schema.root}_id`; } // ============================================================================= @@ -65,306 +253,164 @@ function getLatestTag(): string | null { } } -function getChangedFiles(since: string): Change[] { - const changes: Change[] = []; - +/** Read an entry's nanoid out of git history, for files no longer on disk. */ +function getDeletedEntryId(since: string, filePath: string): string | null { try { - // Get list of changed files - const output = execFileSync("git", ["diff", "--name-status", since, "HEAD", "--", "data/"], { + const content = execFileSync("git", ["show", `${since}:${filePath}`], { encoding: "utf-8", + stdio: ["pipe", "pipe", "pipe"], }); - - for (const line of output.split("\n").filter(Boolean)) { - const [status, file] = line.split("\t"); - if (!file) continue; - - // Parse the category from path - const parts = file.split("/"); - if (parts.length < 3) continue; - - const category = parts[1] as Change["category"]; - const filename = parts[2]; - const slug = filename.replace(/\.ya?ml$/, ""); - - let type: Change["type"]; - switch (status) { - case "A": - type = "added"; - break; - case "M": - type = "modified"; - break; - case "D": - type = "deleted"; - break; - default: - continue; - } - - changes.push({ type, category, file, slug }); - } - } catch (error) { - console.error("Failed to get git changes:", error); + return (parseYaml(content) as { id?: string }).id ?? null; + } catch { + return null; } - - return changes; } -// ============================================================================= -// SQL GENERATORS -// ============================================================================= - -function generateManufacturerSQL( - change: Change, - data: Manufacturer | null, - deletedId: string | null -): string[] { - const sql: string[] = []; - - if (change.type === "deleted") { - if (!deletedId) { - sql.push(`-- WARNING: Could not resolve ID for deleted manufacturer ${change.slug}`); - return sql; - } - sql.push(`DELETE FROM manufacturers WHERE id = ${escapeSQL(deletedId)};`); - } else if (change.type === "added" && data) { - sql.push( - `INSERT INTO manufacturers (id, name, company_name, parent_company, url, description, updated_at) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, ${escapeSQL(data.companyName)}, ${escapeSQL(data.parentCompany)}, ${escapeSQL(data.url)}, ${escapeSQL(data.description)}, datetime('now'));` - ); - } else if (change.type === "modified" && data) { - sql.push( - `UPDATE manufacturers SET name = ${escapeSQL(data.name)}, company_name = ${escapeSQL(data.companyName)}, parent_company = ${escapeSQL(data.parentCompany)}, url = ${escapeSQL(data.url)}, description = ${escapeSQL(data.description)}, updated_at = datetime('now') WHERE id = ${escapeSQL(data.id)};` - ); - } - - return sql; -} - -function generateSoftwareSQL( - change: Change, - data: Software | null, - deletedId: string | null, - manufacturerIds: Map -): string[] { - const sql: string[] = []; +function getChangedFiles(since: string): Change[] { + const changes: Change[] = []; + const output = execFileSync("git", ["diff", "--name-status", since, "HEAD", "--", "data/"], { + encoding: "utf-8", + }); - if (change.type === "deleted") { - if (!deletedId) { - sql.push(`-- WARNING: Could not resolve ID for deleted software ${change.slug}`); - return sql; - } - // Cascade deletes handle related tables - sql.push(`DELETE FROM software_fts WHERE id = ${escapeSQL(deletedId)};`); - sql.push(`DELETE FROM software WHERE id = ${escapeSQL(deletedId)};`); - } else if (data) { - const mfgId = manufacturerIds.get(data.manufacturer) ?? null; - - if (change.type === "modified") { - // Delete existing related data first - sql.push(`DELETE FROM software_categories WHERE software_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM software_formats WHERE software_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM software_platforms WHERE software_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM software_fts WHERE id = ${escapeSQL(data.id)};`); - sql.push( - `UPDATE software SET name = ${escapeSQL(data.name)}, manufacturer_id = ${escapeSQL(mfgId)}, url = ${escapeSQL(data.url)}, description = ${escapeSQL(data.description)}, release_date = ${escapeSQL(data.releaseDate)}, primary_category = ${escapeSQL(data.primaryCategory)}, secondary_category = ${escapeSQL(data.secondaryCategory)}, details = ${escapeSQL(normalizeMarkdown(data.details))}, specs = ${escapeSQL(normalizeMarkdown(data.specs))}, updated_at = datetime('now') WHERE id = ${escapeSQL(data.id)};` - ); - } else { - sql.push( - `INSERT INTO software (id, name, manufacturer_id, url, description, release_date, primary_category, secondary_category, details, specs, updated_at) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, ${escapeSQL(mfgId)}, ${escapeSQL(data.url)}, ${escapeSQL(data.description)}, ${escapeSQL(data.releaseDate)}, ${escapeSQL(data.primaryCategory)}, ${escapeSQL(data.secondaryCategory)}, ${escapeSQL(normalizeMarkdown(data.details))}, ${escapeSQL(normalizeMarkdown(data.specs))}, datetime('now'));` - ); - } + for (const line of output.split("\n").filter(Boolean)) { + const [status, file] = line.split("\t"); + if (!file) continue; - // Insert categories - if (data.categories) { - for (const category of data.categories) { - sql.push( - `INSERT INTO software_categories (software_id, category) VALUES (${escapeSQL(data.id)}, ${escapeSQL(category)});` - ); - } - } + const parts = file.split("/"); + if (parts.length < 3) continue; - // Insert formats - if (data.formats) { - for (const format of data.formats) { - const identifier = data.identifiers?.[format]; - sql.push( - `INSERT INTO software_formats (software_id, format, identifier) VALUES (${escapeSQL(data.id)}, ${escapeSQL(format)}, ${escapeSQL(identifier)});` - ); - } - } + const category = parts[1] as Change["category"]; + if (!(category in COLLECTIONS)) continue; + if (!/\.ya?ml$/.test(parts[2])) continue; - // Insert platforms - if (data.platforms) { - for (const platform of data.platforms) { - sql.push( - `INSERT INTO software_platforms (software_id, platform) VALUES (${escapeSQL(data.id)}, ${escapeSQL(platform)});` - ); - } - } + const type = status.startsWith("A") + ? "added" + : status.startsWith("M") + ? "modified" + : status.startsWith("D") + ? "deleted" + : null; + if (!type) continue; - // Insert FTS - sql.push( - `INSERT INTO software_fts (id, name, manufacturer_name, categories) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, (SELECT name FROM manufacturers WHERE id = ${escapeSQL(mfgId)}), ${escapeSQL(data.categories?.join(" ") ?? "")});` - ); + changes.push({ type, category, file, slug: parts[2].replace(/\.ya?ml$/, "") }); } - return sql; + return changes; } -function generateHardwareSQL( - change: Change, - data: Hardware | null, - deletedId: string | null, - manufacturerIds: Map -): string[] { - const sql: string[] = []; - - if (change.type === "deleted") { - if (!deletedId) { - sql.push(`-- WARNING: Could not resolve ID for deleted hardware ${change.slug}`); - return sql; - } - sql.push(`DELETE FROM hardware_fts WHERE id = ${escapeSQL(deletedId)};`); - sql.push(`DELETE FROM hardware WHERE id = ${escapeSQL(deletedId)};`); - } else if (data) { - const mfgId = manufacturerIds.get(data.manufacturer) ?? null; - - if (change.type === "modified") { - // Delete existing related data first - sql.push(`DELETE FROM hardware_categories WHERE hardware_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM hardware_capabilities WHERE hardware_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM hardware_fts WHERE id = ${escapeSQL(data.id)};`); - sql.push( - `UPDATE hardware SET name = ${escapeSQL(data.name)}, manufacturer_id = ${escapeSQL(mfgId)}, url = ${escapeSQL(data.url)}, description = ${escapeSQL(data.description)}, release_date = ${escapeSQL(data.releaseDate)}, primary_category = ${escapeSQL(data.primaryCategory)}, secondary_category = ${escapeSQL(data.secondaryCategory)}, details = ${escapeSQL(normalizeMarkdown(data.details))}, specs = ${escapeSQL(normalizeMarkdown(data.specs))}, updated_at = datetime('now') WHERE id = ${escapeSQL(data.id)};` - ); - } else { - sql.push( - `INSERT INTO hardware (id, name, manufacturer_id, url, description, release_date, primary_category, secondary_category, details, specs, updated_at) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, ${escapeSQL(mfgId)}, ${escapeSQL(data.url)}, ${escapeSQL(data.description)}, ${escapeSQL(data.releaseDate)}, ${escapeSQL(data.primaryCategory)}, ${escapeSQL(data.secondaryCategory)}, ${escapeSQL(normalizeMarkdown(data.details))}, ${escapeSQL(normalizeMarkdown(data.specs))}, datetime('now'));` - ); - } - - // Insert categories - if (data.categories) { - for (const category of data.categories) { - sql.push( - `INSERT INTO hardware_categories (hardware_id, category) VALUES (${escapeSQL(data.id)}, ${escapeSQL(category)});` - ); - } - } - - // Insert capabilities. A full `pnpm build` writes these, so omitting them - // here would let a patched database drift from a rebuilt one. Deduplicated - // because this path reads YAML without running validation, and - // hardware_capabilities is keyed on (hardware_id, capability), so a - // repeated value would abort the patch on a constraint violation. - for (const capability of new Set(data.capabilities ?? [])) { - sql.push( - `INSERT INTO hardware_capabilities (hardware_id, capability) VALUES (${escapeSQL(data.id)}, ${escapeSQL(capability)});` - ); - } - - // Insert FTS - const categories = data.categories?.join(" ") ?? ""; - sql.push( - `INSERT INTO hardware_fts (id, name, manufacturer_name, description, categories) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, (SELECT name FROM manufacturers WHERE id = ${escapeSQL(mfgId)}), ${escapeSQL(data.description)}, ${escapeSQL(categories)});` - ); - } - - return sql; -} +// ============================================================================= +// STATEMENT GENERATION +// ============================================================================= -function generateContentSQL( - change: Change, - data: Content | null, - deletedId: string | null, - manufacturerIds: Map +/** + * Remove an entry's rows, deepest table first. + * + * `includeRoot` is false for an entry being rewritten rather than removed: + * another entry's `supersedes_id` may point at the root row, and dropping it + * trips that foreign key even though the row is about to come straight back. + * Such a row is upserted in place instead, so it never disappears. + */ +export function deleteStatements( + schema: CollectionSchema, + id: string, + includeRoot: boolean ): string[] { const sql: string[] = []; - if (change.type === "deleted") { - if (!deletedId) { - sql.push(`-- WARNING: Could not resolve ID for deleted content ${change.slug}`); - return sql; - } - sql.push(`DELETE FROM content_fts WHERE id = ${escapeSQL(deletedId)};`); - sql.push(`DELETE FROM content WHERE id = ${escapeSQL(deletedId)};`); - } else if (data) { - const mfgId = manufacturerIds.get(data.manufacturer) ?? null; - - if (change.type === "modified") { - sql.push(`DELETE FROM content_categories WHERE content_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM content_fts WHERE id = ${escapeSQL(data.id)};`); + for (const child of [...schema.children].sort((a, b) => b.depth - a.depth)) { + if (child.parent === schema.root) { sql.push( - `UPDATE content SET name = ${escapeSQL(data.name)}, manufacturer_id = ${escapeSQL(mfgId)}, url = ${escapeSQL(data.url)}, description = ${escapeSQL(data.description)}, release_date = ${escapeSQL(data.releaseDate)}, primary_category = ${escapeSQL(data.primaryCategory)}, secondary_category = ${escapeSQL(data.secondaryCategory)}, details = ${escapeSQL(normalizeMarkdown(data.details))}, specs = ${escapeSQL(normalizeMarkdown(data.specs))}, updated_at = datetime('now') WHERE id = ${escapeSQL(data.id)};` + `DELETE FROM ${quoteIdent(child.name)} WHERE ${quoteIdent(child.fkColumn)} = ${escapeSQL(id)};` ); } else { sql.push( - `INSERT INTO content (id, name, manufacturer_id, url, description, release_date, primary_category, secondary_category, details, specs, updated_at) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, ${escapeSQL(mfgId)}, ${escapeSQL(data.url)}, ${escapeSQL(data.description)}, ${escapeSQL(data.releaseDate)}, ${escapeSQL(data.primaryCategory)}, ${escapeSQL(data.secondaryCategory)}, ${escapeSQL(normalizeMarkdown(data.details))}, ${escapeSQL(normalizeMarkdown(data.specs))}, datetime('now'));` + `DELETE FROM ${quoteIdent(child.name)} WHERE ${quoteIdent(child.fkColumn)} IN ` + + `(SELECT id FROM ${quoteIdent(child.parent)} ` + + `WHERE ${quoteIdent(linkToRoot(schema, child.parent))} = ${escapeSQL(id)});` ); } - - // Insert categories - if (data.categories) { - for (const category of data.categories) { - sql.push( - `INSERT INTO content_categories (content_id, category) VALUES (${escapeSQL(data.id)}, ${escapeSQL(category)});` - ); - } - } - - // Insert FTS - sql.push( - `INSERT INTO content_fts (id, name, manufacturer_name, categories) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, (SELECT name FROM manufacturers WHERE id = ${escapeSQL(mfgId)}), ${escapeSQL(data.categories?.join(" ") ?? "")});` - ); } + // Full-text tables carry no foreign keys, so they are always replaced whole. + if (schema.fts) { + sql.push(`DELETE FROM ${quoteIdent(schema.fts.name)} WHERE id = ${escapeSQL(id)};`); + } + if (includeRoot) { + sql.push(`DELETE FROM ${quoteIdent(schema.root)} WHERE id = ${escapeSQL(id)};`); + } return sql; } -function generateAccessorySQL( - change: Change, - data: Accessory | null, - deletedId: string | null, - manufacturerIds: Map -): string[] { - const sql: string[] = []; - - if (change.type === "deleted") { - if (!deletedId) { - sql.push(`-- WARNING: Could not resolve ID for deleted accessory ${change.slug}`); - return sql; - } - sql.push(`DELETE FROM accessories_fts WHERE id = ${escapeSQL(deletedId)};`); - sql.push(`DELETE FROM accessories WHERE id = ${escapeSQL(deletedId)};`); - } else if (data) { - const mfgId = manufacturerIds.get(data.manufacturer) ?? null; - - if (change.type === "modified") { - sql.push(`DELETE FROM accessories_categories WHERE accessory_id = ${escapeSQL(data.id)};`); - sql.push(`DELETE FROM accessories_fts WHERE id = ${escapeSQL(data.id)};`); - sql.push( - `UPDATE accessories SET name = ${escapeSQL(data.name)}, manufacturer_id = ${escapeSQL(mfgId)}, url = ${escapeSQL(data.url)}, description = ${escapeSQL(data.description)}, release_date = ${escapeSQL(data.releaseDate)}, primary_category = ${escapeSQL(data.primaryCategory)}, secondary_category = ${escapeSQL(data.secondaryCategory)}, details = ${escapeSQL(normalizeMarkdown(data.details))}, specs = ${escapeSQL(normalizeMarkdown(data.specs))}, updated_at = datetime('now') WHERE id = ${escapeSQL(data.id)};` - ); - } else { +/** Re-create an entry from the freshly built database. */ +export function insertStatements( + db: Database.Database, + schema: CollectionSchema, + id: string +): string[] | null { + const root = db.prepare(`SELECT * FROM ${quoteIdent(schema.root)} WHERE id = ?`).get(id) as + | Record + | undefined; + if (!root) return null; + + const assignments = schema.rootColumns + .filter((c) => c !== "id") + .map((c) => `${quoteIdent(c)} = excluded.${quoteIdent(c)}`) + .join(", "); + + const sql: string[] = [ + insertRow( + schema.root, + schema.rootColumns, + schema.rootColumns.map((c) => sqlLiteral(root[c])) + ).replace(/;$/, ` ON CONFLICT(id) DO UPDATE SET ${assignments};`), + ]; + + if (schema.fts) { + const ftsRow = db + .prepare(`SELECT * FROM ${quoteIdent(schema.fts.name)} WHERE id = ?`) + .get(id) as Record | undefined; + if (ftsRow) { sql.push( - `INSERT INTO accessories (id, name, manufacturer_id, url, description, release_date, primary_category, secondary_category, details, specs, updated_at) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, ${escapeSQL(mfgId)}, ${escapeSQL(data.url)}, ${escapeSQL(data.description)}, ${escapeSQL(data.releaseDate)}, ${escapeSQL(data.primaryCategory)}, ${escapeSQL(data.secondaryCategory)}, ${escapeSQL(normalizeMarkdown(data.details))}, ${escapeSQL(normalizeMarkdown(data.specs))}, datetime('now'));` + insertRow( + schema.fts.name, + schema.fts.columns, + schema.fts.columns.map((c) => sqlLiteral(ftsRow[c])) + ) ); } + } - // Insert categories - if (data.categories) { - for (const category of data.categories) { - sql.push( - `INSERT INTO accessories_categories (accessory_id, category) VALUES (${escapeSQL(data.id)}, ${escapeSQL(category)});` - ); - } + for (const child of [...schema.children].sort((a, b) => a.depth - b.depth)) { + const direct = child.parent === schema.root; + const keys = direct ? [] : schema.naturalKeys.get(child.parent)!; + + const rows = direct + ? (db + .prepare( + `SELECT * FROM ${quoteIdent(child.name)} WHERE ${quoteIdent(child.fkColumn)} = ?` + ) + .all(id) as Record[]) + : (db + .prepare( + `SELECT c.*, ${keys.map((k) => `p.${quoteIdent(k)} AS ${quoteIdent(`__key_${k}`)}`).join(", ")} ` + + `FROM ${quoteIdent(child.name)} c ` + + `JOIN ${quoteIdent(child.parent)} p ON p.id = c.${quoteIdent(child.fkColumn)} ` + + `WHERE p.${quoteIdent(linkToRoot(schema, child.parent))} = ?` + ) + .all(id) as Record[]); + + for (const row of rows) { + const values = child.columns.map((column) => { + if (direct || column !== child.fkColumn) return sqlLiteral(row[column]); + // The parent's id is assigned by whichever database applies this patch, + // so resolve it there rather than baking in this build's numbering. + const where = keys + .map((k) => `${quoteIdent(k)} = ${sqlLiteral(row[`__key_${k}`])}`) + .join(" AND "); + return `(SELECT id FROM ${quoteIdent(child.parent)} WHERE ${where})`; + }); + sql.push(insertRow(child.name, child.columns, values)); } - - // Insert FTS - const categories = data.categories?.join(" ") ?? ""; - sql.push( - `INSERT INTO accessories_fts (id, name, manufacturer_name, description, categories) VALUES (${escapeSQL(data.id)}, ${escapeSQL(data.name)}, (SELECT name FROM manufacturers WHERE id = ${escapeSQL(mfgId)}), ${escapeSQL(data.description)}, ${escapeSQL(categories)});` - ); } return sql; @@ -374,7 +420,16 @@ function generateAccessorySQL( // MAIN // ============================================================================= -function generatePatch(fromTag: string, toVersion: string): void { +function readEntryId(category: Change["category"], slug: string): string | null { + try { + const file = path.join(DATA_DIR, category, `${slug}.yaml`); + return (parseYaml(fs.readFileSync(file, "utf-8")) as { id?: string }).id ?? null; + } catch { + return null; + } +} + +function generatePatch(fromTag: string, toVersion: string, dbPath: string): void { const changes = getChangedFiles(fromTag); if (changes.length === 0) { @@ -385,109 +440,136 @@ function generatePatch(fromTag: string, toVersion: string): void { console.log(`\n📝 Generating patch ${fromTag} → ${toVersion}\n`); console.log(` Found ${changes.length} changes\n`); - // Build manufacturer slug -> nanoid map for FK resolution - const manufacturerIds = buildManufacturerIdMap(); + const db = new Database(dbPath, { readonly: true }); + const schemas = new Map( + COLLECTION_ORDER.map((category) => [category, reflectCollection(db, category)]) + ); - const sql: string[] = []; + const body: string[] = []; + const failures: string[] = []; - // Add header - sql.push(`-- Catalog patch: ${fromTag} → ${toVersion}`); - sql.push(`-- Generated: ${new Date().toISOString()}`); - sql.push(`-- Changes: ${changes.length}`); - sql.push(""); - sql.push("BEGIN TRANSACTION;"); - sql.push(""); - - // Process changes in order: manufacturers first (for foreign keys) - const sortedChanges = [...changes].sort((a, b) => { - const order: Record = { - manufacturers: 0, - software: 1, - content: 2, - hardware: 3, - accessories: 4, - }; - return (order[a.category] ?? 99) - (order[b.category] ?? 99); - }); + const ordered = [...changes].sort( + (a, b) => COLLECTION_ORDER.indexOf(a.category) - COLLECTION_ORDER.indexOf(b.category) + ); + + for (const change of ordered) { + const schema = schemas.get(change.category)!; + const id = + change.type === "deleted" + ? getDeletedEntryId(fromTag, change.file) + : readEntryId(change.category, change.slug); + + if (!id) { + failures.push(`${change.category}/${change.slug}: could not resolve entry id`); + continue; + } - for (const change of sortedChanges) { - sql.push(`-- ${change.type.toUpperCase()}: ${change.category}/${change.slug}`); + const statements = deleteStatements(schema, id, change.type === "deleted"); - // Load current data for non-deletions - let data = null; if (change.type !== "deleted") { - try { - const filePath = path.join(DATA_DIR, change.category, `${change.slug}.yaml`); - const content = fs.readFileSync(filePath, "utf-8"); - data = parseYaml(content); - } catch { - console.warn(` ⚠️ Could not read file for ${change.slug}`); + const inserts = insertStatements(db, schema, id); + if (!inserts) { + failures.push( + `${change.category}/${change.slug}: id ${id} is not in ${path.basename(dbPath)} ` + + `(rebuild with 'pnpm build')` + ); continue; } + statements.push(...inserts); } - // For deletions, resolve the nanoid from the base version - const deletedId = change.type === "deleted" ? getDeletedEntryId(fromTag, change.file) : null; - - let statements: string[] = []; - switch (change.category) { - case "manufacturers": - statements = generateManufacturerSQL(change, data as Manufacturer, deletedId); - break; - case "software": - statements = generateSoftwareSQL(change, data as Software, deletedId, manufacturerIds); - break; - case "content": - statements = generateContentSQL(change, data as Content, deletedId, manufacturerIds); - break; - case "hardware": - statements = generateHardwareSQL(change, data as Hardware, deletedId, manufacturerIds); - break; - case "accessories": - statements = generateAccessorySQL(change, data as Accessory, deletedId, manufacturerIds); - break; - } + body.push(`-- ${change.type.toUpperCase()}: ${change.category}/${change.slug}`); + body.push(...statements); + body.push(""); + } - sql.push(...statements); - sql.push(""); + if (failures.length > 0) { + // Stamping a new version while silently dropping a change leaves a database + // claiming content it does not have, so refuse to write the patch at all. + console.error(`\n❌ Refusing to write a patch, ${failures.length} change(s) unresolved:\n`); + for (const failure of failures) console.error(` - ${failure}`); + process.exitCode = 1; + db.close(); + return; } - // Update version metadata - sql.push("-- Update catalog version"); - sql.push(`UPDATE catalog_meta SET value = '${toVersion}' WHERE key = 'version';`); - sql.push(`UPDATE catalog_meta SET value = datetime('now') WHERE key = 'updated_at';`); - sql.push(""); - sql.push("COMMIT;"); + const schemaVersion = ( + db.prepare(`SELECT value FROM catalog_meta WHERE key = 'schema_version'`).get() as + | { value: string } + | undefined + )?.value; + db.close(); + + const sql: string[] = [ + `-- Catalog patch: ${fromTag} → ${toVersion}`, + `-- Generated: ${new Date().toISOString()}`, + `-- Changes: ${changes.length}`, + "", + "BEGIN TRANSACTION;", + "", + ...body, + "-- Update catalog version", + `UPDATE catalog_meta SET value = ${escapeSQL(toVersion)} WHERE key = 'version';`, + `UPDATE catalog_meta SET value = datetime('now') WHERE key = 'updated_at';`, + ]; + + if (schemaVersion) { + sql.push( + `INSERT INTO catalog_meta (key, value) VALUES ('schema_version', ${escapeSQL(schemaVersion)}) ` + + `ON CONFLICT(key) DO UPDATE SET value = excluded.value;` + ); + } - // Ensure output directory exists - fs.mkdirSync(PATCHES_DIR, { recursive: true }); + sql.push("", "COMMIT;"); - // Write patch file + fs.mkdirSync(PATCHES_DIR, { recursive: true }); const patchFile = path.join(PATCHES_DIR, `patch-${fromTag}-${toVersion}.sql`); fs.writeFileSync(patchFile, sql.join("\n")); - const stats = fs.statSync(patchFile); - const sizeKB = (stats.size / 1024).toFixed(2); - + const sizeKB = (fs.statSync(patchFile).size / 1024).toFixed(2); console.log(`✅ Patch generated!`); console.log(` Output: ${patchFile}`); console.log(` Size: ${sizeKB} KB`); } -// Parse arguments -const fromArg = process.argv[2]; -const toArg = process.argv[3]; - -if (fromArg && toArg) { - // Explicit versions provided - generatePatch(fromArg, toArg); -} else { - // Auto-detect from latest tag - const latestTag = getLatestTag(); - if (latestTag) { - const toVersion = toArg ?? "next"; - generatePatch(latestTag, toVersion); - } else { +// ============================================================================= +// CLI +// ============================================================================= + +function main(): void { + const argv = process.argv.slice(2); + const dbFlag = argv.indexOf("--db"); + const dbPath = dbFlag === -1 ? DEFAULT_DB : argv[dbFlag + 1]; + // An explicit database is one the caller vouches for, so do not overwrite it. + const skipBuild = argv.includes("--skip-build") || dbFlag !== -1; + const positional = argv.filter( + (arg, i) => !arg.startsWith("--") && !(dbFlag !== -1 && i === dbFlag + 1) + ); + + const [fromArg, toArg] = positional; + const fromTag = fromArg ?? getLatestTag(); + + if (!fromTag) { console.log("No previous release found. Build a baseline first with: pnpm build"); + } else if (!dbPath) { + console.error("❌ --db needs a path."); + process.exitCode = 1; + } else { + if (!skipBuild) { + // The statements are read out of the built database, so it has to be HEAD's. + console.log("🔨 Building catalog database…"); + execFileSync("tsx", [BUILD_SCRIPT], { stdio: "inherit" }); + } + if (!fs.existsSync(dbPath)) { + console.error(`❌ ${dbPath} not found. Build it with 'pnpm build'.`); + process.exitCode = 1; + } else { + generatePatch(fromTag, toArg ?? "next", dbPath); + } } } + +// Importing this module (from tests) must not run a build. +if (import.meta.url === pathToFileURL(process.argv[1] ?? "").href) { + main(); +} From f853d753de131f74566a9d546e7558e1af7c9a17 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 28 Aug 2026 15:56:09 +0000 Subject: [PATCH 3/3] fix(scripts): harden patch schema reflection CodeRabbit review on #672. None is a live bug against the current schema; all three are traps the reflection would spring on a future one. Assigned primary keys are now read from table_info rather than matched in the CREATE statement, where a quoted column name read as real data and the build's own id would be written into the target database. A nested row's natural key is chosen deterministically (PRAGMA index_list promises no order) and must be NOT NULL, since a null becomes `"col" = NULL` in the lookup and matches nothing. Nesting deeper than two levels is refused rather than silently omitted: the statements resolve one level, so a depth-3 child would be looked up through an integer key compared against the root's text id. Adds a changeset, missing because the PR touches data/*.yaml. Co-Authored-By: Claude --- .changeset/pioneer-manual-links.md | 10 ++ scripts/__tests__/generate-patch.test.ts | 80 ++++++++++++++ scripts/generate-patch.ts | 135 ++++++++++++++++++----- 3 files changed, 200 insertions(+), 25 deletions(-) create mode 100644 .changeset/pioneer-manual-links.md diff --git a/.changeset/pioneer-manual-links.md b/.changeset/pioneer-manual-links.md new file mode 100644 index 000000000..c0be197ba --- /dev/null +++ b/.changeset/pioneer-manual-links.md @@ -0,0 +1,10 @@ +--- +"catalog": patch +--- + +Repoint dead manual links on 7 Pioneer DJ effector entries. The +`downloads.support.alphatheta.com/manuals/dj-effectors/` tree no longer +exists, so every link under it 404s; AlphaTheta moved manual content into +support articles. EFX-1000 keeps its "Specifications" title against a +dedicated specifications article; the other six point at the instruction +manual and are retitled, since the per-page anchor no longer exists. diff --git a/scripts/__tests__/generate-patch.test.ts b/scripts/__tests__/generate-patch.test.ts index 80b9ea802..9171144c5 100644 --- a/scripts/__tests__/generate-patch.test.ts +++ b/scripts/__tests__/generate-patch.test.ts @@ -101,6 +101,86 @@ describe("reflectCollection", () => { expect(content.children.map((c) => c.name)).toContain("content_hardware_compatibility"); }); + it("omits an assigned id declared with a quoted column name", () => { + // The previous check matched the CREATE statement textually, so a quoted + // or table-level primary key read as real data and the build's own id was + // written into the target database. + const quoted = emptyDatabase(); + quoted.exec(`CREATE TABLE hardware_quoted ( + "id" INTEGER PRIMARY KEY AUTOINCREMENT, + hardware_id TEXT NOT NULL REFERENCES hardware(id) ON DELETE CASCADE, + note TEXT + )`); + const child = reflectCollection(quoted, "hardware").children.find( + (c) => c.name === "hardware_quoted" + )!; + expect(child.columns).toEqual(["hardware_id", "note"]); + }); + + it("keeps an id the table owns rather than one SQLite assigns", () => { + const withoutRowid = emptyDatabase(); + withoutRowid.exec(`CREATE TABLE hardware_keyed ( + id INTEGER PRIMARY KEY, + hardware_id TEXT NOT NULL REFERENCES hardware(id) ON DELETE CASCADE + ) WITHOUT ROWID`); + const child = reflectCollection(withoutRowid, "hardware").children.find( + (c) => c.name === "hardware_keyed" + )!; + // No rowid to alias, so the id is data the patch has to carry. + expect(child.columns).toContain("id"); + }); + + it("picks the same natural key whichever order the indexes come back in", () => { + const many = emptyDatabase(); + many.exec(`CREATE TABLE hardware_zones ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + hardware_id TEXT NOT NULL REFERENCES hardware(id) ON DELETE CASCADE, + slug TEXT NOT NULL, + code TEXT NOT NULL + )`); + many.exec(`CREATE UNIQUE INDEX zz_hardware_zones_code ON hardware_zones(hardware_id, code)`); + many.exec(`CREATE UNIQUE INDEX aa_hardware_zones_slug ON hardware_zones(hardware_id, slug)`); + many.exec(`CREATE TABLE hardware_zone_notes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + zone_id INTEGER NOT NULL REFERENCES hardware_zones(id) ON DELETE CASCADE, + note TEXT + )`); + // Sorted by index name, so the choice does not depend on PRAGMA ordering. + expect(reflectCollection(many, "hardware").naturalKeys.get("hardware_zones")).toEqual([ + "hardware_id", + "slug", + ]); + }); + + it("refuses a natural key whose columns can be null", () => { + // A null is copied into the lookup as `"col" = NULL`, which matches + // nothing, so the child would insert a null foreign key or fail outright. + const nullable = emptyDatabase(); + nullable.exec(`CREATE TABLE hardware_zones ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + hardware_id TEXT NOT NULL REFERENCES hardware(id) ON DELETE CASCADE, + slug TEXT, + UNIQUE(hardware_id, slug) + )`); + nullable.exec(`CREATE TABLE hardware_zone_notes ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + zone_id INTEGER NOT NULL REFERENCES hardware_zones(id) ON DELETE CASCADE + )`); + expect(() => reflectCollection(nullable, "hardware")).toThrow(/NOT NULL columns/); + }); + + it("refuses a table nested more than two levels deep", () => { + // The statements resolve one level of nesting, so a depth-3 child would be + // looked up through an integer key compared against the root's text id and + // silently left out of the patch. + const deep = emptyDatabase(); + deep.exec(`CREATE TABLE hardware_variant_parts ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + variant_link_id INTEGER NOT NULL REFERENCES hardware_variant_links(id) ON DELETE CASCADE + )`); + expect(() => reflectCollection(deep, "hardware")).toThrow(/two levels deep/); + }); + it("picks up the collection's full-text table without its shadow tables", () => { expect(schema.fts?.name).toBe("hardware_fts"); expect(schema.children.map((c) => c.name)).not.toContain("hardware_fts_data"); diff --git a/scripts/generate-patch.ts b/scripts/generate-patch.ts index 24431d908..982c6b820 100644 --- a/scripts/generate-patch.ts +++ b/scripts/generate-patch.ts @@ -68,6 +68,7 @@ const FTS_SHADOW_SUFFIX = /_(data|idx|content|docsize|config)$/; // SQL HELPERS // ============================================================================= +/** Quote a table or column name so a reserved word or odd character is safe. */ function quoteIdent(name: string): string { return `"${name.replace(/"/g, '""')}"`; } @@ -81,6 +82,7 @@ function sqlLiteral(value: unknown): string { return escapeSQL(String(value)); } +/** One INSERT over an explicit column list, so column order never matters. */ function insertRow(table: string, columns: string[], values: string[]): string { return ( `INSERT INTO ${quoteIdent(table)} (${columns.map(quoteIdent).join(", ")}) ` + @@ -96,6 +98,7 @@ interface ColumnInfo { name: string; pk: number; type: string; + notnull: number; } export interface ChildTable { @@ -127,13 +130,72 @@ export interface CollectionSchema { naturalKeys: Map; } -/** True for an id SQLite assigns, which must be left out of an INSERT. */ -function isAssignedPk(column: ColumnInfo, createSql: string): boolean { - if (column.pk !== 1 || column.type.toUpperCase() !== "INTEGER") return false; - // A composite primary key or a TEXT id is real data and has to be written. - return new RegExp(`\\b${column.name}\\s+INTEGER\\s+PRIMARY\\s+KEY`, "i").test(createSql); +/** + * The column SQLite assigns itself, which must be left out of an INSERT. + * + * A rowid alias is structurally an INTEGER PRIMARY KEY that is the table's + * only primary-key column, on a table with a rowid. Everything else is real + * data: `hardware_categories` keys on (hardware_id, category) and `hardware` + * on a TEXT nanoid, and both have to be written. Read from `table_info` + * rather than matched in the CREATE statement, where a quoted column name or + * a table-level `PRIMARY KEY (id)` reads as neither. + */ +function assignedPkColumn(columns: ColumnInfo[], createSql: string): string | null { + if (/\bWITHOUT\s+ROWID\b/i.test(createSql)) return null; + const key = columns.filter((c) => c.pk > 0); + if (key.length !== 1 || key[0].type.toUpperCase() !== "INTEGER") return null; + return key[0].name; +} + +/** + * Columns that identify a row in `child.parent` without using its assigned id. + * + * Two properties matter and neither is free. The choice has to be + * deterministic, because `PRAGMA index_list` does not promise an order and a + * parent with two unique indexes would otherwise emit different lookup SQL + * from run to run. And every key column has to be NOT NULL, because a NULL is + * copied into the lookup as `"col" = NULL`, which matches nothing, so the + * child would insert a null foreign key or fail outright. + */ +function naturalKeyFor( + db: Database.Database, + root: string, + child: ChildTable, + parentColumns: ColumnInfo[] +): string[] { + const nullable = new Set(parentColumns.filter((c) => c.notnull === 0).map((c) => c.name)); + const indexes = ( + db.prepare(`PRAGMA index_list(${quoteIdent(child.parent)})`).all() as { + name: string; + unique: number; + origin: string; + }[] + ) + // The primary-key index is the assigned id this exists to avoid. + .filter((i) => i.unique === 1 && i.origin !== "pk") + .sort((a, b) => a.name.localeCompare(b.name)); + + for (const index of indexes) { + const cols = ( + db.prepare(`PRAGMA index_info(${quoteIdent(index.name)})`).all() as { name: string }[] + ).map((c) => c.name); + if (cols.length > 0 && !cols.some((c) => nullable.has(c))) return cols; + } + + throw new Error( + `Cannot patch ${root}: ${child.name} references ${child.parent} by assigned id, ` + + `but ${child.parent} has no unique index over NOT NULL columns to resolve it by.` + ); } +/** + * Describe one collection's tables by reading the built database. + * + * Everything the statements need comes from here: which tables the collection + * owns, which column links each to its parent, how deep each sits, what + * columns it writes, and how a nested row finds its parent. Reflecting it is + * what keeps a new child table or column working without touching this file. + */ export function reflectCollection( db: Database.Database, category: Change["category"] @@ -153,10 +215,9 @@ export function reflectCollection( db.prepare(`PRAGMA table_info(${quoteIdent(table)})`).all() as ColumnInfo[]; const writableColumns = (table: string): string[] => { - const sql = createSql.get(table) ?? ""; - return columnsOf(table) - .filter((c) => !isAssignedPk(c, sql)) - .map((c) => c.name); + const columns = columnsOf(table); + const assigned = assignedPkColumn(columns, createSql.get(table) ?? ""); + return columns.filter((c) => c.name !== assigned).map((c) => c.name); }; const owned = objects @@ -208,32 +269,39 @@ export function reflectCollection( ); } + // The walk above finds a table at any depth, but the statements only resolve + // one level of nesting: a depth-3 child would be looked up through its + // depth-2 parent's assigned integer key compared against the root's text id, + // which matches nothing, so its rows would be silently left out of the patch. + // Fail here instead. Resolving the full chain is the fix if the schema ever + // grows one, and nothing in it does today. + const tooDeep = children.filter((c) => c.depth > 2); + if (tooDeep.length > 0) { + throw new Error( + `Cannot patch ${root}: ${tooDeep.map((c) => c.name).join(", ")} ` + + `${tooDeep.length === 1 ? "is" : "are"} nested more than two levels deep, ` + + `which the statement generator cannot resolve. Extend linkToRoot to walk ` + + `the whole parent chain before adding it.` + ); + } + // A child keyed on something other than the root's text id needs its parent // resolved at apply time. const naturalKeys = new Map(); for (const child of children) { if (child.parent === root || naturalKeys.has(child.parent)) continue; - const indexes = db.prepare(`PRAGMA index_list(${quoteIdent(child.parent)})`).all() as { - name: string; - unique: number; - }[]; - const unique = indexes.find((i) => i.unique === 1); - if (!unique) { - throw new Error( - `Cannot patch ${root}: ${child.name} references ${child.parent} by assigned id, ` + - `but ${child.parent} has no unique index to resolve it by.` - ); - } - const cols = ( - db.prepare(`PRAGMA index_info(${quoteIdent(unique.name)})`).all() as { name: string }[] - ).map((c) => c.name); - naturalKeys.set(child.parent, cols); + naturalKeys.set(child.parent, naturalKeyFor(db, root, child, columnsOf(child.parent))); } return { root, rootColumns: writableColumns(root), fts, children, naturalKeys }; } -/** The column on `parent` that links it back toward the collection root. */ +/** + * The column on `parent` that links it back toward the collection root. + * + * Correct for a parent that is the root or a direct child of it, which + * `reflectCollection` enforces by refusing anything deeper. + */ function linkToRoot(schema: CollectionSchema, parent: string): string { return schema.children.find((c) => c.name === parent)?.fkColumn ?? `${schema.root}_id`; } @@ -242,6 +310,7 @@ function linkToRoot(schema: CollectionSchema, parent: string): string { // GIT HELPERS // ============================================================================= +/** The most recent release tag, used as the default patch base. */ function getLatestTag(): string | null { try { return execFileSync("git", ["describe", "--tags", "--abbrev=0"], { @@ -266,6 +335,12 @@ function getDeletedEntryId(since: string, filePath: string): string | null { } } +/** + * Entry files added, modified or deleted since `since`, one Change each. + * + * Paths outside the five collections, and non-YAML files, are skipped rather + * than guessed at: they carry no entry for a patch to rewrite. + */ function getChangedFiles(since: string): Change[] { const changes: Change[] = []; const output = execFileSync("git", ["diff", "--name-status", since, "HEAD", "--", "data/"], { @@ -420,6 +495,7 @@ export function insertStatements( // MAIN // ============================================================================= +/** An on-disk entry's nanoid, or null when the file is missing or unparsable. */ function readEntryId(category: Change["category"], slug: string): string | null { try { const file = path.join(DATA_DIR, category, `${slug}.yaml`); @@ -429,6 +505,14 @@ function readEntryId(category: Change["category"], slug: string): string | null } } +/** + * Write the SQL that brings a database at `fromTag` up to HEAD. + * + * Collections are emitted manufacturers-first so a product's foreign key + * lands after the row it points at. Nothing is written at all unless every + * change resolved, because a patch that stamps a version while dropping a + * change leaves a database claiming content it does not have. + */ function generatePatch(fromTag: string, toVersion: string, dbPath: string): void { const changes = getChangedFiles(fromTag); @@ -536,6 +620,7 @@ function generatePatch(fromTag: string, toVersion: string, dbPath: string): void // CLI // ============================================================================= +/** Parse the CLI arguments, build the database unless told not to, patch. */ function main(): void { const argv = process.argv.slice(2); const dbFlag = argv.indexOf("--db");