From 4072df8c176207b85e71daca70cec753c6714a46 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Moritz=20M=C3=A4hr?= <14755525+maehr@users.noreply.github.com> Date: Mon, 7 Sep 2026 18:18:39 +0200 Subject: [PATCH 1/2] refactor(compile): remove duplication and detect duplicate registry keys Reduce the repeated code in the registry compiler. Add two integrity guards. The compiled output does not change: all five dump resources keep the same sha256, and the data package descriptor is identical. Compile time does not change either: the median of seven runs is 754ms before and after. Share one validation helper. Four record types repeated the same safeParse, report and throw block. `scripts/validate-data.ts` now prints issues with the same helper, but keeps its own control flow: it counts a failure and continues. Detect a duplicate work key and a duplicate citation system key. A second system file overwrote the first in the Map. A second work file had no check at all, and it produced two Work records under one `@id` with conflicting metadata. `scripts/validate-data.ts` accepted that, because it validates each record on its own and never checks that an `@id` is unique. `release.yml` runs `npm run build:data`, which compiles and validates but never builds the site, so the duplicate reached the release artifact and the Zenodo record with nothing to object. The compiler now fails and names both files. One case was already caught. Two work files that declare different citation systems for one locator collide on the bare `/cite/{work}/{locator}` alias, and `setAlias` rejects that. The guard covers the other cases, which were silent. Count a skipped resolver from `extra_resolvers`. The block resolvers and the per-reference extras run as two loops, and only the first loop warned. A hole in a per-reference map stayed silent. The loops stay separate: one loop over a concatenation allocates an array per reference, 86k of them, to save four lines. Type a resolver target while it is built. TypeScript now checks the field names and the values. The fields stay separate assignments, because a literal of conditional spreads allocates a throwaway object per field and runs three times slower over 172k targets. The published key order does not change, so the dump bytes do not change. Declare the five dump resources once, in `DUMP_SPECS`. `DUMP_MANIFEST` is derived from it, so the two cannot drift. Each body stays a thunk, so `/dump/index.astro` still renders the file list without serialising ~90 MB. Add `standard/iri.ts` for the four IRI prefixes. It is dependency-free, so `src/lib/find.ts` can import it and still ship to the browser. Keep the UUID seeding in `scripts/validate-data.ts` duplicated on purpose. The gate proves the compiler's identifiers are deterministic. A shared function would make each assertion a tautology. Extract `emitMappings` and `systemBlocksOf`. Make `enforceRegistryInvariants` return void; it always returned 0. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VLqvaFc9mqQYcjb9SiKGXd --- scripts/compile.test.ts | 51 ++++- scripts/compile.ts | 480 +++++++++++++++++++++++++-------------- scripts/validate-data.ts | 21 +- src/lib/find.ts | 5 +- standard/iri.ts | 38 ++++ 5 files changed, 406 insertions(+), 189 deletions(-) create mode 100644 standard/iri.ts diff --git a/scripts/compile.test.ts b/scripts/compile.test.ts index 99233d5..a45f7aa 100644 --- a/scripts/compile.test.ts +++ b/scripts/compile.test.ts @@ -147,6 +147,43 @@ additional_systems: assert.match(message, /invalid source file/); }); +// A key is the whole identity of a work or a citation system. Two files +// claiming one used to pass: the systems Map kept the last file read and +// dropped the other, and works had no check at all — so the second work would +// mint the same reference UUIDs as the first, because ADR-0002 seeds them on +// `(work_key, citation_system_key, locator)`. `setAlias` cannot catch that, +// since both files produce the same alias target. + +test('two files declaring one citation system key fail the build', (t) => { + const message = expectCompileError(t, { + systems: { + 'first-file': system('shared'), + 'second-file': system('shared'), + }, + works: { + 'test.work': `${workHeader()} +citation_system: shared +references: + - '1' +`, + }, + }); + assert.match(message, /citation system key "shared" is declared twice/); +}); + +test('two files declaring one work key fail the build', (t) => { + const body = `${workHeader()} +citation_system: primary-section +references: + - '1' +`; + const message = expectCompileError(t, { + systems: twoSystems, + works: { 'first-file': body, 'second-file': body }, + }); + assert.match(message, /work key "test.work" is declared twice/); +}); + test('a locator containing "/" is rejected before any alias is minted', (t) => { const message = expectCompileError(t, { systems: { @@ -950,18 +987,22 @@ test('every descriptor carries the Frictionless fields', () => { }); // `/dump/index.astro` renders DUMP_MANIFEST so that building the page costs no -// serialisation. That is only safe while the manifest names exactly what -// `dumpResources` produces, in the same order. -test('DUMP_MANIFEST matches what dumpResources produces', () => { - const specs = dumpResources(compileFixture(workWithMappings())); +// serialisation. Both it and `dumpResources` are derived from the one +// `DUMP_SPECS` list, so they can no longer disagree about the set or its order. +// What still needs pinning is that the manifest carries no body: forcing one is +// exactly what the split exists to avoid. +test('DUMP_MANIFEST lists the files without carrying a body', () => { assert.deepEqual( - specs.map(({ name, filename, format }) => ({ name, filename, format })), DUMP_MANIFEST.map(({ name, filename, format }) => ({ name, filename, format, })), + dumpResources(compileFixture(workWithMappings())).map( + ({ name, filename, format }) => ({ name, filename, format }), + ), ); + for (const entry of DUMP_MANIFEST) assert.ok(!('body' in entry)); }); test('packageVersion drops a leading v and leaves the rest alone', () => { diff --git a/scripts/compile.ts b/scripts/compile.ts index a686a62..05c9efc 100644 --- a/scripts/compile.ts +++ b/scripts/compile.ts @@ -9,12 +9,15 @@ import { join, basename, resolve } from 'node:path'; import { createHash } from 'node:crypto'; import { v5 as uuidv5 } from 'uuid'; import { parse as parseYaml } from 'yaml'; +import type { z, ZodType } from 'zod'; import { Work, CitationSystem, CanonicalReference, MappingAssertion, } from '../standard/schema/index.js'; +import type { ResolverTargetEntry } from '../standard/schema/canonical-reference.js'; +import { workIri, systemIri, refIri, mappingIri } from '../standard/iri.js'; import { parseSource, SystemSource, @@ -260,7 +263,7 @@ function applyResolverVars( function buildResolverEntry( resolver: ResolverEntry, locatorVars: Record, -): Record | null { +): ResolverTargetEntry | null { const vars = applyResolverVars(resolver, locatorVars); if (!vars) return null; let url: string | null = null; @@ -276,29 +279,48 @@ function buildResolverEntry( url = Object.hasOwn(byMap, key) ? (byMap[key] ?? null) : null; } if (!url) return null; - const entry: Record = { url }; + if (resolver.license !== undefined && !SPDX_IDS.has(resolver.license)) { + // Unreachable via authored YAML — ResolverEntrySource rejects it at + // parse time. Kept so a future caller that skips the parser cannot + // drop a licence statement silently. + throw new Error( + `license "${resolver.license}" is not an SPDX id (record the provider's rights statement in license_url)`, + ); + } + // TypeScript checks every field name and value below, rather than deferring + // a typo to `safeParse`. The assignment order is the published key order, + // and therefore the byte order of the JSONL dump: do not reorder these + // lines. + // + // Assignment rather than one object literal of conditional spreads. The + // literal reads better, but each `...(cond && { k: v })` allocates a + // throwaway object, and this runs 172k times per compile — measured at 3x + // the cost of the assignments for no gain, since the type below checks + // these just as well. + // + // `Pick<…, 'url'>` keeps `url` required, so dropping it from the initialiser + // is a compile error. `access` cannot be covered the same way: it is + // required too, but it is assigned fifth to hold the key order, and + // TypeScript does not let a narrowed property satisfy a required one — no + // arrangement of guards makes the object assignable without the cast. So + // the cast asserts exactly one thing, that `access` was assigned. It is + // assigned unconditionally three lines down, and `CanonicalReference` + // rejects the record on the first reference if that ever stops being true. + const entry: Pick & Partial = + { url }; if (resolver.language !== undefined) entry.language = resolver.language; if (resolver.edition !== undefined) entry.edition = resolver.edition; if (resolver.provider !== undefined) entry.provider = resolver.provider; entry.access = resolver.access ?? 'unknown'; - if (resolver.license !== undefined) { - if (!SPDX_IDS.has(resolver.license)) { - // Unreachable via authored YAML — ResolverEntrySource rejects it at - // parse time. Kept so a future caller that skips the parser cannot - // drop a licence statement silently. - throw new Error( - `license "${resolver.license}" is not an SPDX id (record the provider's rights statement in license_url)`, - ); - } - // Emit the canonical SPDX IRI so dcterms:license has a single - // IRI-typed range in the JSON-LD output. + // Emit the canonical SPDX IRI so dcterms:license has a single IRI-typed + // range in the JSON-LD output. + if (resolver.license !== undefined) entry.license = `${SPDX_LICENSE_BASE}${resolver.license}`; - } if (resolver.license_url !== undefined) entry.license_url = resolver.license_url; if (resolver.last_checked !== undefined) entry.last_checked = resolver.last_checked; - return entry; + return entry as ResolverTargetEntry; } function referenceUuid( @@ -319,6 +341,47 @@ function mappingUuid( return uuidv5(seed, MAPPING_NS); } +/** + * Print the indented issue lines of a failed `safeParse`, in the one form both + * registry gates use. + * + * Only the issue lines are shared. The header above them, and what happens + * after, differ by caller and must: the compiler names the record and throws, + * because a malformed record must never reach the dump, while + * `scripts/validate-data.ts` counts the failure and continues, so that one run + * reports every bad record instead of only the first. + */ +export function printIssues( + issues: readonly { path: readonly PropertyKey[]; message: string }[], +): void { + for (const issue of issues) { + const path = issue.path.map((p) => String(p)).join('.'); + console.error(` ${path || '(root)'}: ${issue.message}`); + } +} + +/** + * Validate one compiled record against its canonical schema, or fail the build. + * + * Every record type ran this same block before: report the path and message of + * each issue, then throw. `prefix` names the record the way its IRI does + * (`ref`, `system`, `work`, `mapping`); `kind` names it the way the thrown + * error reads. The two differ only for a reference, whose IRI says `ref`. + */ +function parseRecord( + schema: S, + record: unknown, + kind: string, + prefix: string, + id: string, +): z.infer { + const parsed = schema.safeParse(record); + if (parsed.success) return parsed.data; + console.error(`✗ ${prefix}/${id}: invalid`); + printIssues(parsed.error.issues); + throw new Error(`invalid ${kind}: ${id}`); +} + function setAlias( aliases: Record, alias: string, @@ -373,7 +436,15 @@ function emitBlockReferences(opts: { const extraResolvers = typeof refSrc === 'string' ? [] : (refSrc.extra_resolvers ?? []); const vars = deriveLocatorVars(locator, system); - const targets: Record[] = []; + // The block's resolvers first, then this reference's own extras. A + // skipped entry warns whichever list it came from: only the first loop + // counted before, so a hole in a per-reference `extra_resolvers` map + // stayed silent — the one outcome `applyResolverVars` exists to prevent. + // + // Two loops rather than one over a concatenation. Joining the lists + // reads better and allocates an array per reference, 86k of them, to + // save four lines. The duplication is the cheaper half of that trade. + const targets: ResolverTargetEntry[] = []; for (const resolver of block.resolvers ?? []) { const entry = buildResolverEntry(resolver, vars); if (entry) targets.push(entry); @@ -382,10 +453,11 @@ function emitBlockReferences(opts: { for (const resolver of extraResolvers) { const entry = buildResolverEntry(resolver, vars); if (entry) targets.push(entry); + else warnings++; } const uuid = referenceUuid(workKey, systemKey, locator); const record = { - id: `https://textrefs.org/id/ref/${uuid}`, + id: refIri(uuid), type: 'CanonicalReference' as const, work_key: workKey, citation_system_key: systemKey, @@ -395,17 +467,15 @@ function emitBlockReferences(opts: { created: opts.created, modified: opts.modified, }; - const parsed = CanonicalReference.safeParse(record); - if (!parsed.success) { - console.error(`✗ ref/${workKey}/${systemKey}/${locator}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid reference: ${workKey}/${systemKey}/${locator}`); - } - outReferences.push(parsed.data); + outReferences.push( + parseRecord( + CanonicalReference, + record, + 'reference', + 'ref', + `${workKey}/${systemKey}/${locator}`, + ), + ); // Qualified alias: always. Keyed by the same tuple that seeds the UUID, // so it can never collide. setAlias(aliases, `${workKey}/${systemKey}/${locator}`, record.id); @@ -417,6 +487,76 @@ function emitBlockReferences(opts: { return warnings; } +/** + * Emit one work's reified `MappingAssertion` records, and the lookup alias each + * one earns. + * + * Separate from the `alternateOf` / `isReferencedBy` projection onto the Work: + * that projection drops retired assertions (#45), because an edge carries no + * status and would advertise a mapping the registry has taken out of use. A + * record keeps its own status, so every assertion becomes one, retired or not. + */ +function emitMappings( + src: WorkSource, + thisWorkIri: string, + outMappings: MappingAssertion[], + aliases: Record, +): void { + for (const mapping of src.mappings ?? []) { + const uuid = mappingUuid(thisWorkIri, mapping.relation, mapping.identifier); + const record = { + id: mappingIri(uuid), + type: 'MappingAssertion' as const, + subject: thisWorkIri, + relation: mapping.relation, + target: { + identifier: mapping.identifier, + ...(mapping.conforms_to !== undefined && { + conforms_to: mapping.conforms_to, + }), + }, + source: mapping.source, + status: mapping.status, + created: mapping.created, + modified: mapping.modified, + }; + outMappings.push( + parseRecord(MappingAssertion, record, 'mapping', 'mapping', uuid), + ); + // Deliberate under ADR-0006: an `isReferencedBy` target (a page + // *about* the work) stays a lookup alias for it. The alias table is + // a lookup convenience, not an identity claim. + setAlias(aliases, mapping.identifier, thisWorkIri); + } +} + +/** + * A work's citation system blocks: the preferred one first, then any fallback + * systems. Each block is emitted against its own citation system, resolvers, + * and status, and only the first mints the bare `/cite/{work}/{locator}` alias + * (ADR-0005). + */ +function systemBlocksOf( + src: WorkSource, +): Array<{ block: SystemBlockSource; isPreferred: boolean }> { + return [ + { + block: { + citation_system: src.citation_system, + reference_status: src.reference_status, + resolvers: src.resolvers, + references: src.references, + references_range: src.references_range, + }, + isPreferred: true, + }, + ...(src.additional_systems ?? []).map((block) => ({ + block, + isPreferred: false, + })), + ]; +} + export interface CompiledRegistry { works: Work[]; systems: CitationSystem[]; @@ -429,12 +569,23 @@ export interface CompiledRegistry { export function compileRegistry(dataRootOverride?: string): CompiledRegistry { const root = dataRootOverride ?? dataRoot; const systems = new Map(); + // A key is the whole identity of a citation system, so two files claiming + // one is an authoring error, not a merge. `Map.set` would keep the last file + // read and drop the other without a word. + const systemFileByKey = new Map(); for (const f of listYaml(join(root, 'systems'))) { const src = parseSource( SystemSource, parseYaml(readFileSync(f, 'utf8')), basename(f), ); + const firstFile = systemFileByKey.get(src.key); + if (firstFile !== undefined) { + throw new Error( + `citation system key "${src.key}" is declared twice: ${firstFile} and ${basename(f)}`, + ); + } + systemFileByKey.set(src.key, basename(f)); systems.set(src.key, src); } @@ -451,7 +602,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { a.localeCompare(b), )) { const record = { - id: `https://textrefs.org/id/system/${key}`, + id: systemIri(key), key, type: 'CitationSystem' as const, preferred_label: src.preferred_label, @@ -462,19 +613,18 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { modified: src.modified, ...(src.superseded_by ? { superseded_by: src.superseded_by } : {}), }; - const parsed = CitationSystem.safeParse(record); - if (!parsed.success) { - console.error(`✗ system/${key}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid system: ${key}`); - } - outSystems.push(parsed.data); + outSystems.push( + parseRecord(CitationSystem, record, 'system', 'system', key), + ); } + // Two work files claiming one key is never a merge. ADR-0002 seeds every + // reference UUID on `(work_key, citation_system_key, locator)`, so the + // second file would not just duplicate the Work record — it would mint the + // same reference identifiers as the first, and `setAlias` cannot see it + // because both files produce the same alias target. + const workFileByKey = new Map(); + for (const file of workFiles) { const src = parseSource( WorkSource, @@ -482,7 +632,14 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { basename(file), ); const workKey = src.work.key; - const workIri = `https://textrefs.org/id/work/${workKey}`; + const firstWorkFile = workFileByKey.get(workKey); + if (firstWorkFile !== undefined) { + throw new Error( + `work key "${workKey}" is declared twice: ${firstWorkFile} and ${basename(file)}`, + ); + } + workFileByKey.set(workKey, basename(file)); + const thisWorkIri = workIri(workKey); const systemKey = src.citation_system; // Direct mapping edges (prov:alternateOf / dcterms:isReferencedBy via @@ -501,7 +658,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { } const workRecord = { - id: workIri, + id: thisWorkIri, key: workKey, type: 'Work' as const, preferred_label: src.work.preferred_label, @@ -522,73 +679,11 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { Object.entries(mappingEdges).filter(([, targets]) => targets.length), ), }; - const workParsed = Work.safeParse(workRecord); - if (!workParsed.success) { - console.error(`✗ work/${workKey}: invalid`); - for (const issue of workParsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid work: ${workKey}`); - } - outWorks.push(workParsed.data); - - for (const mapping of src.mappings ?? []) { - const uuid = mappingUuid(workIri, mapping.relation, mapping.identifier); - const record = { - id: `https://textrefs.org/id/mapping/${uuid}`, - type: 'MappingAssertion' as const, - subject: workIri, - relation: mapping.relation, - target: { - identifier: mapping.identifier, - ...(mapping.conforms_to !== undefined && { - conforms_to: mapping.conforms_to, - }), - }, - source: mapping.source, - status: mapping.status, - created: mapping.created, - modified: mapping.modified, - }; - const parsed = MappingAssertion.safeParse(record); - if (!parsed.success) { - console.error(`✗ mapping/${uuid}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid mapping: ${uuid}`); - } - outMappings.push(parsed.data); - // Deliberate under ADR-0006: an `isReferencedBy` target (a page - // *about* the work) stays a lookup alias for it. The alias table is - // a lookup convenience, not an identity claim. - setAlias(aliases, mapping.identifier, workIri); - } + outWorks.push(parseRecord(Work, workRecord, 'work', 'work', workKey)); - // The preferred block first, then any fallback systems. Each block is - // emitted against its own citation system, resolvers, and status. - const blocks: Array<{ block: SystemBlockSource; isPreferred: boolean }> = [ - { - block: { - citation_system: systemKey, - reference_status: src.reference_status, - resolvers: src.resolvers, - references: src.references, - references_range: src.references_range, - }, - isPreferred: true, - }, - ...(src.additional_systems ?? []).map((block) => ({ - block, - isPreferred: false, - })), - ]; + emitMappings(src, thisWorkIri, outMappings, aliases); - for (const { block, isPreferred } of blocks) { + for (const { block, isPreferred } of systemBlocksOf(src)) { const system = systems.get(block.citation_system); if (!system) { throw new Error( @@ -618,7 +713,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { outReferences.sort((a, b) => a.id.localeCompare(b.id)); outMappings.sort((a, b) => a.id.localeCompare(b.id)); - warnings += enforceRegistryInvariants({ + enforceRegistryInvariants({ works: outWorks, systems: outSystems, references: outReferences, @@ -651,7 +746,7 @@ function enforceRegistryInvariants(reg: { systems: CitationSystem[]; references: CanonicalReference[]; mappings: MappingAssertion[]; -}): number { +}): void { const all: StatusRecord[] = [ ...reg.works, ...reg.systems, @@ -682,15 +777,15 @@ function enforceRegistryInvariants(reg: { // not by MappingAssertions, which are reserved for work-level equivalence. for (const ref of reg.references) { if (TOMBSTONE_STATUSES.has(ref.status)) continue; - const workIri = `https://textrefs.org/id/work/${ref.work_key}`; - const systemIri = `https://textrefs.org/id/system/${ref.citation_system_key}`; - if (tombstoneIris.has(workIri)) + const refWorkIri = workIri(ref.work_key); + const refSystemIri = systemIri(ref.citation_system_key); + if (tombstoneIris.has(refWorkIri)) errors.push( - `${ref.id}: live reference points at tombstoned work ${workIri}`, + `${ref.id}: live reference points at tombstoned work ${refWorkIri}`, ); - if (tombstoneIris.has(systemIri)) + if (tombstoneIris.has(refSystemIri)) errors.push( - `${ref.id}: live reference points at tombstoned system ${systemIri}`, + `${ref.id}: live reference points at tombstoned system ${refSystemIri}`, ); } @@ -716,7 +811,7 @@ function enforceRegistryInvariants(reg: { ); continue; } - const preferredIri = `https://textrefs.org/id/system/${key}`; + const preferredIri = systemIri(key); if (!TOMBSTONE_STATUSES.has(work.status) && tombstoneIris.has(preferredIri)) errors.push( `${work.id}: live work points at tombstoned preferred citation system ${preferredIri}`, @@ -730,17 +825,17 @@ function enforceRegistryInvariants(reg: { for (const ref of reg.references) { if (ref.status !== 'active') continue; - const workIri = `https://textrefs.org/id/work/${ref.work_key}`; - const systemIri = `https://textrefs.org/id/system/${ref.citation_system_key}`; - const workStatus = statusByIri.get(workIri); - const systemStatus = statusByIri.get(systemIri); + const refWorkIri = workIri(ref.work_key); + const refSystemIri = systemIri(ref.citation_system_key); + const workStatus = statusByIri.get(refWorkIri); + const systemStatus = statusByIri.get(refSystemIri); if (workStatus !== 'active') errors.push( - `${ref.id}: active reference requires an active work, but ${workIri} is ${workStatus ?? 'missing'}`, + `${ref.id}: active reference requires an active work, but ${refWorkIri} is ${workStatus ?? 'missing'}`, ); if (systemStatus !== 'active') errors.push( - `${ref.id}: active reference requires an active citation system, but ${systemIri} is ${systemStatus ?? 'missing'}`, + `${ref.id}: active reference requires an active citation system, but ${refSystemIri} is ${systemStatus ?? 'missing'}`, ); } @@ -761,7 +856,6 @@ function enforceRegistryInvariants(reg: { .join('\n')}`, ); } - return 0; } export function readPackageVersion(): string { @@ -809,24 +903,93 @@ function jsonlBody(records: ReadonlyArray): string { } /** - * What `/dump/` publishes, without any body. + * The complete alias table (#84). Two kinds of entry share it: a `/cite/` alias + * targeting a reference IRI, and an external mapping identifier targeting a + * work IRI. Values stay full IRIs so a consumer can tell the two apart; a `://` + * in the key marks the second kind. * - * Split out so a caller that only needs the file list pays nothing for it. - * `/dump/index.astro` renders this during `astro build`; calling - * `dumpResources` there instead would serialise and hash ~90 MB that - * `scripts/compile.ts` then serialises and hashes again. + * Keys are sorted by code unit, so the body — and therefore its sha256 — + * depends on the registry content alone, never on the order the compiler + * happened to visit the work files in. No indentation: the body is ~17 MB. */ -export const DUMP_MANIFEST = [ - { name: 'works', filename: 'works.jsonl', format: 'jsonl' }, +function aliasBody(registry: CompiledRegistry): string { + return ( + JSON.stringify( + Object.fromEntries( + Object.entries(registry.aliases).sort(([a], [b]) => + a < b ? -1 : a > b ? 1 : 0, + ), + ), + ) + '\n' + ); +} + +/** + * The five `/dump/` resources, declared once, in descriptor order. + * + * `body` is a thunk rather than a string because the two consumers need + * different halves of this list. `DUMP_MANIFEST` below is the file list alone, + * which `/dump/index.astro` renders during `astro build`; forcing the bodies + * there would serialise and hash ~90 MB that `scripts/compile.ts` then + * serialises and hashes again. `dumpResources` is the same list with every + * thunk called. + * + * Declaring the set twice is what this replaces: the manifest and the resource + * builder each listed all five, and only a test kept them in step. + */ +const DUMP_SPECS: ReadonlyArray<{ + name: string; + filename: string; + format: string; + mediatype: string; + body: (registry: CompiledRegistry) => string; +}> = [ + { + name: 'works', + filename: 'works.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.works), + }, { name: 'citation-systems', filename: 'citation-systems.jsonl', format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.systems), + }, + { + name: 'references', + filename: 'references.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.references), + }, + { + name: 'mappings', + filename: 'mappings.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.mappings), + }, + { + name: 'aliases', + filename: 'aliases.json', + format: 'json', + mediatype: 'application/json', + body: aliasBody, }, - { name: 'references', filename: 'references.jsonl', format: 'jsonl' }, - { name: 'mappings', filename: 'mappings.jsonl', format: 'jsonl' }, - { name: 'aliases', filename: 'aliases.json', format: 'json' }, -] as const; +]; + +/** + * What `/dump/` publishes, without any body. Derived from `DUMP_SPECS`, so it + * cannot drift from what `dumpResources` actually writes. + */ +export const DUMP_MANIFEST = DUMP_SPECS.map(({ name, filename, format }) => ({ + name, + filename, + format, +})); /** * Every `/dump/` resource body, in descriptor order. Pure — `writeDump` does @@ -834,48 +997,13 @@ export const DUMP_MANIFEST = [ * without a filesystem. */ export function dumpResources(registry: CompiledRegistry): ResourceSpec[] { - const jsonl = ( - name: string, - filename: string, - records: ReadonlyArray, - ): ResourceSpec => ({ - name, - filename, - format: 'jsonl', - mediatype: 'application/x-ndjson', - body: jsonlBody(records), - }); - - // The complete alias table (#84). Two kinds of entry share it: a `/cite/` - // alias targeting a reference IRI, and an external mapping identifier - // targeting a work IRI. Values stay full IRIs so a consumer can tell the two - // apart; a `://` in the key marks the second kind. - // - // Keys are sorted by code unit, so the body — and therefore its sha256 — - // depends on the registry content alone, never on the order the compiler - // happened to visit the work files in. No indentation: the body is ~13 MB. - const aliasBody = - JSON.stringify( - Object.fromEntries( - Object.entries(registry.aliases).sort(([a], [b]) => - a < b ? -1 : a > b ? 1 : 0, - ), - ), - ) + '\n'; - - return [ - jsonl('works', 'works.jsonl', registry.works), - jsonl('citation-systems', 'citation-systems.jsonl', registry.systems), - jsonl('references', 'references.jsonl', registry.references), - jsonl('mappings', 'mappings.jsonl', registry.mappings), - { - name: 'aliases', - filename: 'aliases.json', - format: 'json', - mediatype: 'application/json', - body: aliasBody, - }, - ]; + return DUMP_SPECS.map((spec) => ({ + name: spec.name, + filename: spec.filename, + format: spec.format, + mediatype: spec.mediatype, + body: spec.body(registry), + })); } /** diff --git a/scripts/validate-data.ts b/scripts/validate-data.ts index 5813a2d..b8561cc 100644 --- a/scripts/validate-data.ts +++ b/scripts/validate-data.ts @@ -2,7 +2,8 @@ import { readFileSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; import { dirname, join } from 'node:path'; import { v5 as uuidv5 } from 'uuid'; -import { compileRegistry } from './compile.js'; +import { compileRegistry, printIssues } from './compile.js'; +import { refIri, mappingIri } from '../standard/iri.js'; import { Work, CitationSystem, @@ -12,6 +13,15 @@ import { const projectRoot = join(dirname(fileURLToPath(import.meta.url)), '..'); +// The namespaces and the seed strings below are deliberately a second, +// independent implementation of what `scripts/compile.ts` does. Do not "fix" +// this by importing `referenceUuid` and `mappingUuid` from the compiler. +// +// This gate exists to prove the compiler's identifiers are deterministic from +// the ADR-0002 tuple. Calling the compiler's own function to compute the +// expected value would make every assertion below a tautology that passes +// whatever the compiler does — including silently re-minting all 86k +// identifiers. Two implementations that must agree is the whole check. const REFERENCE_NS = 'b1a3670e-2ac7-544c-a1b9-396e0dc193f7'; const MAPPING_NS = 'f16bb214-4241-549d-ad41-7b011f02befb'; @@ -25,10 +35,7 @@ function reportIssue( issues: readonly { path: readonly PropertyKey[]; message: string }[], ): void { console.error(`✗ ${label}:`); - for (const issue of issues) { - const path = issue.path.map((p) => String(p)).join('.'); - console.error(` ${path || '(root)'}: ${issue.message}`); - } + printIssues(issues); failed++; } @@ -52,7 +59,7 @@ for (const ref of registry.references) { continue; } const seed = [ref.work_key, ref.citation_system_key, ref.locator].join('\n'); - const expected = `https://textrefs.org/id/ref/${uuidv5(seed, REFERENCE_NS)}`; + const expected = refIri(uuidv5(seed, REFERENCE_NS)); if (ref.id !== expected) { console.error( `✗ ref/${ref.work_key}/${ref.locator}: UUID not deterministic from seed (got ${ref.id}, expected ${expected})`, @@ -69,7 +76,7 @@ for (const m of registry.mappings) { continue; } const seed = [m.subject, m.relation, m.target.identifier].join('\n'); - const expected = `https://textrefs.org/id/mapping/${uuidv5(seed, MAPPING_NS)}`; + const expected = mappingIri(uuidv5(seed, MAPPING_NS)); if (m.id !== expected) { console.error( `✗ mapping/${m.id}: UUID not deterministic from seed (expected ${expected})`, diff --git a/src/lib/find.ts b/src/lib/find.ts index e45bf2b..b8e8b53 100644 --- a/src/lib/find.ts +++ b/src/lib/find.ts @@ -29,6 +29,9 @@ import type { WorkAliasIndex } from './alias-index.js'; import { byKey } from './collection.js'; +// `standard/iri.js` is dependency-free by design, so importing it here does not +// breach the rule stated above about what may travel into the browser bundle. +import { refIri } from '../../standard/iri.js'; export type FindCreator = | { kind: 'person'; family: string; given?: string } @@ -518,5 +521,5 @@ export function resolveInIndex( /** The canonical TextRefs URI of a reference. */ export function referenceIri(uuid: string): string { - return `https://textrefs.org/id/ref/${uuid}`; + return refIri(uuid); } diff --git a/standard/iri.ts b/standard/iri.ts new file mode 100644 index 0000000..dabaff0 --- /dev/null +++ b/standard/iri.ts @@ -0,0 +1,38 @@ +// How a TextRefs IRI is spelled, in one place. +// +// The four record types put their key or their UUID after a fixed prefix, and +// that prefix was written out at ten call sites across `scripts/compile.ts`, +// `scripts/validate-data.ts` and `src/lib/find.ts` before this module existed. +// A prefix is not interesting enough to get wrong twice. +// +// This module MUST stay dependency-free. `src/lib/find.ts` ships to the browser +// inside the `/find/` bundle, so anything imported here travels with it; that +// is the same rule its own header states about `standard/schema/`, which pulls +// in Zod. Nothing below imports anything. +// +// The canonical form is fixed by the specification and by the `id` regexes in +// `standard/schema/`. Those regexes stay written out: a pattern is checked +// against a string, and deriving one from the other would make each half prove +// the other rather than prove the spelling. + +export const BASE = 'https://textrefs.org'; + +/** `https://textrefs.org/id/work/{key}` */ +export function workIri(key: string): string { + return `${BASE}/id/work/${key}`; +} + +/** `https://textrefs.org/id/system/{key}` */ +export function systemIri(key: string): string { + return `${BASE}/id/system/${key}`; +} + +/** `https://textrefs.org/id/ref/{uuid}` */ +export function refIri(uuid: string): string { + return `${BASE}/id/ref/${uuid}`; +} + +/** `https://textrefs.org/id/mapping/{uuid}` */ +export function mappingIri(uuid: string): string { + return `${BASE}/id/mapping/${uuid}`; +} From 0df184871b71ac67f52bef35b8f39897c4e9781c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Moritz=20M=C3=A4hr?= <14755525+maehr@users.noreply.github.com> Date: Mon, 7 Sep 2026 19:26:33 +0200 Subject: [PATCH 2/2] test(registry): freeze fifteen published identifiers against silent re-minting `scripts/validate-data.ts` recomputes every reference and mapping UUID from the ADR-0002 seed, independently of the compiler. That check proves the two implementations agree. It cannot prove either one still mints the identifiers the registry published, because both are code. Change the seed rule in `scripts/compile.ts` and in this file, which is the natural move when the check goes red, and every assertion passes while all 86,397 reference identifiers move. Nothing else in the repository records what those identifiers are. Freeze fifteen of them as data: one reference per citation system, and one mapping per relation, because the relation is part of the mapping seed. No edit to any implementation can satisfy a literal. The values come from the v0.1.0 baseline that PR #4 tags. The check runs where the registry is already compiled, so it costs nothing. `release.yml` runs `npm run build:data`, so it guards the exact path that produces the release artifact and the Zenodo record. A failure here is not a test to fix. It reports that a published citation has stopped resolving. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01VLqvaFc9mqQYcjb9SiKGXd --- scripts/validate-data.ts | 151 ++++++++++++++++++++++++++++++++++++++- 1 file changed, 150 insertions(+), 1 deletion(-) diff --git a/scripts/validate-data.ts b/scripts/validate-data.ts index b8561cc..6f529b5 100644 --- a/scripts/validate-data.ts +++ b/scripts/validate-data.ts @@ -120,7 +120,156 @@ if (unmapped.size > 0) { failed += unmapped.size; } +// --- Frozen identifiers ------------------------------------------------------ +// +// The recomputation above proves the compiler agrees with a second +// implementation of the ADR-0002 seed rule. It does not prove either one still +// produces the identifiers the registry published, because both are code: edit +// the seed rule in `scripts/compile.ts` and in this file — the natural move +// when the check goes red — and every assertion passes while all 86k +// identifiers silently re-mint. +// +// These are data, so no edit to any implementation can satisfy them. Each pair +// was taken from the v0.1.0 baseline: one reference per citation system, and +// one mapping per relation, because the relation is part of the mapping seed. +// +// A failure here is not a test to fix. It means the identifiers the registry +// promised have moved, and a published citation has stopped resolving. ADR-0003 +// allows that for a `draft` record; after the v0.1.0 tag it is a breaking +// change, and the tombstone rules in specification §12 apply. +const FROZEN_REFERENCE_IDS: ReadonlyArray<[string, string, string, string]> = [ + [ + 'confucius.analects', + 'analects-book-chapter', + '11.3', + '00e5c911-4eb8-5309-b38e-a409d8bf3c2b', + ], + [ + 'aristotle.nicomachean-ethics', + 'bekker', + '1103b26', + '000112c4-c060-5823-8fe9-c4632d7c9b2b', + ], + [ + 'new-testament', + 'bible-book-chapter-verse', + 'Luke.1.77', + '0001f355-684b-5c1d-978d-351ed95824c0', + ], + [ + 'dante.commedia', + 'dante-cantica-canto-verse', + 'Purg.9.41', + '00084ac1-eb8c-50c5-abe7-d90e44909314', + ], + [ + 'laozi.daodejing', + 'daodejing-chapter', + '61', + '024cf6b6-b1d0-5b90-b0fa-5b99ce612a35', + ], + [ + 'dhammapada', + 'dhammapada-chapter-verse', + '14.1', + '003e3b7a-8893-5917-a731-62a2fa02621b', + ], + [ + 'murasaki-shikibu.genji', + 'genji-chapter', + '47', + '013ade7c-241d-5685-9c7f-97ae56a860ee', + ], + [ + 'homer.iliad', + 'homer-book-line', + '7.18', + '0002587e-7044-516a-9825-f740950b9663', + ], + [ + 'hume.treatise', + 'hume-book-part-section-paragraph', + '2.2.10.4', + '001ef489-c761-53b1-8a07-ce51c95ea4c1', + ], + [ + 'hume.enquiry-human-understanding', + 'hume-section-paragraph', + '1.13', + '0053bd60-ae8c-59c6-bf37-fbf44241b0a8', + ], + [ + 'wittgenstein.philosophical-investigations', + 'integer-section', + '76', + '00164dc6-f185-5081-b9e4-84d59a9ab0e1', + ], + [ + 'plato.republic', + 'stephanus', + '509b', + '00182f77-1597-5f2f-b215-59af4c645640', + ], + [ + 'wittgenstein.tractatus', + 'tractatus-proposition', + '2.0231', + '007377e5-fe05-545a-a4ff-24ee9f016aa9', + ], +]; + +const FROZEN_MAPPING_IDS: ReadonlyArray<[string, string, string, string]> = [ + [ + 'homer.iliad', + 'alternateOf', + 'https://www.wikidata.org/entity/Q8275', + '158a91fb-ac10-5c1c-b048-1fca5d421289', + ], + [ + 'homer.iliad', + 'isReferencedBy', + 'https://en.wikipedia.org/wiki/Iliad', + '14fe951b-bb68-5b82-943a-ca122b2bb61b', + ], +]; + +const refByTuple = new Map( + registry.references.map((r) => [ + [r.work_key, r.citation_system_key, r.locator].join('\n'), + r.id, + ]), +); +for (const [work, system, locator, uuid] of FROZEN_REFERENCE_IDS) { + checked++; + const actual = refByTuple.get([work, system, locator].join('\n')); + const expected = refIri(uuid); + if (actual === undefined) { + console.error( + `✗ frozen ${work}/${system}/${locator}: no such reference in the registry`, + ); + failed++; + } else if (actual !== expected) { + console.error( + `✗ frozen ${work}/${system}/${locator}: identifier moved\n was ${expected}\n compiled ${actual}`, + ); + failed++; + } +} + +const mappingById = new Map(registry.mappings.map((m) => [m.id, m])); +for (const [work, relation, identifier, uuid] of FROZEN_MAPPING_IDS) { + checked++; + const expected = mappingIri(uuid); + const found = mappingById.get(expected); + if (!found) { + console.error( + `✗ frozen mapping ${work} ${relation} ${identifier}: ${expected} is no longer minted`, + ); + failed++; + } +} + console.log( - `\n${checked - failed}/${checked} records valid (works=${registry.works.length}, systems=${registry.systems.length}, refs=${registry.references.length}, mappings=${registry.mappings.length})`, + `\n${checked - failed}/${checked} records valid (works=${registry.works.length}, systems=${registry.systems.length}, refs=${registry.references.length}, mappings=${registry.mappings.length}); ${FROZEN_REFERENCE_IDS.length + FROZEN_MAPPING_IDS.length} frozen identifiers unchanged`, ); if (failed > 0) process.exit(1);