diff --git a/scripts/compile.test.ts b/scripts/compile.test.ts index 99233d5..a45f7aa 100644 --- a/scripts/compile.test.ts +++ b/scripts/compile.test.ts @@ -147,6 +147,43 @@ additional_systems: assert.match(message, /invalid source file/); }); +// A key is the whole identity of a work or a citation system. Two files +// claiming one used to pass: the systems Map kept the last file read and +// dropped the other, and works had no check at all — so the second work would +// mint the same reference UUIDs as the first, because ADR-0002 seeds them on +// `(work_key, citation_system_key, locator)`. `setAlias` cannot catch that, +// since both files produce the same alias target. + +test('two files declaring one citation system key fail the build', (t) => { + const message = expectCompileError(t, { + systems: { + 'first-file': system('shared'), + 'second-file': system('shared'), + }, + works: { + 'test.work': `${workHeader()} +citation_system: shared +references: + - '1' +`, + }, + }); + assert.match(message, /citation system key "shared" is declared twice/); +}); + +test('two files declaring one work key fail the build', (t) => { + const body = `${workHeader()} +citation_system: primary-section +references: + - '1' +`; + const message = expectCompileError(t, { + systems: twoSystems, + works: { 'first-file': body, 'second-file': body }, + }); + assert.match(message, /work key "test.work" is declared twice/); +}); + test('a locator containing "/" is rejected before any alias is minted', (t) => { const message = expectCompileError(t, { systems: { @@ -950,18 +987,22 @@ test('every descriptor carries the Frictionless fields', () => { }); // `/dump/index.astro` renders DUMP_MANIFEST so that building the page costs no -// serialisation. That is only safe while the manifest names exactly what -// `dumpResources` produces, in the same order. -test('DUMP_MANIFEST matches what dumpResources produces', () => { - const specs = dumpResources(compileFixture(workWithMappings())); +// serialisation. Both it and `dumpResources` are derived from the one +// `DUMP_SPECS` list, so they can no longer disagree about the set or its order. +// What still needs pinning is that the manifest carries no body: forcing one is +// exactly what the split exists to avoid. +test('DUMP_MANIFEST lists the files without carrying a body', () => { assert.deepEqual( - specs.map(({ name, filename, format }) => ({ name, filename, format })), DUMP_MANIFEST.map(({ name, filename, format }) => ({ name, filename, format, })), + dumpResources(compileFixture(workWithMappings())).map( + ({ name, filename, format }) => ({ name, filename, format }), + ), ); + for (const entry of DUMP_MANIFEST) assert.ok(!('body' in entry)); }); test('packageVersion drops a leading v and leaves the rest alone', () => { diff --git a/scripts/compile.ts b/scripts/compile.ts index a686a62..05c9efc 100644 --- a/scripts/compile.ts +++ b/scripts/compile.ts @@ -9,12 +9,15 @@ import { join, basename, resolve } from 'node:path'; import { createHash } from 'node:crypto'; import { v5 as uuidv5 } from 'uuid'; import { parse as parseYaml } from 'yaml'; +import type { z, ZodType } from 'zod'; import { Work, CitationSystem, CanonicalReference, MappingAssertion, } from '../standard/schema/index.js'; +import type { ResolverTargetEntry } from '../standard/schema/canonical-reference.js'; +import { workIri, systemIri, refIri, mappingIri } from '../standard/iri.js'; import { parseSource, SystemSource, @@ -260,7 +263,7 @@ function applyResolverVars( function buildResolverEntry( resolver: ResolverEntry, locatorVars: Record, -): Record | null { +): ResolverTargetEntry | null { const vars = applyResolverVars(resolver, locatorVars); if (!vars) return null; let url: string | null = null; @@ -276,29 +279,48 @@ function buildResolverEntry( url = Object.hasOwn(byMap, key) ? (byMap[key] ?? null) : null; } if (!url) return null; - const entry: Record = { url }; + if (resolver.license !== undefined && !SPDX_IDS.has(resolver.license)) { + // Unreachable via authored YAML — ResolverEntrySource rejects it at + // parse time. Kept so a future caller that skips the parser cannot + // drop a licence statement silently. + throw new Error( + `license "${resolver.license}" is not an SPDX id (record the provider's rights statement in license_url)`, + ); + } + // TypeScript checks every field name and value below, rather than deferring + // a typo to `safeParse`. The assignment order is the published key order, + // and therefore the byte order of the JSONL dump: do not reorder these + // lines. + // + // Assignment rather than one object literal of conditional spreads. The + // literal reads better, but each `...(cond && { k: v })` allocates a + // throwaway object, and this runs 172k times per compile — measured at 3x + // the cost of the assignments for no gain, since the type below checks + // these just as well. + // + // `Pick<…, 'url'>` keeps `url` required, so dropping it from the initialiser + // is a compile error. `access` cannot be covered the same way: it is + // required too, but it is assigned fifth to hold the key order, and + // TypeScript does not let a narrowed property satisfy a required one — no + // arrangement of guards makes the object assignable without the cast. So + // the cast asserts exactly one thing, that `access` was assigned. It is + // assigned unconditionally three lines down, and `CanonicalReference` + // rejects the record on the first reference if that ever stops being true. + const entry: Pick & Partial = + { url }; if (resolver.language !== undefined) entry.language = resolver.language; if (resolver.edition !== undefined) entry.edition = resolver.edition; if (resolver.provider !== undefined) entry.provider = resolver.provider; entry.access = resolver.access ?? 'unknown'; - if (resolver.license !== undefined) { - if (!SPDX_IDS.has(resolver.license)) { - // Unreachable via authored YAML — ResolverEntrySource rejects it at - // parse time. Kept so a future caller that skips the parser cannot - // drop a licence statement silently. - throw new Error( - `license "${resolver.license}" is not an SPDX id (record the provider's rights statement in license_url)`, - ); - } - // Emit the canonical SPDX IRI so dcterms:license has a single - // IRI-typed range in the JSON-LD output. + // Emit the canonical SPDX IRI so dcterms:license has a single IRI-typed + // range in the JSON-LD output. + if (resolver.license !== undefined) entry.license = `${SPDX_LICENSE_BASE}${resolver.license}`; - } if (resolver.license_url !== undefined) entry.license_url = resolver.license_url; if (resolver.last_checked !== undefined) entry.last_checked = resolver.last_checked; - return entry; + return entry as ResolverTargetEntry; } function referenceUuid( @@ -319,6 +341,47 @@ function mappingUuid( return uuidv5(seed, MAPPING_NS); } +/** + * Print the indented issue lines of a failed `safeParse`, in the one form both + * registry gates use. + * + * Only the issue lines are shared. The header above them, and what happens + * after, differ by caller and must: the compiler names the record and throws, + * because a malformed record must never reach the dump, while + * `scripts/validate-data.ts` counts the failure and continues, so that one run + * reports every bad record instead of only the first. + */ +export function printIssues( + issues: readonly { path: readonly PropertyKey[]; message: string }[], +): void { + for (const issue of issues) { + const path = issue.path.map((p) => String(p)).join('.'); + console.error(` ${path || '(root)'}: ${issue.message}`); + } +} + +/** + * Validate one compiled record against its canonical schema, or fail the build. + * + * Every record type ran this same block before: report the path and message of + * each issue, then throw. `prefix` names the record the way its IRI does + * (`ref`, `system`, `work`, `mapping`); `kind` names it the way the thrown + * error reads. The two differ only for a reference, whose IRI says `ref`. + */ +function parseRecord( + schema: S, + record: unknown, + kind: string, + prefix: string, + id: string, +): z.infer { + const parsed = schema.safeParse(record); + if (parsed.success) return parsed.data; + console.error(`✗ ${prefix}/${id}: invalid`); + printIssues(parsed.error.issues); + throw new Error(`invalid ${kind}: ${id}`); +} + function setAlias( aliases: Record, alias: string, @@ -373,7 +436,15 @@ function emitBlockReferences(opts: { const extraResolvers = typeof refSrc === 'string' ? [] : (refSrc.extra_resolvers ?? []); const vars = deriveLocatorVars(locator, system); - const targets: Record[] = []; + // The block's resolvers first, then this reference's own extras. A + // skipped entry warns whichever list it came from: only the first loop + // counted before, so a hole in a per-reference `extra_resolvers` map + // stayed silent — the one outcome `applyResolverVars` exists to prevent. + // + // Two loops rather than one over a concatenation. Joining the lists + // reads better and allocates an array per reference, 86k of them, to + // save four lines. The duplication is the cheaper half of that trade. + const targets: ResolverTargetEntry[] = []; for (const resolver of block.resolvers ?? []) { const entry = buildResolverEntry(resolver, vars); if (entry) targets.push(entry); @@ -382,10 +453,11 @@ function emitBlockReferences(opts: { for (const resolver of extraResolvers) { const entry = buildResolverEntry(resolver, vars); if (entry) targets.push(entry); + else warnings++; } const uuid = referenceUuid(workKey, systemKey, locator); const record = { - id: `https://textrefs.org/id/ref/${uuid}`, + id: refIri(uuid), type: 'CanonicalReference' as const, work_key: workKey, citation_system_key: systemKey, @@ -395,17 +467,15 @@ function emitBlockReferences(opts: { created: opts.created, modified: opts.modified, }; - const parsed = CanonicalReference.safeParse(record); - if (!parsed.success) { - console.error(`✗ ref/${workKey}/${systemKey}/${locator}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid reference: ${workKey}/${systemKey}/${locator}`); - } - outReferences.push(parsed.data); + outReferences.push( + parseRecord( + CanonicalReference, + record, + 'reference', + 'ref', + `${workKey}/${systemKey}/${locator}`, + ), + ); // Qualified alias: always. Keyed by the same tuple that seeds the UUID, // so it can never collide. setAlias(aliases, `${workKey}/${systemKey}/${locator}`, record.id); @@ -417,6 +487,76 @@ function emitBlockReferences(opts: { return warnings; } +/** + * Emit one work's reified `MappingAssertion` records, and the lookup alias each + * one earns. + * + * Separate from the `alternateOf` / `isReferencedBy` projection onto the Work: + * that projection drops retired assertions (#45), because an edge carries no + * status and would advertise a mapping the registry has taken out of use. A + * record keeps its own status, so every assertion becomes one, retired or not. + */ +function emitMappings( + src: WorkSource, + thisWorkIri: string, + outMappings: MappingAssertion[], + aliases: Record, +): void { + for (const mapping of src.mappings ?? []) { + const uuid = mappingUuid(thisWorkIri, mapping.relation, mapping.identifier); + const record = { + id: mappingIri(uuid), + type: 'MappingAssertion' as const, + subject: thisWorkIri, + relation: mapping.relation, + target: { + identifier: mapping.identifier, + ...(mapping.conforms_to !== undefined && { + conforms_to: mapping.conforms_to, + }), + }, + source: mapping.source, + status: mapping.status, + created: mapping.created, + modified: mapping.modified, + }; + outMappings.push( + parseRecord(MappingAssertion, record, 'mapping', 'mapping', uuid), + ); + // Deliberate under ADR-0006: an `isReferencedBy` target (a page + // *about* the work) stays a lookup alias for it. The alias table is + // a lookup convenience, not an identity claim. + setAlias(aliases, mapping.identifier, thisWorkIri); + } +} + +/** + * A work's citation system blocks: the preferred one first, then any fallback + * systems. Each block is emitted against its own citation system, resolvers, + * and status, and only the first mints the bare `/cite/{work}/{locator}` alias + * (ADR-0005). + */ +function systemBlocksOf( + src: WorkSource, +): Array<{ block: SystemBlockSource; isPreferred: boolean }> { + return [ + { + block: { + citation_system: src.citation_system, + reference_status: src.reference_status, + resolvers: src.resolvers, + references: src.references, + references_range: src.references_range, + }, + isPreferred: true, + }, + ...(src.additional_systems ?? []).map((block) => ({ + block, + isPreferred: false, + })), + ]; +} + export interface CompiledRegistry { works: Work[]; systems: CitationSystem[]; @@ -429,12 +569,23 @@ export interface CompiledRegistry { export function compileRegistry(dataRootOverride?: string): CompiledRegistry { const root = dataRootOverride ?? dataRoot; const systems = new Map(); + // A key is the whole identity of a citation system, so two files claiming + // one is an authoring error, not a merge. `Map.set` would keep the last file + // read and drop the other without a word. + const systemFileByKey = new Map(); for (const f of listYaml(join(root, 'systems'))) { const src = parseSource( SystemSource, parseYaml(readFileSync(f, 'utf8')), basename(f), ); + const firstFile = systemFileByKey.get(src.key); + if (firstFile !== undefined) { + throw new Error( + `citation system key "${src.key}" is declared twice: ${firstFile} and ${basename(f)}`, + ); + } + systemFileByKey.set(src.key, basename(f)); systems.set(src.key, src); } @@ -451,7 +602,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { a.localeCompare(b), )) { const record = { - id: `https://textrefs.org/id/system/${key}`, + id: systemIri(key), key, type: 'CitationSystem' as const, preferred_label: src.preferred_label, @@ -462,19 +613,18 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { modified: src.modified, ...(src.superseded_by ? { superseded_by: src.superseded_by } : {}), }; - const parsed = CitationSystem.safeParse(record); - if (!parsed.success) { - console.error(`✗ system/${key}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid system: ${key}`); - } - outSystems.push(parsed.data); + outSystems.push( + parseRecord(CitationSystem, record, 'system', 'system', key), + ); } + // Two work files claiming one key is never a merge. ADR-0002 seeds every + // reference UUID on `(work_key, citation_system_key, locator)`, so the + // second file would not just duplicate the Work record — it would mint the + // same reference identifiers as the first, and `setAlias` cannot see it + // because both files produce the same alias target. + const workFileByKey = new Map(); + for (const file of workFiles) { const src = parseSource( WorkSource, @@ -482,7 +632,14 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { basename(file), ); const workKey = src.work.key; - const workIri = `https://textrefs.org/id/work/${workKey}`; + const firstWorkFile = workFileByKey.get(workKey); + if (firstWorkFile !== undefined) { + throw new Error( + `work key "${workKey}" is declared twice: ${firstWorkFile} and ${basename(file)}`, + ); + } + workFileByKey.set(workKey, basename(file)); + const thisWorkIri = workIri(workKey); const systemKey = src.citation_system; // Direct mapping edges (prov:alternateOf / dcterms:isReferencedBy via @@ -501,7 +658,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { } const workRecord = { - id: workIri, + id: thisWorkIri, key: workKey, type: 'Work' as const, preferred_label: src.work.preferred_label, @@ -522,73 +679,11 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { Object.entries(mappingEdges).filter(([, targets]) => targets.length), ), }; - const workParsed = Work.safeParse(workRecord); - if (!workParsed.success) { - console.error(`✗ work/${workKey}: invalid`); - for (const issue of workParsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid work: ${workKey}`); - } - outWorks.push(workParsed.data); - - for (const mapping of src.mappings ?? []) { - const uuid = mappingUuid(workIri, mapping.relation, mapping.identifier); - const record = { - id: `https://textrefs.org/id/mapping/${uuid}`, - type: 'MappingAssertion' as const, - subject: workIri, - relation: mapping.relation, - target: { - identifier: mapping.identifier, - ...(mapping.conforms_to !== undefined && { - conforms_to: mapping.conforms_to, - }), - }, - source: mapping.source, - status: mapping.status, - created: mapping.created, - modified: mapping.modified, - }; - const parsed = MappingAssertion.safeParse(record); - if (!parsed.success) { - console.error(`✗ mapping/${uuid}: invalid`); - for (const issue of parsed.error.issues) { - console.error( - ` ${issue.path.join('.') || '(root)'}: ${issue.message}`, - ); - } - throw new Error(`invalid mapping: ${uuid}`); - } - outMappings.push(parsed.data); - // Deliberate under ADR-0006: an `isReferencedBy` target (a page - // *about* the work) stays a lookup alias for it. The alias table is - // a lookup convenience, not an identity claim. - setAlias(aliases, mapping.identifier, workIri); - } + outWorks.push(parseRecord(Work, workRecord, 'work', 'work', workKey)); - // The preferred block first, then any fallback systems. Each block is - // emitted against its own citation system, resolvers, and status. - const blocks: Array<{ block: SystemBlockSource; isPreferred: boolean }> = [ - { - block: { - citation_system: systemKey, - reference_status: src.reference_status, - resolvers: src.resolvers, - references: src.references, - references_range: src.references_range, - }, - isPreferred: true, - }, - ...(src.additional_systems ?? []).map((block) => ({ - block, - isPreferred: false, - })), - ]; + emitMappings(src, thisWorkIri, outMappings, aliases); - for (const { block, isPreferred } of blocks) { + for (const { block, isPreferred } of systemBlocksOf(src)) { const system = systems.get(block.citation_system); if (!system) { throw new Error( @@ -618,7 +713,7 @@ export function compileRegistry(dataRootOverride?: string): CompiledRegistry { outReferences.sort((a, b) => a.id.localeCompare(b.id)); outMappings.sort((a, b) => a.id.localeCompare(b.id)); - warnings += enforceRegistryInvariants({ + enforceRegistryInvariants({ works: outWorks, systems: outSystems, references: outReferences, @@ -651,7 +746,7 @@ function enforceRegistryInvariants(reg: { systems: CitationSystem[]; references: CanonicalReference[]; mappings: MappingAssertion[]; -}): number { +}): void { const all: StatusRecord[] = [ ...reg.works, ...reg.systems, @@ -682,15 +777,15 @@ function enforceRegistryInvariants(reg: { // not by MappingAssertions, which are reserved for work-level equivalence. for (const ref of reg.references) { if (TOMBSTONE_STATUSES.has(ref.status)) continue; - const workIri = `https://textrefs.org/id/work/${ref.work_key}`; - const systemIri = `https://textrefs.org/id/system/${ref.citation_system_key}`; - if (tombstoneIris.has(workIri)) + const refWorkIri = workIri(ref.work_key); + const refSystemIri = systemIri(ref.citation_system_key); + if (tombstoneIris.has(refWorkIri)) errors.push( - `${ref.id}: live reference points at tombstoned work ${workIri}`, + `${ref.id}: live reference points at tombstoned work ${refWorkIri}`, ); - if (tombstoneIris.has(systemIri)) + if (tombstoneIris.has(refSystemIri)) errors.push( - `${ref.id}: live reference points at tombstoned system ${systemIri}`, + `${ref.id}: live reference points at tombstoned system ${refSystemIri}`, ); } @@ -716,7 +811,7 @@ function enforceRegistryInvariants(reg: { ); continue; } - const preferredIri = `https://textrefs.org/id/system/${key}`; + const preferredIri = systemIri(key); if (!TOMBSTONE_STATUSES.has(work.status) && tombstoneIris.has(preferredIri)) errors.push( `${work.id}: live work points at tombstoned preferred citation system ${preferredIri}`, @@ -730,17 +825,17 @@ function enforceRegistryInvariants(reg: { for (const ref of reg.references) { if (ref.status !== 'active') continue; - const workIri = `https://textrefs.org/id/work/${ref.work_key}`; - const systemIri = `https://textrefs.org/id/system/${ref.citation_system_key}`; - const workStatus = statusByIri.get(workIri); - const systemStatus = statusByIri.get(systemIri); + const refWorkIri = workIri(ref.work_key); + const refSystemIri = systemIri(ref.citation_system_key); + const workStatus = statusByIri.get(refWorkIri); + const systemStatus = statusByIri.get(refSystemIri); if (workStatus !== 'active') errors.push( - `${ref.id}: active reference requires an active work, but ${workIri} is ${workStatus ?? 'missing'}`, + `${ref.id}: active reference requires an active work, but ${refWorkIri} is ${workStatus ?? 'missing'}`, ); if (systemStatus !== 'active') errors.push( - `${ref.id}: active reference requires an active citation system, but ${systemIri} is ${systemStatus ?? 'missing'}`, + `${ref.id}: active reference requires an active citation system, but ${refSystemIri} is ${systemStatus ?? 'missing'}`, ); } @@ -761,7 +856,6 @@ function enforceRegistryInvariants(reg: { .join('\n')}`, ); } - return 0; } export function readPackageVersion(): string { @@ -809,24 +903,93 @@ function jsonlBody(records: ReadonlyArray): string { } /** - * What `/dump/` publishes, without any body. + * The complete alias table (#84). Two kinds of entry share it: a `/cite/` alias + * targeting a reference IRI, and an external mapping identifier targeting a + * work IRI. Values stay full IRIs so a consumer can tell the two apart; a `://` + * in the key marks the second kind. * - * Split out so a caller that only needs the file list pays nothing for it. - * `/dump/index.astro` renders this during `astro build`; calling - * `dumpResources` there instead would serialise and hash ~90 MB that - * `scripts/compile.ts` then serialises and hashes again. + * Keys are sorted by code unit, so the body — and therefore its sha256 — + * depends on the registry content alone, never on the order the compiler + * happened to visit the work files in. No indentation: the body is ~17 MB. */ -export const DUMP_MANIFEST = [ - { name: 'works', filename: 'works.jsonl', format: 'jsonl' }, +function aliasBody(registry: CompiledRegistry): string { + return ( + JSON.stringify( + Object.fromEntries( + Object.entries(registry.aliases).sort(([a], [b]) => + a < b ? -1 : a > b ? 1 : 0, + ), + ), + ) + '\n' + ); +} + +/** + * The five `/dump/` resources, declared once, in descriptor order. + * + * `body` is a thunk rather than a string because the two consumers need + * different halves of this list. `DUMP_MANIFEST` below is the file list alone, + * which `/dump/index.astro` renders during `astro build`; forcing the bodies + * there would serialise and hash ~90 MB that `scripts/compile.ts` then + * serialises and hashes again. `dumpResources` is the same list with every + * thunk called. + * + * Declaring the set twice is what this replaces: the manifest and the resource + * builder each listed all five, and only a test kept them in step. + */ +const DUMP_SPECS: ReadonlyArray<{ + name: string; + filename: string; + format: string; + mediatype: string; + body: (registry: CompiledRegistry) => string; +}> = [ + { + name: 'works', + filename: 'works.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.works), + }, { name: 'citation-systems', filename: 'citation-systems.jsonl', format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.systems), + }, + { + name: 'references', + filename: 'references.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.references), + }, + { + name: 'mappings', + filename: 'mappings.jsonl', + format: 'jsonl', + mediatype: 'application/x-ndjson', + body: (r) => jsonlBody(r.mappings), + }, + { + name: 'aliases', + filename: 'aliases.json', + format: 'json', + mediatype: 'application/json', + body: aliasBody, }, - { name: 'references', filename: 'references.jsonl', format: 'jsonl' }, - { name: 'mappings', filename: 'mappings.jsonl', format: 'jsonl' }, - { name: 'aliases', filename: 'aliases.json', format: 'json' }, -] as const; +]; + +/** + * What `/dump/` publishes, without any body. Derived from `DUMP_SPECS`, so it + * cannot drift from what `dumpResources` actually writes. + */ +export const DUMP_MANIFEST = DUMP_SPECS.map(({ name, filename, format }) => ({ + name, + filename, + format, +})); /** * Every `/dump/` resource body, in descriptor order. Pure — `writeDump` does @@ -834,48 +997,13 @@ export const DUMP_MANIFEST = [ * without a filesystem. */ export function dumpResources(registry: CompiledRegistry): ResourceSpec[] { - const jsonl = ( - name: string, - filename: string, - records: ReadonlyArray, - ): ResourceSpec => ({ - name, - filename, - format: 'jsonl', - mediatype: 'application/x-ndjson', - body: jsonlBody(records), - }); - - // The complete alias table (#84). Two kinds of entry share it: a `/cite/` - // alias targeting a reference IRI, and an external mapping identifier - // targeting a work IRI. Values stay full IRIs so a consumer can tell the two - // apart; a `://` in the key marks the second kind. - // - // Keys are sorted by code unit, so the body — and therefore its sha256 — - // depends on the registry content alone, never on the order the compiler - // happened to visit the work files in. No indentation: the body is ~13 MB. - const aliasBody = - JSON.stringify( - Object.fromEntries( - Object.entries(registry.aliases).sort(([a], [b]) => - a < b ? -1 : a > b ? 1 : 0, - ), - ), - ) + '\n'; - - return [ - jsonl('works', 'works.jsonl', registry.works), - jsonl('citation-systems', 'citation-systems.jsonl', registry.systems), - jsonl('references', 'references.jsonl', registry.references), - jsonl('mappings', 'mappings.jsonl', registry.mappings), - { - name: 'aliases', - filename: 'aliases.json', - format: 'json', - mediatype: 'application/json', - body: aliasBody, - }, - ]; + return DUMP_SPECS.map((spec) => ({ + name: spec.name, + filename: spec.filename, + format: spec.format, + mediatype: spec.mediatype, + body: spec.body(registry), + })); } /** diff --git a/scripts/validate-data.ts b/scripts/validate-data.ts index 5813a2d..6f529b5 100644 --- a/scripts/validate-data.ts +++ b/scripts/validate-data.ts @@ -2,7 +2,8 @@ import { readFileSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; import { dirname, join } from 'node:path'; import { v5 as uuidv5 } from 'uuid'; -import { compileRegistry } from './compile.js'; +import { compileRegistry, printIssues } from './compile.js'; +import { refIri, mappingIri } from '../standard/iri.js'; import { Work, CitationSystem, @@ -12,6 +13,15 @@ import { const projectRoot = join(dirname(fileURLToPath(import.meta.url)), '..'); +// The namespaces and the seed strings below are deliberately a second, +// independent implementation of what `scripts/compile.ts` does. Do not "fix" +// this by importing `referenceUuid` and `mappingUuid` from the compiler. +// +// This gate exists to prove the compiler's identifiers are deterministic from +// the ADR-0002 tuple. Calling the compiler's own function to compute the +// expected value would make every assertion below a tautology that passes +// whatever the compiler does — including silently re-minting all 86k +// identifiers. Two implementations that must agree is the whole check. const REFERENCE_NS = 'b1a3670e-2ac7-544c-a1b9-396e0dc193f7'; const MAPPING_NS = 'f16bb214-4241-549d-ad41-7b011f02befb'; @@ -25,10 +35,7 @@ function reportIssue( issues: readonly { path: readonly PropertyKey[]; message: string }[], ): void { console.error(`✗ ${label}:`); - for (const issue of issues) { - const path = issue.path.map((p) => String(p)).join('.'); - console.error(` ${path || '(root)'}: ${issue.message}`); - } + printIssues(issues); failed++; } @@ -52,7 +59,7 @@ for (const ref of registry.references) { continue; } const seed = [ref.work_key, ref.citation_system_key, ref.locator].join('\n'); - const expected = `https://textrefs.org/id/ref/${uuidv5(seed, REFERENCE_NS)}`; + const expected = refIri(uuidv5(seed, REFERENCE_NS)); if (ref.id !== expected) { console.error( `✗ ref/${ref.work_key}/${ref.locator}: UUID not deterministic from seed (got ${ref.id}, expected ${expected})`, @@ -69,7 +76,7 @@ for (const m of registry.mappings) { continue; } const seed = [m.subject, m.relation, m.target.identifier].join('\n'); - const expected = `https://textrefs.org/id/mapping/${uuidv5(seed, MAPPING_NS)}`; + const expected = mappingIri(uuidv5(seed, MAPPING_NS)); if (m.id !== expected) { console.error( `✗ mapping/${m.id}: UUID not deterministic from seed (expected ${expected})`, @@ -113,7 +120,156 @@ if (unmapped.size > 0) { failed += unmapped.size; } +// --- Frozen identifiers ------------------------------------------------------ +// +// The recomputation above proves the compiler agrees with a second +// implementation of the ADR-0002 seed rule. It does not prove either one still +// produces the identifiers the registry published, because both are code: edit +// the seed rule in `scripts/compile.ts` and in this file — the natural move +// when the check goes red — and every assertion passes while all 86k +// identifiers silently re-mint. +// +// These are data, so no edit to any implementation can satisfy them. Each pair +// was taken from the v0.1.0 baseline: one reference per citation system, and +// one mapping per relation, because the relation is part of the mapping seed. +// +// A failure here is not a test to fix. It means the identifiers the registry +// promised have moved, and a published citation has stopped resolving. ADR-0003 +// allows that for a `draft` record; after the v0.1.0 tag it is a breaking +// change, and the tombstone rules in specification §12 apply. +const FROZEN_REFERENCE_IDS: ReadonlyArray<[string, string, string, string]> = [ + [ + 'confucius.analects', + 'analects-book-chapter', + '11.3', + '00e5c911-4eb8-5309-b38e-a409d8bf3c2b', + ], + [ + 'aristotle.nicomachean-ethics', + 'bekker', + '1103b26', + '000112c4-c060-5823-8fe9-c4632d7c9b2b', + ], + [ + 'new-testament', + 'bible-book-chapter-verse', + 'Luke.1.77', + '0001f355-684b-5c1d-978d-351ed95824c0', + ], + [ + 'dante.commedia', + 'dante-cantica-canto-verse', + 'Purg.9.41', + '00084ac1-eb8c-50c5-abe7-d90e44909314', + ], + [ + 'laozi.daodejing', + 'daodejing-chapter', + '61', + '024cf6b6-b1d0-5b90-b0fa-5b99ce612a35', + ], + [ + 'dhammapada', + 'dhammapada-chapter-verse', + '14.1', + '003e3b7a-8893-5917-a731-62a2fa02621b', + ], + [ + 'murasaki-shikibu.genji', + 'genji-chapter', + '47', + '013ade7c-241d-5685-9c7f-97ae56a860ee', + ], + [ + 'homer.iliad', + 'homer-book-line', + '7.18', + '0002587e-7044-516a-9825-f740950b9663', + ], + [ + 'hume.treatise', + 'hume-book-part-section-paragraph', + '2.2.10.4', + '001ef489-c761-53b1-8a07-ce51c95ea4c1', + ], + [ + 'hume.enquiry-human-understanding', + 'hume-section-paragraph', + '1.13', + '0053bd60-ae8c-59c6-bf37-fbf44241b0a8', + ], + [ + 'wittgenstein.philosophical-investigations', + 'integer-section', + '76', + '00164dc6-f185-5081-b9e4-84d59a9ab0e1', + ], + [ + 'plato.republic', + 'stephanus', + '509b', + '00182f77-1597-5f2f-b215-59af4c645640', + ], + [ + 'wittgenstein.tractatus', + 'tractatus-proposition', + '2.0231', + '007377e5-fe05-545a-a4ff-24ee9f016aa9', + ], +]; + +const FROZEN_MAPPING_IDS: ReadonlyArray<[string, string, string, string]> = [ + [ + 'homer.iliad', + 'alternateOf', + 'https://www.wikidata.org/entity/Q8275', + '158a91fb-ac10-5c1c-b048-1fca5d421289', + ], + [ + 'homer.iliad', + 'isReferencedBy', + 'https://en.wikipedia.org/wiki/Iliad', + '14fe951b-bb68-5b82-943a-ca122b2bb61b', + ], +]; + +const refByTuple = new Map( + registry.references.map((r) => [ + [r.work_key, r.citation_system_key, r.locator].join('\n'), + r.id, + ]), +); +for (const [work, system, locator, uuid] of FROZEN_REFERENCE_IDS) { + checked++; + const actual = refByTuple.get([work, system, locator].join('\n')); + const expected = refIri(uuid); + if (actual === undefined) { + console.error( + `✗ frozen ${work}/${system}/${locator}: no such reference in the registry`, + ); + failed++; + } else if (actual !== expected) { + console.error( + `✗ frozen ${work}/${system}/${locator}: identifier moved\n was ${expected}\n compiled ${actual}`, + ); + failed++; + } +} + +const mappingById = new Map(registry.mappings.map((m) => [m.id, m])); +for (const [work, relation, identifier, uuid] of FROZEN_MAPPING_IDS) { + checked++; + const expected = mappingIri(uuid); + const found = mappingById.get(expected); + if (!found) { + console.error( + `✗ frozen mapping ${work} ${relation} ${identifier}: ${expected} is no longer minted`, + ); + failed++; + } +} + console.log( - `\n${checked - failed}/${checked} records valid (works=${registry.works.length}, systems=${registry.systems.length}, refs=${registry.references.length}, mappings=${registry.mappings.length})`, + `\n${checked - failed}/${checked} records valid (works=${registry.works.length}, systems=${registry.systems.length}, refs=${registry.references.length}, mappings=${registry.mappings.length}); ${FROZEN_REFERENCE_IDS.length + FROZEN_MAPPING_IDS.length} frozen identifiers unchanged`, ); if (failed > 0) process.exit(1); diff --git a/src/lib/find.ts b/src/lib/find.ts index e45bf2b..b8e8b53 100644 --- a/src/lib/find.ts +++ b/src/lib/find.ts @@ -29,6 +29,9 @@ import type { WorkAliasIndex } from './alias-index.js'; import { byKey } from './collection.js'; +// `standard/iri.js` is dependency-free by design, so importing it here does not +// breach the rule stated above about what may travel into the browser bundle. +import { refIri } from '../../standard/iri.js'; export type FindCreator = | { kind: 'person'; family: string; given?: string } @@ -518,5 +521,5 @@ export function resolveInIndex( /** The canonical TextRefs URI of a reference. */ export function referenceIri(uuid: string): string { - return `https://textrefs.org/id/ref/${uuid}`; + return refIri(uuid); } diff --git a/standard/iri.ts b/standard/iri.ts new file mode 100644 index 0000000..dabaff0 --- /dev/null +++ b/standard/iri.ts @@ -0,0 +1,38 @@ +// How a TextRefs IRI is spelled, in one place. +// +// The four record types put their key or their UUID after a fixed prefix, and +// that prefix was written out at ten call sites across `scripts/compile.ts`, +// `scripts/validate-data.ts` and `src/lib/find.ts` before this module existed. +// A prefix is not interesting enough to get wrong twice. +// +// This module MUST stay dependency-free. `src/lib/find.ts` ships to the browser +// inside the `/find/` bundle, so anything imported here travels with it; that +// is the same rule its own header states about `standard/schema/`, which pulls +// in Zod. Nothing below imports anything. +// +// The canonical form is fixed by the specification and by the `id` regexes in +// `standard/schema/`. Those regexes stay written out: a pattern is checked +// against a string, and deriving one from the other would make each half prove +// the other rather than prove the spelling. + +export const BASE = 'https://textrefs.org'; + +/** `https://textrefs.org/id/work/{key}` */ +export function workIri(key: string): string { + return `${BASE}/id/work/${key}`; +} + +/** `https://textrefs.org/id/system/{key}` */ +export function systemIri(key: string): string { + return `${BASE}/id/system/${key}`; +} + +/** `https://textrefs.org/id/ref/{uuid}` */ +export function refIri(uuid: string): string { + return `${BASE}/id/ref/${uuid}`; +} + +/** `https://textrefs.org/id/mapping/{uuid}` */ +export function mappingIri(uuid: string): string { + return `${BASE}/id/mapping/${uuid}`; +}