From 238b7a9b008ae6fc1e1c574ace36e411ec05a006 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 10 Oct 2025 13:34:55 +0200 Subject: [PATCH 001/105] refactor(Match): Clean up constructor --- src/string-searchers/match.ts | 2 -- 1 file changed, 2 deletions(-) diff --git a/src/string-searchers/match.ts b/src/string-searchers/match.ts index e504481..3ca865d 100644 --- a/src/string-searchers/match.ts +++ b/src/string-searchers/match.ts @@ -11,7 +11,5 @@ export class Match { public readonly index: number, public quality: number ) { - this.index = index; - this.quality = quality; } } From 51021615498aa78931be921265c1fec97eb79faf Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 13 Oct 2025 08:58:29 +0200 Subject: [PATCH 002/105] feat: suffix array (first version) --- src/suffix-array-searchers/suffix-array.ts | 142 +++++++++++++++++++++ 1 file changed, 142 insertions(+) create mode 100644 src/suffix-array-searchers/suffix-array.ts diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts new file mode 100644 index 0000000..7beff27 --- /dev/null +++ b/src/suffix-array-searchers/suffix-array.ts @@ -0,0 +1,142 @@ +export class SuffixArray { + + private readonly eoc: number = Number.MAX_SAFE_INTEGER; + private m_str: string; + private m_sa: Int32Array; + private m_isa: Int32Array; + private m_chainHeadsDict: Map; + private m_chainStack: Chain[] = []; + private m_subChains: Chain[] = []; + private m_nextRank: number = 1; + + public constructor(private readonly str: string) { + const l = str.length; + this.m_str = str; + this.m_sa = new Int32Array(l); + this.m_isa = new Int32Array(l); + this.m_chainHeadsDict = new Map(); + + this.FormInitialChains(); + this.BuildSufixArray(); + } + + private FormInitialChains(): void { + this.FindInitialChains(); + this.SortAndPushSubchains(); + } + + private FindInitialChains(): void { + + for (let i = 0; i < this.m_str.length; i++) { + const char_code = this.m_str.charCodeAt(i); + const chain_head_index = this.m_chainHeadsDict.get(char_code); + if (chain_head_index !== undefined) { + this.m_isa[i] = chain_head_index; + } + else { + this.m_isa[i] = this.eoc; + } + this.m_chainHeadsDict.set(char_code, i); + } + + for (const headIndex of this.m_chainHeadsDict.values()) { + const newChain = new Chain(this.m_str, headIndex, 1); + this.m_subChains.push(newChain); + } + } + + private BuildSufixArray(): void { + while (this.m_chainStack.length > 0) { + const chain: Chain = this.m_chainStack.pop() as Chain; + + if (this.m_isa[chain.head] === this.eoc) { + this.RankSuffix(chain.head); + } + else { + this.RefineChainWithInductionSorting(chain); + } + } + } + + private RankSuffix(index: number): void { + this.m_isa[index] = -this.m_nextRank; + this.m_sa[this.m_nextRank - 1] = index; + this.m_nextRank++; + } + + private RefineChainWithInductionSorting(chain: Chain): void { + const notedSuffixes: SuffixRank[] = []; + this.m_chainHeadsDict.clear(); + this.m_subChains = []; + + while (chain.head !== this.eoc) { + const nextIndex: number = this.m_isa[chain.head]; + if (chain.head + chain.length > this.m_str.length - 1) { + this.RankSuffix(chain.head); + } + else if (this.m_isa[chain.head + chain.length] < 0) { + const sr: SuffixRank = new SuffixRank(chain.head, -this.m_isa[chain.head + chain.length]); + notedSuffixes.push(sr); + } + else { + this.ExtendChain(chain); + } + chain.head = nextIndex; + } + + this.SortAndPushSubchains(); + this.SortAndRankNotedSuffixes(notedSuffixes); + } + + private ExtendChain(chain: Chain): void { + const sym: number = this.m_str.charCodeAt(chain.head + chain.length); + if (this.m_chainHeadsDict.has(sym)) { + this.m_isa[this.m_chainHeadsDict.get(sym) as number] = chain.head; + this.m_isa[chain.head] = this.eoc; + } + else { + this.m_isa[chain.head] = this.eoc; + const newChain: Chain = new Chain(this.m_str, chain.head, chain.length + 1); + this.m_subChains.push(newChain); + } + + this.m_chainHeadsDict.set(sym, chain.head); + } + + private SortAndRankNotedSuffixes(notedSuffixes: SuffixRank[]): void { + notedSuffixes.sort((a, b) => { + return a.rank - b.rank; + }); + + for (let i = 0; i < notedSuffixes.length; i++) { + this.RankSuffix(notedSuffixes[i].head); + } + } + + private SortAndPushSubchains(): void { + this.m_subChains.sort(); + for (let i = this.m_subChains.length - 1; i >= 0; i--) { + this.m_chainStack.push(this.m_subChains[i]); + } + } +} + +class SuffixRank { + public constructor( + public readonly head: number, + public readonly rank: number + ) { } +} + +class Chain { + + public readonly m_str: string; + public head: number; + public length: number; + + public constructor(m_str: string, head: number, length: number) { + this.m_str = m_str; + this.head = head; + this.length = length; + } +} \ No newline at end of file From cafe2a7ca04ace60ec240063068bcc5537439a0e Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 14 Oct 2025 11:16:50 +0200 Subject: [PATCH 003/105] feat: suffix array --- src/suffix-array-searchers/index.ts | 2 + .../string-comparison.ts | 41 +++++++++++++++++++ .../suffix-array.test.ts | 28 +++++++++++++ src/suffix-array-searchers/suffix-array.ts | 25 ++++++++--- 4 files changed, 90 insertions(+), 6 deletions(-) create mode 100644 src/suffix-array-searchers/index.ts create mode 100644 src/suffix-array-searchers/string-comparison.ts create mode 100644 src/suffix-array-searchers/suffix-array.test.ts diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts new file mode 100644 index 0000000..d207668 --- /dev/null +++ b/src/suffix-array-searchers/index.ts @@ -0,0 +1,2 @@ +export { StringComparison } from './string-comparison.js'; +export { SuffixArray } from './suffix-array.js'; \ No newline at end of file diff --git a/src/suffix-array-searchers/string-comparison.ts b/src/suffix-array-searchers/string-comparison.ts new file mode 100644 index 0000000..ee05abc --- /dev/null +++ b/src/suffix-array-searchers/string-comparison.ts @@ -0,0 +1,41 @@ + +export class StringComparison { + + public static compareSubstringsOrdinal( + strA: string, + indexA: number, + strB: string, + indexB: number, + length: number + ): number { + + const endA = Math.min(indexA + length, strA.length); + const endB = Math.min(indexB + length, strB.length); + + let iA = indexA; + let iB = indexB; + + while (iA < endA && iB < endB) { + const codeA = strA.charCodeAt(iA); + const codeB = strB.charCodeAt(iB); + + if (codeA < codeB) { + return -1; + } + if (codeA > codeB) { + return 1; + } + + iA++; + iB++; + } + + const lenComparedA = endA - indexA; + const lenComparedB = endB - indexB; + + if (lenComparedA === lenComparedB) { + return 0; + } + return lenComparedA < lenComparedB ? -1 : 1; + } +} diff --git a/src/suffix-array-searchers/suffix-array.test.ts b/src/suffix-array-searchers/suffix-array.test.ts new file mode 100644 index 0000000..1a1d318 --- /dev/null +++ b/src/suffix-array-searchers/suffix-array.test.ts @@ -0,0 +1,28 @@ +import { SuffixArray } from './suffix-array.js'; + +test('can create suffix array test 1', () => { + const result = SuffixArray.Create('banana$'); + expect(result).toEqual(new Int32Array([6, 5, 3, 1, 0, 4, 2])); +}); + +test('can create suffix array test 2', () => { + const result = SuffixArray.Create('mississippi'); + expect(result).toEqual(new Int32Array([10, 7, 4, 1, 0, 9, 8, 6, 3, 5, 2])); +}); + +test('can create suffix array of empty string', () => { + const result = SuffixArray.Create(''); + expect(result).toEqual(new Int32Array([])); +}); + +test('can not create suffix array of null', () => { + expect(() => { + SuffixArray.Create(null!); + }).toThrow(); +}); + +test('can not create suffix array of undefined', () => { + expect(() => { + SuffixArray.Create(undefined!); + }).toThrow(); +}); \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts index 7beff27..3e8d92d 100644 --- a/src/suffix-array-searchers/suffix-array.ts +++ b/src/suffix-array-searchers/suffix-array.ts @@ -1,6 +1,8 @@ +import { StringComparison } from './string-comparison.js'; + export class SuffixArray { - private readonly eoc: number = Number.MAX_SAFE_INTEGER; + private readonly eoc: number = 2147483647; private m_str: string; private m_sa: Int32Array; private m_isa: Int32Array; @@ -9,15 +11,22 @@ export class SuffixArray { private m_subChains: Chain[] = []; private m_nextRank: number = 1; - public constructor(private readonly str: string) { + public static Create(str: string): Int32Array { + if (str == null) { + throw new Error('Input string cannot be null.'); + } + const suffixArray: SuffixArray = new SuffixArray(str); + suffixArray.FormInitialChains(); + suffixArray.BuildSufixArray(); + return suffixArray.m_sa; + } + + private constructor(private readonly str: string) { const l = str.length; this.m_str = str; this.m_sa = new Int32Array(l); this.m_isa = new Int32Array(l); this.m_chainHeadsDict = new Map(); - - this.FormInitialChains(); - this.BuildSufixArray(); } private FormInitialChains(): void { @@ -114,7 +123,11 @@ export class SuffixArray { } private SortAndPushSubchains(): void { - this.m_subChains.sort(); + this.m_subChains.sort((c1: Chain, c2: Chain): number => { + const len = Math.min(c1.length, c2.length); + return StringComparison.compareSubstringsOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); + + }); for (let i = this.m_subChains.length - 1; i >= 0; i--) { this.m_chainStack.push(this.m_subChains[i]); } From c9982bc1c2650ebed1686bf85c3b4e41722b7bf9 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 14 Oct 2025 11:35:08 +0200 Subject: [PATCH 004/105] wip: suffix array searcher --- src/string-searchers/normalizing-searcher.ts | 2 +- src/suffix-array-searchers/index.ts | 3 ++- .../suffix-array-searcher.ts | 18 ++++++++++++++++++ 3 files changed, 21 insertions(+), 2 deletions(-) create mode 100644 src/suffix-array-searchers/suffix-array-searcher.ts diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index 1f015b4..efdcd93 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -17,7 +17,7 @@ export class NormalizingSearcher implements StringSearcher { public constructor( private readonly stringSearcher: StringSearcher, private readonly normalizer: Normalizer - ) {} + ) { } /** * {@inheritDoc StringSearcher.index} diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index d207668..5bdb7cf 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -1,2 +1,3 @@ export { StringComparison } from './string-comparison.js'; -export { SuffixArray } from './suffix-array.js'; \ No newline at end of file +export { SuffixArray } from './suffix-array.js'; +export { SuffixArraySearcher } from './suffix-array-searcher.js'; \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts new file mode 100644 index 0000000..648570c --- /dev/null +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -0,0 +1,18 @@ +import { Memento, Meta, Query, Result } from '../fuzzy-search.js'; +import { StringSearcher } from '../interfaces/string-searcher.js'; + +export class SuffixArraySearcher implements StringSearcher { + index(terms: string[]): Meta { + throw new Error('Method not implemented.'); + } + getMatches(query: Query): Result { + throw new Error('Method not implemented.'); + } + save(memento: Memento): void { + throw new Error('Method not implemented.'); + } + load(memento: Memento): void { + throw new Error('Method not implemented.'); + } + +} \ No newline at end of file From 61e5675f18c6771fc512779315487b8846ef7681 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 14 Oct 2025 11:41:03 +0200 Subject: [PATCH 005/105] refactor: meta merger --- src/commons/index.ts | 1 + src/commons/meta-merger.ts | 37 ++++++++++++++++++++++++ src/dynamic-searchers/result-merger.ts | 40 +++----------------------- 3 files changed, 42 insertions(+), 36 deletions(-) create mode 100644 src/commons/meta-merger.ts diff --git a/src/commons/index.ts b/src/commons/index.ts index dfff72c..6e1b7aa 100644 --- a/src/commons/index.ts +++ b/src/commons/index.ts @@ -1,3 +1,4 @@ export { ArrayUtilities } from './array-utilities.js'; export { HashUtilities } from './hash-utilities.js'; +export { MetaMerger } from './meta-merger.js'; export { StringUtilities } from './string-utilities.js'; diff --git a/src/commons/meta-merger.ts b/src/commons/meta-merger.ts new file mode 100644 index 0000000..1859fc3 --- /dev/null +++ b/src/commons/meta-merger.ts @@ -0,0 +1,37 @@ +/* eslint-disable @typescript-eslint/no-explicit-any */ + +import { Meta } from '../interfaces/meta.js'; + +export class MetaMerger { + + /** + * Merges two {@link Meta} objects into a new {@link Meta} object. + * @param meta1 The first meta object. + * @param meta2 The second meta object. + * @returns The merged meta object. + */ + public static mergeMeta(meta1: Meta, meta2: Meta): Meta { + const newMetaEntries: Map = new Map(); + + for (const [key, value] of meta1.allEntries) { + newMetaEntries.set(key, value); + } + + for (const [key, value] of meta2.allEntries) { + const presentValue = newMetaEntries.get(key); + if (presentValue === undefined) { + newMetaEntries.set(key, value); + continue; + } + if (typeof presentValue === 'number' && typeof value === 'number') { + newMetaEntries.set(key, presentValue + value); + continue; + } + newMetaEntries.delete(key); + newMetaEntries.set(`${key}_0`, presentValue); + newMetaEntries.set(`${key}_1`, value); + } + + return new Meta(newMetaEntries); + } +} \ No newline at end of file diff --git a/src/dynamic-searchers/result-merger.ts b/src/dynamic-searchers/result-merger.ts index 9b9ffae..25a6d2d 100644 --- a/src/dynamic-searchers/result-merger.ts +++ b/src/dynamic-searchers/result-merger.ts @@ -1,8 +1,7 @@ -/* eslint-disable @typescript-eslint/no-explicit-any */ - import { EntityMatch } from '../interfaces/entity-match.js'; import { EntityResult } from '../interfaces/entity-result.js'; import { Meta } from '../interfaces/meta.js'; +import { MetaMerger } from '../commons/meta-merger.js'; import { Query } from '../interfaces/query.js'; /** @@ -22,7 +21,7 @@ export class ResultMerger { ): EntityResult { const query: Query = result1.query; const newMatches = this.mergeMatches(result1.matches, result2.matches, query.topN); - const newMeta: Meta = this.mergeMeta(result1.meta, result2.meta); + const newMeta: Meta = MetaMerger.mergeMeta(result1.meta, result2.meta); return new EntityResult(newMatches, query, newMeta); } @@ -48,40 +47,9 @@ export class ResultMerger { const newMatches = [...matches1, ...matches2]; newMatches.sort((m1, m2) => m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : 0 ); return newMatches.length <= topN ? newMatches : newMatches.slice(0, topN); } - - /** - * Merges two {@link Meta} objects into a new {@link Meta} object. - * @param meta1 The first meta object. - * @param meta2 The second meta object. - * @returns The merged meta object. - */ - private static mergeMeta(meta1: Meta, meta2: Meta): Meta { - const newMetaEntries: Map = new Map(); - - for (const [key, value] of meta1.allEntries) { - newMetaEntries.set(key, value); - } - - for (const [key, value] of meta2.allEntries) { - const presentValue = newMetaEntries.get(key); - if (presentValue === undefined) { - newMetaEntries.set(key, value); - continue; - } - if (typeof presentValue === 'number' && typeof value === 'number') { - newMetaEntries.set(key, presentValue + value); - continue; - } - newMetaEntries.delete(key); - newMetaEntries.set(`${key}_0`, presentValue); - newMetaEntries.set(`${key}_1`, value); - } - - return new Meta(newMetaEntries); - } } From 1a5a4834460f35b6e5099a45a4c235e6bbd9915d Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 15 Oct 2025 11:28:01 +0200 Subject: [PATCH 006/105] feat: suffix array searcher --- src/interfaces/query.ts | 6 +- .../string-comparison.ts | 2 +- .../suffix-array-searcher.ts | 108 +++++++++++++++++- src/suffix-array-searchers/suffix-array.ts | 4 +- 4 files changed, 112 insertions(+), 8 deletions(-) diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index 65b798b..2bfa1af 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -13,8 +13,8 @@ export class Query { public readonly topN: number; /** - * The minimum quality of matches to return. Increasing this value will increase the performance but make - * the searcher less fuzzy. The value must be between 0 and 1. + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * number of matches. The value must be between 0 and 1. */ public readonly minQuality: number; @@ -23,7 +23,7 @@ export class Query { * @param string The query string. * @param topN The maximum number of matches to return. Provide Infinity to return all matches. * @param minQuality The minimum quality of matches to return. Increasing this value will increase the - * performance but make the searcher less fuzzy. The value must be between 0 and 1; lower or larger values will be + * performance but reduce the number of matches. The value must be between 0 and 1; lower or larger values will be * clamped. */ public constructor(string: string, topN: number = 10, minQuality: number = 0.3) { diff --git a/src/suffix-array-searchers/string-comparison.ts b/src/suffix-array-searchers/string-comparison.ts index ee05abc..243b4a4 100644 --- a/src/suffix-array-searchers/string-comparison.ts +++ b/src/suffix-array-searchers/string-comparison.ts @@ -1,7 +1,7 @@ export class StringComparison { - public static compareSubstringsOrdinal( + public static compareOrdinal( strA: string, indexA: number, strB: string, diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 648570c..fab7651 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -1,13 +1,117 @@ import { Memento, Meta, Query, Result } from '../fuzzy-search.js'; +import { Match } from '../string-searchers/match.js'; +import { StringComparison } from './string-comparison.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; +import { SuffixArray } from './suffix-array.js'; + +// todo: make sure the terms don't have the separating character. Move to a suffix array config. export class SuffixArraySearcher implements StringSearcher { + + private readonly separator = 'µ'; + private str: string; + private suffixArray: Int32Array; + private indexToTermIndex: Int32Array; + private termLengths: Int32Array; + + public constructor() { + this.str = ''; + this.suffixArray = new Int32Array(0); + this.indexToTermIndex = new Int32Array(0); + this.termLengths = new Int32Array(0); + } + index(terms: string[]): Meta { - throw new Error('Method not implemented.'); + const start = performance.now(); + this.str = this.separator + terms.join(this.separator) + this.separator; + this.suffixArray = SuffixArray.create(this.str); + this.indexToTermIndex = new Int32Array(this.suffixArray.length); + this.termLengths = new Int32Array(terms.length); + + let i = 0; + for (let j = 0; j < terms.length; j++) { + this.termLengths[j] = terms[j].length; + for (let k = 0; k <= terms[j].length; k++) { + this.indexToTermIndex[i++] = j; + } + } + + this.indexToTermIndex[i++] = -1; + const duration = Math.round(performance.now() - start); + + const meta = new Meta(); + meta.add('suffixArraySearcherIndexing', duration); + return meta; } + getMatches(query: Query): Result { - throw new Error('Method not implemented.'); + if (query.string == null || query.string === '') { + return new Result([], query, new Meta()); + } + + // todo prefix: pass query.string modified + const [start, end] = this.GetPositionsInSuffixArray(query.string); + const matchedTermIds = new Int32Array(end - start); + + let i = 0; + for (let j = start; j < end; j++) { + const termIndex = this.indexToTermIndex[this.suffixArray[j]]; + matchedTermIds[i++] = termIndex; + } + + const matches: Match[] = []; + + let quality = 0; + for (let k = 0; k < matchedTermIds.length; k++) { + quality = this.computeQuality(query.string.length, this.termLengths[matchedTermIds[k]]); + if (quality > query.minQuality) { + matches.push(new Match(matchedTermIds[k], quality)); + } + } + + // todo: remove duplicate matches, measure performance + return new Result(matches, query, new Meta()); + } + + private computeQuality(queryLength: number, termLength: number): number { + return queryLength / termLength; } + + private GetPositionsInSuffixArray(substring: string): number[] { + let left = 0; + let right = this.suffixArray.length; + let middle = 0; + + while (left < right) { + middle = Math.floor((left + right) / 2); + + if (StringComparison.compareOrdinal( + this.str, this.suffixArray[middle], substring, 0, substring.length) < 0) { + left = middle + 1; + } + else { + right = middle; + } + } + + const start = left; + right = this.suffixArray.length; + + while (left < right) { + middle = Math.floor((left + right) / 2); + if (StringComparison.compareOrdinal( + this.str, this.suffixArray[middle], substring, 0, substring.length) <= 0) { + left = middle + 1; + } + else { + right = middle; + } + } + + return [start, right]; + } + + save(memento: Memento): void { throw new Error('Method not implemented.'); } diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts index 3e8d92d..6813060 100644 --- a/src/suffix-array-searchers/suffix-array.ts +++ b/src/suffix-array-searchers/suffix-array.ts @@ -11,7 +11,7 @@ export class SuffixArray { private m_subChains: Chain[] = []; private m_nextRank: number = 1; - public static Create(str: string): Int32Array { + public static create(str: string): Int32Array { if (str == null) { throw new Error('Input string cannot be null.'); } @@ -125,7 +125,7 @@ export class SuffixArray { private SortAndPushSubchains(): void { this.m_subChains.sort((c1: Chain, c2: Chain): number => { const len = Math.min(c1.length, c2.length); - return StringComparison.compareSubstringsOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); + return StringComparison.compareOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); }); for (let i = this.m_subChains.length - 1; i >= 0; i--) { From 82b8e6fa9d27b536368ef4caf36efbf4b30b6f5c Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 15 Oct 2025 17:34:23 +0200 Subject: [PATCH 007/105] test: suffix array searcher --- src/suffix-array-searchers/suffix-array-searcher.ts | 1 + src/suffix-array-searchers/suffix-array.test.ts | 10 +++++----- 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index fab7651..c044ccf 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -51,6 +51,7 @@ export class SuffixArraySearcher implements StringSearcher { // todo prefix: pass query.string modified const [start, end] = this.GetPositionsInSuffixArray(query.string); + // todo: refactor such that end is included. const matchedTermIds = new Int32Array(end - start); let i = 0; diff --git a/src/suffix-array-searchers/suffix-array.test.ts b/src/suffix-array-searchers/suffix-array.test.ts index 1a1d318..a5c3798 100644 --- a/src/suffix-array-searchers/suffix-array.test.ts +++ b/src/suffix-array-searchers/suffix-array.test.ts @@ -1,28 +1,28 @@ import { SuffixArray } from './suffix-array.js'; test('can create suffix array test 1', () => { - const result = SuffixArray.Create('banana$'); + const result = SuffixArray.create('banana$'); expect(result).toEqual(new Int32Array([6, 5, 3, 1, 0, 4, 2])); }); test('can create suffix array test 2', () => { - const result = SuffixArray.Create('mississippi'); + const result = SuffixArray.create('mississippi'); expect(result).toEqual(new Int32Array([10, 7, 4, 1, 0, 9, 8, 6, 3, 5, 2])); }); test('can create suffix array of empty string', () => { - const result = SuffixArray.Create(''); + const result = SuffixArray.create(''); expect(result).toEqual(new Int32Array([])); }); test('can not create suffix array of null', () => { expect(() => { - SuffixArray.Create(null!); + SuffixArray.create(null!); }).toThrow(); }); test('can not create suffix array of undefined', () => { expect(() => { - SuffixArray.Create(undefined!); + SuffixArray.create(undefined!); }).toThrow(); }); \ No newline at end of file From a26ee1f3b8f60b301e4e21bf075464255707105e Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 15 Oct 2025 17:37:03 +0200 Subject: [PATCH 008/105] more tests --- .../suffix-array-searcher.test.ts | 69 +++++++++++++++++++ 1 file changed, 69 insertions(+) create mode 100644 src/suffix-array-searchers/suffix-array-searcher.test.ts diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts new file mode 100644 index 0000000..0fda005 --- /dev/null +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -0,0 +1,69 @@ +import { Match } from '../string-searchers/match.js'; +import { Query } from '../interfaces/query.js'; +import { SuffixArraySearcher } from './suffix-array-searcher.js'; + +const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher(); +suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); + +function getMatches(queryString: string): Match[] { + const query = new Query(queryString, 10, 0); + const matches = suffixArraySearcher.getMatches(query).matches; + matches.sort((m1, m2) => m1.index - m2.index); + return matches; +} + +test('can find exact match test 1', () => { + expect(getMatches('Alice')).toEqual([new Match(0, 1)]); +}); + +test('can find exact match test 2', () => { + expect(getMatches('Bob')).toEqual([new Match(1, 1)]); +}); + +test('can find prefix matches test 1', () => { + expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); +}); + +test('can find prefix matches test 2', () => { + expect(getMatches('Cha')).toEqual([new Match(4, 3 / 7)]); +}); + +test('can find prefix matches test 3', () => { + expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); +}); + +test('can find suffix matches test 1', () => { + expect(getMatches('lice')).toEqual([new Match(0, 4 / 5)]); +}); + +test('can find suffix matches test 2', () => { + expect(getMatches('ol')).toEqual([new Match(3, 2 / 5)]); +}); + +test('can find infix matches test 1', () => { + expect(getMatches('arl')).toEqual([new Match(2, 3 / 6), new Match(4, 3 / 7)]); +}); + +test('can find infix matches test 2', () => { + expect(getMatches('li')).toEqual( + [new Match(0, 2 / 5), new Match(4, 2 / 7)]); +}); + +test('can find infix matches test 3', () => { + expect(getMatches('l')).toEqual( + [new Match(0, 1 / 5), new Match(2, 1 / 6), new Match(3, 1 / 5), new Match(4, 1 / 7)]); +}); + +test('empty query returns no matches', () => { + expect(getMatches('')).toEqual([]); +}); + +test('null query returns no matches', () => { + expect(getMatches(null!)).toEqual([]); +}); + +test('undefined query returns no matches', () => { + expect(getMatches(undefined!)).toEqual([]); +}); + + From 8f6430e970efae9423d4999e1a7feb73be0abd0f Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 15 Oct 2025 20:48:30 +0200 Subject: [PATCH 009/105] fix --- .../suffix-array-searcher.ts | 32 +++++++++---------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index c044ccf..fb4f75f 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -79,37 +79,37 @@ export class SuffixArraySearcher implements StringSearcher { } private GetPositionsInSuffixArray(substring: string): number[] { - let left = 0; - let right = this.suffixArray.length; - let middle = 0; + let l = 0; + let r = this.suffixArray.length; + let mid = 0; - while (left < right) { - middle = Math.floor((left + right) / 2); + while (l < r) { + mid = Math.floor((l + r) / 2); if (StringComparison.compareOrdinal( - this.str, this.suffixArray[middle], substring, 0, substring.length) < 0) { - left = middle + 1; + substring, 0, this.str, this.suffixArray[mid], substring.length) > 0) { + l = mid + 1; } else { - right = middle; + r = mid; } } - const start = left; - right = this.suffixArray.length; + const start = l; + r = this.suffixArray.length; - while (left < right) { - middle = Math.floor((left + right) / 2); + while (l < r) { + mid = Math.floor((l + r) / 2); if (StringComparison.compareOrdinal( - this.str, this.suffixArray[middle], substring, 0, substring.length) <= 0) { - left = middle + 1; + substring, 0, this.str, this.suffixArray[mid], substring.length) == 0) { + l = mid + 1; } else { - right = middle; + r = mid; } } - return [start, right]; + return [start, r]; } From 838c180a8122b9a6c426e737384ce1e9060c776d Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 16 Oct 2025 11:57:45 +0200 Subject: [PATCH 010/105] feat: regression test --- src/regression-test/main.ts | 88 +++++++++++++++++++++++++++++++++++++ 1 file changed, 88 insertions(+) create mode 100644 src/regression-test/main.ts diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts new file mode 100644 index 0000000..5e5939d --- /dev/null +++ b/src/regression-test/main.ts @@ -0,0 +1,88 @@ +/* + Regression tests. Run this file and check the git diff of the output files to see what changed. +*/ + +import { readFileSync, writeFileSync } from 'fs'; +import { EntityMatch } from '../interfaces/entity-match.js'; +import { EntityResult } from '../interfaces/entity-result.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; +import { SearcherFactory } from '../searcher-factory.js'; + +interface GeoEntity { + id: number; + name: string; +} + +const outputPath = './src/regression-test/output'; +const outputColumnWidth = 40; +const text = readFileSync('./data/world-ctvs.txt', 'utf-8'); +const lines = text.split('\n').slice(1); +const entities = lines.map((l, index) => ({ id: index, name: l })); + +const searcher = SearcherFactory.createDefaultSearcher(); + +console.log(`Indexing ${entities.length} entities...`); +const indexingMeta: Meta = searcher.indexEntities( + entities, + (e) => e.id, + (e) => e.name.split(';') +); + +writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta)); + +console.log("Running queries..."); + +runQuery('carcassonne-prefix', 'carcasso'); +runQuery('carcassonne-infix', 'cassonn'); +runQuery('carcassonne-suffix', 'sonne'); +runQuery('munich-insertion', 'muniich'); +runQuery('boston-deletion', 'bostn') +runQuery('boulder-creek-substitution', 'boulder creak'); +runQuery('tübingen-transposition', 'tübignen'); +runQuery('tokyo', '東京都'); +runQuery('tokyo-prefix', '東京'); +runQuery('tbilisi', 'თბილისი'); +runQuery('tbilisi-deletion', 'თბიისი'); +runQuery('kuwait-city', 'مدينة الكويت'); +runQuery('kuwait-city-prefix', 'مدينة الك'); + +console.log("Finished.") + +function runQuery(queryName: string, queryString: string) { + const query: Query = new Query(queryString); + const result: EntityResult = searcher.getMatches(query); + const queryJson = JSON.stringify(query, null, 2); + const metaJson = metaToJson(result.meta); + const matchesString = matchesToString(result.matches); + const output = `${queryJson}\n\n${metaJson}\n\n${matchesString}`; + writeFileSync(`${outputPath}/${queryName}.txt`, output); +} + +function metaToJson(meta: Meta): string { + return JSON.stringify(Object.fromEntries(meta.allEntries), null, 2); +} + +function matchesToString(matches: EntityMatch[]): string { + const header = + padRight("Rank", 8) + + padRight("Entity", outputColumnWidth) + + padRight("Matched String", outputColumnWidth) + + padRight("Quality", 8) + + "\n\n"; + const matchesString = matches.map((m, i) => matchToString(m, i + 1)).join('\n'); + return header + matchesString; +} + +function matchToString(match: EntityMatch, rank: number): string { + return ( + padRight(rank.toString(), 8) + + padRight(match.entity.name, outputColumnWidth) + + padRight(match.matchedString, outputColumnWidth) + + padRight(match.quality.toFixed(2), 8) + ); +} + +function padRight(s: string, targetWidth: number): string { + return s + ' '.repeat(Math.max(0, targetWidth - s.length)); +} From 214b8c192cb81420b8e302111de1f57b832958aa Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 16 Oct 2025 11:58:45 +0200 Subject: [PATCH 011/105] work work --- .vscode/launch.json | 24 ++++++++++--- notes.md | 84 +++++++++++++++++++++++++++++++++++++++++++++ src/fuzzy-search.ts | 1 + 3 files changed, 105 insertions(+), 4 deletions(-) create mode 100644 notes.md diff --git a/.vscode/launch.json b/.vscode/launch.json index 77fba74..e796d7a 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -4,10 +4,26 @@ { "type": "node", "request": "launch", - "name": "Launch Program", - "skipFiles": ["/**"], + "name": "Launch Basic Usage", + "skipFiles": [ + "/**" + ], "program": "${workspaceFolder}/dist/basic-usage.js", - "outFiles": ["${workspaceFolder}/**/*.js"] + "outFiles": [ + "${workspaceFolder}/**/*.js" + ] + }, + { + "type": "node", + "request": "launch", + "name": "Run Regression Test", + "skipFiles": [ + "/**" + ], + "program": "${workspaceFolder}/dist/regression-test/main.js", + "outFiles": [ + "${workspaceFolder}/**/*.js" + ] } ] -} +} \ No newline at end of file diff --git a/notes.md b/notes.md new file mode 100644 index 0000000..b654e79 --- /dev/null +++ b/notes.md @@ -0,0 +1,84 @@ +- Implement Suffix Array searcher. + +https://en.wikipedia.org/wiki/Suffix_array + +```python +n = len(S) + +def search(P: str) -> tuple[int, int]: + """ + Return indices (s, r) such that the interval A[s:r] (including the end + index) represents all suffixes of S that start with the pattern P. + """ + # Find starting position of interval + l = 0 # in Python, arrays are indexed starting at 0 + r = n + while l < r: + mid = (l + r) // 2 # division rounding down to nearest integer + # suffixAt(A[i]) is the ith smallest suffix + if P > suffixAt(A[mid]): + l = mid + 1 + else: + r = mid + s = l + + # Find ending position of interval + r = n + while l < r: + mid = (l + r) // 2 + if suffixAt(A[mid]).startswith(P): + l = mid + 1 + else: + r = mid + return (s, r) +``` + +Chat GPT (verify!) + +```js +function compareSubstringsOrdinal(strA, indexA, strB, indexB, length) { + const lenA = strA.length; + const lenB = strB.length; + const endA = Math.min(indexA + length, lenA); + const endB = Math.min(indexB + length, lenB); + + let iA = indexA; + let iB = indexB; + + while (iA < endA && iB < endB) { + const codeA = strA.charCodeAt(iA); + const codeB = strB.charCodeAt(iB); + + if (codeA < codeB) return -1; + if (codeA > codeB) return 1; + + iA++; + iB++; + } + + // If both ran out at the same time, they're equal + const lenComparedA = endA - indexA; + const lenComparedB = endB - indexB; + + if (lenComparedA === lenComparedB) return 0; + return lenComparedA < lenComparedB ? -1 : 1; +} +``` + +Attribution (Chat GPT) + +```text +/* + Original: https://github.com/eranmeir/Sufa-Suffix-Array-Csharp + Copyright (c) 2012 Eran Meir + SPDX-License-Identifier: MIT + + Translation to TypeScript, modifications and refactoring + (c) 2025 Kevin Schaal +*/ +``` + +Todo +==== + +Test empty term. \ No newline at end of file diff --git a/src/fuzzy-search.ts b/src/fuzzy-search.ts index 55b8b6a..06c3f1b 100644 --- a/src/fuzzy-search.ts +++ b/src/fuzzy-search.ts @@ -6,5 +6,6 @@ export * from './interfaces/index.js'; export * from './normalization/index.js'; export * from './performance/index.js'; export * from './string-searchers/index.js'; +export * from './suffix-array-searchers/index.js'; export { Config } from './config.js'; export { SearcherFactory } from './searcher-factory.js'; From c334faa0c10fc13812d0a8754eeadc6a0e2ddbe1 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 16 Oct 2025 12:15:14 +0200 Subject: [PATCH 012/105] fix: dependencies --- src/suffix-array-searchers/suffix-array-searcher.ts | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index fb4f75f..20d3458 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -1,5 +1,8 @@ -import { Memento, Meta, Query, Result } from '../fuzzy-search.js'; import { Match } from '../string-searchers/match.js'; +import { Memento } from '../interfaces/memento.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; +import { Result } from '../string-searchers/result.js'; import { StringComparison } from './string-comparison.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SuffixArray } from './suffix-array.js'; @@ -116,6 +119,7 @@ export class SuffixArraySearcher implements StringSearcher { save(memento: Memento): void { throw new Error('Method not implemented.'); } + load(memento: Memento): void { throw new Error('Method not implemented.'); } From a12b3c60ef85e58e1bd0adb6c77b9b2acca1e66e Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 16 Oct 2025 19:20:21 +0200 Subject: [PATCH 013/105] fixes --- src/regression-test/main.ts | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts index 5e5939d..3401256 100644 --- a/src/regression-test/main.ts +++ b/src/regression-test/main.ts @@ -3,6 +3,7 @@ */ import { readFileSync, writeFileSync } from 'fs'; +import { Config } from '../config.js'; import { EntityMatch } from '../interfaces/entity-match.js'; import { EntityResult } from '../interfaces/entity-result.js'; import { Meta } from '../interfaces/meta.js'; @@ -16,11 +17,13 @@ interface GeoEntity { const outputPath = './src/regression-test/output'; const outputColumnWidth = 40; -const text = readFileSync('./data/world-ctvs.txt', 'utf-8'); +const text = readFileSync('./data/world-ctvs.txt', 'utf8'); const lines = text.split('\n').slice(1); const entities = lines.map((l, index) => ({ id: index, name: l })); -const searcher = SearcherFactory.createDefaultSearcher(); +const config = Config.createDefaultConfig(); +config.normalizerConfig.allowCharacter = (_) => true; +const searcher = SearcherFactory.createSearcher(config); console.log(`Indexing ${entities.length} entities...`); const indexingMeta: Meta = searcher.indexEntities( @@ -52,6 +55,7 @@ console.log("Finished.") function runQuery(queryName: string, queryString: string) { const query: Query = new Query(queryString); const result: EntityResult = searcher.getMatches(query); + console.log(`'${queryString}' (${queryName}): ${result.matches.length} matches.`); const queryJson = JSON.stringify(query, null, 2); const metaJson = metaToJson(result.meta); const matchesString = matchesToString(result.matches); From b2ee929b50f737669b0dc54a27d7c0cb971e66eb Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 17 Oct 2025 10:49:25 +0200 Subject: [PATCH 014/105] fix: regression test --- .vscode/settings.json | 11 +++++-- src/regression-test/main.ts | 66 ++++++++++++++++++++++++------------- 2 files changed, 51 insertions(+), 26 deletions(-) diff --git a/.vscode/settings.json b/.vscode/settings.json index d632a35..e8947c0 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -5,9 +5,14 @@ "editor.formatOnSave": false }, "editor.defaultFormatter": "esbenp.prettier-vscode", - "editor.rulers": [120], + "editor.rulers": [ + 120 + ], "liveServer.settings.port": 5501, "[json]": { "editor.defaultFormatter": "esbenp.prettier-vscode" - } -} + }, + "[plaintext]": { + "editor.renderControlCharacters": false + }, +} \ No newline at end of file diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts index 3401256..ab724d7 100644 --- a/src/regression-test/main.ts +++ b/src/regression-test/main.ts @@ -10,17 +10,20 @@ import { Meta } from '../interfaces/meta.js'; import { Query } from '../interfaces/query.js'; import { SearcherFactory } from '../searcher-factory.js'; -interface GeoEntity { - id: number; - name: string; -} - const outputPath = './src/regression-test/output'; -const outputColumnWidth = 40; +const shortColumnWidth = 8; +const wideColumnWidth = 40; +const RLI = "\u2067", LRI = "\u2066", PDI = "\u2069"; + const text = readFileSync('./data/world-ctvs.txt', 'utf8'); const lines = text.split('\n').slice(1); const entities = lines.map((l, index) => ({ id: index, name: l })); +interface GeoEntity { + id: number; + name: string; +} + const config = Config.createDefaultConfig(); config.normalizerConfig.allowCharacter = (_) => true; const searcher = SearcherFactory.createSearcher(config); @@ -31,8 +34,7 @@ const indexingMeta: Meta = searcher.indexEntities( (e) => e.id, (e) => e.name.split(';') ); - -writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta)); +writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { encoding: 'utf8' }); console.log("Running queries..."); @@ -47,18 +49,18 @@ runQuery('tokyo', '東京都'); runQuery('tokyo-prefix', '東京'); runQuery('tbilisi', 'თბილისი'); runQuery('tbilisi-deletion', 'თბიისი'); -runQuery('kuwait-city', 'مدينة الكويت'); -runQuery('kuwait-city-prefix', 'مدينة الك'); +runQuery('kuwait-city', 'مدينة الكويت', true); +runQuery('kuwait-city-prefix', 'مدينة الك', true); console.log("Finished.") -function runQuery(queryName: string, queryString: string) { +function runQuery(queryName: string, queryString: string, rtl: boolean = false) { const query: Query = new Query(queryString); const result: EntityResult = searcher.getMatches(query); console.log(`'${queryString}' (${queryName}): ${result.matches.length} matches.`); const queryJson = JSON.stringify(query, null, 2); const metaJson = metaToJson(result.meta); - const matchesString = matchesToString(result.matches); + const matchesString = matchesToString(result.matches, rtl); const output = `${queryJson}\n\n${metaJson}\n\n${matchesString}`; writeFileSync(`${outputPath}/${queryName}.txt`, output); } @@ -67,26 +69,44 @@ function metaToJson(meta: Meta): string { return JSON.stringify(Object.fromEntries(meta.allEntries), null, 2); } -function matchesToString(matches: EntityMatch[]): string { +function matchesToString(matches: EntityMatch[], rtl: boolean): string { const header = - padRight("Rank", 8) + - padRight("Entity", outputColumnWidth) + - padRight("Matched String", outputColumnWidth) + - padRight("Quality", 8) + + padRight("Rank", shortColumnWidth) + + padRight("Entity", wideColumnWidth) + + padRight("Matched String", wideColumnWidth) + + padRight("Quality", shortColumnWidth) + "\n\n"; - const matchesString = matches.map((m, i) => matchToString(m, i + 1)).join('\n'); + const matchesString = matches.map((m, i) => matchToString(m, i + 1, rtl)).join('\n'); return header + matchesString; } -function matchToString(match: EntityMatch, rank: number): string { +function matchToString(match: EntityMatch, rank: number, rtl: boolean): string { + if (!rtl) { + return ( + padRight(rank.toString(), shortColumnWidth) + + padRight(match.entity.name, wideColumnWidth) + + padRight(match.matchedString, wideColumnWidth) + + padRight(match.quality.toFixed(2), shortColumnWidth) + ); + } return ( - padRight(rank.toString(), 8) + - padRight(match.entity.name, outputColumnWidth) + - padRight(match.matchedString, outputColumnWidth) + - padRight(match.quality.toFixed(2), 8) + padAndMark(rank.toString(), shortColumnWidth, false) + + padAndMark(match.entity.name, wideColumnWidth, true) + + padAndMark(match.matchedString, wideColumnWidth, true) + + padAndMark(match.quality.toFixed(2), shortColumnWidth, false) ); } +function padAndMark(s: string, targetWidth: number, rtl: boolean): string { + const padded = rtl ? padLeft(s, targetWidth) : padRight(s, targetWidth); + const mark = rtl ? RLI : LRI; + return mark + padded + PDI; +} + function padRight(s: string, targetWidth: number): string { return s + ' '.repeat(Math.max(0, targetWidth - s.length)); } + +function padLeft(s: string, targetWidth: number): string { + return ' '.repeat(Math.max(0, targetWidth - s.length)) + s; +} From 7cdbccc1e8ff35b2070a9f1fa688082779d3613d Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 17 Oct 2025 11:17:04 +0200 Subject: [PATCH 015/105] prettier --- .vscode/launch.json | 22 +- .vscode/settings.json | 13 +- package-lock.json | 384 ++++++++++-------- package.json | 2 +- src/basic-usage.ts | 60 +-- src/commons/meta-merger.ts | 57 ++- src/dynamic-searchers/result-merger.ts | 4 +- src/interfaces/query.ts | 2 +- src/regression-test/main.ts | 112 ----- src/string-searchers/match.ts | 3 +- src/string-searchers/normalizing-searcher.ts | 2 +- src/suffix-array-searchers/index.ts | 2 +- .../string-comparison.ts | 63 ++- .../suffix-array-searcher.test.ts | 38 +- .../suffix-array-searcher.ts | 195 +++++---- .../suffix-array.test.ts | 26 +- src/suffix-array-searchers/suffix-array.ts | 253 ++++++------ 17 files changed, 531 insertions(+), 707 deletions(-) delete mode 100644 src/regression-test/main.ts diff --git a/.vscode/launch.json b/.vscode/launch.json index e796d7a..bd015c4 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -5,25 +5,9 @@ "type": "node", "request": "launch", "name": "Launch Basic Usage", - "skipFiles": [ - "/**" - ], + "skipFiles": ["/**"], "program": "${workspaceFolder}/dist/basic-usage.js", - "outFiles": [ - "${workspaceFolder}/**/*.js" - ] - }, - { - "type": "node", - "request": "launch", - "name": "Run Regression Test", - "skipFiles": [ - "/**" - ], - "program": "${workspaceFolder}/dist/regression-test/main.js", - "outFiles": [ - "${workspaceFolder}/**/*.js" - ] + "outFiles": ["${workspaceFolder}/**/*.js"] } ] -} \ No newline at end of file +} diff --git a/.vscode/settings.json b/.vscode/settings.json index e8947c0..452875b 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -5,14 +5,5 @@ "editor.formatOnSave": false }, "editor.defaultFormatter": "esbenp.prettier-vscode", - "editor.rulers": [ - 120 - ], - "liveServer.settings.port": 5501, - "[json]": { - "editor.defaultFormatter": "esbenp.prettier-vscode" - }, - "[plaintext]": { - "editor.renderControlCharacters": false - }, -} \ No newline at end of file + "editor.rulers": [120] +} diff --git a/package-lock.json b/package-lock.json index cd0a715..0767e04 100644 --- a/package-lock.json +++ b/package-lock.json @@ -40,15 +40,15 @@ } }, "node_modules/@babel/code-frame": { - "version": "7.26.2", - "resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.26.2.tgz", - "integrity": "sha512-RJlIHRueQgwWitWgF8OdFYGZX328Ax5BCemNGlqHfplnRT9ESi8JkFlvaVYbS+UubVY6dpv87Fs2u5M29iNFVQ==", + "version": "7.27.1", + "resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.27.1.tgz", + "integrity": "sha512-cjQ7ZlQ0Mv3b47hABuTevyTuYN4i+loJKGeV9flcCgIK37cCXRh+L1bd3iBHlynerhQ7BhCkn2BPbQUL+rGqFg==", "dev": true, "license": "MIT", "dependencies": { - "@babel/helper-validator-identifier": "^7.25.9", + "@babel/helper-validator-identifier": "^7.27.1", "js-tokens": "^4.0.0", - "picocolors": "^1.0.0" + "picocolors": "^1.1.1" }, "engines": { "node": ">=6.9.0" @@ -70,6 +70,7 @@ "integrity": "sha512-i1SLeK+DzNnQ3LL/CswPCa/E5u4lh1k6IAEphON8F+cXt0t9euTshDru0q7/IqMa1PMPz5RnHuHscF8/ZJsStg==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@ampproject/remapping": "^2.2.0", "@babel/code-frame": "^7.26.0", @@ -387,9 +388,9 @@ } }, "node_modules/@babel/helper-string-parser": { - "version": "7.25.9", - "resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.25.9.tgz", - "integrity": "sha512-4A/SCr/2KLd5jrtOMFzaKjVtAei3+2r/NChoBNoZ3EyP/+GlhoaEGoWOZUmFmoITP7zOJyHIMm+DYRd8o3PvHA==", + "version": "7.27.1", + "resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.27.1.tgz", + "integrity": "sha512-qMlSxKbpRlAridDExk92nSobyDdpPijUq2DW6oDnUqd0iOGxmQjyqhMIihI9+zv4LPyZdRje2cavWPbCbWm3eA==", "dev": true, "license": "MIT", "engines": { @@ -397,9 +398,9 @@ } }, "node_modules/@babel/helper-validator-identifier": { - "version": "7.25.9", - "resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.25.9.tgz", - "integrity": "sha512-Ed61U6XJc3CVRfkERJWDz4dJwKe7iLmmJsbOGu9wSloNSFttHV0I8g6UAgb7qnK5ly5bGLPd4oXZlxCdANBOWQ==", + "version": "7.27.1", + "resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-7.27.1.tgz", + "integrity": "sha512-D2hP9eA+Sqx1kBZgzxZh0y1trbuU+JoDkiEwqhQ36nodYqJwyEIhPSdMNd7lOm/4io72luTPWH20Yda0xOuUow==", "dev": true, "license": "MIT", "engines": { @@ -432,27 +433,27 @@ } }, "node_modules/@babel/helpers": { - "version": "7.26.0", - "resolved": "https://registry.npmjs.org/@babel/helpers/-/helpers-7.26.0.tgz", - "integrity": "sha512-tbhNuIxNcVb21pInl3ZSjksLCvgdZy9KwJ8brv993QtIVKJBBkYXz4q4ZbAv31GdnC+R90np23L5FbEBlthAEw==", + "version": "7.28.4", + "resolved": "https://registry.npmjs.org/@babel/helpers/-/helpers-7.28.4.tgz", + "integrity": "sha512-HFN59MmQXGHVyYadKLVumYsA9dBFun/ldYxipEjzA4196jpLZd8UjEEBLkbEkvfYreDqJhZxYAWFPtrfhNpj4w==", "dev": true, "license": "MIT", "dependencies": { - "@babel/template": "^7.25.9", - "@babel/types": "^7.26.0" + "@babel/template": "^7.27.2", + "@babel/types": "^7.28.4" }, "engines": { "node": ">=6.9.0" } }, "node_modules/@babel/parser": { - "version": "7.26.2", - "resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.26.2.tgz", - "integrity": "sha512-DWMCZH9WA4Maitz2q21SRKHo9QXZxkDsbNZoVD62gusNtNBBqDg9i7uOhASfTfIGNzW+O+r7+jAlM8dwphcJKQ==", + "version": "7.28.4", + "resolved": "https://registry.npmjs.org/@babel/parser/-/parser-7.28.4.tgz", + "integrity": "sha512-yZbBqeM6TkpP9du/I2pUZnJsRMGGvOuIrhjzC1AwHwW+6he4mni6Bp/m8ijn0iOuZuPI2BfkCoSRunpyjnrQKg==", "dev": true, "license": "MIT", "dependencies": { - "@babel/types": "^7.26.0" + "@babel/types": "^7.28.4" }, "bin": { "parser": "bin/babel-parser.js" @@ -1945,28 +1946,25 @@ } }, "node_modules/@babel/runtime": { - "version": "7.26.0", - "resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.26.0.tgz", - "integrity": "sha512-FDSOghenHTiToteC/QRlv2q3DhPZ/oOXTBoirfWNx1Cx3TMVcGWQtMMmQcSvb/JjpNeGzx8Pq/b4fKEJuWm1sw==", + "version": "7.28.4", + "resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.28.4.tgz", + "integrity": "sha512-Q/N6JNWvIvPnLDvjlE1OUBLPQHH6l3CltCEsHIujp45zQUSSh8K+gHnaEX45yAT1nyngnINhvWtzN+Nb9D8RAQ==", "dev": true, "license": "MIT", - "dependencies": { - "regenerator-runtime": "^0.14.0" - }, "engines": { "node": ">=6.9.0" } }, "node_modules/@babel/template": { - "version": "7.25.9", - "resolved": "https://registry.npmjs.org/@babel/template/-/template-7.25.9.tgz", - "integrity": "sha512-9DGttpmPvIxBb/2uwpVo3dqJ+O6RooAFOS+lB+xDqoE2PVCE8nfoHMdZLpfCQRLwvohzXISPZcgxt80xLfsuwg==", + "version": "7.27.2", + "resolved": "https://registry.npmjs.org/@babel/template/-/template-7.27.2.tgz", + "integrity": "sha512-LPDZ85aEJyYSd18/DkjNh4/y1ntkE5KwUHWTiqgRxruuZL2F1yuHligVHLvcHY2vMHXttKFpJn6LwfI7cw7ODw==", "dev": true, "license": "MIT", "dependencies": { - "@babel/code-frame": "^7.25.9", - "@babel/parser": "^7.25.9", - "@babel/types": "^7.25.9" + "@babel/code-frame": "^7.27.1", + "@babel/parser": "^7.27.2", + "@babel/types": "^7.27.1" }, "engines": { "node": ">=6.9.0" @@ -2002,14 +2000,14 @@ } }, "node_modules/@babel/types": { - "version": "7.26.0", - "resolved": "https://registry.npmjs.org/@babel/types/-/types-7.26.0.tgz", - "integrity": "sha512-Z/yiTPj+lDVnF7lWeKCIJzaIkI0vYO87dMpZ4bg4TDrFe4XXLFWL1TbXU27gBP3QccxV9mZICCrnjnYlJjXHOA==", + "version": "7.28.4", + "resolved": "https://registry.npmjs.org/@babel/types/-/types-7.28.4.tgz", + "integrity": "sha512-bkFqkLhh3pMBUQQkpVgWDWq/lqzc2678eUyDlTBhRqhCHFguYYGM0Efga7tYk4TogG/3x0EEl66/OQ+WGbWB/Q==", "dev": true, "license": "MIT", "dependencies": { - "@babel/helper-string-parser": "^7.25.9", - "@babel/helper-validator-identifier": "^7.25.9" + "@babel/helper-string-parser": "^7.27.1", + "@babel/helper-validator-identifier": "^7.27.1" }, "engines": { "node": ">=6.9.0" @@ -2023,9 +2021,9 @@ "license": "MIT" }, "node_modules/@esbuild/aix-ppc64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.23.1.tgz", - "integrity": "sha512-6VhYk1diRqrhBAqpJEdjASR/+WVRtfjpqKuNw11cLiaWpAT/Uu+nokB+UJnevzy/P9C/ty6AOe0dwueMrGh/iQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.25.10.tgz", + "integrity": "sha512-0NFWnA+7l41irNuaSVlLfgNT12caWJVLzp5eAVhZ0z1qpxbockccEt3s+149rE64VUI3Ml2zt8Nv5JVc4QXTsw==", "cpu": [ "ppc64" ], @@ -2040,9 +2038,9 @@ } }, "node_modules/@esbuild/android-arm": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.23.1.tgz", - "integrity": "sha512-uz6/tEy2IFm9RYOyvKl88zdzZfwEfKZmnX9Cj1BHjeSGNuGLuMD1kR8y5bteYmwqKm1tj8m4cb/aKEorr6fHWQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.25.10.tgz", + "integrity": "sha512-dQAxF1dW1C3zpeCDc5KqIYuZ1tgAdRXNoZP7vkBIRtKZPYe2xVr/d3SkirklCHudW1B45tGiUlz2pUWDfbDD4w==", "cpu": [ "arm" ], @@ -2057,9 +2055,9 @@ } }, "node_modules/@esbuild/android-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.23.1.tgz", - "integrity": "sha512-xw50ipykXcLstLeWH7WRdQuysJqejuAGPd30vd1i5zSyKK3WE+ijzHmLKxdiCMtH1pHz78rOg0BKSYOSB/2Khw==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.25.10.tgz", + "integrity": "sha512-LSQa7eDahypv/VO6WKohZGPSJDq5OVOo3UoFR1E4t4Gj1W7zEQMUhI+lo81H+DtB+kP+tDgBp+M4oNCwp6kffg==", "cpu": [ "arm64" ], @@ -2074,9 +2072,9 @@ } }, "node_modules/@esbuild/android-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.23.1.tgz", - "integrity": "sha512-nlN9B69St9BwUoB+jkyU090bru8L0NA3yFvAd7k8dNsVH8bi9a8cUAUSEcEEgTp2z3dbEDGJGfP6VUnkQnlReg==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.25.10.tgz", + "integrity": "sha512-MiC9CWdPrfhibcXwr39p9ha1x0lZJ9KaVfvzA0Wxwz9ETX4v5CHfF09bx935nHlhi+MxhA63dKRRQLiVgSUtEg==", "cpu": [ "x64" ], @@ -2091,9 +2089,9 @@ } }, "node_modules/@esbuild/darwin-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.23.1.tgz", - "integrity": "sha512-YsS2e3Wtgnw7Wq53XXBLcV6JhRsEq8hkfg91ESVadIrzr9wO6jJDMZnCQbHm1Guc5t/CdDiFSSfWP58FNuvT3Q==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.25.10.tgz", + "integrity": "sha512-JC74bdXcQEpW9KkV326WpZZjLguSZ3DfS8wrrvPMHgQOIEIG/sPXEN/V8IssoJhbefLRcRqw6RQH2NnpdprtMA==", "cpu": [ "arm64" ], @@ -2108,9 +2106,9 @@ } }, "node_modules/@esbuild/darwin-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.23.1.tgz", - "integrity": "sha512-aClqdgTDVPSEGgoCS8QDG37Gu8yc9lTHNAQlsztQ6ENetKEO//b8y31MMu2ZaPbn4kVsIABzVLXYLhCGekGDqw==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.25.10.tgz", + "integrity": "sha512-tguWg1olF6DGqzws97pKZ8G2L7Ig1vjDmGTwcTuYHbuU6TTjJe5FXbgs5C1BBzHbJ2bo1m3WkQDbWO2PvamRcg==", "cpu": [ "x64" ], @@ -2125,9 +2123,9 @@ } }, "node_modules/@esbuild/freebsd-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.23.1.tgz", - "integrity": "sha512-h1k6yS8/pN/NHlMl5+v4XPfikhJulk4G+tKGFIOwURBSFzE8bixw1ebjluLOjfwtLqY0kewfjLSrO6tN2MgIhA==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.25.10.tgz", + "integrity": "sha512-3ZioSQSg1HT2N05YxeJWYR+Libe3bREVSdWhEEgExWaDtyFbbXWb49QgPvFH8u03vUPX10JhJPcz7s9t9+boWg==", "cpu": [ "arm64" ], @@ -2142,9 +2140,9 @@ } }, "node_modules/@esbuild/freebsd-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.23.1.tgz", - "integrity": "sha512-lK1eJeyk1ZX8UklqFd/3A60UuZ/6UVfGT2LuGo3Wp4/z7eRTRYY+0xOu2kpClP+vMTi9wKOfXi2vjUpO1Ro76g==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.25.10.tgz", + "integrity": "sha512-LLgJfHJk014Aa4anGDbh8bmI5Lk+QidDmGzuC2D+vP7mv/GeSN+H39zOf7pN5N8p059FcOfs2bVlrRr4SK9WxA==", "cpu": [ "x64" ], @@ -2159,9 +2157,9 @@ } }, "node_modules/@esbuild/linux-arm": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.23.1.tgz", - "integrity": "sha512-CXXkzgn+dXAPs3WBwE+Kvnrf4WECwBdfjfeYHpMeVxWE0EceB6vhWGShs6wi0IYEqMSIzdOF1XjQ/Mkm5d7ZdQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.25.10.tgz", + "integrity": "sha512-oR31GtBTFYCqEBALI9r6WxoU/ZofZl962pouZRTEYECvNF/dtXKku8YXcJkhgK/beU+zedXfIzHijSRapJY3vg==", "cpu": [ "arm" ], @@ -2176,9 +2174,9 @@ } }, "node_modules/@esbuild/linux-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.23.1.tgz", - "integrity": "sha512-/93bf2yxencYDnItMYV/v116zff6UyTjo4EtEQjUBeGiVpMmffDNUyD9UN2zV+V3LRV3/on4xdZ26NKzn6754g==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.25.10.tgz", + "integrity": "sha512-5luJWN6YKBsawd5f9i4+c+geYiVEw20FVW5x0v1kEMWNq8UctFjDiMATBxLvmmHA4bf7F6hTRaJgtghFr9iziQ==", "cpu": [ "arm64" ], @@ -2193,9 +2191,9 @@ } }, "node_modules/@esbuild/linux-ia32": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.23.1.tgz", - "integrity": "sha512-VTN4EuOHwXEkXzX5nTvVY4s7E/Krz7COC8xkftbbKRYAl96vPiUssGkeMELQMOnLOJ8k3BY1+ZY52tttZnHcXQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.25.10.tgz", + "integrity": "sha512-NrSCx2Kim3EnnWgS4Txn0QGt0Xipoumb6z6sUtl5bOEZIVKhzfyp/Lyw4C1DIYvzeW/5mWYPBFJU3a/8Yr75DQ==", "cpu": [ "ia32" ], @@ -2210,9 +2208,9 @@ } }, "node_modules/@esbuild/linux-loong64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.23.1.tgz", - "integrity": "sha512-Vx09LzEoBa5zDnieH8LSMRToj7ir/Jeq0Gu6qJ/1GcBq9GkfoEAoXvLiW1U9J1qE/Y/Oyaq33w5p2ZWrNNHNEw==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.25.10.tgz", + "integrity": "sha512-xoSphrd4AZda8+rUDDfD9J6FUMjrkTz8itpTITM4/xgerAZZcFW7Dv+sun7333IfKxGG8gAq+3NbfEMJfiY+Eg==", "cpu": [ "loong64" ], @@ -2227,9 +2225,9 @@ } }, "node_modules/@esbuild/linux-mips64el": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.23.1.tgz", - "integrity": "sha512-nrFzzMQ7W4WRLNUOU5dlWAqa6yVeI0P78WKGUo7lg2HShq/yx+UYkeNSE0SSfSure0SqgnsxPvmAUu/vu0E+3Q==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.25.10.tgz", + "integrity": "sha512-ab6eiuCwoMmYDyTnyptoKkVS3k8fy/1Uvq7Dj5czXI6DF2GqD2ToInBI0SHOp5/X1BdZ26RKc5+qjQNGRBelRA==", "cpu": [ "mips64el" ], @@ -2244,9 +2242,9 @@ } }, "node_modules/@esbuild/linux-ppc64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.23.1.tgz", - "integrity": "sha512-dKN8fgVqd0vUIjxuJI6P/9SSSe/mB9rvA98CSH2sJnlZ/OCZWO1DJvxj8jvKTfYUdGfcq2dDxoKaC6bHuTlgcw==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.25.10.tgz", + "integrity": "sha512-NLinzzOgZQsGpsTkEbdJTCanwA5/wozN9dSgEl12haXJBzMTpssebuXR42bthOF3z7zXFWH1AmvWunUCkBE4EA==", "cpu": [ "ppc64" ], @@ -2261,9 +2259,9 @@ } }, "node_modules/@esbuild/linux-riscv64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.23.1.tgz", - "integrity": "sha512-5AV4Pzp80fhHL83JM6LoA6pTQVWgB1HovMBsLQ9OZWLDqVY8MVobBXNSmAJi//Csh6tcY7e7Lny2Hg1tElMjIA==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.25.10.tgz", + "integrity": "sha512-FE557XdZDrtX8NMIeA8LBJX3dC2M8VGXwfrQWU7LB5SLOajfJIxmSdyL/gU1m64Zs9CBKvm4UAuBp5aJ8OgnrA==", "cpu": [ "riscv64" ], @@ -2278,9 +2276,9 @@ } }, "node_modules/@esbuild/linux-s390x": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.23.1.tgz", - "integrity": "sha512-9ygs73tuFCe6f6m/Tb+9LtYxWR4c9yg7zjt2cYkjDbDpV/xVn+68cQxMXCjUpYwEkze2RcU/rMnfIXNRFmSoDw==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.25.10.tgz", + "integrity": "sha512-3BBSbgzuB9ajLoVZk0mGu+EHlBwkusRmeNYdqmznmMc9zGASFjSsxgkNsqmXugpPk00gJ0JNKh/97nxmjctdew==", "cpu": [ "s390x" ], @@ -2295,9 +2293,9 @@ } }, "node_modules/@esbuild/linux-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.23.1.tgz", - "integrity": "sha512-EV6+ovTsEXCPAp58g2dD68LxoP/wK5pRvgy0J/HxPGB009omFPv3Yet0HiaqvrIrgPTBuC6wCH1LTOY91EO5hQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.25.10.tgz", + "integrity": "sha512-QSX81KhFoZGwenVyPoberggdW1nrQZSvfVDAIUXr3WqLRZGZqWk/P4T8p2SP+de2Sr5HPcvjhcJzEiulKgnxtA==", "cpu": [ "x64" ], @@ -2311,10 +2309,27 @@ "node": ">=18" } }, + "node_modules/@esbuild/netbsd-arm64": { + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.25.10.tgz", + "integrity": "sha512-AKQM3gfYfSW8XRk8DdMCzaLUFB15dTrZfnX8WXQoOUpUBQ+NaAFCP1kPS/ykbbGYz7rxn0WS48/81l9hFl3u4A==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "netbsd" + ], + "engines": { + "node": ">=18" + } + }, "node_modules/@esbuild/netbsd-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.23.1.tgz", - "integrity": "sha512-aevEkCNu7KlPRpYLjwmdcuNz6bDFiE7Z8XC4CPqExjTvrHugh28QzUXVOZtiYghciKUacNktqxdpymplil1beA==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.25.10.tgz", + "integrity": "sha512-7RTytDPGU6fek/hWuN9qQpeGPBZFfB4zZgcz2VK2Z5VpdUxEI8JKYsg3JfO0n/Z1E/6l05n0unDCNc4HnhQGig==", "cpu": [ "x64" ], @@ -2329,9 +2344,9 @@ } }, "node_modules/@esbuild/openbsd-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.23.1.tgz", - "integrity": "sha512-3x37szhLexNA4bXhLrCC/LImN/YtWis6WXr1VESlfVtVeoFJBRINPJ3f0a/6LV8zpikqoUg4hyXw0sFBt5Cr+Q==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.25.10.tgz", + "integrity": "sha512-5Se0VM9Wtq797YFn+dLimf2Zx6McttsH2olUBsDml+lm0GOCRVebRWUvDtkY4BWYv/3NgzS8b/UM3jQNh5hYyw==", "cpu": [ "arm64" ], @@ -2346,9 +2361,9 @@ } }, "node_modules/@esbuild/openbsd-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.23.1.tgz", - "integrity": "sha512-aY2gMmKmPhxfU+0EdnN+XNtGbjfQgwZj43k8G3fyrDM/UdZww6xrWxmDkuz2eCZchqVeABjV5BpildOrUbBTqA==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.25.10.tgz", + "integrity": "sha512-XkA4frq1TLj4bEMB+2HnI0+4RnjbuGZfet2gs/LNs5Hc7D89ZQBHQ0gL2ND6Lzu1+QVkjp3x1gIcPKzRNP8bXw==", "cpu": [ "x64" ], @@ -2362,10 +2377,27 @@ "node": ">=18" } }, + "node_modules/@esbuild/openharmony-arm64": { + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.25.10.tgz", + "integrity": "sha512-AVTSBhTX8Y/Fz6OmIVBip9tJzZEUcY8WLh7I59+upa5/GPhh2/aM6bvOMQySspnCCHvFi79kMtdJS1w0DXAeag==", + "cpu": [ + "arm64" + ], + "dev": true, + "license": "MIT", + "optional": true, + "os": [ + "openharmony" + ], + "engines": { + "node": ">=18" + } + }, "node_modules/@esbuild/sunos-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.23.1.tgz", - "integrity": "sha512-RBRT2gqEl0IKQABT4XTj78tpk9v7ehp+mazn2HbUeZl1YMdaGAQqhapjGTCe7uw7y0frDi4gS0uHzhvpFuI1sA==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.25.10.tgz", + "integrity": "sha512-fswk3XT0Uf2pGJmOpDB7yknqhVkJQkAQOcW/ccVOtfx05LkbWOaRAtn5SaqXypeKQra1QaEa841PgrSL9ubSPQ==", "cpu": [ "x64" ], @@ -2380,9 +2412,9 @@ } }, "node_modules/@esbuild/win32-arm64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.23.1.tgz", - "integrity": "sha512-4O+gPR5rEBe2FpKOVyiJ7wNDPA8nGzDuJ6gN4okSA1gEOYZ67N8JPk58tkWtdtPeLz7lBnY6I5L3jdsr3S+A6A==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.25.10.tgz", + "integrity": "sha512-ah+9b59KDTSfpaCg6VdJoOQvKjI33nTaQr4UluQwW7aEwZQsbMCfTmfEO4VyewOxx4RaDT/xCy9ra2GPWmO7Kw==", "cpu": [ "arm64" ], @@ -2397,9 +2429,9 @@ } }, "node_modules/@esbuild/win32-ia32": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.23.1.tgz", - "integrity": "sha512-BcaL0Vn6QwCwre3Y717nVHZbAa4UBEigzFm6VdsVdT/MbZ38xoj1X9HPkZhbmaBGUD1W8vxAfffbDe8bA6AKnQ==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.25.10.tgz", + "integrity": "sha512-QHPDbKkrGO8/cz9LKVnJU22HOi4pxZnZhhA2HYHez5Pz4JeffhDjf85E57Oyco163GnzNCVkZK0b/n4Y0UHcSw==", "cpu": [ "ia32" ], @@ -2414,9 +2446,9 @@ } }, "node_modules/@esbuild/win32-x64": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.23.1.tgz", - "integrity": "sha512-BHpFFeslkWrXWyUPnbKm+xYYVYruCinGcftSBaa8zoF9hZO4BcSCFUvHVTtzpIY6YzUnYtuEhZ+C9iEXjxnasg==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.25.10.tgz", + "integrity": "sha512-9KpxSVFCu0iK1owoez6aC/s/EdUQLDN3adTxGCqxMVhrPDj6bt5dbrHDXUuq+Bs2vATFBBrQS5vdQ/Ed2P+nbw==", "cpu": [ "x64" ], @@ -2497,9 +2529,9 @@ } }, "node_modules/@eslint/eslintrc/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -2595,9 +2627,9 @@ } }, "node_modules/@humanwhocodes/config-array/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -3417,6 +3449,7 @@ "integrity": "sha512-qoQprZvz5wQFJwMDqeseRXWv3rqMvhgpbXFfVyWhbx9X47POIA6i/+dXefEmZKoAgOaTdaIgNSMqMIU61yRyzA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@babel/parser": "^7.20.7", "@babel/types": "^7.20.7", @@ -3629,6 +3662,7 @@ "integrity": "sha512-7n59qFpghG4uazrF9qtGKBZXn7Oz4sOMm8dwNWDQY96Xlm2oX67eipqcblDj+oY1lLCbf1oltMZFpUso66Kl1A==", "dev": true, "license": "BSD-2-Clause", + "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.15.0", "@typescript-eslint/types": "8.15.0", @@ -3800,6 +3834,7 @@ "integrity": "sha512-cl669nCJTZBsL97OF4kUQm5g5hC2uihk0NxY3WENAC0TYdILVkAyHymAntgxGkl7K+t0cXIrH5siy5S4XkFycA==", "dev": true, "license": "MIT", + "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -4260,9 +4295,9 @@ "license": "ISC" }, "node_modules/brace-expansion": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.0.1.tgz", - "integrity": "sha512-XnAIvQ8eM+kC6aULx6wuQiwVsnzsi9d3WxzV3FpWTGA19F621kwdbsAcFKXgKUHZWsy+mY6iL1sHTxWEFCytDA==", + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-2.0.2.tgz", + "integrity": "sha512-Jt0vHyM+jmUBqojB7E1NIYadt0vI0Qxjxd2TErW94wDz+E2LAm5vKMXXwg6ZZBTHPuUlDgQHKXvjGBdfcF1ZDQ==", "dev": true, "license": "MIT", "dependencies": { @@ -4315,6 +4350,7 @@ } ], "license": "MIT", + "peer": true, "dependencies": { "caniuse-lite": "^1.0.30001669", "electron-to-chromium": "^1.5.41", @@ -4425,9 +4461,9 @@ } }, "node_modules/caniuse-lite": { - "version": "1.0.30001683", - "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001683.tgz", - "integrity": "sha512-iqmNnThZ0n70mNwvxpEC2nBJ037ZHZUoBI5Gorh1Mw6IlEAZujEoU1tXA628iZfzm7R9FvFzxbfdgml82a3k8Q==", + "version": "1.0.30001749", + "resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001749.tgz", + "integrity": "sha512-0rw2fJOmLfnzCRbkm8EyHL8SvI2Apu5UbnQuTsJ0ClgrH8hcwFooJ1s5R0EP8o8aVrFu8++ae29Kt9/gZAZp/Q==", "dev": true, "funding": [ { @@ -5343,9 +5379,9 @@ } }, "node_modules/esbuild": { - "version": "0.23.1", - "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.23.1.tgz", - "integrity": "sha512-VVNz/9Sa0bs5SELtn3f7qhJCDPCF5oMEl5cO9/SSinpE9hbPVvxbd572HH5AKiP7WD8INO53GgfDDhRjkylHEg==", + "version": "0.25.10", + "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.25.10.tgz", + "integrity": "sha512-9RiGKvCwaqxO2owP61uQ4BgNborAQskMR6QusfWzQqv7AZOg5oGehdY2pRJMTKuwxd1IDBP4rSbI5lHzU7SMsQ==", "dev": true, "hasInstallScript": true, "license": "MIT", @@ -5356,30 +5392,32 @@ "node": ">=18" }, "optionalDependencies": { - "@esbuild/aix-ppc64": "0.23.1", - "@esbuild/android-arm": "0.23.1", - "@esbuild/android-arm64": "0.23.1", - "@esbuild/android-x64": "0.23.1", - "@esbuild/darwin-arm64": "0.23.1", - "@esbuild/darwin-x64": "0.23.1", - "@esbuild/freebsd-arm64": "0.23.1", - "@esbuild/freebsd-x64": "0.23.1", - "@esbuild/linux-arm": "0.23.1", - "@esbuild/linux-arm64": "0.23.1", - "@esbuild/linux-ia32": "0.23.1", - "@esbuild/linux-loong64": "0.23.1", - "@esbuild/linux-mips64el": "0.23.1", - "@esbuild/linux-ppc64": "0.23.1", - "@esbuild/linux-riscv64": "0.23.1", - "@esbuild/linux-s390x": "0.23.1", - "@esbuild/linux-x64": "0.23.1", - "@esbuild/netbsd-x64": "0.23.1", - "@esbuild/openbsd-arm64": "0.23.1", - "@esbuild/openbsd-x64": "0.23.1", - "@esbuild/sunos-x64": "0.23.1", - "@esbuild/win32-arm64": "0.23.1", - "@esbuild/win32-ia32": "0.23.1", - "@esbuild/win32-x64": "0.23.1" + "@esbuild/aix-ppc64": "0.25.10", + "@esbuild/android-arm": "0.25.10", + "@esbuild/android-arm64": "0.25.10", + "@esbuild/android-x64": "0.25.10", + "@esbuild/darwin-arm64": "0.25.10", + "@esbuild/darwin-x64": "0.25.10", + "@esbuild/freebsd-arm64": "0.25.10", + "@esbuild/freebsd-x64": "0.25.10", + "@esbuild/linux-arm": "0.25.10", + "@esbuild/linux-arm64": "0.25.10", + "@esbuild/linux-ia32": "0.25.10", + "@esbuild/linux-loong64": "0.25.10", + "@esbuild/linux-mips64el": "0.25.10", + "@esbuild/linux-ppc64": "0.25.10", + "@esbuild/linux-riscv64": "0.25.10", + "@esbuild/linux-s390x": "0.25.10", + "@esbuild/linux-x64": "0.25.10", + "@esbuild/netbsd-arm64": "0.25.10", + "@esbuild/netbsd-x64": "0.25.10", + "@esbuild/openbsd-arm64": "0.25.10", + "@esbuild/openbsd-x64": "0.25.10", + "@esbuild/openharmony-arm64": "0.25.10", + "@esbuild/sunos-x64": "0.25.10", + "@esbuild/win32-arm64": "0.25.10", + "@esbuild/win32-ia32": "0.25.10", + "@esbuild/win32-x64": "0.25.10" } }, "node_modules/escalade": { @@ -5412,6 +5450,7 @@ "deprecated": "This version is no longer supported. Please see https://eslint.org/version-support for other options.", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.2.0", "@eslint-community/regexpp": "^4.6.1", @@ -5493,9 +5532,9 @@ } }, "node_modules/eslint/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -6173,9 +6212,9 @@ } }, "node_modules/glob/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -7183,9 +7222,9 @@ } }, "node_modules/jake/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -7212,6 +7251,7 @@ "integrity": "sha512-NIy3oAFp9shda19hy4HK0HRTWKtPJmGdnvywu01nOqNC2vZg+Z+fvJDxpMQA88eb2I9EcafcdjYgsDthnYTvGw==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@jest/core": "^29.7.0", "@jest/types": "^29.6.3", @@ -8523,9 +8563,9 @@ "license": "MIT" }, "node_modules/nanoid": { - "version": "3.3.7", - "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.7.tgz", - "integrity": "sha512-eSRppjcPIatRIMC1U6UngP8XFcz8MQWGQdt1MTBQ7NaAmvXDfvNxbvWV3x2y6CdEUciCSsDHDQZbhYaB8QEo2g==", + "version": "3.3.11", + "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.11.tgz", + "integrity": "sha512-N8SpfPUnUp1bK+PMYW8qSWdl9U+wwNWI4QKxOYDy9JAro3WMX7p2OeVRF9v+347pnakNevPmiHhNmZ2HbFA76w==", "dev": true, "funding": [ { @@ -9060,6 +9100,7 @@ } ], "license": "MIT", + "peer": true, "dependencies": { "nanoid": "^3.3.7", "picocolors": "^1.1.1", @@ -9892,13 +9933,6 @@ "node": ">=4" } }, - "node_modules/regenerator-runtime": { - "version": "0.14.1", - "resolved": "https://registry.npmjs.org/regenerator-runtime/-/regenerator-runtime-0.14.1.tgz", - "integrity": "sha512-dYnhHh0nJoMfnkZs6GmmhFknAGRrLznOu5nc9ML+EJxGvrx6H7teuevqVqCuPcPK//3eDrrjQhehXVx9cnkGdw==", - "dev": true, - "license": "MIT" - }, "node_modules/regenerator-transform": { "version": "0.15.2", "resolved": "https://registry.npmjs.org/regenerator-transform/-/regenerator-transform-0.15.2.tgz", @@ -10108,6 +10142,7 @@ "integrity": "sha512-fS6iqSPZDs3dr/y7Od6y5nha8dW1YnbgtsyotCVvoFGKbERG++CVRFv1meyGDE1SNItQA8BrnCw7ScdAhRJ3XQ==", "dev": true, "license": "MIT", + "peer": true, "bin": { "rollup": "dist/bin/rollup" }, @@ -11000,9 +11035,9 @@ } }, "node_modules/test-exclude/node_modules/brace-expansion": { - "version": "1.1.11", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.11.tgz", - "integrity": "sha512-iCuPHDFgrHX7H2vEI/5xpz07zSHB00TpugqhmYtVmMO6518mCuRMoOYFldEBl0g187ufozdaHgWKcYFb61qGiA==", + "version": "1.1.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.12.tgz", + "integrity": "sha512-9T9UjW3r0UW5c1Q7GTwllptXwhvYmEzFhzMfZ9H7FQWt+uZePjZPjBP/W1ZEyZ1twGWom5/56TF4lPcqjnDHcg==", "dev": true, "license": "MIT", "dependencies": { @@ -11142,13 +11177,13 @@ "license": "0BSD" }, "node_modules/tsx": { - "version": "4.19.2", - "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.19.2.tgz", - "integrity": "sha512-pOUl6Vo2LUq/bSa8S5q7b91cgNSjctn9ugq/+Mvow99qW6x/UZYwzxy/3NmqoT66eHYfCVvFvACC58UBPFf28g==", + "version": "4.20.6", + "resolved": "https://registry.npmjs.org/tsx/-/tsx-4.20.6.tgz", + "integrity": "sha512-ytQKuwgmrrkDTFP4LjR0ToE2nqgy886GpvRSpU0JAnrdBYppuY5rLkRUYPU1yCryb24SsKBTL/hlDQAEFVwtZg==", "dev": true, "license": "MIT", "dependencies": { - "esbuild": "~0.23.0", + "esbuild": "~0.25.0", "get-tsconfig": "^4.7.5" }, "bin": { @@ -11317,6 +11352,7 @@ "integrity": "sha512-hjcS1mhfuyi4WW8IWtjP7brDrG2cuDZukyrYrSauoXGNgx0S7zceP07adYkJycEr56BOUTNPzbInooiN3fn1qw==", "dev": true, "license": "Apache-2.0", + "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" diff --git a/package.json b/package.json index 745d8a4..97040fa 100644 --- a/package.json +++ b/package.json @@ -74,4 +74,4 @@ "dist", "usage-examples" ] -} \ No newline at end of file +} diff --git a/src/basic-usage.ts b/src/basic-usage.ts index 13d6731..df975db 100644 --- a/src/basic-usage.ts +++ b/src/basic-usage.ts @@ -1,55 +1,9 @@ -// import { Config } from './config'; +import { Match } from './string-searchers/match.js'; import { Query } from './interfaces/query.js'; -import { SearcherFactory } from './searcher-factory.js'; +import { SuffixArraySearcher } from './suffix-array-searchers/suffix-array-searcher.js'; -class Person { - constructor( - public id: number, - public firstName: string, - public lastName: string - ) {} -} - -const searcher = SearcherFactory.createDefaultSearcher(); - -// If your dataset contains non-latin characters, build the searcher in the following way instead: -/* const config = Config.createDefaultConfig(); -config.normalizerConfig.allowCharacter = (_c) => true; -const searcher = SearcherFactory.createSearcher(config); */ - -const persons = [ - { id: 23501, firstName: 'Alice', lastName: 'King' }, - { id: 99234, firstName: 'Bob', lastName: 'Bishop' }, - { id: 5823, firstName: 'Carol', lastName: 'Queen' }, - { id: 11923, firstName: 'Charlie', lastName: 'Rook' } -]; - -const indexingMeta = searcher.indexEntities( - persons, - (e) => e.id, - (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] -); -console.dir(indexingMeta); - -const result = searcher.getMatches(new Query('alice kign')); -console.dir(result); - -const removalResult = searcher.removeEntities([99234, 5823]); -console.dir(removalResult); - -const persons2 = [ - { id: 723, firstName: 'David', lastName: 'Knight' }, // new - { id: 2634, firstName: 'Eve', lastName: 'Pawn' }, // new - { id: 23501, firstName: 'Allie', lastName: 'King' }, // updated - { id: 11923, firstName: 'Charles', lastName: 'Rook' } // updated -]; - -const upsertMeta = searcher.upsertEntities( - persons2, - (e) => e.id, - (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] -); -console.dir(upsertMeta); - -const result2 = searcher.getMatches(new Query('allie')); -console.dir(result2); +const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher(); +suffixArraySearcher.index(['Alice', 'Bob', 'Carol', 'Charlie']); +const matches = suffixArraySearcher.getMatches(new Query('li')).matches; +console.log(matches); +console.log('finished'); diff --git a/src/commons/meta-merger.ts b/src/commons/meta-merger.ts index 1859fc3..37fdfa9 100644 --- a/src/commons/meta-merger.ts +++ b/src/commons/meta-merger.ts @@ -3,35 +3,34 @@ import { Meta } from '../interfaces/meta.js'; export class MetaMerger { + /** + * Merges two {@link Meta} objects into a new {@link Meta} object. + * @param meta1 The first meta object. + * @param meta2 The second meta object. + * @returns The merged meta object. + */ + public static mergeMeta(meta1: Meta, meta2: Meta): Meta { + const newMetaEntries: Map = new Map(); - /** - * Merges two {@link Meta} objects into a new {@link Meta} object. - * @param meta1 The first meta object. - * @param meta2 The second meta object. - * @returns The merged meta object. - */ - public static mergeMeta(meta1: Meta, meta2: Meta): Meta { - const newMetaEntries: Map = new Map(); - - for (const [key, value] of meta1.allEntries) { - newMetaEntries.set(key, value); - } - - for (const [key, value] of meta2.allEntries) { - const presentValue = newMetaEntries.get(key); - if (presentValue === undefined) { - newMetaEntries.set(key, value); - continue; - } - if (typeof presentValue === 'number' && typeof value === 'number') { - newMetaEntries.set(key, presentValue + value); - continue; - } - newMetaEntries.delete(key); - newMetaEntries.set(`${key}_0`, presentValue); - newMetaEntries.set(`${key}_1`, value); - } + for (const [key, value] of meta1.allEntries) { + newMetaEntries.set(key, value); + } - return new Meta(newMetaEntries); + for (const [key, value] of meta2.allEntries) { + const presentValue = newMetaEntries.get(key); + if (presentValue === undefined) { + newMetaEntries.set(key, value); + continue; + } + if (typeof presentValue === 'number' && typeof value === 'number') { + newMetaEntries.set(key, presentValue + value); + continue; + } + newMetaEntries.delete(key); + newMetaEntries.set(`${key}_0`, presentValue); + newMetaEntries.set(`${key}_1`, value); } -} \ No newline at end of file + + return new Meta(newMetaEntries); + } +} diff --git a/src/dynamic-searchers/result-merger.ts b/src/dynamic-searchers/result-merger.ts index 25a6d2d..e80645a 100644 --- a/src/dynamic-searchers/result-merger.ts +++ b/src/dynamic-searchers/result-merger.ts @@ -47,8 +47,8 @@ export class ResultMerger { const newMatches = [...matches1, ...matches2]; newMatches.sort((m1, m2) => m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : 0 ); return newMatches.length <= topN ? newMatches : newMatches.slice(0, topN); } diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index 2bfa1af..5a6bba7 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -13,7 +13,7 @@ export class Query { public readonly topN: number; /** - * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the * number of matches. The value must be between 0 and 1. */ public readonly minQuality: number; diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts deleted file mode 100644 index ab724d7..0000000 --- a/src/regression-test/main.ts +++ /dev/null @@ -1,112 +0,0 @@ -/* - Regression tests. Run this file and check the git diff of the output files to see what changed. -*/ - -import { readFileSync, writeFileSync } from 'fs'; -import { Config } from '../config.js'; -import { EntityMatch } from '../interfaces/entity-match.js'; -import { EntityResult } from '../interfaces/entity-result.js'; -import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; -import { SearcherFactory } from '../searcher-factory.js'; - -const outputPath = './src/regression-test/output'; -const shortColumnWidth = 8; -const wideColumnWidth = 40; -const RLI = "\u2067", LRI = "\u2066", PDI = "\u2069"; - -const text = readFileSync('./data/world-ctvs.txt', 'utf8'); -const lines = text.split('\n').slice(1); -const entities = lines.map((l, index) => ({ id: index, name: l })); - -interface GeoEntity { - id: number; - name: string; -} - -const config = Config.createDefaultConfig(); -config.normalizerConfig.allowCharacter = (_) => true; -const searcher = SearcherFactory.createSearcher(config); - -console.log(`Indexing ${entities.length} entities...`); -const indexingMeta: Meta = searcher.indexEntities( - entities, - (e) => e.id, - (e) => e.name.split(';') -); -writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { encoding: 'utf8' }); - -console.log("Running queries..."); - -runQuery('carcassonne-prefix', 'carcasso'); -runQuery('carcassonne-infix', 'cassonn'); -runQuery('carcassonne-suffix', 'sonne'); -runQuery('munich-insertion', 'muniich'); -runQuery('boston-deletion', 'bostn') -runQuery('boulder-creek-substitution', 'boulder creak'); -runQuery('tübingen-transposition', 'tübignen'); -runQuery('tokyo', '東京都'); -runQuery('tokyo-prefix', '東京'); -runQuery('tbilisi', 'თბილისი'); -runQuery('tbilisi-deletion', 'თბიისი'); -runQuery('kuwait-city', 'مدينة الكويت', true); -runQuery('kuwait-city-prefix', 'مدينة الك', true); - -console.log("Finished.") - -function runQuery(queryName: string, queryString: string, rtl: boolean = false) { - const query: Query = new Query(queryString); - const result: EntityResult = searcher.getMatches(query); - console.log(`'${queryString}' (${queryName}): ${result.matches.length} matches.`); - const queryJson = JSON.stringify(query, null, 2); - const metaJson = metaToJson(result.meta); - const matchesString = matchesToString(result.matches, rtl); - const output = `${queryJson}\n\n${metaJson}\n\n${matchesString}`; - writeFileSync(`${outputPath}/${queryName}.txt`, output); -} - -function metaToJson(meta: Meta): string { - return JSON.stringify(Object.fromEntries(meta.allEntries), null, 2); -} - -function matchesToString(matches: EntityMatch[], rtl: boolean): string { - const header = - padRight("Rank", shortColumnWidth) + - padRight("Entity", wideColumnWidth) + - padRight("Matched String", wideColumnWidth) + - padRight("Quality", shortColumnWidth) + - "\n\n"; - const matchesString = matches.map((m, i) => matchToString(m, i + 1, rtl)).join('\n'); - return header + matchesString; -} - -function matchToString(match: EntityMatch, rank: number, rtl: boolean): string { - if (!rtl) { - return ( - padRight(rank.toString(), shortColumnWidth) + - padRight(match.entity.name, wideColumnWidth) + - padRight(match.matchedString, wideColumnWidth) + - padRight(match.quality.toFixed(2), shortColumnWidth) - ); - } - return ( - padAndMark(rank.toString(), shortColumnWidth, false) + - padAndMark(match.entity.name, wideColumnWidth, true) + - padAndMark(match.matchedString, wideColumnWidth, true) + - padAndMark(match.quality.toFixed(2), shortColumnWidth, false) - ); -} - -function padAndMark(s: string, targetWidth: number, rtl: boolean): string { - const padded = rtl ? padLeft(s, targetWidth) : padRight(s, targetWidth); - const mark = rtl ? RLI : LRI; - return mark + padded + PDI; -} - -function padRight(s: string, targetWidth: number): string { - return s + ' '.repeat(Math.max(0, targetWidth - s.length)); -} - -function padLeft(s: string, targetWidth: number): string { - return ' '.repeat(Math.max(0, targetWidth - s.length)) + s; -} diff --git a/src/string-searchers/match.ts b/src/string-searchers/match.ts index 3ca865d..6f4a206 100644 --- a/src/string-searchers/match.ts +++ b/src/string-searchers/match.ts @@ -10,6 +10,5 @@ export class Match { public constructor( public readonly index: number, public quality: number - ) { - } + ) {} } diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index efdcd93..1f015b4 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -17,7 +17,7 @@ export class NormalizingSearcher implements StringSearcher { public constructor( private readonly stringSearcher: StringSearcher, private readonly normalizer: Normalizer - ) { } + ) {} /** * {@inheritDoc StringSearcher.index} diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index 5bdb7cf..f58863b 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -1,3 +1,3 @@ export { StringComparison } from './string-comparison.js'; export { SuffixArray } from './suffix-array.js'; -export { SuffixArraySearcher } from './suffix-array-searcher.js'; \ No newline at end of file +export { SuffixArraySearcher } from './suffix-array-searcher.js'; diff --git a/src/suffix-array-searchers/string-comparison.ts b/src/suffix-array-searchers/string-comparison.ts index 243b4a4..d61eb10 100644 --- a/src/suffix-array-searchers/string-comparison.ts +++ b/src/suffix-array-searchers/string-comparison.ts @@ -1,41 +1,32 @@ - export class StringComparison { + public static compareOrdinal(strA: string, indexA: number, strB: string, indexB: number, length: number): number { + const endA = Math.min(indexA + length, strA.length); + const endB = Math.min(indexB + length, strB.length); + + let iA = indexA; + let iB = indexB; + + while (iA < endA && iB < endB) { + const codeA = strA.charCodeAt(iA); + const codeB = strB.charCodeAt(iB); + + if (codeA < codeB) { + return -1; + } + if (codeA > codeB) { + return 1; + } + + iA++; + iB++; + } - public static compareOrdinal( - strA: string, - indexA: number, - strB: string, - indexB: number, - length: number - ): number { - - const endA = Math.min(indexA + length, strA.length); - const endB = Math.min(indexB + length, strB.length); - - let iA = indexA; - let iB = indexB; - - while (iA < endA && iB < endB) { - const codeA = strA.charCodeAt(iA); - const codeB = strB.charCodeAt(iB); - - if (codeA < codeB) { - return -1; - } - if (codeA > codeB) { - return 1; - } - - iA++; - iB++; - } - - const lenComparedA = endA - indexA; - const lenComparedB = endB - indexB; + const lenComparedA = endA - indexA; + const lenComparedB = endB - indexB; - if (lenComparedA === lenComparedB) { - return 0; - } - return lenComparedA < lenComparedB ? -1 : 1; + if (lenComparedA === lenComparedB) { + return 0; } + return lenComparedA < lenComparedB ? -1 : 1; + } } diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts index 0fda005..9ccbd0f 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.test.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -6,64 +6,62 @@ const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher(); suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); function getMatches(queryString: string): Match[] { - const query = new Query(queryString, 10, 0); - const matches = suffixArraySearcher.getMatches(query).matches; - matches.sort((m1, m2) => m1.index - m2.index); - return matches; + const query = new Query(queryString, 10, 0); + const matches = suffixArraySearcher.getMatches(query).matches; + matches.sort((m1, m2) => m1.index - m2.index); + return matches; } test('can find exact match test 1', () => { - expect(getMatches('Alice')).toEqual([new Match(0, 1)]); + expect(getMatches('Alice')).toEqual([new Match(0, 1)]); }); test('can find exact match test 2', () => { - expect(getMatches('Bob')).toEqual([new Match(1, 1)]); + expect(getMatches('Bob')).toEqual([new Match(1, 1)]); }); test('can find prefix matches test 1', () => { - expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); + expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); }); test('can find prefix matches test 2', () => { - expect(getMatches('Cha')).toEqual([new Match(4, 3 / 7)]); + expect(getMatches('Cha')).toEqual([new Match(4, 3 / 7)]); }); test('can find prefix matches test 3', () => { - expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); + expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); }); test('can find suffix matches test 1', () => { - expect(getMatches('lice')).toEqual([new Match(0, 4 / 5)]); + expect(getMatches('lice')).toEqual([new Match(0, 4 / 5)]); }); test('can find suffix matches test 2', () => { - expect(getMatches('ol')).toEqual([new Match(3, 2 / 5)]); + expect(getMatches('ol')).toEqual([new Match(3, 2 / 5)]); }); test('can find infix matches test 1', () => { - expect(getMatches('arl')).toEqual([new Match(2, 3 / 6), new Match(4, 3 / 7)]); + expect(getMatches('arl')).toEqual([new Match(2, 3 / 6), new Match(4, 3 / 7)]); }); test('can find infix matches test 2', () => { - expect(getMatches('li')).toEqual( - [new Match(0, 2 / 5), new Match(4, 2 / 7)]); + expect(getMatches('li')).toEqual([new Match(0, 2 / 5), new Match(4, 2 / 7)]); }); test('can find infix matches test 3', () => { - expect(getMatches('l')).toEqual( - [new Match(0, 1 / 5), new Match(2, 1 / 6), new Match(3, 1 / 5), new Match(4, 1 / 7)]); + expect(getMatches('l')).toEqual([new Match(0, 1 / 5), new Match(2, 1 / 6), new Match(3, 1 / 5), new Match(4, 1 / 7)]); }); test('empty query returns no matches', () => { - expect(getMatches('')).toEqual([]); + expect(getMatches('')).toEqual([]); }); test('null query returns no matches', () => { - expect(getMatches(null!)).toEqual([]); + expect(getMatches(null!)).toEqual([]); }); test('undefined query returns no matches', () => { - expect(getMatches(undefined!)).toEqual([]); + expect(getMatches(undefined!)).toEqual([]); }); - +// todo: test substring that is not present diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 20d3458..2fe8513 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -10,118 +10,111 @@ import { SuffixArray } from './suffix-array.js'; // todo: make sure the terms don't have the separating character. Move to a suffix array config. export class SuffixArraySearcher implements StringSearcher { - - private readonly separator = 'µ'; - private str: string; - private suffixArray: Int32Array; - private indexToTermIndex: Int32Array; - private termLengths: Int32Array; - - public constructor() { - this.str = ''; - this.suffixArray = new Int32Array(0); - this.indexToTermIndex = new Int32Array(0); - this.termLengths = new Int32Array(0); + private readonly separator = 'µ'; + private str: string; + private suffixArray: Int32Array; + private indexToTermIndex: Int32Array; + private termLengths: Int32Array; + + public constructor() { + this.str = ''; + this.suffixArray = new Int32Array(0); + this.indexToTermIndex = new Int32Array(0); + this.termLengths = new Int32Array(0); + } + + index(terms: string[]): Meta { + const start = performance.now(); + this.str = this.separator + terms.join(this.separator) + this.separator; + this.suffixArray = SuffixArray.create(this.str); + this.indexToTermIndex = new Int32Array(this.suffixArray.length); + this.termLengths = new Int32Array(terms.length); + + let i = 0; + for (let j = 0; j < terms.length; j++) { + this.termLengths[j] = terms[j].length; + for (let k = 0; k <= terms[j].length; k++) { + this.indexToTermIndex[i++] = j; + } } - index(terms: string[]): Meta { - const start = performance.now(); - this.str = this.separator + terms.join(this.separator) + this.separator; - this.suffixArray = SuffixArray.create(this.str); - this.indexToTermIndex = new Int32Array(this.suffixArray.length); - this.termLengths = new Int32Array(terms.length); - - let i = 0; - for (let j = 0; j < terms.length; j++) { - this.termLengths[j] = terms[j].length; - for (let k = 0; k <= terms[j].length; k++) { - this.indexToTermIndex[i++] = j; - } - } - - this.indexToTermIndex[i++] = -1; - const duration = Math.round(performance.now() - start); - - const meta = new Meta(); - meta.add('suffixArraySearcherIndexing', duration); - return meta; - } + this.indexToTermIndex[i++] = -1; + const duration = Math.round(performance.now() - start); + + const meta = new Meta(); + meta.add('suffixArraySearcherIndexing', duration); + return meta; + } - getMatches(query: Query): Result { - if (query.string == null || query.string === '') { - return new Result([], query, new Meta()); - } - - // todo prefix: pass query.string modified - const [start, end] = this.GetPositionsInSuffixArray(query.string); - // todo: refactor such that end is included. - const matchedTermIds = new Int32Array(end - start); - - let i = 0; - for (let j = start; j < end; j++) { - const termIndex = this.indexToTermIndex[this.suffixArray[j]]; - matchedTermIds[i++] = termIndex; - } - - const matches: Match[] = []; - - let quality = 0; - for (let k = 0; k < matchedTermIds.length; k++) { - quality = this.computeQuality(query.string.length, this.termLengths[matchedTermIds[k]]); - if (quality > query.minQuality) { - matches.push(new Match(matchedTermIds[k], quality)); - } - } - - // todo: remove duplicate matches, measure performance - return new Result(matches, query, new Meta()); + getMatches(query: Query): Result { + if (query.string == null || query.string === '') { + return new Result([], query, new Meta()); } - private computeQuality(queryLength: number, termLength: number): number { - return queryLength / termLength; + // todo prefix: pass query.string modified + const [start, end] = this.GetPositionsInSuffixArray(query.string); + // todo: refactor such that end is included. + const matchedTermIds = new Int32Array(end - start); + + let i = 0; + for (let j = start; j < end; j++) { + const termIndex = this.indexToTermIndex[this.suffixArray[j]]; + matchedTermIds[i++] = termIndex; } - private GetPositionsInSuffixArray(substring: string): number[] { - let l = 0; - let r = this.suffixArray.length; - let mid = 0; - - while (l < r) { - mid = Math.floor((l + r) / 2); - - if (StringComparison.compareOrdinal( - substring, 0, this.str, this.suffixArray[mid], substring.length) > 0) { - l = mid + 1; - } - else { - r = mid; - } - } - - const start = l; - r = this.suffixArray.length; - - while (l < r) { - mid = Math.floor((l + r) / 2); - if (StringComparison.compareOrdinal( - substring, 0, this.str, this.suffixArray[mid], substring.length) == 0) { - l = mid + 1; - } - else { - r = mid; - } - } - - return [start, r]; + const matches: Match[] = []; + + let quality = 0; + for (let k = 0; k < matchedTermIds.length; k++) { + quality = this.computeQuality(query.string.length, this.termLengths[matchedTermIds[k]]); + if (quality > query.minQuality) { + matches.push(new Match(matchedTermIds[k], quality)); + } } + // todo: remove duplicate matches, measure performance + return new Result(matches, query, new Meta()); + } + + private computeQuality(queryLength: number, termLength: number): number { + return queryLength / termLength; + } - save(memento: Memento): void { - throw new Error('Method not implemented.'); + private GetPositionsInSuffixArray(substring: string): number[] { + let l = 0; + let r = this.suffixArray.length; + let mid = 0; + + while (l < r) { + mid = Math.floor((l + r) / 2); + + if (StringComparison.compareOrdinal(substring, 0, this.str, this.suffixArray[mid], substring.length) > 0) { + l = mid + 1; + } else { + r = mid; + } } - load(memento: Memento): void { - throw new Error('Method not implemented.'); + const start = l; + r = this.suffixArray.length; + + while (l < r) { + mid = Math.floor((l + r) / 2); + if (StringComparison.compareOrdinal(substring, 0, this.str, this.suffixArray[mid], substring.length) == 0) { + l = mid + 1; + } else { + r = mid; + } } -} \ No newline at end of file + return [start, r]; + } + + save(memento: Memento): void { + throw new Error('Method not implemented.'); + } + + load(memento: Memento): void { + throw new Error('Method not implemented.'); + } +} diff --git a/src/suffix-array-searchers/suffix-array.test.ts b/src/suffix-array-searchers/suffix-array.test.ts index a5c3798..9b4ec0c 100644 --- a/src/suffix-array-searchers/suffix-array.test.ts +++ b/src/suffix-array-searchers/suffix-array.test.ts @@ -1,28 +1,28 @@ import { SuffixArray } from './suffix-array.js'; test('can create suffix array test 1', () => { - const result = SuffixArray.create('banana$'); - expect(result).toEqual(new Int32Array([6, 5, 3, 1, 0, 4, 2])); + const result = SuffixArray.create('banana$'); + expect(result).toEqual(new Int32Array([6, 5, 3, 1, 0, 4, 2])); }); test('can create suffix array test 2', () => { - const result = SuffixArray.create('mississippi'); - expect(result).toEqual(new Int32Array([10, 7, 4, 1, 0, 9, 8, 6, 3, 5, 2])); + const result = SuffixArray.create('mississippi'); + expect(result).toEqual(new Int32Array([10, 7, 4, 1, 0, 9, 8, 6, 3, 5, 2])); }); test('can create suffix array of empty string', () => { - const result = SuffixArray.create(''); - expect(result).toEqual(new Int32Array([])); + const result = SuffixArray.create(''); + expect(result).toEqual(new Int32Array([])); }); test('can not create suffix array of null', () => { - expect(() => { - SuffixArray.create(null!); - }).toThrow(); + expect(() => { + SuffixArray.create(null!); + }).toThrow(); }); test('can not create suffix array of undefined', () => { - expect(() => { - SuffixArray.create(undefined!); - }).toThrow(); -}); \ No newline at end of file + expect(() => { + SuffixArray.create(undefined!); + }).toThrow(); +}); diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts index 6813060..0ec437c 100644 --- a/src/suffix-array-searchers/suffix-array.ts +++ b/src/suffix-array-searchers/suffix-array.ts @@ -1,155 +1,146 @@ import { StringComparison } from './string-comparison.js'; export class SuffixArray { - - private readonly eoc: number = 2147483647; - private m_str: string; - private m_sa: Int32Array; - private m_isa: Int32Array; - private m_chainHeadsDict: Map; - private m_chainStack: Chain[] = []; - private m_subChains: Chain[] = []; - private m_nextRank: number = 1; - - public static create(str: string): Int32Array { - if (str == null) { - throw new Error('Input string cannot be null.'); - } - const suffixArray: SuffixArray = new SuffixArray(str); - suffixArray.FormInitialChains(); - suffixArray.BuildSufixArray(); - return suffixArray.m_sa; + private readonly eoc: number = 2147483647; + private m_str: string; + private m_sa: Int32Array; + private m_isa: Int32Array; + private m_chainHeadsDict: Map; + private m_chainStack: Chain[] = []; + private m_subChains: Chain[] = []; + private m_nextRank: number = 1; + + public static create(str: string): Int32Array { + if (str == null) { + throw new Error('Input string cannot be null.'); } - - private constructor(private readonly str: string) { - const l = str.length; - this.m_str = str; - this.m_sa = new Int32Array(l); - this.m_isa = new Int32Array(l); - this.m_chainHeadsDict = new Map(); + const suffixArray: SuffixArray = new SuffixArray(str); + suffixArray.FormInitialChains(); + suffixArray.BuildSufixArray(); + return suffixArray.m_sa; + } + + private constructor(private readonly str: string) { + const l = str.length; + this.m_str = str; + this.m_sa = new Int32Array(l); + this.m_isa = new Int32Array(l); + this.m_chainHeadsDict = new Map(); + } + + private FormInitialChains(): void { + this.FindInitialChains(); + this.SortAndPushSubchains(); + } + + private FindInitialChains(): void { + for (let i = 0; i < this.m_str.length; i++) { + const char_code = this.m_str.charCodeAt(i); + const chain_head_index = this.m_chainHeadsDict.get(char_code); + if (chain_head_index !== undefined) { + this.m_isa[i] = chain_head_index; + } else { + this.m_isa[i] = this.eoc; + } + this.m_chainHeadsDict.set(char_code, i); } - private FormInitialChains(): void { - this.FindInitialChains(); - this.SortAndPushSubchains(); + for (const headIndex of this.m_chainHeadsDict.values()) { + const newChain = new Chain(this.m_str, headIndex, 1); + this.m_subChains.push(newChain); } + } - private FindInitialChains(): void { - - for (let i = 0; i < this.m_str.length; i++) { - const char_code = this.m_str.charCodeAt(i); - const chain_head_index = this.m_chainHeadsDict.get(char_code); - if (chain_head_index !== undefined) { - this.m_isa[i] = chain_head_index; - } - else { - this.m_isa[i] = this.eoc; - } - this.m_chainHeadsDict.set(char_code, i); - } - - for (const headIndex of this.m_chainHeadsDict.values()) { - const newChain = new Chain(this.m_str, headIndex, 1); - this.m_subChains.push(newChain); - } - } + private BuildSufixArray(): void { + while (this.m_chainStack.length > 0) { + const chain: Chain = this.m_chainStack.pop() as Chain; - private BuildSufixArray(): void { - while (this.m_chainStack.length > 0) { - const chain: Chain = this.m_chainStack.pop() as Chain; - - if (this.m_isa[chain.head] === this.eoc) { - this.RankSuffix(chain.head); - } - else { - this.RefineChainWithInductionSorting(chain); - } - } + if (this.m_isa[chain.head] === this.eoc) { + this.RankSuffix(chain.head); + } else { + this.RefineChainWithInductionSorting(chain); + } } - - private RankSuffix(index: number): void { - this.m_isa[index] = -this.m_nextRank; - this.m_sa[this.m_nextRank - 1] = index; - this.m_nextRank++; + } + + private RankSuffix(index: number): void { + this.m_isa[index] = -this.m_nextRank; + this.m_sa[this.m_nextRank - 1] = index; + this.m_nextRank++; + } + + private RefineChainWithInductionSorting(chain: Chain): void { + const notedSuffixes: SuffixRank[] = []; + this.m_chainHeadsDict.clear(); + this.m_subChains = []; + + while (chain.head !== this.eoc) { + const nextIndex: number = this.m_isa[chain.head]; + if (chain.head + chain.length > this.m_str.length - 1) { + this.RankSuffix(chain.head); + } else if (this.m_isa[chain.head + chain.length] < 0) { + const sr: SuffixRank = new SuffixRank(chain.head, -this.m_isa[chain.head + chain.length]); + notedSuffixes.push(sr); + } else { + this.ExtendChain(chain); + } + chain.head = nextIndex; } - private RefineChainWithInductionSorting(chain: Chain): void { - const notedSuffixes: SuffixRank[] = []; - this.m_chainHeadsDict.clear(); - this.m_subChains = []; - - while (chain.head !== this.eoc) { - const nextIndex: number = this.m_isa[chain.head]; - if (chain.head + chain.length > this.m_str.length - 1) { - this.RankSuffix(chain.head); - } - else if (this.m_isa[chain.head + chain.length] < 0) { - const sr: SuffixRank = new SuffixRank(chain.head, -this.m_isa[chain.head + chain.length]); - notedSuffixes.push(sr); - } - else { - this.ExtendChain(chain); - } - chain.head = nextIndex; - } - - this.SortAndPushSubchains(); - this.SortAndRankNotedSuffixes(notedSuffixes); + this.SortAndPushSubchains(); + this.SortAndRankNotedSuffixes(notedSuffixes); + } + + private ExtendChain(chain: Chain): void { + const sym: number = this.m_str.charCodeAt(chain.head + chain.length); + if (this.m_chainHeadsDict.has(sym)) { + this.m_isa[this.m_chainHeadsDict.get(sym) as number] = chain.head; + this.m_isa[chain.head] = this.eoc; + } else { + this.m_isa[chain.head] = this.eoc; + const newChain: Chain = new Chain(this.m_str, chain.head, chain.length + 1); + this.m_subChains.push(newChain); } - private ExtendChain(chain: Chain): void { - const sym: number = this.m_str.charCodeAt(chain.head + chain.length); - if (this.m_chainHeadsDict.has(sym)) { - this.m_isa[this.m_chainHeadsDict.get(sym) as number] = chain.head; - this.m_isa[chain.head] = this.eoc; - } - else { - this.m_isa[chain.head] = this.eoc; - const newChain: Chain = new Chain(this.m_str, chain.head, chain.length + 1); - this.m_subChains.push(newChain); - } - - this.m_chainHeadsDict.set(sym, chain.head); - } + this.m_chainHeadsDict.set(sym, chain.head); + } - private SortAndRankNotedSuffixes(notedSuffixes: SuffixRank[]): void { - notedSuffixes.sort((a, b) => { - return a.rank - b.rank; - }); + private SortAndRankNotedSuffixes(notedSuffixes: SuffixRank[]): void { + notedSuffixes.sort((a, b) => { + return a.rank - b.rank; + }); - for (let i = 0; i < notedSuffixes.length; i++) { - this.RankSuffix(notedSuffixes[i].head); - } + for (let i = 0; i < notedSuffixes.length; i++) { + this.RankSuffix(notedSuffixes[i].head); } - - private SortAndPushSubchains(): void { - this.m_subChains.sort((c1: Chain, c2: Chain): number => { - const len = Math.min(c1.length, c2.length); - return StringComparison.compareOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); - - }); - for (let i = this.m_subChains.length - 1; i >= 0; i--) { - this.m_chainStack.push(this.m_subChains[i]); - } + } + + private SortAndPushSubchains(): void { + this.m_subChains.sort((c1: Chain, c2: Chain): number => { + const len = Math.min(c1.length, c2.length); + return StringComparison.compareOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); + }); + for (let i = this.m_subChains.length - 1; i >= 0; i--) { + this.m_chainStack.push(this.m_subChains[i]); } + } } class SuffixRank { - public constructor( - public readonly head: number, - public readonly rank: number - ) { } + public constructor( + public readonly head: number, + public readonly rank: number + ) {} } class Chain { - - public readonly m_str: string; - public head: number; - public length: number; - - public constructor(m_str: string, head: number, length: number) { - this.m_str = m_str; - this.head = head; - this.length = length; - } -} \ No newline at end of file + public readonly m_str: string; + public head: number; + public length: number; + + public constructor(m_str: string, head: number, length: number) { + this.m_str = m_str; + this.head = head; + this.length = length; + } +} From 236de057fefc8edd6557512cdc474a6d08800637 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 17 Oct 2025 11:22:13 +0200 Subject: [PATCH 016/105] prettier --- .vscode/settings.json | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/.vscode/settings.json b/.vscode/settings.json index b35c98c..d0f2161 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -5,9 +5,7 @@ "editor.formatOnSave": false }, "editor.defaultFormatter": "esbenp.prettier-vscode", - "editor.rulers": [ - 120 - ], + "editor.rulers": [120], "liveServer.settings.port": 5501, "[json]": { "editor.defaultFormatter": "esbenp.prettier-vscode" @@ -15,4 +13,4 @@ "[plaintext]": { "editor.renderControlCharacters": false } -} \ No newline at end of file +} From c6a7af0e2f5402afb2addbd74255ba429bc3a379 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 17 Oct 2025 14:17:20 +0200 Subject: [PATCH 017/105] chore: update prettier --- .vscode/settings.json | 6 ++++-- package-lock.json | 8 ++++---- package.json | 2 +- 3 files changed, 9 insertions(+), 7 deletions(-) diff --git a/.vscode/settings.json b/.vscode/settings.json index d0f2161..b35c98c 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -5,7 +5,9 @@ "editor.formatOnSave": false }, "editor.defaultFormatter": "esbenp.prettier-vscode", - "editor.rulers": [120], + "editor.rulers": [ + 120 + ], "liveServer.settings.port": 5501, "[json]": { "editor.defaultFormatter": "esbenp.prettier-vscode" @@ -13,4 +15,4 @@ "[plaintext]": { "editor.renderControlCharacters": false } -} +} \ No newline at end of file diff --git a/package-lock.json b/package-lock.json index 0767e04..375c4c3 100644 --- a/package-lock.json +++ b/package-lock.json @@ -18,7 +18,7 @@ "file-saver": "^2.0.5", "jest": "^29.7.0", "microbundle": "^0.15.1", - "prettier": "^3.3.3", + "prettier": "^3.6.2", "ts-jest": "^29.2.5", "tsx": "^4.19.2", "typedoc": "^0.26.11", @@ -9715,9 +9715,9 @@ } }, "node_modules/prettier": { - "version": "3.3.3", - "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.3.3.tgz", - "integrity": "sha512-i2tDNA0O5IrMO757lfrdQZCc2jPNDVntV0m/+4whiDfWaTKfMNgR7Qz0NAeGz/nRqF4m5/6CLzbP4/liHt12Ew==", + "version": "3.6.2", + "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.6.2.tgz", + "integrity": "sha512-I7AIg5boAr5R0FFtJ6rCfD+LFsWHp81dolrFD8S79U9tb8Az2nGrJncnMSnys+bpQJfRUzqs9hnA81OAA3hCuQ==", "dev": true, "license": "MIT", "bin": { diff --git a/package.json b/package.json index 97040fa..ea2dbc8 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "file-saver": "^2.0.5", "jest": "^29.7.0", "microbundle": "^0.15.1", - "prettier": "^3.3.3", + "prettier": "^3.6.2", "ts-jest": "^29.2.5", "tsx": "^4.19.2", "typedoc": "^0.26.11", From 52be726739df64ddc8bbe51829c366faffcc2756 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 17 Oct 2025 14:28:07 +0200 Subject: [PATCH 018/105] test: no matches for non-existing substring --- src/suffix-array-searchers/suffix-array-searcher.test.ts | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts index 9ccbd0f..8552a9b 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.test.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -52,6 +52,10 @@ test('can find infix matches test 3', () => { expect(getMatches('l')).toEqual([new Match(0, 1 / 5), new Match(2, 1 / 6), new Match(3, 1 / 5), new Match(4, 1 / 7)]); }); +test('no matches for non-existing substring', () => { + expect(getMatches('xyz')).toEqual([]); +}); + test('empty query returns no matches', () => { expect(getMatches('')).toEqual([]); }); @@ -63,5 +67,3 @@ test('null query returns no matches', () => { test('undefined query returns no matches', () => { expect(getMatches(undefined!)).toEqual([]); }); - -// todo: test substring that is not present From 4a4c27bbb5bcb2009907667990373e9d4e084de0 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 18 Oct 2025 12:57:08 +0200 Subject: [PATCH 019/105] feat: character normalizer and tests --- .../character-normalizer.test.ts | 82 +++++++++++++ src/normalization/character-normalizer.ts | 116 ++++++++++++++++++ 2 files changed, 198 insertions(+) create mode 100644 src/normalization/character-normalizer.test.ts create mode 100644 src/normalization/character-normalizer.ts diff --git a/src/normalization/character-normalizer.test.ts b/src/normalization/character-normalizer.test.ts new file mode 100644 index 0000000..0e0f8a7 --- /dev/null +++ b/src/normalization/character-normalizer.test.ts @@ -0,0 +1,82 @@ +import { CharacterNormalizer } from './character-normalizer.js'; +import { StringUtilities } from '../commons/string-utilities.js'; + +const spaceEquivalentCharacters = new Set(['_', '-', '–', '/', ',', '\t']); +const normalizer = new CharacterNormalizer( + c => spaceEquivalentCharacters.has(c), c => StringUtilities.isAlphanumeric(c)); + +test('can normalize empty string', () => { + expect(normalizer.normalize('')).toBe(''); +}); + +test('can normalize single lower case letter', () => { + expect(normalizer.normalize('h')).toBe('h'); +}); + +test('can normalize single upper case letter', () => { + expect(normalizer.normalize('H')).toBe('h'); +}); + +test('can normalize two letters', () => { + expect(normalizer.normalize('He')).toBe('he'); +}); + +test('can normalize three letters', () => { + expect(normalizer.normalize('Hel')).toBe('hel'); +}); + +test('can normalize single word', () => { + expect(normalizer.normalize('HELLO')).toBe('hello'); +}); + +test('can normalize two words', () => { + expect(normalizer.normalize('Hello World')).toBe('hello world'); +}); + +test('can normalize two words with forbidden character', () => { + expect(normalizer.normalize('Hello$World')).toBe('helloworld'); +}); + +test('can normalize single word with spaces', () => { + expect(normalizer.normalize(' hello ')).toBe('hello'); +}); + +test('can normalize two words with spaces', () => { + expect(normalizer.normalize(' hello world ')).toBe('hello world'); +}); + +test('can normalize emoji', () => { + expect(normalizer.normalize('👩‍💻')).toBe(''); +}); + +test('can normalize string with emoji', () => { + expect(normalizer.normalize('hello 👩‍💻 world')).toBe('hello world'); +}); + +test('can normalize string with emoji in the beginning', () => { + expect(normalizer.normalize('👩‍💻helloworld')).toBe('helloworld'); +}); + +test('can normalize string with emoji in the middle', () => { + expect(normalizer.normalize('hello👩‍💻world')).toBe('helloworld'); +}); + +test('can normalize string with emoji in the end', () => { + expect(normalizer.normalize('helloworld👩‍💻')).toBe('helloworld'); +}); + +test('can normalize string with several emojis', () => { + expect(normalizer.normalize('👋hello👋👋world👩‍💻')).toBe('helloworld'); +}); + +test('can normalize string with padding characters', () => { + expect(normalizer.normalize('%hello!world$')).toBe('helloworld'); +}); + +test('can normalize string with space equivalent characters', () => { + expect(normalizer.normalize('Lorem-ipsum_dolor,sit amet')).toBe('lorem ipsum dolor sit amet'); +}); + +test('can normalize string with space equivalent characters and additional spaces', () => { + expect(normalizer.normalize('Lorem- ipsum _dolor , sit amet')).toBe('lorem ipsum dolor sit amet'); +}); diff --git a/src/normalization/character-normalizer.ts b/src/normalization/character-normalizer.ts new file mode 100644 index 0000000..f98aa60 --- /dev/null +++ b/src/normalization/character-normalizer.ts @@ -0,0 +1,116 @@ +import { Meta } from '../interfaces/meta.js'; +import { NormalizationResult } from './normalization-result.js'; +import { Normalizer } from '../interfaces/normalizer.js'; +import { StringUtilities } from '../commons/string-utilities.js'; + +/** + * Normalizes every character according to the configuration. + */ +export class CharacterNormalizer implements Normalizer { + /** + * A function that determines whether a character is treated as a space. + */ + private readonly treatCharacterAsSpace: (c: string) => boolean; + + /** + * A function that determines whether a character is allowed. Surrogate characters are disallowed by default. + */ + private readonly allowCharacter: (c: string) => boolean; + + /** + * The number of encountered surrogate characters in a bulk normalization. + */ + private numberOfSurrogateCharacters: number = 0; + + /** + * Creates a new instance of the CharacterNormalizer class. + * @param treatCharacterAsSpace A function that determines whether a character is treated as a space. + * @param allowCharacter A function that determines whether a character is allowed. + */ + public constructor( + treatCharacterAsSpace: (c: string) => boolean, + allowCharacter: (c: string) => boolean) { + + this.treatCharacterAsSpace = treatCharacterAsSpace; + this.allowCharacter = allowCharacter; + } + + /** + * {@inheritDoc Normalizer.normalize} + */ + public normalize(input: string): string { + const normalized: string[] = new Array(input.length); + let j = 0; + + let previousIsSkippedEmptyChar = false; + let properCharacterAdded = false; + + for (let i = 0, l = input.length; i < l; i++) { + const normalizedChar = this.getNormalizedCharacter(input[i]); + + if (normalizedChar === '') { + continue; + } + + if (normalizedChar === ' ') { + previousIsSkippedEmptyChar = true; + } else { + if (previousIsSkippedEmptyChar && properCharacterAdded) { + normalized[j++] = ' '; + } + + normalized[j++] = normalizedChar; + properCharacterAdded = true; + previousIsSkippedEmptyChar = false; + } + } + + if (!properCharacterAdded) { + return ''; + } + + return normalized.join(''); + } + + /** + * Normalizes the given character. Space equivalent characters are replaced by a space. Characters + * that are not allowed are removed, in addition to surrogate characters and padding characters. + * @param character The character to normalize. + * @returns The normalized character. + */ + private getNormalizedCharacter(character: string): string { + if (character === ' ' || this.treatCharacterAsSpace(character)) { + return ' '; + } + + if (this.isSurrogate(character) || !this.allowCharacter(character)) { + return ''; + } + + return character.toLowerCase(); + } + + /** + * Checks if the given character is part of a surrogate pair. + * @param character The character to check. + * @returns True if the character is part of a surrogate pair, false otherwise. + */ + private isSurrogate(character: string): boolean { + if (StringUtilities.isSurrogate(character)) { + this.numberOfSurrogateCharacters++; + return true; + } + return false; + } + + /** + * {@inheritDoc Normalizer.normalizeBulk} + */ + public normalizeBulk(input: string[]): NormalizationResult { + this.numberOfSurrogateCharacters = 0; + const normalized = input.map((s) => this.normalize(s)); + const meta = new Meta(); + meta.add('numberOfSurrogateCharacters', this.numberOfSurrogateCharacters); + return new NormalizationResult(normalized, meta); + } +} From e7a29ca7cbdcad5ef8f268c13f4ed7f2860d64e6 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 15:26:37 +0200 Subject: [PATCH 020/105] notes --- notes.md | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/notes.md b/notes.md index b654e79..8a51ce0 100644 --- a/notes.md +++ b/notes.md @@ -81,4 +81,20 @@ Attribution (Chat GPT) Todo ==== -Test empty term. \ No newline at end of file +Test empty term. +Test second suffix array implementation. + +First: +suffixArraySearcherIndexing: 3663 +suffixArraySearcherIndexing: 3638 +suffixArraySearcherIndexing: 3655 + + +Changelog: + +paddingLeft, paddingRight and paddingMiddle were moved from the NormalizerConfig to the +NgramNormalizerConfig. + +// '$$', +// '!', +// '!$$', \ No newline at end of file From 7189bc77710719aeb83966f94e0d0a916b274ef5 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 15:26:55 +0200 Subject: [PATCH 021/105] notes --- notes.md | 1 - 1 file changed, 1 deletion(-) diff --git a/notes.md b/notes.md index 8a51ce0..b2b841a 100644 --- a/notes.md +++ b/notes.md @@ -81,7 +81,6 @@ Attribution (Chat GPT) Todo ==== -Test empty term. Test second suffix array implementation. First: From 3436e6717e3312bc5dcefef900b0f777436f9b38 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 15:31:45 +0200 Subject: [PATCH 022/105] feat: refactor normalization --- .../character-normalizer.test.ts | 2 +- .../default-normalizer-non-latin.test.ts | 3 - src/normalization/default-normalizer.test.ts | 99 ++++--------- src/normalization/default-normalizer.ts | 5 +- src/normalization/index.ts | 1 + .../ngram-full-normalizer.test.ts | 49 ++++++ src/normalization/ngram-normalizer.test.ts | 69 +-------- src/normalization/ngram-normalizer.ts | 140 +++--------------- src/normalization/normalizer-config.ts | 25 +--- .../normalizing-searcher.test.ts | 10 +- .../suffix-array-searcher.test.ts | 2 +- .../suffix-array-searcher.ts | 15 +- 12 files changed, 132 insertions(+), 288 deletions(-) create mode 100644 src/normalization/ngram-full-normalizer.test.ts diff --git a/src/normalization/character-normalizer.test.ts b/src/normalization/character-normalizer.test.ts index 0e0f8a7..b6a5770 100644 --- a/src/normalization/character-normalizer.test.ts +++ b/src/normalization/character-normalizer.test.ts @@ -79,4 +79,4 @@ test('can normalize string with space equivalent characters', () => { test('can normalize string with space equivalent characters and additional spaces', () => { expect(normalizer.normalize('Lorem- ipsum _dolor , sit amet')).toBe('lorem ipsum dolor sit amet'); -}); +}); \ No newline at end of file diff --git a/src/normalization/default-normalizer-non-latin.test.ts b/src/normalization/default-normalizer-non-latin.test.ts index 6eb0410..851762d 100644 --- a/src/normalization/default-normalizer-non-latin.test.ts +++ b/src/normalization/default-normalizer-non-latin.test.ts @@ -2,9 +2,6 @@ import { DefaultNormalizer } from './default-normalizer.js'; import { NormalizerConfig } from './normalizer-config.js'; const config = NormalizerConfig.createDefaultConfig(); -config.paddingLeft = ''; -config.paddingRight = ''; -config.paddingMiddle = ' '; config.allowCharacter = (_c: string) => true; const normalizer = DefaultNormalizer.create(config); diff --git a/src/normalization/default-normalizer.test.ts b/src/normalization/default-normalizer.test.ts index 914c2a0..99db49c 100644 --- a/src/normalization/default-normalizer.test.ts +++ b/src/normalization/default-normalizer.test.ts @@ -1,91 +1,46 @@ import { DefaultNormalizer } from './default-normalizer.js'; import { NormalizerConfig } from './normalizer-config.js'; -const normalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); - const config = NormalizerConfig.createDefaultConfig(); -config.paddingLeft = ''; -config.paddingRight = ''; -config.paddingMiddle = ' '; -const noPaddingNormalizer = DefaultNormalizer.create(config); - -test('normalization test 1', () => { - expect(normalizer.normalize('Mikael Håkansson')).toBe('$$mikael!$$haakansson!'); -}); - -test('normalization test 2', () => { - expect(normalizer.normalize('Klara Åberg')).toBe('$$klara!$$aaberg!'); -}); - -test('normalization test 3', () => { - expect(normalizer.normalize('Andel Hadžić')).toBe('$$andel!$$hadzic!'); -}); - -test('normalization test 4', () => { - expect(normalizer.normalize('Lenni Gilliéron')).toBe('$$lenni!$$gillieron!'); -}); - -test('normalization test 5', () => { - expect(normalizer.normalize('Julieta Nieto Ríos')).toBe('$$julieta!$$nieto!$$rios!'); -}); - -test('normalization test 6', () => { - expect(normalizer.normalize('Æstrid Ærenlund')).toBe('$$aestrid!$$aerenlund!'); -}); - -test('normalization test 7', () => { - expect(normalizer.normalize('Ømer Østergaard')).toBe('$$omer!$$ostergaard!'); -}); - -test('normalization test 8', () => { - expect(normalizer.normalize('Sơn Lâm Đặng')).toBe('$$son!$$lam!$$dang!'); -}); - -test('normalization test 9', () => { - expect(normalizer.normalize('Thanh Việt Đoàn')).toBe('$$thanh!$$viet!$$doan!'); -}); - -test('normalization test 10', () => { - expect(normalizer.normalize('Thiên Duyên Tô')).toBe('$$thien!$$duyen!$$to!'); -}); +const normalizer = DefaultNormalizer.create(config); test('German Eszett', () => { - expect(noPaddingNormalizer.normalize('Fußball')).toBe('fussball'); + expect(normalizer.normalize('Fußball')).toBe('fussball'); }); test('German umlauts', () => { - expect(noPaddingNormalizer.normalize('ä ö ü')).toBe('ae oe ue'); + expect(normalizer.normalize('ä ö ü')).toBe('ae oe ue'); }); test('variatons of a', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect( - noPaddingNormalizer.normalize( - 'Áá Àà Ââ Ǎǎ Ăă Ãã Ảả Ȧȧ Ạạ Ää Åå Ḁḁ Āā Ąą ᶏ Ⱥⱥ Ȁȁ Ấấ Ầầ Ẫẫ Ẩẩ Ậậ Ắắ Ằằ Ẵẵ Ẳẳ Ặặ Ǻǻ Ǡǡ Ǟǟ Ȃȃ Ɑɑ ᴀ Ɐɐ ɒ Aa' + - ' Ææ ᴁ ᴭ ᵆ Ǽǽ Ǣǣ ᴂ' - ) - ).toBe( - 'aa aa aa aa aa aa aa aa aa aeae aaaa aa aa aa a aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa a aa a aa' + - ' aeae ae ae ae aeae aeae ae' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect( + normalizer.normalize( + 'Áá Àà Ââ Ǎǎ Ăă Ãã Ảả Ȧȧ Ạạ Ää Åå Ḁḁ Āā Ąą ᶏ Ⱥⱥ Ȁȁ Ấấ Ầầ Ẫẫ Ẩẩ Ậậ Ắắ Ằằ Ẵẵ Ẳẳ Ặặ Ǻǻ Ǡǡ Ǟǟ Ȃȃ Ɑɑ ᴀ Ɐɐ ɒ Aa' + + ' Ææ ᴁ ᴭ ᵆ Ǽǽ Ǣǣ ᴂ' + ) + ).toBe( + 'aa aa aa aa aa aa aa aa aa aeae aaaa aa aa aa a aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa a aa a aa' + + ' aeae ae ae ae aeae aeae ae' + ); }); test('variatons of u', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect( - noPaddingNormalizer.normalize( - 'Úú Ùù Ŭŭ Ûû Ǔǔ Ůů Üü Ǘǘ Ǜǜ Ǚǚ Ǖǖ Űű Ũũ Ṹṹ Ųų Ūū Ṻṻ Ủủ Ȕȕ Ȗȗ Ưư Ứứ Ừừ Ữữ Ửử Ựự Ụụ Ṳṳ Ṷṷ Ṵṵ Ʉʉ Ʊʊ Ȣȣ ᵾ ᶙ ᴜ Uu' + - ' ᵫ ɯ' - ) - ).toBe( - 'uu uu uu uu uu uu ueue uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu ouou u u u uu' + - ' ue m' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect( + normalizer.normalize( + 'Úú Ùù Ŭŭ Ûû Ǔǔ Ůů Üü Ǘǘ Ǜǜ Ǚǚ Ǖǖ Űű Ũũ Ṹṹ Ųų Ūū Ṻṻ Ủủ Ȕȕ Ȗȗ Ưư Ứứ Ừừ Ữữ Ửử Ựự Ụụ Ṳṳ Ṷṷ Ṵṵ Ʉʉ Ʊʊ Ȣȣ ᵾ ᶙ ᴜ Uu' + + ' ᵫ ɯ' + ) + ).toBe( + 'uu uu uu uu uu uu ueue uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu ouou u u u uu' + + ' ue m' + ); }); test('variatons of s', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect(noPaddingNormalizer.normalize('ſ ẞß Śś Ṥṥ Ŝŝ Šš Ṧṧ Ṡṡẛ Şş Ṣṣ Ṩṩ Șș S̩s̩ ᵴ ᶊ ʂ ȿ ꜱ Ʃʃ Ss')).toBe( - 's ssss ss ss ss ss ss sss ss ss ss ss ss s s s s s ss ss' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect(normalizer.normalize('ſ ẞß Śś Ṥṥ Ŝŝ Šš Ṧṧ Ṡṡẛ Şş Ṣṣ Ṩṩ Șș S̩s̩ ᵴ ᶊ ʂ ȿ ꜱ Ʃʃ Ss')).toBe( + 's ssss ss ss ss ss ss sss ss ss ss ss ss s s s s s ss ss' + ); }); diff --git a/src/normalization/default-normalizer.ts b/src/normalization/default-normalizer.ts index 712245a..9b080d1 100644 --- a/src/normalization/default-normalizer.ts +++ b/src/normalization/default-normalizer.ts @@ -1,6 +1,6 @@ +import { CharacterNormalizer } from './character-normalizer.js'; import { GenericNormalizer } from './generic-normalizer.js'; import { MultiNormalizer } from './multi-normalizer.js'; -import { NgramNormalizer } from './ngram-normalizer.js'; import { Normalizer } from '../interfaces/normalizer.js'; import { NormalizerConfig } from './normalizer-config.js'; import { SanitizingNormalizer } from './sanitizing-normalizer.js'; @@ -21,7 +21,8 @@ export class DefaultNormalizer { variation.toLowerCase().normalize('NFKC') ); const normalizer3 = new GenericNormalizer((input: string): string => input.normalize('NFKD')); - const normalizer4 = new NgramNormalizer(normalizerConfig); + const normalizer4 = new CharacterNormalizer( + normalizerConfig.treatCharacterAsSpace, normalizerConfig.allowCharacter); const multiNormalizer = new MultiNormalizer([normalizer1, normalizer2, normalizer3, normalizer4]); return multiNormalizer; } diff --git a/src/normalization/index.ts b/src/normalization/index.ts index 5d5f47d..7be12ef 100644 --- a/src/normalization/index.ts +++ b/src/normalization/index.ts @@ -1,3 +1,4 @@ +export { CharacterNormalizer } from './character-normalizer.js'; export { DefaultNormalizer } from './default-normalizer.js'; export { GenericNormalizer } from './generic-normalizer.js'; export { LatinReplacements } from './latin-replacements.js'; diff --git a/src/normalization/ngram-full-normalizer.test.ts b/src/normalization/ngram-full-normalizer.test.ts new file mode 100644 index 0000000..05d1ab5 --- /dev/null +++ b/src/normalization/ngram-full-normalizer.test.ts @@ -0,0 +1,49 @@ +import { DefaultNormalizer } from './default-normalizer.js'; +import { MultiNormalizer } from './multi-normalizer.js'; +import { NgramNormalizer } from './ngram-normalizer.js'; +import { NormalizerConfig } from './normalizer-config.js'; + +const defaultNormalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); +const ngramNormalizer = NgramNormalizer.createDefault(); +const normalizer = new MultiNormalizer([defaultNormalizer, ngramNormalizer]); + +test('normalization test 1', () => { + expect(normalizer.normalize('Mikael Håkansson')).toBe('$$mikael!$$haakansson!'); +}); + +test('normalization test 2', () => { + expect(normalizer.normalize('Klara Åberg')).toBe('$$klara!$$aaberg!'); +}); + +test('normalization test 3', () => { + expect(normalizer.normalize('Andel Hadžić')).toBe('$$andel!$$hadzic!'); +}); + +test('normalization test 4', () => { + expect(normalizer.normalize('Lenni Gilliéron')).toBe('$$lenni!$$gillieron!'); +}); + +test('normalization test 5', () => { + expect(normalizer.normalize('Julieta Nieto Ríos')).toBe('$$julieta!$$nieto!$$rios!'); +}); + +test('normalization test 6', () => { + expect(normalizer.normalize('Æstrid Ærenlund')).toBe('$$aestrid!$$aerenlund!'); +}); + +test('normalization test 7', () => { + expect(normalizer.normalize('Ømer Østergaard')).toBe('$$omer!$$ostergaard!'); +}); + +test('normalization test 8', () => { + expect(normalizer.normalize('Sơn Lâm Đặng')).toBe('$$son!$$lam!$$dang!'); +}); + +test('normalization test 9', () => { + expect(normalizer.normalize('Thanh Việt Đoàn')).toBe('$$thanh!$$viet!$$doan!'); +}); + +test('normalization test 10', () => { + expect(normalizer.normalize('Thiên Duyên Tô')).toBe('$$thien!$$duyen!$$to!'); +}); + diff --git a/src/normalization/ngram-normalizer.test.ts b/src/normalization/ngram-normalizer.test.ts index 643e5f1..c51aad9 100644 --- a/src/normalization/ngram-normalizer.test.ts +++ b/src/normalization/ngram-normalizer.test.ts @@ -1,80 +1,23 @@ import { NgramNormalizer } from './ngram-normalizer.js'; -import { NormalizerConfig } from './normalizer-config.js'; -const config = NormalizerConfig.createDefaultConfig(); -config.paddingLeft = '$$'; -config.paddingRight = '!!'; -config.paddingMiddle = '%%'; -const normalizer = new NgramNormalizer(config); +const normalizer = new NgramNormalizer('$$', '!!', '%%'); test('can normalize empty string', () => { expect(normalizer.normalize('')).toBe(''); }); -test('can normalize single lower case letter', () => { +test('can normalize single letter', () => { expect(normalizer.normalize('h')).toBe('$$h!!'); }); -test('can normalize single upper case letter', () => { - expect(normalizer.normalize('H')).toBe('$$h!!'); -}); - -test('can normalize two letters', () => { - expect(normalizer.normalize('He')).toBe('$$he!!'); -}); - -test('can normalize three letters', () => { - expect(normalizer.normalize('Hel')).toBe('$$hel!!'); -}); - test('can normalize single word', () => { - expect(normalizer.normalize('HELLO')).toBe('$$hello!!'); + expect(normalizer.normalize('hello')).toBe('$$hello!!'); }); test('can normalize two words', () => { - expect(normalizer.normalize('Hello World')).toBe('$$hello%%world!!'); -}); - -test('can normalize single word with spaces', () => { - expect(normalizer.normalize(' hello ')).toBe('$$hello!!'); -}); - -test('can normalize two words with spaces', () => { - expect(normalizer.normalize(' hello world ')).toBe('$$hello%%world!!'); -}); - -test('can normalize emoji', () => { - expect(normalizer.normalize('👩‍💻')).toBe(''); -}); - -test('can normalize string with emoji', () => { - expect(normalizer.normalize('hello 👩‍💻 world')).toBe('$$hello%%world!!'); -}); - -test('can normalize string with emoji in the beginning', () => { - expect(normalizer.normalize('👩‍💻helloworld')).toBe('$$helloworld!!'); -}); - -test('can normalize string with emoji in the middle', () => { - expect(normalizer.normalize('hello👩‍💻world')).toBe('$$helloworld!!'); -}); - -test('can normalize string with emoji in the end', () => { - expect(normalizer.normalize('helloworld👩‍💻')).toBe('$$helloworld!!'); -}); - -test('can normalize string with several emojis', () => { - expect(normalizer.normalize('👋hello👋👋world👩‍💻')).toBe('$$helloworld!!'); -}); - -test('can normalize string with padding characters', () => { - expect(normalizer.normalize('%hello!world$')).toBe('$$helloworld!!'); -}); - -test('can normalize string with space equivalent characters', () => { - expect(normalizer.normalize('Lorem-ipsum_dolor,sit amet')).toBe('$$lorem%%ipsum%%dolor%%sit%%amet!!'); + expect(normalizer.normalize('hello world')).toBe('$$hello%%world!!'); }); -test('can normalize string with space equivalent characters and additional spaces', () => { - expect(normalizer.normalize('Lorem- ipsum _dolor , sit amet')).toBe('$$lorem%%ipsum%%dolor%%sit%%amet!!'); +test('can normalize three words', () => { + expect(normalizer.normalize('hello new world')).toBe('$$hello%%new%%world!!'); }); diff --git a/src/normalization/ngram-normalizer.ts b/src/normalization/ngram-normalizer.ts index ba38d23..9d8a1cf 100644 --- a/src/normalization/ngram-normalizer.ts +++ b/src/normalization/ngram-normalizer.ts @@ -1,149 +1,49 @@ import { Meta } from '../interfaces/meta.js'; import { NormalizationResult } from './normalization-result.js'; import { Normalizer } from '../interfaces/normalizer.js'; -import { NormalizerConfig } from './normalizer-config.js'; -import { StringUtilities } from '../commons/string-utilities.js'; /** * Normalization for creating proper n-grams. */ export class NgramNormalizer implements Normalizer { - /** - * The string that is appended to the left of the input string. - */ - private readonly paddingLeft: string; - - /** - * The string that is appended to the right of the input string. - */ - private readonly paddingRight: string; - - /** - * The string that is inserted for spaces in the input string. - */ - private readonly paddingMiddle: string; - - /** - * A set of all padding characters. - */ - private readonly paddingCharacters: Set; - - /** - * A function that determines whether a character is treated as a space. - */ - private readonly treatCharacterAsSpace: (c: string) => boolean; - - /** - * A function that determines whether a character is allowed. Padding characters. - * and surrogate characters are disallowed by default. - */ - private readonly allowCharacter: (c: string) => boolean; - - /** - * The number of encountered surrogate characters in a bulk normalization. - */ - private numberOfSurrogateCharacters: number = 0; /** * Creates a new instance of the NgramNormalizer class. - * @param normalizerConfig The configuration for the normalizer. - */ - public constructor(normalizerConfig: NormalizerConfig) { - this.paddingLeft = normalizerConfig.paddingLeft; - this.paddingRight = normalizerConfig.paddingRight; - this.paddingMiddle = normalizerConfig.paddingMiddle; - this.paddingCharacters = new Set( - [ - normalizerConfig.paddingLeft.split(''), - normalizerConfig.paddingRight.split(''), - normalizerConfig.paddingMiddle.split('') - ].flat() - ); - this.treatCharacterAsSpace = normalizerConfig.treatCharacterAsSpace; - this.allowCharacter = normalizerConfig.allowCharacter; + * @param paddingLeft The string that is appended to the left of the input string. + * @param paddingRight The string that is appended to the right of the input string. + * @param paddingMiddle The string that is inserted for spaces in the input string. + */ + public constructor( + public readonly paddingLeft: string, + public readonly paddingRight: string, + public readonly paddingMiddle: string) { } /** * {@inheritDoc Normalizer.normalize} */ public normalize(input: string): string { - const normalized: string[] = new Array(input.length + 2); - let j = 0; - - normalized[j++] = this.paddingLeft; - let previousIsPadding = true; - let previousIsSkippedEmptyChar = false; - let properCharacterAdded = false; - - for (let i = 0, l = input.length; i < l; i++) { - const normalizedChar = this.getNormalizedCharacter(input[i]); - - if (normalizedChar === '') { - continue; - } - - if (normalizedChar === ' ') { - previousIsSkippedEmptyChar = true; - } else { - if (previousIsSkippedEmptyChar && !previousIsPadding) { - normalized[j++] = this.paddingMiddle; - } - - normalized[j++] = normalizedChar; - properCharacterAdded = true; - previousIsPadding = false; - previousIsSkippedEmptyChar = false; - } - } - - normalized[j++] = this.paddingRight; - - if (!properCharacterAdded) { - return ''; + if (!input) { + return input; } - - return normalized.join(''); - } - - /** - * Normalizes the given character. Space equivalent characters are replaced by a space. Characters - * that are not allowed are removed, in addition to surrogate characters and padding characters. - * @param character The character to normalize. - * @returns The normalized character. - */ - private getNormalizedCharacter(character: string): string { - if (character === ' ' || this.treatCharacterAsSpace(character)) { - return ' '; - } - - if (this.isSurrogate(character) || this.paddingCharacters.has(character) || !this.allowCharacter(character)) { - return ''; - } - - return character.toLowerCase(); - } - - /** - * Checks if the given character is part of a surrogate pair. - * @param character The character to check. - * @returns True if the character is part of a surrogate pair, false otherwise. - */ - private isSurrogate(character: string): boolean { - if (StringUtilities.isSurrogate(character)) { - this.numberOfSurrogateCharacters++; - return true; - } - return false; + return `${this.paddingLeft}${input.split(' ').join(this.paddingMiddle)}${this.paddingRight}`; } /** * {@inheritDoc Normalizer.normalizeBulk} */ public normalizeBulk(input: string[]): NormalizationResult { - this.numberOfSurrogateCharacters = 0; const normalized = input.map((s) => this.normalize(s)); const meta = new Meta(); - meta.add('numberOfSurrogateCharacters', this.numberOfSurrogateCharacters); return new NormalizationResult(normalized, meta); } + + /** + * Creates an opinionated n-gram normalizer. Strings are padded with '$$' on the left, '!' on the right, and '!$$' in + * the middle. + * @returns An opinionated default n-gram normalizer. + */ + public static createDefault(): NgramNormalizer { + return new NgramNormalizer('$$', '!', '!$$'); + } } diff --git a/src/normalization/normalizer-config.ts b/src/normalization/normalizer-config.ts index e3ee737..d5f29a8 100644 --- a/src/normalization/normalizer-config.ts +++ b/src/normalization/normalizer-config.ts @@ -7,31 +7,23 @@ import { StringUtilities } from '../commons/string-utilities.js'; export class NormalizerConfig { /** * Creates a new instance of the NormalizerConfig class. - * @param paddingLeft The string that is appended to the left of the input string. - * @param paddingRight The string that is appended to the right of the input string. - * @param paddingMiddle The string that is inserted for spaces in the input string. * @param replacements A list of replacement maps. Each map maps from the variation character to the base * character(s). * @param treatCharacterAsSpace A function that determines whether a character is treated as a space. - * @param allowCharacter A function that determines whether a character is allowed. Padding characters and surrogate - * characters are disallowed by default. + * @param allowCharacter A function that determines whether a character is allowed. Surrogate characters are + * disallowed by default. */ public constructor( - public paddingLeft: string, - public paddingRight: string, - public paddingMiddle: string, public replacements: Map[], public treatCharacterAsSpace: (c: string) => boolean, public allowCharacter: (c: string) => boolean - ) {} + ) { } /** * Creates an opinionated default normalizer config. Applies latin replacements and filters out non-alphanumeric - * characters. Strings are padded with '$$' on the left, '!' on the right, and '!$$' in the middle. The config is - * closely related to the default NgramComputerConfig. - * The full normalization pipeline is built in the class DefaultNormalizer and includes a lowercasing and an NFKD - * normalization step. - * @returns The default normalizer config. + * characters. The full normalization pipeline is built in the class DefaultNormalizer and includes a lowercasing and + * an NFKD normalization step. + * @returns The opiniated default normalizer config. */ public static createDefaultConfig(): NormalizerConfig { const spaceEquivalentCharacters = new Set(['_', '-', '–', '/', ',', '\t']); @@ -41,12 +33,11 @@ export class NormalizerConfig { }; return new NormalizerConfig( - '$$', - '!', - '!$$', [LatinReplacements.Value], (c) => spaceEquivalentCharacters.has(c), allowCharacter ); } } + + diff --git a/src/string-searchers/normalizing-searcher.test.ts b/src/string-searchers/normalizing-searcher.test.ts index 235ac5f..e7ec209 100644 --- a/src/string-searchers/normalizing-searcher.test.ts +++ b/src/string-searchers/normalizing-searcher.test.ts @@ -1,16 +1,16 @@ +import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { LiteralSearcher } from './literal-searcher.js'; import { Match } from './match.js'; +import { MultiNormalizer } from '../normalization/multi-normalizer.js'; import { NgramNormalizer } from '../normalization/ngram-normalizer.js'; import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from './normalizing-searcher.js'; import { Query } from '../interfaces/query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; -const config = NormalizerConfig.createDefaultConfig(); -config.paddingLeft = '$$'; -config.paddingRight = '!!'; -config.paddingMiddle = '%%'; -const normalizer = new NgramNormalizer(config); +const defaultNormalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); +const ngramNormalizer = new NgramNormalizer('$$', '!!', '%%'); +const normalizer = new MultiNormalizer([defaultNormalizer, ngramNormalizer]); const literalSearcher: StringSearcher = new LiteralSearcher(); const normalizingSearcher: StringSearcher = new NormalizingSearcher(literalSearcher, normalizer); normalizingSearcher.index(['Hello world!']); diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts index 8552a9b..7034f0b 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.test.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -2,7 +2,7 @@ import { Match } from '../string-searchers/match.js'; import { Query } from '../interfaces/query.js'; import { SuffixArraySearcher } from './suffix-array-searcher.js'; -const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher(); +const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher('$'); suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); function getMatches(queryString: string): Match[] { diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 2fe8513..9f95bfd 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -10,13 +10,14 @@ import { SuffixArray } from './suffix-array.js'; // todo: make sure the terms don't have the separating character. Move to a suffix array config. export class SuffixArraySearcher implements StringSearcher { - private readonly separator = 'µ'; + private separator: string; private str: string; private suffixArray: Int32Array; private indexToTermIndex: Int32Array; private termLengths: Int32Array; - public constructor() { + public constructor(separator: string) { + this.separator = separator; this.str = ''; this.suffixArray = new Int32Array(0); this.indexToTermIndex = new Int32Array(0); @@ -111,10 +112,16 @@ export class SuffixArraySearcher implements StringSearcher { } save(memento: Memento): void { - throw new Error('Method not implemented.'); + memento.add(this.str); + memento.add(this.suffixArray); + memento.add(this.indexToTermIndex); + memento.add(this.termLengths); } load(memento: Memento): void { - throw new Error('Method not implemented.'); + this.str = memento.get(); + this.suffixArray = memento.get(); + this.indexToTermIndex = memento.get(); + this.termLengths = memento.get(); } } From 25c7fd3f4d0b919927833396674438829b543062 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 17:09:04 +0200 Subject: [PATCH 023/105] refactor: restore tests --- notes.md | 2 + src/config.ts | 14 +++---- src/fuzzy-searchers/fuzzy-search-config.ts | 37 +++++++++++++++++++ src/fuzzy-searchers/fuzzy-searcher.test.ts | 4 +- src/fuzzy-searchers/index.ts | 2 +- src/fuzzy-searchers/ngram-computer-config.ts | 28 -------------- src/fuzzy-searchers/ngram-computer.test.ts | 14 +++---- src/fuzzy-searchers/ngram-computer.ts | 12 +++--- .../ngram-full-normalizer.test.ts | 2 +- src/normalization/ngram-normalizer.ts | 11 +----- src/normalization/normalizer-config.ts | 2 +- src/string-searchers/normalizing-searcher.ts | 8 ++-- 12 files changed, 66 insertions(+), 70 deletions(-) create mode 100644 src/fuzzy-searchers/fuzzy-search-config.ts delete mode 100644 src/fuzzy-searchers/ngram-computer-config.ts diff --git a/notes.md b/notes.md index b2b841a..8fb4088 100644 --- a/notes.md +++ b/notes.md @@ -83,6 +83,8 @@ Todo Test second suffix array implementation. +Check todos. + First: suffixArraySearcherIndexing: 3663 suffixArraySearcherIndexing: 3638 diff --git a/src/config.ts b/src/config.ts index 3346c09..1b11eea 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,4 +1,4 @@ -import { NgramComputerConfig } from './fuzzy-searchers/ngram-computer-config.js'; +import { FuzzySearchConfig } from './fuzzy-searchers/fuzzy-search-config.js'; import { NormalizerConfig } from './normalization/normalizer-config.js'; /** @@ -8,16 +8,14 @@ export class Config { /** * Creates a new instance of the Config class. * @param normalizerConfig The configuration for the default normalizer. - * @param ngramComputerConfig The configuration for the n-gram computer. * @param maxQueryLength The maximum query length. - * @param inequalityPenalty The inequality penalty. + * @param fuzzySearchConfig The fuzzy search configuration. */ public constructor( public normalizerConfig: NormalizerConfig, - public ngramComputerConfig: NgramComputerConfig, public maxQueryLength: number, - public inequalityPenalty: number - ) {} + public fuzzySearchConfig: FuzzySearchConfig, + ) { } /** * Creates an opinionated default configuration. @@ -25,7 +23,7 @@ export class Config { */ public static createDefaultConfig(): Config { const normalizerConfig = NormalizerConfig.createDefaultConfig(); - const ngramComputerConfig = NgramComputerConfig.createDefaultConfig(); - return new Config(normalizerConfig, ngramComputerConfig, 150, 0.05); + const fuzzySearchConfig = FuzzySearchConfig.createDefaultConfig(); + return new Config(normalizerConfig, 150, fuzzySearchConfig); } } diff --git a/src/fuzzy-searchers/fuzzy-search-config.ts b/src/fuzzy-searchers/fuzzy-search-config.ts new file mode 100644 index 0000000..60cb65a --- /dev/null +++ b/src/fuzzy-searchers/fuzzy-search-config.ts @@ -0,0 +1,37 @@ +/** + * Holds fuzzy search configuration values. + */ +export class FuzzySearchConfig { + /** + * Creates a new instance of the FuzzySearchConfig class. + * @param paddingLeft The string that is appended to the left of the input string. + * @param paddingRight The string that is appended to the right of the input string. + * @param paddingMiddle The string that is inserted for spaces in the input string. + * @param ngramN The number of characters in each n-gram. + * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be + * removed. + * @param inequalityPenalty The inequality penalty. + */ + public constructor( + public paddingLeft: string, + public paddingRight: string, + public paddingMiddle: string, + public ngramN: number, + public transformNgram: (ngram: string) => string | null, + public inequalityPenalty: number, + ) { } + + /** + * Creates an opinionated default configuration with n=3. Strings are padded with '$$' on the left, '!' on the + * right, and '!$$' in the middle. N-grams that end with '$' are removed. N-grams that don't contain '$' are sorted. + * @returns The default configuration. + */ + public static createDefaultConfig(): FuzzySearchConfig { + const transformNgram = (ngram: string): string | null => { + return ngram.endsWith('$') ? null + : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') + : ngram + } + return new FuzzySearchConfig('$$', '!', '!$$', 3, transformNgram, 0.05); + } +} \ No newline at end of file diff --git a/src/fuzzy-searchers/fuzzy-searcher.test.ts b/src/fuzzy-searchers/fuzzy-searcher.test.ts index 26e3b71..08c61f1 100644 --- a/src/fuzzy-searchers/fuzzy-searcher.test.ts +++ b/src/fuzzy-searchers/fuzzy-searcher.test.ts @@ -1,12 +1,10 @@ import { FuzzySearcher } from './fuzzy-searcher.js'; import { Match } from '../string-searchers/match.js'; import { NgramComputer } from './ngram-computer.js'; -import { NgramComputerConfig } from './ngram-computer-config.js'; import { Query } from '../interfaces/query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; -const commonNgramComputerConfig = new NgramComputerConfig(3); -const commonNgramComputer = new NgramComputer(commonNgramComputerConfig); +const commonNgramComputer = new NgramComputer(3); const fuzzySearcher: StringSearcher = new FuzzySearcher(commonNgramComputer); fuzzySearcher.index(['Alice', 'Bob', 'Carol', 'Charlie']); diff --git a/src/fuzzy-searchers/index.ts b/src/fuzzy-searchers/index.ts index afaf41d..431c7d5 100644 --- a/src/fuzzy-searchers/index.ts +++ b/src/fuzzy-searchers/index.ts @@ -1,6 +1,6 @@ export { FuzzySearcher } from './fuzzy-searcher.js'; +export { FuzzySearchConfig } from './fuzzy-search-config.js'; export { InvertedIndex } from './inverted-index.js'; export { NgramComputer } from './ngram-computer.js'; -export { NgramComputerConfig } from './ngram-computer-config.js'; export { QualityComputer } from './quality-computer.js'; export { TermIds } from './term-ids.js'; diff --git a/src/fuzzy-searchers/ngram-computer-config.ts b/src/fuzzy-searchers/ngram-computer-config.ts deleted file mode 100644 index 593d2d5..0000000 --- a/src/fuzzy-searchers/ngram-computer-config.ts +++ /dev/null @@ -1,28 +0,0 @@ -/** - * Holds configuration values for the n-gram computer. - */ -export class NgramComputerConfig { - /** - * Creates a new instance of the NgramComputerConfig class. - * @param ngramN The number of characters in each n-gram. - * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be - * removed. - */ - public constructor( - public ngramN: number, - public transformNgram?: (ngram: string) => string | null - ) {} - - /** - * Creates an opinionated default n-gram computer config. Removes n-grams that end with '$', sorts n-grams that don't - * contain '$'. The config is closely related to the default NormalizerConfig. - * @returns The default n-gram computer config. - */ - public static createDefaultConfig(): NgramComputerConfig { - return new NgramComputerConfig(3, (ngram) => - ngram.endsWith('$') ? null - : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') - : ngram - ); - } -} diff --git a/src/fuzzy-searchers/ngram-computer.test.ts b/src/fuzzy-searchers/ngram-computer.test.ts index 65a8b6e..14ea137 100644 --- a/src/fuzzy-searchers/ngram-computer.test.ts +++ b/src/fuzzy-searchers/ngram-computer.test.ts @@ -1,21 +1,17 @@ import { NgramComputer } from './ngram-computer.js'; -import { NgramComputerConfig } from './ngram-computer-config.js'; // Common n-grams. -const commonNgramComputerConfig = new NgramComputerConfig(3); -const commonNgramComputer = new NgramComputer(commonNgramComputerConfig); +const commonNgramComputer = new NgramComputer(3); // Sorted n-grams. -const sortedNgramComputerConfig = new NgramComputerConfig(3, (ngram) => ngram.split('').sort().join('')); -const sortedNgramComputer = new NgramComputer(sortedNgramComputerConfig); +const sortedNgramComputer = new NgramComputer(3, (ngram) => ngram.split('').sort().join('')); // Default n-grams: remove n-grams that end with $, sort n-grams that don't contain $. -const defaultNgramComputerConfig = new NgramComputerConfig(3, (ngram) => +const defaultNgramComputer = new NgramComputer(3, (ngram) => ngram.endsWith('$') ? null - : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') - : ngram + : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') + : ngram ); -const defaultNgramComputer = new NgramComputer(defaultNgramComputerConfig); test('can compute n-grams of empty string', () => { expect(commonNgramComputer.computeNgrams('')).toEqual([]); diff --git a/src/fuzzy-searchers/ngram-computer.ts b/src/fuzzy-searchers/ngram-computer.ts index d3cc9ca..0866e0a 100644 --- a/src/fuzzy-searchers/ngram-computer.ts +++ b/src/fuzzy-searchers/ngram-computer.ts @@ -1,5 +1,3 @@ -import { NgramComputerConfig } from './ngram-computer-config.js'; - /** * Computes the n-grams of a string. */ @@ -16,11 +14,13 @@ export class NgramComputer { /** * Creates a new instance of the NgramComputer class. - * @param ngramComputerConfig The configuration for the n-gram computer. + * @param ngramN The number of characters in each n-gram. + * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be + * removed. */ - public constructor(ngramComputerConfig: NgramComputerConfig) { - this.ngramN = ngramComputerConfig.ngramN; - this.transformNgram = ngramComputerConfig.transformNgram ?? ((ngram): string => ngram); + public constructor(ngramN: number, transformNgram?: (ngram: string) => string | null) { + this.ngramN = ngramN; + this.transformNgram = transformNgram ?? ((ngram): string => ngram); } /** diff --git a/src/normalization/ngram-full-normalizer.test.ts b/src/normalization/ngram-full-normalizer.test.ts index 05d1ab5..528b8fe 100644 --- a/src/normalization/ngram-full-normalizer.test.ts +++ b/src/normalization/ngram-full-normalizer.test.ts @@ -4,7 +4,7 @@ import { NgramNormalizer } from './ngram-normalizer.js'; import { NormalizerConfig } from './normalizer-config.js'; const defaultNormalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); -const ngramNormalizer = NgramNormalizer.createDefault(); +const ngramNormalizer = new NgramNormalizer('$$', '!', '!$$'); const normalizer = new MultiNormalizer([defaultNormalizer, ngramNormalizer]); test('normalization test 1', () => { diff --git a/src/normalization/ngram-normalizer.ts b/src/normalization/ngram-normalizer.ts index 9d8a1cf..6edbe2f 100644 --- a/src/normalization/ngram-normalizer.ts +++ b/src/normalization/ngram-normalizer.ts @@ -1,7 +1,7 @@ import { Meta } from '../interfaces/meta.js'; import { NormalizationResult } from './normalization-result.js'; import { Normalizer } from '../interfaces/normalizer.js'; - +// todo: move to fuzzy-searchers /** * Normalization for creating proper n-grams. */ @@ -37,13 +37,4 @@ export class NgramNormalizer implements Normalizer { const meta = new Meta(); return new NormalizationResult(normalized, meta); } - - /** - * Creates an opinionated n-gram normalizer. Strings are padded with '$$' on the left, '!' on the right, and '!$$' in - * the middle. - * @returns An opinionated default n-gram normalizer. - */ - public static createDefault(): NgramNormalizer { - return new NgramNormalizer('$$', '!', '!$$'); - } } diff --git a/src/normalization/normalizer-config.ts b/src/normalization/normalizer-config.ts index d5f29a8..4e74c1d 100644 --- a/src/normalization/normalizer-config.ts +++ b/src/normalization/normalizer-config.ts @@ -23,7 +23,7 @@ export class NormalizerConfig { * Creates an opinionated default normalizer config. Applies latin replacements and filters out non-alphanumeric * characters. The full normalization pipeline is built in the class DefaultNormalizer and includes a lowercasing and * an NFKD normalization step. - * @returns The opiniated default normalizer config. + * @returns The default normalizer config. */ public static createDefaultConfig(): NormalizerConfig { const spaceEquivalentCharacters = new Set(['_', '-', '–', '/', ',', '\t']); diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index 1f015b4..b723336 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -13,11 +13,13 @@ export class NormalizingSearcher implements StringSearcher { * Creates a new instance of the NormalizingStringSearcher class. * @param stringSearcher The string searcher to use. * @param normalizer The normalizer to use. + * @param normalizationDurationMetaKey The meta key under which the normalization duration is reported. */ public constructor( private readonly stringSearcher: StringSearcher, - private readonly normalizer: Normalizer - ) {} + private readonly normalizer: Normalizer, + private readonly normalizationDurationMetaKey: string = 'normalizationDuration' + ) { } /** * {@inheritDoc StringSearcher.index} @@ -27,7 +29,7 @@ export class NormalizingSearcher implements StringSearcher { const result = this.normalizer.normalizeBulk(terms); const duration = Math.round(performance.now() - start); const meta = this.stringSearcher.index(result.strings); - meta.add('normalizationDuration', duration); + meta.add(this.normalizationDurationMetaKey, duration); for (const entry of result.meta.allEntries) { meta.add(entry[0], entry[1]); } From f3dc0a17f2ba3eaa40c56f3529d6ba3cf5498df4 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 17:10:38 +0200 Subject: [PATCH 024/105] chore: prettier --- .vscode/settings.json | 6 +- src/config.ts | 4 +- src/fuzzy-searchers/fuzzy-search-config.ts | 66 ++++++++++--------- src/fuzzy-searchers/ngram-computer.test.ts | 4 +- src/fuzzy-searchers/ngram-computer.ts | 2 +- .../character-normalizer.test.ts | 6 +- src/normalization/character-normalizer.ts | 5 +- src/normalization/default-normalizer.test.ts | 52 +++++++-------- src/normalization/default-normalizer.ts | 4 +- .../ngram-full-normalizer.test.ts | 1 - src/normalization/ngram-normalizer.ts | 5 +- src/normalization/normalizer-config.ts | 12 +--- src/string-searchers/normalizing-searcher.ts | 2 +- 13 files changed, 81 insertions(+), 88 deletions(-) diff --git a/.vscode/settings.json b/.vscode/settings.json index b35c98c..d0f2161 100644 --- a/.vscode/settings.json +++ b/.vscode/settings.json @@ -5,9 +5,7 @@ "editor.formatOnSave": false }, "editor.defaultFormatter": "esbenp.prettier-vscode", - "editor.rulers": [ - 120 - ], + "editor.rulers": [120], "liveServer.settings.port": 5501, "[json]": { "editor.defaultFormatter": "esbenp.prettier-vscode" @@ -15,4 +13,4 @@ "[plaintext]": { "editor.renderControlCharacters": false } -} \ No newline at end of file +} diff --git a/src/config.ts b/src/config.ts index 1b11eea..f814fe5 100644 --- a/src/config.ts +++ b/src/config.ts @@ -14,8 +14,8 @@ export class Config { public constructor( public normalizerConfig: NormalizerConfig, public maxQueryLength: number, - public fuzzySearchConfig: FuzzySearchConfig, - ) { } + public fuzzySearchConfig: FuzzySearchConfig + ) {} /** * Creates an opinionated default configuration. diff --git a/src/fuzzy-searchers/fuzzy-search-config.ts b/src/fuzzy-searchers/fuzzy-search-config.ts index 60cb65a..aa860f2 100644 --- a/src/fuzzy-searchers/fuzzy-search-config.ts +++ b/src/fuzzy-searchers/fuzzy-search-config.ts @@ -2,36 +2,38 @@ * Holds fuzzy search configuration values. */ export class FuzzySearchConfig { - /** - * Creates a new instance of the FuzzySearchConfig class. - * @param paddingLeft The string that is appended to the left of the input string. - * @param paddingRight The string that is appended to the right of the input string. - * @param paddingMiddle The string that is inserted for spaces in the input string. - * @param ngramN The number of characters in each n-gram. - * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be - * removed. - * @param inequalityPenalty The inequality penalty. - */ - public constructor( - public paddingLeft: string, - public paddingRight: string, - public paddingMiddle: string, - public ngramN: number, - public transformNgram: (ngram: string) => string | null, - public inequalityPenalty: number, - ) { } + /** + * Creates a new instance of the FuzzySearchConfig class. + * @param paddingLeft The string that is appended to the left of the input string. + * @param paddingRight The string that is appended to the right of the input string. + * @param paddingMiddle The string that is inserted for spaces in the input string. + * @param ngramN The number of characters in each n-gram. + * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be + * removed. + * @param inequalityPenalty The inequality penalty. + */ + public constructor( + public paddingLeft: string, + public paddingRight: string, + public paddingMiddle: string, + public ngramN: number, + public transformNgram: (ngram: string) => string | null, + public inequalityPenalty: number + ) {} - /** - * Creates an opinionated default configuration with n=3. Strings are padded with '$$' on the left, '!' on the - * right, and '!$$' in the middle. N-grams that end with '$' are removed. N-grams that don't contain '$' are sorted. - * @returns The default configuration. - */ - public static createDefaultConfig(): FuzzySearchConfig { - const transformNgram = (ngram: string): string | null => { - return ngram.endsWith('$') ? null - : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') - : ngram - } - return new FuzzySearchConfig('$$', '!', '!$$', 3, transformNgram, 0.05); - } -} \ No newline at end of file + /** + * Creates an opinionated default configuration with n=3. Strings are padded with '$$' on the left, '!' on the + * right, and '!$$' in the middle. N-grams that end with '$' are removed. N-grams that don't contain '$' are sorted. + * @returns The default configuration. + */ + public static createDefaultConfig(): FuzzySearchConfig { + const transformNgram = (ngram: string): string | null => { + return ( + ngram.endsWith('$') ? null + : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') + : ngram + ); + }; + return new FuzzySearchConfig('$$', '!', '!$$', 3, transformNgram, 0.05); + } +} diff --git a/src/fuzzy-searchers/ngram-computer.test.ts b/src/fuzzy-searchers/ngram-computer.test.ts index 14ea137..f940e54 100644 --- a/src/fuzzy-searchers/ngram-computer.test.ts +++ b/src/fuzzy-searchers/ngram-computer.test.ts @@ -9,8 +9,8 @@ const sortedNgramComputer = new NgramComputer(3, (ngram) => ngram.split('').sort // Default n-grams: remove n-grams that end with $, sort n-grams that don't contain $. const defaultNgramComputer = new NgramComputer(3, (ngram) => ngram.endsWith('$') ? null - : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') - : ngram + : ngram.indexOf('$') === -1 ? ngram.split('').sort().join('') + : ngram ); test('can compute n-grams of empty string', () => { diff --git a/src/fuzzy-searchers/ngram-computer.ts b/src/fuzzy-searchers/ngram-computer.ts index 0866e0a..3dbabc4 100644 --- a/src/fuzzy-searchers/ngram-computer.ts +++ b/src/fuzzy-searchers/ngram-computer.ts @@ -15,7 +15,7 @@ export class NgramComputer { /** * Creates a new instance of the NgramComputer class. * @param ngramN The number of characters in each n-gram. - * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be + * @param transformNgram A function for transforming each n-gram. N-grams that are transformed to null will be * removed. */ public constructor(ngramN: number, transformNgram?: (ngram: string) => string | null) { diff --git a/src/normalization/character-normalizer.test.ts b/src/normalization/character-normalizer.test.ts index b6a5770..47de715 100644 --- a/src/normalization/character-normalizer.test.ts +++ b/src/normalization/character-normalizer.test.ts @@ -3,7 +3,9 @@ import { StringUtilities } from '../commons/string-utilities.js'; const spaceEquivalentCharacters = new Set(['_', '-', '–', '/', ',', '\t']); const normalizer = new CharacterNormalizer( - c => spaceEquivalentCharacters.has(c), c => StringUtilities.isAlphanumeric(c)); + (c) => spaceEquivalentCharacters.has(c), + (c) => StringUtilities.isAlphanumeric(c) +); test('can normalize empty string', () => { expect(normalizer.normalize('')).toBe(''); @@ -79,4 +81,4 @@ test('can normalize string with space equivalent characters', () => { test('can normalize string with space equivalent characters and additional spaces', () => { expect(normalizer.normalize('Lorem- ipsum _dolor , sit amet')).toBe('lorem ipsum dolor sit amet'); -}); \ No newline at end of file +}); diff --git a/src/normalization/character-normalizer.ts b/src/normalization/character-normalizer.ts index f98aa60..b5c3710 100644 --- a/src/normalization/character-normalizer.ts +++ b/src/normalization/character-normalizer.ts @@ -27,10 +27,7 @@ export class CharacterNormalizer implements Normalizer { * @param treatCharacterAsSpace A function that determines whether a character is treated as a space. * @param allowCharacter A function that determines whether a character is allowed. */ - public constructor( - treatCharacterAsSpace: (c: string) => boolean, - allowCharacter: (c: string) => boolean) { - + public constructor(treatCharacterAsSpace: (c: string) => boolean, allowCharacter: (c: string) => boolean) { this.treatCharacterAsSpace = treatCharacterAsSpace; this.allowCharacter = allowCharacter; } diff --git a/src/normalization/default-normalizer.test.ts b/src/normalization/default-normalizer.test.ts index 99db49c..1adf14b 100644 --- a/src/normalization/default-normalizer.test.ts +++ b/src/normalization/default-normalizer.test.ts @@ -5,42 +5,42 @@ const config = NormalizerConfig.createDefaultConfig(); const normalizer = DefaultNormalizer.create(config); test('German Eszett', () => { - expect(normalizer.normalize('Fußball')).toBe('fussball'); + expect(normalizer.normalize('Fußball')).toBe('fussball'); }); test('German umlauts', () => { - expect(normalizer.normalize('ä ö ü')).toBe('ae oe ue'); + expect(normalizer.normalize('ä ö ü')).toBe('ae oe ue'); }); test('variatons of a', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect( - normalizer.normalize( - 'Áá Àà Ââ Ǎǎ Ăă Ãã Ảả Ȧȧ Ạạ Ää Åå Ḁḁ Āā Ąą ᶏ Ⱥⱥ Ȁȁ Ấấ Ầầ Ẫẫ Ẩẩ Ậậ Ắắ Ằằ Ẵẵ Ẳẳ Ặặ Ǻǻ Ǡǡ Ǟǟ Ȃȃ Ɑɑ ᴀ Ɐɐ ɒ Aa' + - ' Ææ ᴁ ᴭ ᵆ Ǽǽ Ǣǣ ᴂ' - ) - ).toBe( - 'aa aa aa aa aa aa aa aa aa aeae aaaa aa aa aa a aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa a aa a aa' + - ' aeae ae ae ae aeae aeae ae' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect( + normalizer.normalize( + 'Áá Àà Ââ Ǎǎ Ăă Ãã Ảả Ȧȧ Ạạ Ää Åå Ḁḁ Āā Ąą ᶏ Ⱥⱥ Ȁȁ Ấấ Ầầ Ẫẫ Ẩẩ Ậậ Ắắ Ằằ Ẵẵ Ẳẳ Ặặ Ǻǻ Ǡǡ Ǟǟ Ȃȃ Ɑɑ ᴀ Ɐɐ ɒ Aa' + + ' Ææ ᴁ ᴭ ᵆ Ǽǽ Ǣǣ ᴂ' + ) + ).toBe( + 'aa aa aa aa aa aa aa aa aa aeae aaaa aa aa aa a aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa aa a aa a aa' + + ' aeae ae ae ae aeae aeae ae' + ); }); test('variatons of u', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect( - normalizer.normalize( - 'Úú Ùù Ŭŭ Ûû Ǔǔ Ůů Üü Ǘǘ Ǜǜ Ǚǚ Ǖǖ Űű Ũũ Ṹṹ Ųų Ūū Ṻṻ Ủủ Ȕȕ Ȗȗ Ưư Ứứ Ừừ Ữữ Ửử Ựự Ụụ Ṳṳ Ṷṷ Ṵṵ Ʉʉ Ʊʊ Ȣȣ ᵾ ᶙ ᴜ Uu' + - ' ᵫ ɯ' - ) - ).toBe( - 'uu uu uu uu uu uu ueue uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu ouou u u u uu' + - ' ue m' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect( + normalizer.normalize( + 'Úú Ùù Ŭŭ Ûû Ǔǔ Ůů Üü Ǘǘ Ǜǜ Ǚǚ Ǖǖ Űű Ũũ Ṹṹ Ųų Ūū Ṻṻ Ủủ Ȕȕ Ȗȗ Ưư Ứứ Ừừ Ữữ Ửử Ựự Ụụ Ṳṳ Ṷṷ Ṵṵ Ʉʉ Ʊʊ Ȣȣ ᵾ ᶙ ᴜ Uu' + + ' ᵫ ɯ' + ) + ).toBe( + 'uu uu uu uu uu uu ueue uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu uu ouou u u u uu' + + ' ue m' + ); }); test('variatons of s', () => { - // source: https://en.wiktionary.org/wiki/Appendix:Latin_script - expect(normalizer.normalize('ſ ẞß Śś Ṥṥ Ŝŝ Šš Ṧṧ Ṡṡẛ Şş Ṣṣ Ṩṩ Șș S̩s̩ ᵴ ᶊ ʂ ȿ ꜱ Ʃʃ Ss')).toBe( - 's ssss ss ss ss ss ss sss ss ss ss ss ss s s s s s ss ss' - ); + // source: https://en.wiktionary.org/wiki/Appendix:Latin_script + expect(normalizer.normalize('ſ ẞß Śś Ṥṥ Ŝŝ Šš Ṧṧ Ṡṡẛ Şş Ṣṣ Ṩṩ Șș S̩s̩ ᵴ ᶊ ʂ ȿ ꜱ Ʃʃ Ss')).toBe( + 's ssss ss ss ss ss ss sss ss ss ss ss ss s s s s s ss ss' + ); }); diff --git a/src/normalization/default-normalizer.ts b/src/normalization/default-normalizer.ts index 9b080d1..26d68be 100644 --- a/src/normalization/default-normalizer.ts +++ b/src/normalization/default-normalizer.ts @@ -22,7 +22,9 @@ export class DefaultNormalizer { ); const normalizer3 = new GenericNormalizer((input: string): string => input.normalize('NFKD')); const normalizer4 = new CharacterNormalizer( - normalizerConfig.treatCharacterAsSpace, normalizerConfig.allowCharacter); + normalizerConfig.treatCharacterAsSpace, + normalizerConfig.allowCharacter + ); const multiNormalizer = new MultiNormalizer([normalizer1, normalizer2, normalizer3, normalizer4]); return multiNormalizer; } diff --git a/src/normalization/ngram-full-normalizer.test.ts b/src/normalization/ngram-full-normalizer.test.ts index 528b8fe..3725025 100644 --- a/src/normalization/ngram-full-normalizer.test.ts +++ b/src/normalization/ngram-full-normalizer.test.ts @@ -46,4 +46,3 @@ test('normalization test 9', () => { test('normalization test 10', () => { expect(normalizer.normalize('Thiên Duyên Tô')).toBe('$$thien!$$duyen!$$to!'); }); - diff --git a/src/normalization/ngram-normalizer.ts b/src/normalization/ngram-normalizer.ts index 6edbe2f..b5efa64 100644 --- a/src/normalization/ngram-normalizer.ts +++ b/src/normalization/ngram-normalizer.ts @@ -6,7 +6,6 @@ import { Normalizer } from '../interfaces/normalizer.js'; * Normalization for creating proper n-grams. */ export class NgramNormalizer implements Normalizer { - /** * Creates a new instance of the NgramNormalizer class. * @param paddingLeft The string that is appended to the left of the input string. @@ -16,8 +15,8 @@ export class NgramNormalizer implements Normalizer { public constructor( public readonly paddingLeft: string, public readonly paddingRight: string, - public readonly paddingMiddle: string) { - } + public readonly paddingMiddle: string + ) {} /** * {@inheritDoc Normalizer.normalize} diff --git a/src/normalization/normalizer-config.ts b/src/normalization/normalizer-config.ts index 4e74c1d..a95bbda 100644 --- a/src/normalization/normalizer-config.ts +++ b/src/normalization/normalizer-config.ts @@ -10,14 +10,14 @@ export class NormalizerConfig { * @param replacements A list of replacement maps. Each map maps from the variation character to the base * character(s). * @param treatCharacterAsSpace A function that determines whether a character is treated as a space. - * @param allowCharacter A function that determines whether a character is allowed. Surrogate characters are + * @param allowCharacter A function that determines whether a character is allowed. Surrogate characters are * disallowed by default. */ public constructor( public replacements: Map[], public treatCharacterAsSpace: (c: string) => boolean, public allowCharacter: (c: string) => boolean - ) { } + ) {} /** * Creates an opinionated default normalizer config. Applies latin replacements and filters out non-alphanumeric @@ -32,12 +32,6 @@ export class NormalizerConfig { return StringUtilities.isAlphanumeric(c); }; - return new NormalizerConfig( - [LatinReplacements.Value], - (c) => spaceEquivalentCharacters.has(c), - allowCharacter - ); + return new NormalizerConfig([LatinReplacements.Value], (c) => spaceEquivalentCharacters.has(c), allowCharacter); } } - - diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index b723336..a8c6638 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -19,7 +19,7 @@ export class NormalizingSearcher implements StringSearcher { private readonly stringSearcher: StringSearcher, private readonly normalizer: Normalizer, private readonly normalizationDurationMetaKey: string = 'normalizationDuration' - ) { } + ) {} /** * {@inheritDoc StringSearcher.index} From a62319876c0d1df9a7bb4faa546c6bba60e845b7 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 17:14:21 +0200 Subject: [PATCH 025/105] chore: run regression test --- src/regression-test/output/_indexing-meta.txt | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 74697ef..6664cf5 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,9 +1,10 @@ { "numberOfInvalidTerms": 1, + "ngramNormalizationDuration": 267, "numberOfDistinctTerms": 1190185, - "normalizationDuration": 1342, + "defaultNormalizationDuration": 1350, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 6741 + "indexingDuration": 6687 } \ No newline at end of file From c3d6c0243f605ba6c096d15a795829fd88d9ecfa Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Mon, 20 Oct 2025 19:43:06 +0200 Subject: [PATCH 026/105] feat: SubstringSearchConfig --- src/config.ts | 12 +++++++---- src/suffix-array-searchers/index.ts | 1 + .../substring-search-config.ts | 20 +++++++++++++++++++ 3 files changed, 29 insertions(+), 4 deletions(-) create mode 100644 src/suffix-array-searchers/substring-search-config.ts diff --git a/src/config.ts b/src/config.ts index f814fe5..6a9a1a3 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,5 +1,6 @@ -import { FuzzySearchConfig } from './fuzzy-searchers/fuzzy-search-config.js'; +import { FuzzySearchConfig } from './fuzzy-search.js'; import { NormalizerConfig } from './normalization/normalizer-config.js'; +import { SubstringSearchConfig } from './suffix-array-searchers/substring-search-config.js'; /** * Holds configuration values for the searcher. @@ -10,12 +11,14 @@ export class Config { * @param normalizerConfig The configuration for the default normalizer. * @param maxQueryLength The maximum query length. * @param fuzzySearchConfig The fuzzy search configuration. + * @param substringSearchConfig The substring search configuration. */ public constructor( public normalizerConfig: NormalizerConfig, public maxQueryLength: number, - public fuzzySearchConfig: FuzzySearchConfig - ) {} + public fuzzySearchConfig: FuzzySearchConfig, + public substringSearchConfig: SubstringSearchConfig + ) { } /** * Creates an opinionated default configuration. @@ -24,6 +27,7 @@ export class Config { public static createDefaultConfig(): Config { const normalizerConfig = NormalizerConfig.createDefaultConfig(); const fuzzySearchConfig = FuzzySearchConfig.createDefaultConfig(); - return new Config(normalizerConfig, 150, fuzzySearchConfig); + const substringSearchConfig = SubstringSearchConfig.createDefaultConfig(); + return new Config(normalizerConfig, 150, fuzzySearchConfig, substringSearchConfig); } } diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index f58863b..9b45478 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -1,3 +1,4 @@ export { StringComparison } from './string-comparison.js'; export { SuffixArray } from './suffix-array.js'; export { SuffixArraySearcher } from './suffix-array-searcher.js'; +export { SubstringSearchConfig } from './substring-search-config.js'; diff --git a/src/suffix-array-searchers/substring-search-config.ts b/src/suffix-array-searchers/substring-search-config.ts new file mode 100644 index 0000000..67d4f4a --- /dev/null +++ b/src/suffix-array-searchers/substring-search-config.ts @@ -0,0 +1,20 @@ +/** + * Holds substring search configuration values. + */ +export class SubstringSearchConfig { + /** + * Creates a new instance of the SubstringSearchConfig class. + * @param suffixArraySeparator The suffix array separator character. + */ + public constructor( + public suffixArraySeparator: string + ) { } + + /** + * Creates a default configuration with '$' as the suffix array separator. + * @returns The default configuration. + */ + public static createDefaultConfig(): SubstringSearchConfig { + return new SubstringSearchConfig('$'); + } +} From a4b4dafc81ca5ab94ef52ecea254d99414a19206 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 21 Oct 2025 15:58:30 +0200 Subject: [PATCH 027/105] wip --- notes.md | 7 +- src/config.ts | 2 +- .../entity-searcher-factory.ts | 70 +++++++++++++++++-- src/regression-test/main.ts | 2 +- src/suffix-array-searchers/index.ts | 1 + src/suffix-array-searchers/prefix-searcher.ts | 52 ++++++++++++++ .../suffix-array-searcher.test.ts | 6 +- .../suffix-array-searcher.ts | 16 ++++- 8 files changed, 142 insertions(+), 14 deletions(-) create mode 100644 src/suffix-array-searchers/prefix-searcher.ts diff --git a/notes.md b/notes.md index 8fb4088..d5e60a9 100644 --- a/notes.md +++ b/notes.md @@ -85,11 +85,15 @@ Test second suffix array implementation. Check todos. +Adjust usage examples to new config structure. + First: suffixArraySearcherIndexing: 3663 suffixArraySearcherIndexing: 3638 suffixArraySearcherIndexing: 3655 +Check if doc comments are complete for the new files. + Changelog: @@ -98,4 +102,5 @@ NgramNormalizerConfig. // '$$', // '!', -// '!$$', \ No newline at end of file +// '!$$', + diff --git a/src/config.ts b/src/config.ts index 6a9a1a3..f7c37e5 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,4 +1,4 @@ -import { FuzzySearchConfig } from './fuzzy-search.js'; +import { FuzzySearchConfig } from './fuzzy-searchers/fuzzy-search-config.js'; import { NormalizerConfig } from './normalization/normalizer-config.js'; import { SubstringSearchConfig } from './suffix-array-searchers/substring-search-config.js'; diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 0bb5e88..e6e17ff 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -2,13 +2,19 @@ import { Config } from '../config.js'; import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; +import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; import { NgramComputer } from '../fuzzy-searchers/ngram-computer.js'; +import { NgramNormalizer } from '../normalization/ngram-normalizer.js'; import { Normalizer } from '../interfaces/normalizer.js'; +import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from '../string-searchers/normalizing-searcher.js'; +import { PrefixSearcher } from '../suffix-array-searchers/prefix-searcher.js'; import { SortingSearcher } from '../string-searchers/sorting-searcher.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; +import { SubstringSearchConfig } from '../suffix-array-searchers/substring-search-config.js'; +import { SuffixArraySearcher } from '../suffix-array-searchers/suffix-array-searcher.js'; /** * Factory for creating entity searchers. @@ -22,13 +28,65 @@ export class EntitySearcherFactory { * @returns The default entity searcher. */ public static createSearcher(config: Config): DefaultEntitySearcher { - const ngramComputer: NgramComputer = new NgramComputer(config.ngramComputerConfig); - const normalizer: Normalizer = DefaultNormalizer.create(config.normalizerConfig); - let stringSearcher: StringSearcher = new FuzzySearcher(ngramComputer); - stringSearcher = new InequalityPenalizingSearcher(stringSearcher, config.inequalityPenalty); + const defaultNormalizer: Normalizer = this.createDefaultNormalizer(config); + + // let stringSearcher: StringSearcher = new SuffixArraySearcher('$'); + // let stringSearcher: StringSearcher = this.createFuzzySearcher(config.fuzzySearchConfig); + const suffixArraySearcher: SuffixArraySearcher = this.createSubstringSearcher(config.substringSearchConfig); + // todo: searcher switch. + let stringSearcher = this.createPrefixSearcher(suffixArraySearcher); + stringSearcher = new DistinctSearcher(stringSearcher); stringSearcher = new SortingSearcher(stringSearcher); - stringSearcher = new NormalizingSearcher(stringSearcher, normalizer); + stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'defaultNormalizationDuration'); return new DefaultEntitySearcher(stringSearcher); } -} + + private static createDefaultNormalizer(config: Config): Normalizer { + // todo: suffix array separator. + const forbiddenCharacters = new Set( + [ + config.fuzzySearchConfig.paddingLeft.split(''), + config.fuzzySearchConfig.paddingRight.split(''), + config.fuzzySearchConfig.paddingMiddle.split('') + ].flat() + ); + + const allowCharacter: (c: string) => boolean = (c) => + config.normalizerConfig.allowCharacter(c) && !forbiddenCharacters.has(c); + const modifiedNormalizerConfig = new NormalizerConfig( + config.normalizerConfig.replacements, + config.normalizerConfig.treatCharacterAsSpace, + allowCharacter + ); + + return DefaultNormalizer.create(modifiedNormalizerConfig); + } + + /** + * @param config The fuzzy search configuration. + * @returns The fuzzy string searcher. + */ + private static createFuzzySearcher(config: FuzzySearchConfig): StringSearcher { + const ngramComputer: NgramComputer = new NgramComputer(config.ngramN, config.transformNgram); + const ngramNormalizer: Normalizer = new NgramNormalizer( + config.paddingLeft, + config.paddingRight, + config.paddingMiddle + ); + + let fuzzySearcher: StringSearcher = new FuzzySearcher(ngramComputer); + fuzzySearcher = new InequalityPenalizingSearcher(fuzzySearcher, config.inequalityPenalty); + fuzzySearcher = new NormalizingSearcher(fuzzySearcher, ngramNormalizer, 'ngramNormalizationDuration'); + return fuzzySearcher; + } + + private static createSubstringSearcher(config: SubstringSearchConfig): SuffixArraySearcher { + const substringSearcher = new SuffixArraySearcher(config.suffixArraySeparator); + return substringSearcher; + } + + private static createPrefixSearcher(suffixArraySearcher: SuffixArraySearcher): StringSearcher { + return new PrefixSearcher(suffixArraySearcher); + } +} \ No newline at end of file diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts index c3a315f..5dd2d08 100644 --- a/src/regression-test/main.ts +++ b/src/regression-test/main.ts @@ -41,7 +41,7 @@ writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { en console.log('Running queries...'); runQuery('carcassonne-prefix', 'carcasso'); -runQuery('carcassonne-infix', 'cassonn'); +runQuery('carcassonne-substring', 'cassonn'); runQuery('carcassonne-suffix', 'sonne'); runQuery('munich-insertion', 'muniich'); runQuery('boston-deletion', 'bostn'); diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index 9b45478..bcdacbb 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -2,3 +2,4 @@ export { StringComparison } from './string-comparison.js'; export { SuffixArray } from './suffix-array.js'; export { SuffixArraySearcher } from './suffix-array-searcher.js'; export { SubstringSearchConfig } from './substring-search-config.js'; +export { PrefixSearcher } from './prefix-searcher.js'; diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts new file mode 100644 index 0000000..7669920 --- /dev/null +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -0,0 +1,52 @@ +import { Memento } from '../interfaces/memento.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; +import { Result } from '../string-searchers/result.js'; +import { StringSearcher } from '../interfaces/string-searcher.js'; +import { SuffixArraySearcher } from './suffix-array-searcher.js'; + +export class PrefixSearcher implements StringSearcher { + + public constructor( + private readonly suffixArraySearcher: SuffixArraySearcher, + ) { + } + + /** + * {@inheritDoc StringSearcher.index} + */ + index(terms: string[]): Meta { + return this.suffixArraySearcher.index(terms); + } + + /** + * {@inheritDoc StringSearcher.getMatches} + */ + getMatches(query: Query): Result { + if (!query.string) { + return new Result([], query, new Meta()); + } + const modifiedQueryString = this.modifyQueryString(query.string); + const modifiedQuery = new Query(modifiedQueryString, query.topN, query.minQuality); + return this.suffixArraySearcher.getMatches(modifiedQuery); + } + + private modifyQueryString(original: string): string { + return `${this.suffixArraySearcher.separator}${original}`; + } + + /** + * {@inheritDoc StringSearcher.save} + */ + save(memento: Memento): void { + this.suffixArraySearcher.save(memento); + } + + /** + * {@inheritDoc StringSearcher.load} + */ + load(memento: Memento): void { + this.suffixArraySearcher.load(memento); + } + +} \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts index 7034f0b..e9b246c 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.test.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -40,15 +40,15 @@ test('can find suffix matches test 2', () => { expect(getMatches('ol')).toEqual([new Match(3, 2 / 5)]); }); -test('can find infix matches test 1', () => { +test('can find substring matches test 1', () => { expect(getMatches('arl')).toEqual([new Match(2, 3 / 6), new Match(4, 3 / 7)]); }); -test('can find infix matches test 2', () => { +test('can find substring matches test 2', () => { expect(getMatches('li')).toEqual([new Match(0, 2 / 5), new Match(4, 2 / 7)]); }); -test('can find infix matches test 3', () => { +test('can find substring matches test 3', () => { expect(getMatches('l')).toEqual([new Match(0, 1 / 5), new Match(2, 1 / 6), new Match(3, 1 / 5), new Match(4, 1 / 7)]); }); diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 9f95bfd..143bd34 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -10,7 +10,7 @@ import { SuffixArray } from './suffix-array.js'; // todo: make sure the terms don't have the separating character. Move to a suffix array config. export class SuffixArraySearcher implements StringSearcher { - private separator: string; + public readonly separator: string; private str: string; private suffixArray: Int32Array; private indexToTermIndex: Int32Array; @@ -24,6 +24,9 @@ export class SuffixArraySearcher implements StringSearcher { this.termLengths = new Int32Array(0); } + /** + * {@inheritDoc StringSearcher.index} + */ index(terms: string[]): Meta { const start = performance.now(); this.str = this.separator + terms.join(this.separator) + this.separator; @@ -47,8 +50,11 @@ export class SuffixArraySearcher implements StringSearcher { return meta; } + /** + * {@inheritDoc StringSearcher.getMatches} + */ getMatches(query: Query): Result { - if (query.string == null || query.string === '') { + if (!query.string) { return new Result([], query, new Meta()); } @@ -111,6 +117,9 @@ export class SuffixArraySearcher implements StringSearcher { return [start, r]; } + /** + * {@inheritDoc StringSearcher.save} + */ save(memento: Memento): void { memento.add(this.str); memento.add(this.suffixArray); @@ -118,6 +127,9 @@ export class SuffixArraySearcher implements StringSearcher { memento.add(this.termLengths); } + /** + * {@inheritDoc StringSearcher.load} + */ load(memento: Memento): void { this.str = memento.get(); this.suffixArray = memento.get(); From 6a45c0ff14f22d0f15c5878a35600844f56dda09 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 21 Oct 2025 16:11:18 +0200 Subject: [PATCH 028/105] fix: prefix searcher quality --- src/suffix-array-searchers/prefix-searcher.ts | 2 +- src/suffix-array-searchers/suffix-array-searcher.ts | 5 +++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index 7669920..13cf5e2 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -28,7 +28,7 @@ export class PrefixSearcher implements StringSearcher { } const modifiedQueryString = this.modifyQueryString(query.string); const modifiedQuery = new Query(modifiedQueryString, query.topN, query.minQuality); - return this.suffixArraySearcher.getMatches(modifiedQuery); + return this.suffixArraySearcher.getMatches(modifiedQuery, query.string.length); } private modifyQueryString(original: string): string { diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 143bd34..770dd82 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -53,7 +53,7 @@ export class SuffixArraySearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - getMatches(query: Query): Result { + getMatches(query: Query, queryLength?: number): Result { if (!query.string) { return new Result([], query, new Meta()); } @@ -71,9 +71,10 @@ export class SuffixArraySearcher implements StringSearcher { const matches: Match[] = []; + queryLength = queryLength ?? query.string.length; let quality = 0; for (let k = 0; k < matchedTermIds.length; k++) { - quality = this.computeQuality(query.string.length, this.termLengths[matchedTermIds[k]]); + quality = this.computeQuality(queryLength, this.termLengths[matchedTermIds[k]]); if (quality > query.minQuality) { matches.push(new Match(matchedTermIds[k], quality)); } From c204d62577b5c3e0c21337e5c70faa553a513127 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Tue, 21 Oct 2025 20:20:09 +0200 Subject: [PATCH 029/105] feat: sorting entity searcher --- .../sorting-entity-searcher.ts | 117 ++++++++++++++++++ src/sort-order.ts | 14 +++ 2 files changed, 131 insertions(+) create mode 100644 src/entity-searchers/sorting-entity-searcher.ts create mode 100644 src/sort-order.ts diff --git a/src/entity-searchers/sorting-entity-searcher.ts b/src/entity-searchers/sorting-entity-searcher.ts new file mode 100644 index 0000000..bef70b1 --- /dev/null +++ b/src/entity-searchers/sorting-entity-searcher.ts @@ -0,0 +1,117 @@ +import { EntityMatch } from '../interfaces/entity-match.js'; +import { EntityResult } from '../interfaces/entity-result.js'; +import { EntitySearcher } from '../interfaces/entity-searcher.js'; +import { Memento } from '../interfaces/memento.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; +import { SortOrder } from '../sort-order.js'; + +/** + * Sorts entity matches according to the specified order after retrieval. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + */ +export class SortingEntitySearcher implements EntitySearcher { + + /** + * The collator for string comparisons. + */ + private readonly collator = new Intl.Collator(undefined, { numeric: true }); + + /** + * Creates a new instance of the SortingEntitySearcher class. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + * @param sortOrder The sort order to use for the entity matches. + * @param entitySearcher The entity searcher to wrap and sort results from. + */ + public constructor( + private readonly sortOrder: SortOrder, + private readonly entitySearcher: EntitySearcher) { } + + /** + * {@inheritDoc EntitySearcher.indexEntities} + */ + public indexEntities( + entities: TEntity[], + getId: (entity: TEntity) => TId, + getTerms: (entity: TEntity) => string[] + ): Meta { + return this.entitySearcher.indexEntities(entities, getId, getTerms); + } + + /** + * {@inheritDoc EntitySearcher.getMatches} + */ + public getMatches(query: Query): EntityResult { + const result = this.entitySearcher.getMatches(query); + + switch (this.sortOrder) { + case SortOrder.QualityAndIndex: + // The entity matches are already sorted by quality and index. + return result; + case SortOrder.QualityAndMatchedString: + this.sortMatchesByQualityAndMatchedString(result.matches); + return result; + default: + throw new Error(`Unsupported sort order: ${this.sortOrder}`); + } + } + + /** + * Sorts the entity matches in place by quality and matched string. + * @param matches The entity matches to sort. + */ + private sortMatchesByQualityAndMatchedString(matches: EntityMatch[]): void { + + matches.sort((m1, m2) => { + return ( + m1.quality > m2.quality ? -1 + : m1.quality < m2.quality ? 1 + : this.collator.compare(m1.matchedString, m2.matchedString) + ); + }); + } + + /** + * {@inheritDoc EntitySearcher.tryGetEntity} + */ + public tryGetEntity(id: TId): TEntity | null { + return this.entitySearcher.tryGetEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.getEntities} + */ + public getEntities(): TEntity[] { + return this.entitySearcher.getEntities(); + } + + /** + * {@inheritDoc EntitySearcher.tryGetTerms} + */ + public tryGetTerms(id: TId): string[] | null { + return this.entitySearcher.tryGetTerms(id); + } + + /** + * {@inheritDoc EntitySearcher.getTerms} + */ + public getTerms(): string[] { + return this.entitySearcher.getTerms(); + } + + /** + * {@inheritDoc EntitySearcher.save} + */ + public save(memento: Memento): void { + this.entitySearcher.save(memento); + } + + /** + * {@inheritDoc EntitySearcher.load} + */ + public load(memento: Memento): void { + this.entitySearcher.load(memento); + } +} diff --git a/src/sort-order.ts b/src/sort-order.ts new file mode 100644 index 0000000..33165c7 --- /dev/null +++ b/src/sort-order.ts @@ -0,0 +1,14 @@ +/** + * Specifies the sort order for the entity matches. + */ +export enum SortOrder { + /** + * Sort by quality first, then by index. + */ + QualityAndIndex = 'qualityAndIndex', + + /** + * Sort by quality first, then by matched string. + */ + QualityAndMatchedString = 'qualityAndMatchedString', +} \ No newline at end of file From 4c97f3f2af4c7689141cd72bd423058f68cd60c8 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 09:45:52 +0200 Subject: [PATCH 030/105] refactor: dynamic searcher --- .../default-dynamic-searcher.ts | 26 +++++++++++++++---- .../timing-dynamic-searcher.ts | 16 +++++++++++- .../default-entity-searcher.ts | 10 ++----- .../sorting-entity-searcher.ts | 14 ++++++++++ src/interfaces/dynamic-searcher.ts | 2 +- src/interfaces/entity-searcher.ts | 18 ++++++++++++- 6 files changed, 70 insertions(+), 16 deletions(-) diff --git a/src/dynamic-searchers/default-dynamic-searcher.ts b/src/dynamic-searchers/default-dynamic-searcher.ts index 0c3549d..eead8b9 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.ts @@ -1,6 +1,6 @@ -import { DefaultEntitySearcher } from '../entity-searchers/default-entity-searcher.js'; import { DynamicSearcher } from '../interfaces/dynamic-searcher.js'; import { EntityResult } from '../interfaces/entity-result.js'; +import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; import { Query } from '../interfaces/query.js'; @@ -23,9 +23,9 @@ export class DefaultDynamicSearcher implements DynamicSearcher, - private readonly secondarySearcher: DefaultEntitySearcher - ) {} + private readonly mainSearcher: EntitySearcher, + private readonly secondarySearcher: EntitySearcher + ) { } /** * {@inheritDoc DynamicSearcher.indexEntities} @@ -114,6 +114,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher 0 ? this.reindexSecondarySearcher(entitiesToInsert, getId, getTerms) : new Meta(); } @@ -163,7 +164,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher string[], - searcher: DefaultEntitySearcher + searcher: EntitySearcher ): boolean { const presentTerms: string[] | null = searcher.tryGetTerms(entityId); if (presentTerms === null) { @@ -191,6 +192,21 @@ export class DefaultDynamicSearcher implements DynamicSearcher term === terms2[index]); } + /** + * {@inheritDoc DynamicSearcher.removeEntity} + */ + public removeEntity(id: TId): boolean { + return this.mainSearcher.removeEntity(id) || this.secondarySearcher.removeEntity(id); + } + + /** + * {@inheritDoc DynamicSearcher.replaceEntity} + */ + public replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { + return this.mainSearcher.replaceEntity(id, newEntity, newEntityId) || + this.secondarySearcher.replaceEntity(id, newEntity, newEntityId); + } + /** * {@inheritDoc DynamicSearcher.save} */ diff --git a/src/dynamic-searchers/timing-dynamic-searcher.ts b/src/dynamic-searchers/timing-dynamic-searcher.ts index ea14037..f16727a 100644 --- a/src/dynamic-searchers/timing-dynamic-searcher.ts +++ b/src/dynamic-searchers/timing-dynamic-searcher.ts @@ -17,7 +17,7 @@ export class TimingDynamicSearcher implements DynamicSearcher) {} + public constructor(private readonly dynamicSearcher: DynamicSearcher) { } /** * {@inheritDoc DynamicSearcher.removeEntities} @@ -99,6 +99,20 @@ export class TimingDynamicSearcher implements DynamicSearcher implements EntitySearcher implements EntitySearcher implements EntitySearcher extends MementoSerializable { * @returns The terms of the entities. */ getTerms(): string[]; + + /** + * Removes the entity with the given id. + * @param id The id of the entity to remove. + * @returns True if the entity was present, false otherwise. + */ + removeEntity(id: TId): boolean; + + /** + * Replaces the entity with the given id. The terms of the entity are not updated. + * @param id The id of the entity to replace. + * @param newEntity The new entity. + * @param newEntityId The id of the new entity. + * @returns True if the id to replace was present, false otherwise. + */ + replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean; } From c3393790d402cfd8c384c511e6ab91cb3192c22a Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 09:46:48 +0200 Subject: [PATCH 031/105] fix: check --- src/dynamic-searchers/default-dynamic-searcher.ts | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/src/dynamic-searchers/default-dynamic-searcher.ts b/src/dynamic-searchers/default-dynamic-searcher.ts index eead8b9..d44afee 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.ts @@ -114,8 +114,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher 0 ? this.reindexSecondarySearcher(entitiesToInsert, getId, getTerms) : new Meta(); + return entitiesToInsert.length > 0 ? this.reindexSecondarySearcher(entitiesToInsert, getId, getTerms) : new Meta(); } /** From 7a76e073c92ceee9518d62e274040a17a3f93735 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 09:47:09 +0200 Subject: [PATCH 032/105] notes --- notes.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/notes.md b/notes.md index d5e60a9..179dcc9 100644 --- a/notes.md +++ b/notes.md @@ -94,12 +94,15 @@ suffixArraySearcherIndexing: 3655 Check if doc comments are complete for the new files. +Readme: mention that matches are sorted by their internal index. Changelog: paddingLeft, paddingRight and paddingMiddle were moved from the NormalizerConfig to the NgramNormalizerConfig. +check what happens for unknown enum values. + // '$$', // '!', // '!$$', From c263ad975dd7391b95b5e36eab99090ecd150827 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 09:50:56 +0200 Subject: [PATCH 033/105] regression test --- src/regression-test/output/_indexing-meta.txt | 6 +++--- src/regression-test/output/boston-deletion.txt | 16 ++++++++-------- .../output/boulder-creek-substitution.txt | 12 ++++++------ .../output/carcassonne-prefix.txt | 14 +++++++------- ...nne-infix.txt => carcassonne-substring.txt} | 6 +++--- .../output/carcassonne-suffix.txt | 6 +++--- .../output/kuwait-city-prefix.txt | 16 ++++++++-------- src/regression-test/output/kuwait-city.txt | 16 ++++++++-------- .../output/munich-insertion.txt | 14 +++++++------- .../output/tbilisi-deletion.txt | 4 ++-- src/regression-test/output/tbilisi.txt | 6 +++--- src/regression-test/output/tokyo-prefix.txt | 18 +++++++++--------- .../output/t\303\274bingen-transposition.txt" | 14 +++++++------- 13 files changed, 74 insertions(+), 74 deletions(-) rename src/regression-test/output/{carcassonne-infix.txt => carcassonne-substring.txt} (87%) diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 6664cf5..2e974c3 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,10 +1,10 @@ { "numberOfInvalidTerms": 1, - "ngramNormalizationDuration": 267, + "ngramNormalizationDuration": 290, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1350, + "defaultNormalizationDuration": 1445, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 6687 + "indexingDuration": 7022 } \ No newline at end of file diff --git a/src/regression-test/output/boston-deletion.txt b/src/regression-test/output/boston-deletion.txt index e79d565..f3d59ab 100644 --- a/src/regression-test/output/boston-deletion.txt +++ b/src/regression-test/output/boston-deletion.txt @@ -5,18 +5,18 @@ } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality -1 Bosti Bosti 0.67 -2 Bosto Bosto 0.67 -3 Bosta Bosta 0.67 +1 Bosta Bosta 0.67 +2 Bosti Bosti 0.67 +3 Bosto Bosto 0.67 4 Bost Bost 0.63 -5 Boston Boston 0.54 -6 Boston Boston 0.54 +5 Bostan Bostan 0.54 +6 Bostel Bostel 0.54 7 Boston Boston 0.54 8 Boston Boston 0.54 -9 Bostel Bostel 0.54 -10 Bostan Bostan 0.54 \ No newline at end of file +9 Boston Boston 0.54 +10 Boston Boston 0.54 \ No newline at end of file diff --git a/src/regression-test/output/boulder-creek-substitution.txt b/src/regression-test/output/boulder-creek-substitution.txt index 5ea8a03..e44b95e 100644 --- a/src/regression-test/output/boulder-creek-substitution.txt +++ b/src/regression-test/output/boulder-creek-substitution.txt @@ -13,10 +13,10 @@ Rank Entity Matched String 1 Boulder Creek Boulder Creek 0.75 2 Boulder City Boulder City 0.61 3 Boulder Boulder 0.54 -4 Bouldercombe Bouldercombe 0.54 -5 South Boulder South Boulder 0.54 -6 Boulder Boulder 0.54 -7 Boulder Hill Boulder Hill 0.54 -8 The Boulders The Boulders 0.47 -9 Boulders Boulders 0.47 +4 Boulder Boulder 0.54 +5 Boulder Hill Boulder Hill 0.54 +6 Bouldercombe Bouldercombe 0.54 +7 South Boulder South Boulder 0.54 +8 Boulders Boulders 0.47 +9 The Boulders The Boulders 0.47 10 Boulder Junction Boulder Junction 0.45 \ No newline at end of file diff --git a/src/regression-test/output/carcassonne-prefix.txt b/src/regression-test/output/carcassonne-prefix.txt index e591958..c8b7710 100644 --- a/src/regression-test/output/carcassonne-prefix.txt +++ b/src/regression-test/output/carcassonne-prefix.txt @@ -5,18 +5,18 @@ } { - "queryDuration": 9 + "queryDuration": 8 } Rank Entity Matched String Quality 1 Carcasse Carcasse 0.74 -2 Carcassonne Carcassonne 0.63 -3 Carasso Carasso 0.63 -4 Carcas Carcas 0.63 +2 Carasso Carasso 0.63 +3 Carcas Carcas 0.63 +4 Carcasi Carcasi 0.63 5 Carcasí Carcasí 0.63 -6 Carcasi Carcasi 0.63 +6 Carcassonne Carcassonne 0.63 7 Carcaboso Carcaboso 0.57 8 Casaracra Casaracra 0.57 -9 Cassou Cassou 0.53 -10 Caracase Caracase 0.53 \ No newline at end of file +9 Caracase Caracase 0.53 +10 Cassou Cassou 0.53 \ No newline at end of file diff --git a/src/regression-test/output/carcassonne-infix.txt b/src/regression-test/output/carcassonne-substring.txt similarity index 87% rename from src/regression-test/output/carcassonne-infix.txt rename to src/regression-test/output/carcassonne-substring.txt index 50596ad..a5d710e 100644 --- a/src/regression-test/output/carcassonne-infix.txt +++ b/src/regression-test/output/carcassonne-substring.txt @@ -5,17 +5,17 @@ } { - "queryDuration": 4 + "queryDuration": 5 } Rank Entity Matched String Quality 1 Cassone Cassone 0.75 2 Casson Casson 0.71 -3 Cassou Cassou 0.59 +3 Canossa Canossa 0.59 4 Cassola Cassola 0.59 5 Cassop Cassop 0.59 -6 Canossa Canossa 0.59 +6 Cassou Cassou 0.59 7 Cassoneca Cassoneca 0.57 8 Cassongue Cassongue 0.57 9 Carcassonne Carcassonne 0.55 diff --git a/src/regression-test/output/carcassonne-suffix.txt b/src/regression-test/output/carcassonne-suffix.txt index 6b22d4a..dec75b6 100644 --- a/src/regression-test/output/carcassonne-suffix.txt +++ b/src/regression-test/output/carcassonne-suffix.txt @@ -5,7 +5,7 @@ } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality @@ -15,8 +15,8 @@ Rank Entity Matched String 3 Sonna Sonna 0.67 4 Sonna Sonna 0.67 5 Sone Sone 0.63 -6 Sonon Sonon 0.63 +6 Sone Sone 0.63 7 Sone Sone 0.63 -8 Sone Sone 0.63 +8 Sonon Sonon 0.63 9 Sonnega Sonnega 0.59 10 Sonnar Sonnar 0.54 \ No newline at end of file diff --git a/src/regression-test/output/kuwait-city-prefix.txt b/src/regression-test/output/kuwait-city-prefix.txt index 5a11bc0..512a974 100644 --- a/src/regression-test/output/kuwait-city-prefix.txt +++ b/src/regression-test/output/kuwait-city-prefix.txt @@ -10,13 +10,13 @@ Rank Entity Matched String Quality -⁦1 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.69 ⁩ -⁦2 ⁩⁧ مدينة الحق⁩⁧ مدينة الحق⁩⁦0.69 ⁩ +⁦1 ⁩⁧ مدينة الحق⁩⁧ مدينة الحق⁩⁦0.69 ⁩ +⁦2 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.69 ⁩ ⁦3 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦0.66 ⁩ -⁦4 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.63 ⁩ -⁦5 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.63 ⁩ -⁦6 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.63 ⁩ -⁦7 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.63 ⁩ -⁦8 ⁩⁧ مدينة المجد⁩⁧ مدينة المجد⁩⁦0.63 ⁩ +⁦4 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.63 ⁩ +⁦5 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.63 ⁩ +⁦6 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.63 ⁩ +⁦7 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.63 ⁩ +⁦8 ⁩⁧ مدينة الطيب⁩⁧ مدينة الطيب⁩⁦0.63 ⁩ ⁦9 ⁩⁧ مدينة العرب⁩⁧ مدينة العرب⁩⁦0.63 ⁩ -⁦10 ⁩⁧ مدينة الطيب⁩⁧ مدينة الطيب⁩⁦0.63 ⁩ \ No newline at end of file +⁦10 ⁩⁧ مدينة المجد⁩⁧ مدينة المجد⁩⁦0.63 ⁩ \ No newline at end of file diff --git a/src/regression-test/output/kuwait-city.txt b/src/regression-test/output/kuwait-city.txt index fadcd48..f6e7af7 100644 --- a/src/regression-test/output/kuwait-city.txt +++ b/src/regression-test/output/kuwait-city.txt @@ -11,12 +11,12 @@ Rank Entity Matched String Quality ⁦1 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦1.00 ⁩ -⁦2 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.58 ⁩ -⁦3 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.58 ⁩ -⁦4 ⁩⁧ مدينة الوردي⁩⁧ مدينة الوردي⁩⁦0.58 ⁩ -⁦5 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.58 ⁩ -⁦6 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.58 ⁩ -⁦7 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.58 ⁩ +⁦2 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.58 ⁩ +⁦3 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.58 ⁩ +⁦4 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.58 ⁩ +⁦5 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.58 ⁩ +⁦6 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.58 ⁩ +⁦7 ⁩⁧ مدينة الضباط⁩⁧ مدينة الضباط⁩⁦0.58 ⁩ ⁦8 ⁩⁧ مدينة العمال⁩⁧ مدينة العمال⁩⁦0.58 ⁩ -⁦9 ⁩⁧ مدينة الضباط⁩⁧ مدينة الضباط⁩⁦0.58 ⁩ -⁦10 ⁩⁧ مدينة المجد⁩⁧ مدينة المجد⁩⁦0.58 ⁩ \ No newline at end of file +⁦9 ⁩⁧ مدينة المجد⁩⁧ مدينة المجد⁩⁦0.58 ⁩ +⁦10 ⁩⁧ مدينة الوردي⁩⁧ مدينة الوردي⁩⁦0.58 ⁩ \ No newline at end of file diff --git a/src/regression-test/output/munich-insertion.txt b/src/regression-test/output/munich-insertion.txt index 42706cb..b3b6efd 100644 --- a/src/regression-test/output/munich-insertion.txt +++ b/src/regression-test/output/munich-insertion.txt @@ -5,7 +5,7 @@ } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality @@ -14,9 +14,9 @@ Rank Entity Matched String 2 Munini Munini 0.59 3 Munichis Munichis 0.53 4 New Munich New Munich 0.52 -5 Mutuini Mutuini 0.47 -6 Muniene Muniene 0.47 -7 Mumuni Mumuni 0.47 -8 Muninga Muninga 0.47 -9 Munichoco Munichoco 0.47 -10 Munhini Munhini 0.47 \ No newline at end of file +5 Mumuni Mumuni 0.47 +6 Munhini Munhini 0.47 +7 Munichoco Munichoco 0.47 +8 Muniene Muniene 0.47 +9 Muninga Muninga 0.47 +10 Mutuini Mutuini 0.47 \ No newline at end of file diff --git a/src/regression-test/output/tbilisi-deletion.txt b/src/regression-test/output/tbilisi-deletion.txt index b7cc493..0bcd146 100644 --- a/src/regression-test/output/tbilisi-deletion.txt +++ b/src/regression-test/output/tbilisi-deletion.txt @@ -18,5 +18,5 @@ Rank Entity Matched String 6 თეჯისი თეჯისი 0.41 7 ყვიბისი ყვიბისი 0.36 8 ყვიბისი ყვიბისი 0.36 -9 თერგვისი თერგვისი 0.32 -10 თამარისი თამარისი 0.32 \ No newline at end of file +9 თამარისი თამარისი 0.32 +10 თერგვისი თერგვისი 0.32 \ No newline at end of file diff --git a/src/regression-test/output/tbilisi.txt b/src/regression-test/output/tbilisi.txt index ae9eb2d..130aea4 100644 --- a/src/regression-test/output/tbilisi.txt +++ b/src/regression-test/output/tbilisi.txt @@ -17,6 +17,6 @@ Rank Entity Matched String 5 სხვილისი სხვილისი 0.42 6 სხვილისი სხვილისი 0.42 7 თეჯისი თეჯისი 0.36 -8 წნელისი წნელისი 0.36 -9 ლისი ლისი 0.36 -10 ძალისი ძალისი 0.36 \ No newline at end of file +8 ლისი ლისი 0.36 +9 ძალისი ძალისი 0.36 +10 წნელისი წნელისი 0.36 \ No newline at end of file diff --git a/src/regression-test/output/tokyo-prefix.txt b/src/regression-test/output/tokyo-prefix.txt index 1e16145..813032a 100644 --- a/src/regression-test/output/tokyo-prefix.txt +++ b/src/regression-test/output/tokyo-prefix.txt @@ -11,12 +11,12 @@ Rank Entity Matched String Quality 1 東京都 東京都 0.47 -2 東區 東區 0.32 -3 東澳 東澳 0.32 -4 東区 東区 0.32 -5 東村 東村 0.32 -6 東豐 東豐 0.32 -7 東勢 東勢 0.32 -8 東興 東興 0.32 -9 東園 東園 0.32 -10 東片 東片 0.32 \ No newline at end of file +2 東勢 東勢 0.32 +3 東区 東区 0.32 +4 東區 東區 0.32 +5 東園 東園 0.32 +6 東村 東村 0.32 +7 東澳 東澳 0.32 +8 東片 東片 0.32 +9 東興 東興 0.32 +10 東豐 東豐 0.32 \ No newline at end of file diff --git "a/src/regression-test/output/t\303\274bingen-transposition.txt" "b/src/regression-test/output/t\303\274bingen-transposition.txt" index d57bfda..15dddf3 100644 --- "a/src/regression-test/output/t\303\274bingen-transposition.txt" +++ "b/src/regression-test/output/t\303\274bingen-transposition.txt" @@ -12,11 +12,11 @@ Rank Entity Matched String 1 Tübingen Tübingen 0.76 2 Tüfingen Tüfingen 0.57 -3 Tusingene Tusingene 0.47 -4 Täbingen Täbingen 0.47 -5 Hübingen Hübingen 0.47 -6 Tuningen Tuningen 0.47 -7 Tumlingen Tumlingen 0.47 -8 Bübingen Bübingen 0.47 -9 Tumringen Tumringen 0.47 +3 Bübingen Bübingen 0.47 +4 Hübingen Hübingen 0.47 +5 Täbingen Täbingen 0.47 +6 Tumlingen Tumlingen 0.47 +7 Tumringen Tumringen 0.47 +8 Tuningen Tuningen 0.47 +9 Tusingene Tusingene 0.47 10 Tuttlingen Tuttlingen 0.43 \ No newline at end of file From afd1f98fd83ccbb2474ba086944018c106fe34b2 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 12:18:38 +0200 Subject: [PATCH 034/105] feat: warnings for the prefix searcher --- src/suffix-array-searchers/prefix-searcher.ts | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index 13cf5e2..abbc33d 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -16,6 +16,7 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.index} */ index(terms: string[]): Meta { + console.warn('PrefixSearcher.index was called. Call SuffixArraySearcher.index instead.'); return this.suffixArraySearcher.index(terms); } @@ -39,6 +40,7 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.save} */ save(memento: Memento): void { + console.warn('PrefixSearcher.save was called. Call SuffixArraySearcher.save instead.'); this.suffixArraySearcher.save(memento); } @@ -46,6 +48,7 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.load} */ load(memento: Memento): void { + console.warn('PrefixSearcher.load was called. Call SuffixArraySearcher.load instead.'); this.suffixArraySearcher.load(memento); } From 6b2962735f158cfe8fcc9478f6e8919e5f67c15e Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 12:19:18 +0200 Subject: [PATCH 035/105] feat: sort order for the config --- src/config.ts | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/src/config.ts b/src/config.ts index f7c37e5..054e364 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,5 +1,6 @@ import { FuzzySearchConfig } from './fuzzy-searchers/fuzzy-search-config.js'; import { NormalizerConfig } from './normalization/normalizer-config.js'; +import { SortOrder } from './sort-order.js'; import { SubstringSearchConfig } from './suffix-array-searchers/substring-search-config.js'; /** @@ -10,12 +11,14 @@ export class Config { * Creates a new instance of the Config class. * @param normalizerConfig The configuration for the default normalizer. * @param maxQueryLength The maximum query length. + * @param sortOrder The sort order for the entity matches. * @param fuzzySearchConfig The fuzzy search configuration. * @param substringSearchConfig The substring search configuration. */ public constructor( public normalizerConfig: NormalizerConfig, public maxQueryLength: number, + public sortOrder: SortOrder, public fuzzySearchConfig: FuzzySearchConfig, public substringSearchConfig: SubstringSearchConfig ) { } @@ -26,8 +29,10 @@ export class Config { */ public static createDefaultConfig(): Config { const normalizerConfig = NormalizerConfig.createDefaultConfig(); + const maxQueryLength = 150; + const sortOrder = SortOrder.QualityAndMatchedString; const fuzzySearchConfig = FuzzySearchConfig.createDefaultConfig(); const substringSearchConfig = SubstringSearchConfig.createDefaultConfig(); - return new Config(normalizerConfig, 150, fuzzySearchConfig, substringSearchConfig); + return new Config(normalizerConfig, maxQueryLength, sortOrder, fuzzySearchConfig, substringSearchConfig); } } From 376e1f4cefd5ae42da47e47168d270bf782d7740 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 16:38:02 +0200 Subject: [PATCH 036/105] feat: searcher switch --- .../default-dynamic-searcher.ts | 3 +- .../default-entity-searcher.ts | 5 +- .../entity-searcher-factory.ts | 29 +++++--- src/interfaces/index.ts | 1 + src/interfaces/query.ts | 20 ++++- src/interfaces/searcher-type.ts | 20 +++++ src/string-searchers/index.ts | 1 + src/string-searchers/normalizing-searcher.ts | 5 +- src/string-searchers/searcher-switch.ts | 74 +++++++++++++++++++ src/suffix-array-searchers/prefix-searcher.ts | 2 +- 10 files changed, 140 insertions(+), 20 deletions(-) create mode 100644 src/interfaces/searcher-type.ts create mode 100644 src/string-searchers/searcher-switch.ts diff --git a/src/dynamic-searchers/default-dynamic-searcher.ts b/src/dynamic-searchers/default-dynamic-searcher.ts index d44afee..1ed82a5 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.ts @@ -47,7 +47,8 @@ export class DefaultDynamicSearcher implements DynamicSearcher this.maxQueryLength) { - query = new Query(query.string.substring(0, this.maxQueryLength), query.topN, query.minQuality); + query = new Query( + query.string.substring(0, this.maxQueryLength), query.topN, query.minQuality, query.searcherTypes); } return ResultMerger.mergeResults(this.mainSearcher.getMatches(query), this.secondarySearcher.getMatches(query)); } diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 87fa7f6..f52400f 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -6,6 +6,7 @@ import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -92,8 +93,8 @@ export class DefaultEntitySearcher implements EntitySearcher { - const stringSearcherQuery: Query = new Query(query.string, Infinity, query.minQuality); - const result = this.stringSearcher.getMatches(stringSearcherQuery); + query = new Query(query.string, Infinity, query.minQuality, [SearcherType.Fuzzy]); + const result = this.stringSearcher.getMatches(query); const matches: EntityMatch[] = this.getMatchesFromResult(result, query.topN); return new EntityResult(matches, query, result.meta); } diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index e6e17ff..7ee0a0c 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -2,6 +2,7 @@ import { Config } from '../config.js'; import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; +import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; @@ -11,6 +12,8 @@ import { Normalizer } from '../interfaces/normalizer.js'; import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from '../string-searchers/normalizing-searcher.js'; import { PrefixSearcher } from '../suffix-array-searchers/prefix-searcher.js'; +import { SearcherSwitch } from '../string-searchers/searcher-switch.js'; +import { SortingEntitySearcher } from './sorting-entity-searcher.js'; import { SortingSearcher } from '../string-searchers/sorting-searcher.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SubstringSearchConfig } from '../suffix-array-searchers/substring-search-config.js'; @@ -21,34 +24,40 @@ import { SuffixArraySearcher } from '../suffix-array-searchers/suffix-array-sear */ export class EntitySearcherFactory { /** - * Creates a new default entity searcher. + * Creates a new entity searcher. * @typeParam TEntity The type of the entities. * @typeParam TId The type of the entity ids. * @param config The config to use. - * @returns The default entity searcher. + * @returns The entity searcher. */ - public static createSearcher(config: Config): DefaultEntitySearcher { + public static createSearcher(config: Config): EntitySearcher { const defaultNormalizer: Normalizer = this.createDefaultNormalizer(config); - // let stringSearcher: StringSearcher = new SuffixArraySearcher('$'); - // let stringSearcher: StringSearcher = this.createFuzzySearcher(config.fuzzySearchConfig); + const fuzzySearcher: StringSearcher = this.createFuzzySearcher(config.fuzzySearchConfig); const suffixArraySearcher: SuffixArraySearcher = this.createSubstringSearcher(config.substringSearchConfig); - // todo: searcher switch. - let stringSearcher = this.createPrefixSearcher(suffixArraySearcher); + const prefixSearcher = this.createPrefixSearcher(suffixArraySearcher); + + let stringSearcher: StringSearcher = new SearcherSwitch( + prefixSearcher, + suffixArraySearcher, + fuzzySearcher + ); stringSearcher = new DistinctSearcher(stringSearcher); stringSearcher = new SortingSearcher(stringSearcher); stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'defaultNormalizationDuration'); - return new DefaultEntitySearcher(stringSearcher); + let entitySearcher: EntitySearcher = new DefaultEntitySearcher(stringSearcher); + entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); + return entitySearcher; } private static createDefaultNormalizer(config: Config): Normalizer { - // todo: suffix array separator. const forbiddenCharacters = new Set( [ config.fuzzySearchConfig.paddingLeft.split(''), config.fuzzySearchConfig.paddingRight.split(''), - config.fuzzySearchConfig.paddingMiddle.split('') + config.fuzzySearchConfig.paddingMiddle.split(''), + config.substringSearchConfig.suffixArraySeparator.split(''), ].flat() ); diff --git a/src/interfaces/index.ts b/src/interfaces/index.ts index 9f3b429..f4daef8 100644 --- a/src/interfaces/index.ts +++ b/src/interfaces/index.ts @@ -7,4 +7,5 @@ export { MementoSerializable } from './memento-serializable.js'; export { Meta } from './meta.js'; export { Normalizer } from './normalizer.js'; export { Query } from './query.js'; +export { SearcherType } from './searcher-type.js'; export { StringSearcher } from './string-searcher.js'; diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index 5a6bba7..1daa811 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -1,3 +1,5 @@ +import { SearcherType } from './searcher-type.js' + /** * Holds the query string and query parameters. */ @@ -18,17 +20,27 @@ export class Query { */ public readonly minQuality: number; + /** + * The searcher types to use. + */ + public readonly searcherTypes: SearcherType[]; + /** * Creates a new instance of the Query class. * @param string The query string. * @param topN The maximum number of matches to return. Provide Infinity to return all matches. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the - * performance but reduce the number of matches. The value must be between 0 and 1; lower or larger values will be - * clamped. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. The value must be between 0 and 1; lower or larger values will be clamped. + * @param searcherTypes The searcher types to use. */ - public constructor(string: string, topN: number = 10, minQuality: number = 0.3) { + public constructor( + string: string, + topN: number = 10, + minQuality: number = 0.3, + searcherTypes: SearcherType[] = [SearcherType.Prefix, SearcherType.Substring, SearcherType.Fuzzy]) { this.string = string; this.topN = Math.max(0, topN); this.minQuality = Math.max(0, Math.min(1, minQuality)); + this.searcherTypes = searcherTypes; } } diff --git a/src/interfaces/searcher-type.ts b/src/interfaces/searcher-type.ts new file mode 100644 index 0000000..3c3a515 --- /dev/null +++ b/src/interfaces/searcher-type.ts @@ -0,0 +1,20 @@ +/** + * Searcher types. + */ +export enum SearcherType { + + /** + * Fuzzy searcher type. + */ + Fuzzy = "fuzzy", + + /** + * Substring searcher type. + */ + Substring = "substring", + + /** + * Prefix searcher type. + */ + Prefix = "prefix" +} \ No newline at end of file diff --git a/src/string-searchers/index.ts b/src/string-searchers/index.ts index ab7f9ea..365a752 100644 --- a/src/string-searchers/index.ts +++ b/src/string-searchers/index.ts @@ -4,4 +4,5 @@ export { LiteralSearcher } from './literal-searcher.js'; export { Match } from './match.js'; export { NormalizingSearcher } from './normalizing-searcher.js'; export { Result } from './result.js'; +export { SearcherSwitch } from './searcher-switch.js'; export { SortingSearcher } from './sorting-searcher.js'; diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index a8c6638..47a50d3 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -19,7 +19,7 @@ export class NormalizingSearcher implements StringSearcher { private readonly stringSearcher: StringSearcher, private readonly normalizer: Normalizer, private readonly normalizationDurationMetaKey: string = 'normalizationDuration' - ) {} + ) { } /** * {@inheritDoc StringSearcher.index} @@ -40,7 +40,8 @@ export class NormalizingSearcher implements StringSearcher { * {@inheritDoc StringSearcher.getMatches} */ public getMatches(query: Query): Result { - const normalizedQuery = new Query(this.normalizer.normalize(query.string), query.topN, query.minQuality); + const normalizedQuery = new Query( + this.normalizer.normalize(query.string), query.topN, query.minQuality, query.searcherTypes); const result = this.stringSearcher.getMatches(normalizedQuery); return new Result(result.matches, query, result.meta); } diff --git a/src/string-searchers/searcher-switch.ts b/src/string-searchers/searcher-switch.ts new file mode 100644 index 0000000..f0ab93e --- /dev/null +++ b/src/string-searchers/searcher-switch.ts @@ -0,0 +1,74 @@ +import { Memento } from '../interfaces/memento.js'; +import { Meta } from '../interfaces/meta.js'; +import { MetaMerger } from '../commons/meta-merger.js'; +import { Query } from '../interfaces/query.js'; +import { Result } from '../string-searchers/result.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; +import { StringSearcher } from '../interfaces/string-searcher.js'; + +/** + * A searcher switch that routes to the prefix searcher, the substring searcher, or the fuzzy searcher. The prefix + * searcher is a simple wrapper around the suffix array searcher and only relevant for getMatches. It will be always up + * to date with the substring searcher. + */ +export class SearcherSwitch implements StringSearcher { + + /** + * Creates a new instance of the SearcherSwitch class. + * @param prefixSearcher The prefix searcher. + * @param substringSearcher The substring searcher. + * @param fuzzySearcher The fuzzy searcher. + */ + public constructor( + private readonly prefixSearcher: StringSearcher, + private readonly substringSearcher: StringSearcher, + private readonly fuzzySearcher: StringSearcher, + ) { + } + + /** + * {@inheritDoc StringSearcher.index} + */ + index(terms: string[]): Meta { + const substringSearcherMeta = this.substringSearcher.index(terms); + const fuzzySearcherMeta = this.fuzzySearcher.index(terms); + return MetaMerger.mergeMeta(fuzzySearcherMeta, substringSearcherMeta); + } + + /** + * {@inheritDoc StringSearcher.getMatches} + */ + getMatches(query: Query): Result { + + if (query.searcherTypes.length != 1) { + throw new Error('SearcherSwitch.getMatches only supports queries with a single searcher type.'); + } + + switch (query.searcherTypes[0]) { + case SearcherType.Prefix: + return this.prefixSearcher.getMatches(query); + case SearcherType.Substring: + return this.substringSearcher.getMatches(query); + case SearcherType.Fuzzy: + return this.fuzzySearcher.getMatches(query); + default: + throw new Error(`Unknown searcher type: ${query.searcherTypes[0]}`); + } + } + + /** + * {@inheritDoc StringSearcher.save} + */ + save(memento: Memento): void { + this.substringSearcher.save(memento); + this.fuzzySearcher.save(memento); + } + + /** + * {@inheritDoc StringSearcher.load} + */ + load(memento: Memento): void { + this.substringSearcher.load(memento); + this.fuzzySearcher.load(memento); + } +} \ No newline at end of file diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index abbc33d..71809e5 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -28,7 +28,7 @@ export class PrefixSearcher implements StringSearcher { return new Result([], query, new Meta()); } const modifiedQueryString = this.modifyQueryString(query.string); - const modifiedQuery = new Query(modifiedQueryString, query.topN, query.minQuality); + const modifiedQuery = new Query(modifiedQueryString, query.topN, query.minQuality, query.searcherTypes); return this.suffixArraySearcher.getMatches(modifiedQuery, query.string.length); } From 5f68843bab8d63ae336c82a0d2db9f7aa870baec Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 16:40:38 +0200 Subject: [PATCH 037/105] test: prefix searcher --- .../prefix-searcher.test.ts | 47 +++++++++++++++++++ 1 file changed, 47 insertions(+) create mode 100644 src/suffix-array-searchers/prefix-searcher.test.ts diff --git a/src/suffix-array-searchers/prefix-searcher.test.ts b/src/suffix-array-searchers/prefix-searcher.test.ts new file mode 100644 index 0000000..2161982 --- /dev/null +++ b/src/suffix-array-searchers/prefix-searcher.test.ts @@ -0,0 +1,47 @@ +import { Match } from '../string-searchers/match.js'; +import { PrefixSearcher } from './prefix-searcher.js'; +import { Query } from '../interfaces/query.js'; +import { SuffixArraySearcher } from './suffix-array-searcher.js'; + +const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher('$'); +suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); +const prefixSearcher: PrefixSearcher = new PrefixSearcher(suffixArraySearcher); + +function getMatches(queryString: string): Match[] { + const query = new Query(queryString, 10, 0); + const matches = prefixSearcher.getMatches(query).matches; + matches.sort((m1, m2) => m1.index - m2.index); + return matches; +} + +test('can find exact match test 1', () => { + expect(getMatches('Alice')).toEqual([new Match(0, 1)]); +}); + +test('can find prefix match test 1', () => { + expect(getMatches('B')).toEqual([new Match(1, 1 / 3)]); +}); + +test('can find prefix match test 2', () => { + expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); +}); + +test('can find prefix match test 2', () => { + expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); +}); + +test('can not find substring matches', () => { + expect(getMatches('lice')).toEqual([]); +}) + +test('empty query returns no matches', () => { + expect(getMatches('')).toEqual([]); +}); + +test('null query returns no matches', () => { + expect(getMatches(null!)).toEqual([]); +}); + +test('undefined query returns no matches', () => { + expect(getMatches(undefined!)).toEqual([]); +}); \ No newline at end of file From dc694d0a36374cf5871469312efb28743dca0bf0 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 16:42:15 +0200 Subject: [PATCH 038/105] basic usage --- src/basic-usage.ts | 17 +++++++++-------- 1 file changed, 9 insertions(+), 8 deletions(-) diff --git a/src/basic-usage.ts b/src/basic-usage.ts index df975db..515761d 100644 --- a/src/basic-usage.ts +++ b/src/basic-usage.ts @@ -1,9 +1,10 @@ -import { Match } from './string-searchers/match.js'; -import { Query } from './interfaces/query.js'; -import { SuffixArraySearcher } from './suffix-array-searchers/suffix-array-searcher.js'; +import { DefaultNormalizer } from './normalization/default-normalizer.js'; +import { NormalizerConfig } from './normalization/normalizer-config.js'; -const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher(); -suffixArraySearcher.index(['Alice', 'Bob', 'Carol', 'Charlie']); -const matches = suffixArraySearcher.getMatches(new Query('li')).matches; -console.log(matches); -console.log('finished'); +const config = NormalizerConfig.createDefaultConfig(); +config.allowCharacter = (_c: string) => true; +const normalizer = DefaultNormalizer.create(config); +// const result = normalizer.normalize('Tô'); +const result = 'Tô'.normalize("NFKD") +console.log("Result:") +console.log(result); \ No newline at end of file From de9476d790342e2aeb25ed62039c4f61ebd93900 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 17:29:19 +0200 Subject: [PATCH 039/105] fix: default entity searcher --- src/entity-searchers/default-entity-searcher.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index f52400f..8ae1580 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -93,8 +93,8 @@ export class DefaultEntitySearcher implements EntitySearcher { - query = new Query(query.string, Infinity, query.minQuality, [SearcherType.Fuzzy]); - const result = this.stringSearcher.getMatches(query); + const stringSearcherQuery: Query = new Query(query.string, Infinity, query.minQuality, [SearcherType.Fuzzy]); + const result = this.stringSearcher.getMatches(stringSearcherQuery); const matches: EntityMatch[] = this.getMatchesFromResult(result, query.topN); return new EntityResult(matches, query, result.meta); } From df50f560300c92d75e0535f7d99f6e9055156ee3 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 17:30:20 +0200 Subject: [PATCH 040/105] regression test --- notes.md | 2 ++ src/regression-test/output/_indexing-meta.txt | 7 ++++--- src/regression-test/output/boston-deletion.txt | 9 +++++++-- .../output/boulder-creek-substitution.txt | 7 ++++++- src/regression-test/output/carcassonne-prefix.txt | 7 ++++++- src/regression-test/output/carcassonne-substring.txt | 9 +++++++-- src/regression-test/output/carcassonne-suffix.txt | 7 ++++++- src/regression-test/output/kuwait-city-prefix.txt | 7 ++++++- src/regression-test/output/kuwait-city.txt | 7 ++++++- src/regression-test/output/munich-insertion.txt | 9 +++++++-- src/regression-test/output/tbilisi-deletion.txt | 7 ++++++- src/regression-test/output/tbilisi.txt | 7 ++++++- src/regression-test/output/tokyo-prefix.txt | 7 ++++++- src/regression-test/output/tokyo.txt | 7 ++++++- .../output/t\303\274bingen-transposition.txt" | 7 ++++++- 15 files changed, 87 insertions(+), 19 deletions(-) diff --git a/notes.md b/notes.md index 179dcc9..ef9510e 100644 --- a/notes.md +++ b/notes.md @@ -92,6 +92,8 @@ suffixArraySearcherIndexing: 3663 suffixArraySearcherIndexing: 3638 suffixArraySearcherIndexing: 3655 +revert basic-usage.ts + Check if doc comments are complete for the new files. Readme: mention that matches are sorted by their internal index. diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 2e974c3..6fd97e1 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,10 +1,11 @@ { "numberOfInvalidTerms": 1, - "ngramNormalizationDuration": 290, + "ngramNormalizationDuration": 276, + "suffixArraySearcherIndexing": 2757, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1445, + "defaultNormalizationDuration": 1288, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 7022 + "indexingDuration": 9406 } \ No newline at end of file diff --git a/src/regression-test/output/boston-deletion.txt b/src/regression-test/output/boston-deletion.txt index f3d59ab..dfe7936 100644 --- a/src/regression-test/output/boston-deletion.txt +++ b/src/regression-test/output/boston-deletion.txt @@ -1,11 +1,16 @@ { "string": "bostn", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { - "queryDuration": 3 + "queryDuration": 2 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/boulder-creek-substitution.txt b/src/regression-test/output/boulder-creek-substitution.txt index e44b95e..d1b71dc 100644 --- a/src/regression-test/output/boulder-creek-substitution.txt +++ b/src/regression-test/output/boulder-creek-substitution.txt @@ -1,7 +1,12 @@ { "string": "boulder creak", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/carcassonne-prefix.txt b/src/regression-test/output/carcassonne-prefix.txt index c8b7710..0b76560 100644 --- a/src/regression-test/output/carcassonne-prefix.txt +++ b/src/regression-test/output/carcassonne-prefix.txt @@ -1,7 +1,12 @@ { "string": "carcasso", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/carcassonne-substring.txt b/src/regression-test/output/carcassonne-substring.txt index a5d710e..080eb04 100644 --- a/src/regression-test/output/carcassonne-substring.txt +++ b/src/regression-test/output/carcassonne-substring.txt @@ -1,11 +1,16 @@ { "string": "cassonn", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { - "queryDuration": 5 + "queryDuration": 4 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/carcassonne-suffix.txt b/src/regression-test/output/carcassonne-suffix.txt index dec75b6..30e687f 100644 --- a/src/regression-test/output/carcassonne-suffix.txt +++ b/src/regression-test/output/carcassonne-suffix.txt @@ -1,7 +1,12 @@ { "string": "sonne", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/kuwait-city-prefix.txt b/src/regression-test/output/kuwait-city-prefix.txt index 512a974..6d59e3a 100644 --- a/src/regression-test/output/kuwait-city-prefix.txt +++ b/src/regression-test/output/kuwait-city-prefix.txt @@ -1,7 +1,12 @@ { "string": "مدينة الك", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/kuwait-city.txt b/src/regression-test/output/kuwait-city.txt index f6e7af7..b5f61ba 100644 --- a/src/regression-test/output/kuwait-city.txt +++ b/src/regression-test/output/kuwait-city.txt @@ -1,7 +1,12 @@ { "string": "مدينة الكويت", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/munich-insertion.txt b/src/regression-test/output/munich-insertion.txt index b3b6efd..582b322 100644 --- a/src/regression-test/output/munich-insertion.txt +++ b/src/regression-test/output/munich-insertion.txt @@ -1,11 +1,16 @@ { "string": "muniich", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { - "queryDuration": 3 + "queryDuration": 2 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/tbilisi-deletion.txt b/src/regression-test/output/tbilisi-deletion.txt index 0bcd146..766cb55 100644 --- a/src/regression-test/output/tbilisi-deletion.txt +++ b/src/regression-test/output/tbilisi-deletion.txt @@ -1,7 +1,12 @@ { "string": "თბიისი", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/tbilisi.txt b/src/regression-test/output/tbilisi.txt index 130aea4..f928388 100644 --- a/src/regression-test/output/tbilisi.txt +++ b/src/regression-test/output/tbilisi.txt @@ -1,7 +1,12 @@ { "string": "თბილისი", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/tokyo-prefix.txt b/src/regression-test/output/tokyo-prefix.txt index 813032a..a814edf 100644 --- a/src/regression-test/output/tokyo-prefix.txt +++ b/src/regression-test/output/tokyo-prefix.txt @@ -1,7 +1,12 @@ { "string": "東京", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git a/src/regression-test/output/tokyo.txt b/src/regression-test/output/tokyo.txt index 6f7757b..d283a4f 100644 --- a/src/regression-test/output/tokyo.txt +++ b/src/regression-test/output/tokyo.txt @@ -1,7 +1,12 @@ { "string": "東京都", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { diff --git "a/src/regression-test/output/t\303\274bingen-transposition.txt" "b/src/regression-test/output/t\303\274bingen-transposition.txt" index 15dddf3..a602a63 100644 --- "a/src/regression-test/output/t\303\274bingen-transposition.txt" +++ "b/src/regression-test/output/t\303\274bingen-transposition.txt" @@ -1,7 +1,12 @@ { "string": "tübignen", "topN": 10, - "minQuality": 0.3 + "minQuality": 0.3, + "searcherTypes": [ + "prefix", + "substring", + "fuzzy" + ] } { From 70cb7a6dc0474e7664038c02503074ffc2ea66dc Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 20:58:49 +0200 Subject: [PATCH 041/105] regression test --- src/regression-test/output/_indexing-meta.txt | 8 +++---- .../output/carcassonne-prefix.txt | 14 ++++++------ .../output/carcassonne-substring.txt | 18 +++++++-------- .../output/carcassonne-suffix.txt | 22 +++++++++---------- .../output/kuwait-city-prefix.txt | 18 +++++++-------- src/regression-test/output/kuwait-city.txt | 2 +- src/regression-test/output/tbilisi.txt | 4 ++-- src/regression-test/output/tokyo-prefix.txt | 16 +++++++------- src/regression-test/output/tokyo.txt | 2 +- 9 files changed, 52 insertions(+), 52 deletions(-) diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 6fd97e1..41409b4 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,11 +1,11 @@ { "numberOfInvalidTerms": 1, - "ngramNormalizationDuration": 276, - "suffixArraySearcherIndexing": 2757, + "ngramNormalizationDuration": 272, + "suffixArraySearcherIndexing": 2768, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1288, + "defaultNormalizationDuration": 1344, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 9406 + "indexingDuration": 9585 } \ No newline at end of file diff --git a/src/regression-test/output/carcassonne-prefix.txt b/src/regression-test/output/carcassonne-prefix.txt index 0b76560..5fc9b06 100644 --- a/src/regression-test/output/carcassonne-prefix.txt +++ b/src/regression-test/output/carcassonne-prefix.txt @@ -10,17 +10,17 @@ } { - "queryDuration": 8 + "queryDuration": 9 } Rank Entity Matched String Quality -1 Carcasse Carcasse 0.74 -2 Carasso Carasso 0.63 -3 Carcas Carcas 0.63 -4 Carcasi Carcasi 0.63 -5 Carcasí Carcasí 0.63 -6 Carcassonne Carcassonne 0.63 +1 Carcassonne Carcassonne 2.73 +2 Carcasse Carcasse 0.74 +3 Carasso Carasso 0.63 +4 Carcas Carcas 0.63 +5 Carcasi Carcasi 0.63 +6 Carcasí Carcasí 0.63 7 Carcaboso Carcaboso 0.57 8 Casaracra Casaracra 0.57 9 Caracase Caracase 0.53 diff --git a/src/regression-test/output/carcassonne-substring.txt b/src/regression-test/output/carcassonne-substring.txt index 080eb04..4718e56 100644 --- a/src/regression-test/output/carcassonne-substring.txt +++ b/src/regression-test/output/carcassonne-substring.txt @@ -15,13 +15,13 @@ Rank Entity Matched String Quality -1 Cassone Cassone 0.75 -2 Casson Casson 0.71 -3 Canossa Canossa 0.59 -4 Cassola Cassola 0.59 -5 Cassop Cassop 0.59 -6 Cassou Cassou 0.59 -7 Cassoneca Cassoneca 0.57 -8 Cassongue Cassongue 0.57 -9 Carcassonne Carcassonne 0.55 +1 Carcassonne Carcassonne 1.64 +2 Cassone Cassone 0.75 +3 Casson Casson 0.71 +4 Canossa Canossa 0.59 +5 Cassola Cassola 0.59 +6 Cassop Cassop 0.59 +7 Cassou Cassou 0.59 +8 Cassoneca Cassoneca 0.57 +9 Cassongue Cassongue 0.57 10 Cassoday Cassoday 0.53 \ No newline at end of file diff --git a/src/regression-test/output/carcassonne-suffix.txt b/src/regression-test/output/carcassonne-suffix.txt index 30e687f..00d3d67 100644 --- a/src/regression-test/output/carcassonne-suffix.txt +++ b/src/regression-test/output/carcassonne-suffix.txt @@ -10,18 +10,18 @@ } { - "queryDuration": 3 + "queryDuration": 0 } Rank Entity Matched String Quality -1 Sonnen Sonnen 0.81 -2 Sonene Sonene 0.68 -3 Sonna Sonna 0.67 -4 Sonna Sonna 0.67 -5 Sone Sone 0.63 -6 Sone Sone 0.63 -7 Sone Sone 0.63 -8 Sonon Sonon 0.63 -9 Sonnega Sonnega 0.59 -10 Sonnar Sonnar 0.54 \ No newline at end of file +1 Sonnen Sonnen 2.83 +2 Sonnega Sonnega 2.71 +3 Sonneberg Sonneberg 2.56 +4 Sonneborn Sonneborn 2.56 +5 Sonnefeld Sonnefeld 2.56 +6 Sonnendal Sonnendal 2.56 +7 Sonneneck Sonneneck 2.56 +8 Sonnenham Sonnenham 2.56 +9 Sonnenhof Sonnenhof 2.56 +10 Sonnental Sonnental 2.56 \ No newline at end of file diff --git a/src/regression-test/output/kuwait-city-prefix.txt b/src/regression-test/output/kuwait-city-prefix.txt index 6d59e3a..83891bc 100644 --- a/src/regression-test/output/kuwait-city-prefix.txt +++ b/src/regression-test/output/kuwait-city-prefix.txt @@ -15,13 +15,13 @@ Rank Entity Matched String Quality -⁦1 ⁩⁧ مدينة الحق⁩⁧ مدينة الحق⁩⁦0.69 ⁩ -⁦2 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.69 ⁩ -⁦3 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦0.66 ⁩ -⁦4 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.63 ⁩ -⁦5 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.63 ⁩ -⁦6 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.63 ⁩ -⁦7 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.63 ⁩ -⁦8 ⁩⁧ مدينة الطيب⁩⁧ مدينة الطيب⁩⁦0.63 ⁩ -⁦9 ⁩⁧ مدينة العرب⁩⁧ مدينة العرب⁩⁦0.63 ⁩ +⁦1 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦2.75 ⁩ +⁦2 ⁩⁧ مدينة الكويرة المهجورة⁩⁧ مدينة الكويرة المهجورة⁩⁦2.41 ⁩ +⁦3 ⁩⁧ المدينة الكرفانيه⁩⁧ المدينة الكرفانيه⁩⁦1.53 ⁩ +⁦4 ⁩⁧ مدينة الحق⁩⁧ مدينة الحق⁩⁦0.69 ⁩ +⁦5 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.69 ⁩ +⁦6 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.63 ⁩ +⁦7 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.63 ⁩ +⁦8 ⁩⁧ مدينة الشعب⁩⁧ مدينة الشعب⁩⁦0.63 ⁩ +⁦9 ⁩⁧ مدينة الصدر⁩⁧ مدينة الصدر⁩⁦0.63 ⁩ ⁦10 ⁩⁧ مدينة المجد⁩⁧ مدينة المجد⁩⁦0.63 ⁩ \ No newline at end of file diff --git a/src/regression-test/output/kuwait-city.txt b/src/regression-test/output/kuwait-city.txt index b5f61ba..e56f10c 100644 --- a/src/regression-test/output/kuwait-city.txt +++ b/src/regression-test/output/kuwait-city.txt @@ -15,7 +15,7 @@ Rank Entity Matched String Quality -⁦1 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦1.00 ⁩ +⁦1 ⁩⁧ مدينة الكويت⁩⁧ مدينة الكويت⁩⁦3.00 ⁩ ⁦2 ⁩⁧ مدينة البعث⁩⁧ مدينة البعث⁩⁦0.58 ⁩ ⁦3 ⁩⁧ مدينة الرد⁩⁧ مدينة الرد⁩⁦0.58 ⁩ ⁦4 ⁩⁧ مدينة الشرق⁩⁧ مدينة الشرق⁩⁦0.58 ⁩ diff --git a/src/regression-test/output/tbilisi.txt b/src/regression-test/output/tbilisi.txt index f928388..34a6d8d 100644 --- a/src/regression-test/output/tbilisi.txt +++ b/src/regression-test/output/tbilisi.txt @@ -15,8 +15,8 @@ Rank Entity Matched String Quality -1 თბილისი თბილისი 1.00 -2 თბილისი თბილისი 1.00 +1 თბილისი თბილისი 3.00 +2 თბილისი თბილისი 3.00 3 მილისი მილისი 0.47 4 მილისი მილისი 0.47 5 სხვილისი სხვილისი 0.42 diff --git a/src/regression-test/output/tokyo-prefix.txt b/src/regression-test/output/tokyo-prefix.txt index a814edf..17eda29 100644 --- a/src/regression-test/output/tokyo-prefix.txt +++ b/src/regression-test/output/tokyo-prefix.txt @@ -15,13 +15,13 @@ Rank Entity Matched String Quality -1 東京都 東京都 0.47 -2 東勢 東勢 0.32 -3 東区 東区 0.32 -4 東區 東區 0.32 -5 東園 東園 0.32 -6 東村 東村 0.32 -7 東澳 東澳 0.32 -8 東片 東片 0.32 +1 東京都 東京都 2.67 +2 西東京市 西東京市 1.50 +3 東勢 東勢 0.32 +4 東区 東区 0.32 +5 東區 東區 0.32 +6 東園 東園 0.32 +7 東村 東村 0.32 +8 東澳 東澳 0.32 9 東興 東興 0.32 10 東豐 東豐 0.32 \ No newline at end of file diff --git a/src/regression-test/output/tokyo.txt b/src/regression-test/output/tokyo.txt index d283a4f..c882aae 100644 --- a/src/regression-test/output/tokyo.txt +++ b/src/regression-test/output/tokyo.txt @@ -15,4 +15,4 @@ Rank Entity Matched String Quality -1 東京都 東京都 1.00 \ No newline at end of file +1 東京都 東京都 3.00 \ No newline at end of file From 13c697bc9901f76c4342c634015eec7a3828cdbb Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Wed, 22 Oct 2025 21:01:02 +0200 Subject: [PATCH 042/105] docs(SearchState) --- src/entity-searchers/search-state.ts | 33 ++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) create mode 100644 src/entity-searchers/search-state.ts diff --git a/src/entity-searchers/search-state.ts b/src/entity-searchers/search-state.ts new file mode 100644 index 0000000..aca98e7 --- /dev/null +++ b/src/entity-searchers/search-state.ts @@ -0,0 +1,33 @@ +import { EntityMatch } from '../interfaces/entity-match.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; + +/** + * The search state during an entity search. + * @typeParam TEntity The type of the entities. + */ +export class SearchState { + /** + * The search query. + */ + public readonly query: Query; + /** + * The indexes of the entities that have already been matched. + */ + public readonly matchedIndexes: Set = new Set(); + /** + * The matched entities. + */ + public readonly matches: EntityMatch[] = []; + /** + * The meta data for each searcher type. + */ + public readonly meta: Meta[] = []; + /** + * Creates a new instance of the SearchState class. + * @param query The search query. + */ + public constructor(query: Query) { + this.query = query; + } +} From 9997aee0ead984cc50d7ba9a52f7627dba65a0b8 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 12:31:35 +0200 Subject: [PATCH 043/105] feat: performance test --- .vscode/launch.json | 18 ++++-- .vscode/tasks.json | 13 ++++- src/performance-test/main.ts | 56 +++++++++++++++++++ .../output/_indexing-meta.txt | 11 ++++ src/performance-test/output/performance.txt | 28 ++++++++++ src/performance/index.ts | 1 + src/performance/performance-test.ts | 39 ++++++++++++- src/performance/query-counts.ts | 25 +++++++++ src/performance/report.ts | 15 ++++- 9 files changed, 194 insertions(+), 12 deletions(-) create mode 100644 src/performance-test/main.ts create mode 100644 src/performance-test/output/_indexing-meta.txt create mode 100644 src/performance-test/output/performance.txt create mode 100644 src/performance/query-counts.ts diff --git a/.vscode/launch.json b/.vscode/launch.json index 82d9d50..e796d7a 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -5,17 +5,25 @@ "type": "node", "request": "launch", "name": "Launch Basic Usage", - "skipFiles": ["/**"], + "skipFiles": [ + "/**" + ], "program": "${workspaceFolder}/dist/basic-usage.js", - "outFiles": ["${workspaceFolder}/**/*.js"] + "outFiles": [ + "${workspaceFolder}/**/*.js" + ] }, { "type": "node", "request": "launch", "name": "Run Regression Test", - "skipFiles": ["/**"], + "skipFiles": [ + "/**" + ], "program": "${workspaceFolder}/dist/regression-test/main.js", - "outFiles": ["${workspaceFolder}/**/*.js"] + "outFiles": [ + "${workspaceFolder}/**/*.js" + ] } ] -} +} \ No newline at end of file diff --git a/.vscode/tasks.json b/.vscode/tasks.json index 0632500..6d68c89 100644 --- a/.vscode/tasks.json +++ b/.vscode/tasks.json @@ -5,12 +5,21 @@ "type": "typescript", "tsconfig": "tsconfig.json", "option": "watch", - "problemMatcher": ["$tsc-watch"], + "problemMatcher": [ + "$tsc-watch" + ], "group": { "kind": "build", "isDefault": true }, "label": "tsc: watch - tsconfig.json" + }, + { + "label": "Performance Test", + "type": "shell", + "command": "node dist/performance-test/main.js", + "problemMatcher": [], + "group": "test" } ] -} +} \ No newline at end of file diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts new file mode 100644 index 0000000..d7b6ab7 --- /dev/null +++ b/src/performance-test/main.ts @@ -0,0 +1,56 @@ +/** + * Performance test. Run this file and check the git diff of the output files to see how the performance changed. +*/ + +import { readFileSync, writeFileSync } from 'fs'; +import { Config } from '../config.js'; +import { Meta } from '../interfaces/meta.js'; +import { PerformanceTest } from '../performance/performance-test.js'; +import { Report } from '../performance/report.js'; +import { SearcherFactory } from '../searcher-factory.js'; +import { TestRunParameters } from '../performance/test-run-parameters.js'; + +const outputPath = './src/performance-test/output'; +const seed = 0; +const numberOfQueries = 1_000; +const topN = 10; +const minQuality = 0; + +const text = readFileSync('./data/world-ctvs.txt', 'utf8'); +const lines = text.split('\n').slice(1); +const entities = lines.map((l, index) => ({ id: index, name: l })); + +interface GeoEntity { + id: number; + name: string; +} + +const config = Config.createDefaultConfig(); +config.normalizerConfig.allowCharacter = (_) => true; +const searcher = SearcherFactory.createSearcher(config); + +console.log(`Indexing ${entities.length} entities...`); +const indexingMeta: Meta = searcher.indexEntities( + entities, + (e) => e.id, + (e) => e.name.split(';') +); +writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { encoding: 'utf8' }); + +const performanceTest: PerformanceTest = new PerformanceTest(searcher); +const testRunParameters: TestRunParameters = + new TestRunParameters(seed, numberOfQueries, topN, minQuality); + +console.log('Running performance test...'); +const report: Report = performanceTest.run(testRunParameters); +const reportJson = reportToJson(report); +writeFileSync(`${outputPath}/performance.txt`, reportJson); +console.log(reportJson); + +function metaToJson(meta: Meta): string { + return JSON.stringify(Object.fromEntries(meta.allEntries), null, 2); +} + +function reportToJson(report: Report): string { + return JSON.stringify(report, null, 2); +} diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt new file mode 100644 index 0000000..7d7c6c1 --- /dev/null +++ b/src/performance-test/output/_indexing-meta.txt @@ -0,0 +1,11 @@ +{ + "numberOfInvalidTerms": 1, + "ngramNormalizationDuration": 296, + "suffixArraySearcherIndexing": 2880, + "numberOfDistinctTerms": 1190185, + "defaultNormalizationDuration": 1377, + "numberOfSurrogateCharacters": 46, + "numberOfEntities": 1237154, + "numberOfTerms": 1237486, + "indexingDuration": 10006 +} \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt new file mode 100644 index 0000000..55d4010 --- /dev/null +++ b/src/performance-test/output/performance.txt @@ -0,0 +1,28 @@ +{ + "testParameters": { + "testSeed": 0, + "numberOfQueries": 1000, + "topN": 10, + "minQuality": 0 + }, + "queryCounts": { + "prefixQueries": 483, + "substringQueries": 517, + "transpositionErrors": 143 + }, + "totalDuration": 11896.186388000013, + "averageDuration": 11.896186388000013, + "standardDeviation": 22.0089920110702, + "fastest": { + "query": "nkh", + "duration": 0.015542000001005363 + }, + "slowest": { + "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", + "duration": 211.1774170000026 + }, + "longest": { + "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", + "duration": 211.1774170000026 + } +} \ No newline at end of file diff --git a/src/performance/index.ts b/src/performance/index.ts index 3a089ec..44a0be4 100644 --- a/src/performance/index.ts +++ b/src/performance/index.ts @@ -1,4 +1,5 @@ export { PerformanceTest } from './performance-test.js'; +export { QueryCounts } from './query-counts.js'; export { Report } from './report.js'; export { TestRunParameters } from './test-run-parameters.js'; export { TimedQuery } from './timed-query.js'; diff --git a/src/performance/performance-test.ts b/src/performance/performance-test.ts index 15d72b1..59825d2 100644 --- a/src/performance/performance-test.ts +++ b/src/performance/performance-test.ts @@ -1,5 +1,6 @@ import { DynamicSearcher } from '../interfaces/dynamic-searcher.js'; import { Query } from '../interfaces/query.js'; +import { QueryCounts } from './query-counts.js'; import { Report } from './report.js'; import { TestRunParameters } from './test-run-parameters.js'; import { TimedQuery } from './timed-query.js'; @@ -18,6 +19,11 @@ export class PerformanceTest { */ private random: () => number; + /** + * The query counts. + */ + private queryCounts: QueryCounts; + /** * Creates a new instance of the PerformanceTest class. * @param dynamicSearcher The searcher to test. @@ -25,6 +31,7 @@ export class PerformanceTest { public constructor(public readonly dynamicSearcher: DynamicSearcher) { this.terms = []; this.random = this.mulberry32(0); + this.queryCounts = new QueryCounts(); } /** @@ -36,6 +43,7 @@ export class PerformanceTest { const measurements: TimedQuery[] = []; this.terms = this.dynamicSearcher.getTerms().filter((t) => t); this.random = this.mulberry32(parameters.testSeed); + this.queryCounts = new QueryCounts(); const numberOfQueries = this.terms.length === 0 ? 0 : parameters.numberOfQueries; for (let i = 0, l = numberOfQueries; i < l; i++) { @@ -49,6 +57,7 @@ export class PerformanceTest { return Report.Create( new TestRunParameters(parameters.testSeed, numberOfQueries, parameters.topN, parameters.minQuality), + this.queryCounts, measurements ); } @@ -60,8 +69,34 @@ export class PerformanceTest { */ private GetRandomQueryString(): string { const term: string = this.GetRandomTerm(); - const length: number = this.getRandomInteger(1, term.length); - return term.substring(0, length); + let start: number; + + if (this.RollPercentage(0.5)) { + start = 0; + this.queryCounts.prefixQueries++; + } + else { + start = this.getRandomInteger(0, term.length); + this.queryCounts.substringQueries++; + } + + const end = this.getRandomInteger(start + 1, term.length + 1); + const substring: string = term.substring(start, end); + + if (substring.length < 2 || this.RollPercentage(0.8)) { + return substring; + } + + const errorPosition = this.getRandomInteger(0, substring.length - 1); + this.queryCounts.transpositionErrors++; + return substring.substring(0, errorPosition) + + substring[errorPosition + 1] + + substring[errorPosition] + + substring.substring(errorPosition + 2); + } + + private RollPercentage(chance: number): boolean { + return this.random() < chance; } /** diff --git a/src/performance/query-counts.ts b/src/performance/query-counts.ts new file mode 100644 index 0000000..68014ce --- /dev/null +++ b/src/performance/query-counts.ts @@ -0,0 +1,25 @@ +/** + * Counts of the generated queries. + */ +export class QueryCounts { + + /** + * The number of generated prefix queries. + */ + public prefixQueries: number = 0; + + /** + * The number of generated substring queries. + */ + public substringQueries: number = 0; + + /** + * The number of queries that included a transposition error. + */ + public transpositionErrors: number = 0; + + /** + * Creates a new instance of the QueryCounts class. + */ + public constructor() { } +} \ No newline at end of file diff --git a/src/performance/report.ts b/src/performance/report.ts index add5e04..5ec5d1c 100644 --- a/src/performance/report.ts +++ b/src/performance/report.ts @@ -1,3 +1,4 @@ +import { QueryCounts } from './query-counts.js'; import { TestRunParameters } from './test-run-parameters.js'; import { TimedQuery } from './timed-query.js'; @@ -8,6 +9,7 @@ export class Report { /** * Creates a new instance of the Report class. * @param testParameters The parameters of the test run. + * @param queryCounts Counts of the generated and run queries. * @param totalDuration The total duration of all queries in milliseconds. * @param averageDuration The average duration of a query in milliseconds. * @param standardDeviation The standard deviation of the query durations in milliseconds. @@ -17,24 +19,30 @@ export class Report { */ public constructor( public readonly testParameters: TestRunParameters, + public readonly queryCounts: QueryCounts, public readonly totalDuration: number, public readonly averageDuration: number, public readonly standardDeviation: number, public readonly fastest: TimedQuery, public readonly slowest: TimedQuery, public readonly longest: TimedQuery - ) {} + ) { } /** * Creates a report from the timed queries. * @param testRunParameters The parameters of the test run. + * @param queryCounts Counts of the generated and run queries. * @param measurements The timed queries. * @returns The performance report. */ - public static Create(testRunParameters: TestRunParameters, measurements: TimedQuery[]): Report { + public static Create( + testRunParameters: TestRunParameters, + queryCounts: QueryCounts, + measurements: TimedQuery[]): Report { if (measurements.length === 0) { return new Report( testRunParameters, + queryCounts, 0, 0, 0, @@ -52,7 +60,8 @@ export class Report { const fastest: TimedQuery = measurements.reduce((prev, cur) => (prev.duration < cur.duration ? prev : cur)); const slowest: TimedQuery = measurements.reduce((prev, cur) => (prev.duration > cur.duration ? prev : cur)); const longest: TimedQuery = measurements.reduce((prev, cur) => (prev.query.length > cur.query.length ? prev : cur)); - return new Report(testRunParameters, totalDuration, averageDuration, standardDeviation, fastest, slowest, longest); + return new Report( + testRunParameters, queryCounts, totalDuration, averageDuration, standardDeviation, fastest, slowest, longest); } /** From 2da9bbf9dda15ce5c65ba86524d0827866e94080 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 12:33:21 +0200 Subject: [PATCH 044/105] readmes --- src/performance-test/readme.md | 1 + src/regression-test/readme.md | 1 + 2 files changed, 2 insertions(+) create mode 100644 src/performance-test/readme.md create mode 100644 src/regression-test/readme.md diff --git a/src/performance-test/readme.md b/src/performance-test/readme.md new file mode 100644 index 0000000..a101d85 --- /dev/null +++ b/src/performance-test/readme.md @@ -0,0 +1 @@ +Run the performance test via the 'Run Task' command in VS Code. \ No newline at end of file diff --git a/src/regression-test/readme.md b/src/regression-test/readme.md new file mode 100644 index 0000000..e1c2e75 --- /dev/null +++ b/src/regression-test/readme.md @@ -0,0 +1 @@ +Run the regression test via the 'Run and Debug' action on the left sidebar in VS Code. \ No newline at end of file From e8627b6b48cd306a2951b3712f30667688af4896 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 12:36:46 +0200 Subject: [PATCH 045/105] readme --- src/performance-test/{readme.md => READM.md} | 0 src/regression-test/{readme.md => READM.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename src/performance-test/{readme.md => READM.md} (100%) rename src/regression-test/{readme.md => READM.md} (100%) diff --git a/src/performance-test/readme.md b/src/performance-test/READM.md similarity index 100% rename from src/performance-test/readme.md rename to src/performance-test/READM.md diff --git a/src/regression-test/readme.md b/src/regression-test/READM.md similarity index 100% rename from src/regression-test/readme.md rename to src/regression-test/READM.md From db1aeebdd04bd8004e3eb0b596aa27d749f11ae4 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 12:37:01 +0200 Subject: [PATCH 046/105] readme --- src/performance-test/{READM.md => README.md} | 0 src/regression-test/{READM.md => README.md} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename src/performance-test/{READM.md => README.md} (100%) rename src/regression-test/{READM.md => README.md} (100%) diff --git a/src/performance-test/READM.md b/src/performance-test/README.md similarity index 100% rename from src/performance-test/READM.md rename to src/performance-test/README.md diff --git a/src/regression-test/READM.md b/src/regression-test/README.md similarity index 100% rename from src/regression-test/READM.md rename to src/regression-test/README.md From 5b4de97e19c9302af4f000214f2017c6ef539d3c Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 12:43:24 +0200 Subject: [PATCH 047/105] performance test --- src/fuzzy-searchers/fuzzy-searcher.ts | 7 ++++++- src/performance-test/output/_indexing-meta.txt | 9 +++++---- src/performance-test/output/performance.txt | 14 +++++++------- .../suffix-array-searcher.ts | 2 +- 4 files changed, 19 insertions(+), 13 deletions(-) diff --git a/src/fuzzy-searchers/fuzzy-searcher.ts b/src/fuzzy-searchers/fuzzy-searcher.ts index ce3c416..1c8bf2f 100644 --- a/src/fuzzy-searchers/fuzzy-searcher.ts +++ b/src/fuzzy-searchers/fuzzy-searcher.ts @@ -42,10 +42,10 @@ export class FuzzySearcher implements StringSearcher { * {@inheritDoc StringSearcher.index} */ public index(terms: string[]): Meta { + const start = performance.now(); this.invertedIndex = new InvertedIndex(); this.commonNgramCounts = new Int32Array(terms.length); this.numberOfNgrams = new Int32Array(terms.length); - const meta = new Meta(); let nofInvalidTerms = 0; for (let i = 0, l = terms.length; i < l; i++) { @@ -66,7 +66,12 @@ export class FuzzySearcher implements StringSearcher { } this.invertedIndex.seal(); + + const duration = Math.round(performance.now() - start); + + const meta = new Meta(); meta.add('numberOfInvalidTerms', nofInvalidTerms); + meta.add('indexingDurationFuzzySearcher', duration); return meta; } diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 7d7c6c1..7ea2e8c 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,11 +1,12 @@ { "numberOfInvalidTerms": 1, - "ngramNormalizationDuration": 296, - "suffixArraySearcherIndexing": 2880, + "indexingDurationFuzzySearcher": 3904, + "ngramNormalizationDuration": 287, + "indexingDurationSuffixArraySearcher": 2858, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1377, + "defaultNormalizationDuration": 1415, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 10006 + "indexingDuration": 9958 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index 55d4010..24ff2ed 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -10,19 +10,19 @@ "substringQueries": 517, "transpositionErrors": 143 }, - "totalDuration": 11896.186388000013, - "averageDuration": 11.896186388000013, - "standardDeviation": 22.0089920110702, + "totalDuration": 11835.207838999999, + "averageDuration": 11.835207838999999, + "standardDeviation": 21.88322318933828, "fastest": { - "query": "nkh", - "duration": 0.015542000001005363 + "query": "eo", + "duration": 0.014999999999417923 }, "slowest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 211.1774170000026 + "duration": 211.47399999999834 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 211.1774170000026 + "duration": 211.47399999999834 } } \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 770dd82..e00808b 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -46,7 +46,7 @@ export class SuffixArraySearcher implements StringSearcher { const duration = Math.round(performance.now() - start); const meta = new Meta(); - meta.add('suffixArraySearcherIndexing', duration); + meta.add('indexingDurationSuffixArraySearcher', duration); return meta; } From 8f878021fd394b3140fb03958a63ae27d9c4f50a Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 14:58:35 +0200 Subject: [PATCH 048/105] work work --- src/entity-searchers/index.ts | 2 ++ src/performance-test/main.ts | 4 +++- src/regression-test/main.ts | 10 ++++++---- 3 files changed, 11 insertions(+), 5 deletions(-) diff --git a/src/entity-searchers/index.ts b/src/entity-searchers/index.ts index 3552aae..19e1bbb 100644 --- a/src/entity-searchers/index.ts +++ b/src/entity-searchers/index.ts @@ -1,2 +1,4 @@ export { DefaultEntitySearcher } from './default-entity-searcher.js'; export { EntitySearcherFactory } from './entity-searcher-factory.js'; +export { SearchState } from './search-state.js'; +export { SortingEntitySearcher } from './sorting-entity-searcher.js'; diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts index d7b6ab7..7eaeb7a 100644 --- a/src/performance-test/main.ts +++ b/src/performance-test/main.ts @@ -35,7 +35,9 @@ const indexingMeta: Meta = searcher.indexEntities( (e) => e.id, (e) => e.name.split(';') ); -writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { encoding: 'utf8' }); +const metaJson = metaToJson(indexingMeta); +writeFileSync(`${outputPath}/_indexing-meta.txt`, metaJson, { encoding: 'utf8' }); +console.log(metaJson); const performanceTest: PerformanceTest = new PerformanceTest(searcher); const testRunParameters: TestRunParameters = diff --git a/src/regression-test/main.ts b/src/regression-test/main.ts index 5dd2d08..6f878cd 100644 --- a/src/regression-test/main.ts +++ b/src/regression-test/main.ts @@ -1,6 +1,6 @@ -/* - Regression tests. Run this file and check the git diff of the output files to see what changed. -*/ +/** + * Regression tests. Run this file and check the git diff of the output files to see what changed. + */ import { readFileSync, writeFileSync } from 'fs'; import { Config } from '../config.js'; @@ -36,7 +36,9 @@ const indexingMeta: Meta = searcher.indexEntities( (e) => e.id, (e) => e.name.split(';') ); -writeFileSync(`${outputPath}/_indexing-meta.txt`, metaToJson(indexingMeta), { encoding: 'utf8' }); +const metaJson = metaToJson(indexingMeta); +writeFileSync(`${outputPath}/_indexing-meta.txt`, metaJson, { encoding: 'utf8' }); +console.log(metaJson); console.log('Running queries...'); From 98bf465d8f8dd432fceec639a63f7f88e549eb78 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 15:00:17 +0200 Subject: [PATCH 049/105] remove notes --- notes.md | 111 ------------------------------------------------------- 1 file changed, 111 deletions(-) delete mode 100644 notes.md diff --git a/notes.md b/notes.md deleted file mode 100644 index ef9510e..0000000 --- a/notes.md +++ /dev/null @@ -1,111 +0,0 @@ -- Implement Suffix Array searcher. - -https://en.wikipedia.org/wiki/Suffix_array - -```python -n = len(S) - -def search(P: str) -> tuple[int, int]: - """ - Return indices (s, r) such that the interval A[s:r] (including the end - index) represents all suffixes of S that start with the pattern P. - """ - # Find starting position of interval - l = 0 # in Python, arrays are indexed starting at 0 - r = n - while l < r: - mid = (l + r) // 2 # division rounding down to nearest integer - # suffixAt(A[i]) is the ith smallest suffix - if P > suffixAt(A[mid]): - l = mid + 1 - else: - r = mid - s = l - - # Find ending position of interval - r = n - while l < r: - mid = (l + r) // 2 - if suffixAt(A[mid]).startswith(P): - l = mid + 1 - else: - r = mid - return (s, r) -``` - -Chat GPT (verify!) - -```js -function compareSubstringsOrdinal(strA, indexA, strB, indexB, length) { - const lenA = strA.length; - const lenB = strB.length; - const endA = Math.min(indexA + length, lenA); - const endB = Math.min(indexB + length, lenB); - - let iA = indexA; - let iB = indexB; - - while (iA < endA && iB < endB) { - const codeA = strA.charCodeAt(iA); - const codeB = strB.charCodeAt(iB); - - if (codeA < codeB) return -1; - if (codeA > codeB) return 1; - - iA++; - iB++; - } - - // If both ran out at the same time, they're equal - const lenComparedA = endA - indexA; - const lenComparedB = endB - indexB; - - if (lenComparedA === lenComparedB) return 0; - return lenComparedA < lenComparedB ? -1 : 1; -} -``` - -Attribution (Chat GPT) - -```text -/* - Original: https://github.com/eranmeir/Sufa-Suffix-Array-Csharp - Copyright (c) 2012 Eran Meir - SPDX-License-Identifier: MIT - - Translation to TypeScript, modifications and refactoring - (c) 2025 Kevin Schaal -*/ -``` - -Todo -==== - -Test second suffix array implementation. - -Check todos. - -Adjust usage examples to new config structure. - -First: -suffixArraySearcherIndexing: 3663 -suffixArraySearcherIndexing: 3638 -suffixArraySearcherIndexing: 3655 - -revert basic-usage.ts - -Check if doc comments are complete for the new files. - -Readme: mention that matches are sorted by their internal index. - -Changelog: - -paddingLeft, paddingRight and paddingMiddle were moved from the NormalizerConfig to the -NgramNormalizerConfig. - -check what happens for unknown enum values. - -// '$$', -// '!', -// '!$$', - From 15fc07d0bf0d36c8eb5c4dadb1b240909a1651d3 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 15:53:31 +0200 Subject: [PATCH 050/105] work work --- .../entity-searcher-factory.ts | 18 +++ src/entity-searchers/fast-entity-searcher.ts | 104 ++++++++++++++++++ src/entity-searchers/index.ts | 1 + .../output/_indexing-meta.txt | 10 +- src/performance-test/output/performance.txt | 22 ++-- src/performance/performance-test.ts | 2 +- 6 files changed, 140 insertions(+), 17 deletions(-) create mode 100644 src/entity-searchers/fast-entity-searcher.ts diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 7ee0a0c..25e6a05 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -3,6 +3,7 @@ import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; +import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; @@ -47,10 +48,16 @@ export class EntitySearcherFactory { stringSearcher = new SortingSearcher(stringSearcher); stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'defaultNormalizationDuration'); let entitySearcher: EntitySearcher = new DefaultEntitySearcher(stringSearcher); + entitySearcher = new FastEntitySearcher(entitySearcher); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; } + /** + * Creates the default normalizer. + * @param config The searcher configuration. + * @returns The default normalizer. + */ private static createDefaultNormalizer(config: Config): Normalizer { const forbiddenCharacters = new Set( [ @@ -73,6 +80,7 @@ export class EntitySearcherFactory { } /** + * Creates the fuzzy string searcher. * @param config The fuzzy search configuration. * @returns The fuzzy string searcher. */ @@ -90,11 +98,21 @@ export class EntitySearcherFactory { return fuzzySearcher; } + /** + * Creates the substring searcher. + * @param config The substring search configuration. + * @returns The substring searcher. + */ private static createSubstringSearcher(config: SubstringSearchConfig): SuffixArraySearcher { const substringSearcher = new SuffixArraySearcher(config.suffixArraySeparator); return substringSearcher; } + /** + * Creates the prefix searcher. + * @param suffixArraySearcher The suffix array searcher to use. + * @returns The prefix searcher. + */ private static createPrefixSearcher(suffixArraySearcher: SuffixArraySearcher): StringSearcher { return new PrefixSearcher(suffixArraySearcher); } diff --git a/src/entity-searchers/fast-entity-searcher.ts b/src/entity-searchers/fast-entity-searcher.ts new file mode 100644 index 0000000..a050d9c --- /dev/null +++ b/src/entity-searchers/fast-entity-searcher.ts @@ -0,0 +1,104 @@ +import { EntityResult } from "../interfaces/entity-result.js"; +import { EntitySearcher } from "../interfaces/entity-searcher.js"; +import { Memento } from "../interfaces/memento.js"; +import { Meta } from "../interfaces/meta.js"; +import { Query } from "../interfaces/query.js"; + +/** + * A entity searcher that tries to optimize performance by querying with an increased quality threshold at first. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + */ +export class FastEntitySearcher implements EntitySearcher { + + /** + * Creates a new instance of the FastEntitySearcher class. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + * @param entitySearcher + */ + public constructor(private readonly entitySearcher: EntitySearcher) { + } + + /** + * {@inheritDoc EntitySearcher.indexEntities} + */ + indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { + return this.entitySearcher.indexEntities(entities, getId, getTerms); + } + + /** + * {@inheritDoc EntitySearcher.getMatches} + */ + getMatches(query: Query): EntityResult { + if (query.topN === Infinity || query.minQuality >= 0.2) { + return this.entitySearcher.getMatches(query); + } + + const firstQuery = new Query(query.string, query.topN, 0.3, query.searcherTypes); + const result = this.entitySearcher.getMatches(firstQuery); + if (result.matches.length == query.topN) { + return new EntityResult(result.matches, query, result.meta); + } + else { + return this.entitySearcher.getMatches(query); + } + } + + /** + * {@inheritDoc EntitySearcher.tryGetEntity} + */ + tryGetEntity(id: TId): TEntity | null { + return this.entitySearcher.tryGetEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.getEntities} + */ + getEntities(): TEntity[] { + return this.entitySearcher.getEntities(); + } + + /** + * {@inheritDoc EntitySearcher.tryGetTerms} + */ + tryGetTerms(id: TId): string[] | null { + return this.entitySearcher.tryGetTerms(id); + } + + /** + * {@inheritDoc EntitySearcher.getTerms} + */ + getTerms(): string[] { + return this.entitySearcher.getTerms(); + } + + /** + * {@inheritDoc EntitySearcher.removeEntity} + */ + removeEntity(id: TId): boolean { + return this.entitySearcher.removeEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.replaceEntity} + */ + replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { + return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + } + + /** + * {@inheritDoc EntitySearcher.save} + */ + save(memento: Memento): void { + return this.entitySearcher.save(memento); + } + + /** + * {@inheritDoc EntitySearcher.load} + */ + load(memento: Memento): void { + return this.entitySearcher.load(memento); + } + +} \ No newline at end of file diff --git a/src/entity-searchers/index.ts b/src/entity-searchers/index.ts index 19e1bbb..03ce00c 100644 --- a/src/entity-searchers/index.ts +++ b/src/entity-searchers/index.ts @@ -1,4 +1,5 @@ export { DefaultEntitySearcher } from './default-entity-searcher.js'; export { EntitySearcherFactory } from './entity-searcher-factory.js'; +export { FastEntitySearcher } from './fast-entity-searcher.js'; export { SearchState } from './search-state.js'; export { SortingEntitySearcher } from './sorting-entity-searcher.js'; diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 7ea2e8c..1816093 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3904, - "ngramNormalizationDuration": 287, - "indexingDurationSuffixArraySearcher": 2858, + "indexingDurationFuzzySearcher": 3828, + "ngramNormalizationDuration": 281, + "indexingDurationSuffixArraySearcher": 2819, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1415, + "defaultNormalizationDuration": 1344, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 9958 + "indexingDuration": 9664 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index 24ff2ed..239ff46 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -1,28 +1,28 @@ { "testParameters": { "testSeed": 0, - "numberOfQueries": 1000, + "numberOfQueries": 2000, "topN": 10, "minQuality": 0 }, "queryCounts": { - "prefixQueries": 483, - "substringQueries": 517, - "transpositionErrors": 143 + "prefixQueries": 1003, + "substringQueries": 997, + "transpositionErrors": 299 }, - "totalDuration": 11835.207838999999, - "averageDuration": 11.835207838999999, - "standardDeviation": 21.88322318933828, + "totalDuration": 2274.5138099999676, + "averageDuration": 1.1372569049999839, + "standardDeviation": 5.465560304604728, "fastest": { - "query": "eo", - "duration": 0.014999999999417923 + "query": "村", + "duration": 0.012208000000100583 }, "slowest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 211.47399999999834 + "duration": 220.9579589999994 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 211.47399999999834 + "duration": 220.9579589999994 } } \ No newline at end of file diff --git a/src/performance/performance-test.ts b/src/performance/performance-test.ts index 59825d2..386e2a6 100644 --- a/src/performance/performance-test.ts +++ b/src/performance/performance-test.ts @@ -50,7 +50,7 @@ export class PerformanceTest { const queryString = this.GetRandomQueryString(); const query = new Query(queryString, parameters.topN, parameters.minQuality); const start = performance.now(); - const _result = this.dynamicSearcher.getMatches(query); + const _ = this.dynamicSearcher.getMatches(query); const duration = performance.now() - start; measurements.push(new TimedQuery(queryString, duration)); } From 0d3c09fac9ef3aa18d1cf9ec65002f2c12c26b7f Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 16:16:39 +0200 Subject: [PATCH 051/105] feat: new meta merger --- src/commons/meta-merger.test.ts | 63 +++++++++++++++++++++++++ src/commons/meta-merger.ts | 59 ++++++++++++++++------- src/dynamic-searchers/result-merger.ts | 6 +-- src/performance-test/main.ts | 2 +- src/string-searchers/searcher-switch.ts | 2 +- 5 files changed, 109 insertions(+), 23 deletions(-) create mode 100644 src/commons/meta-merger.test.ts diff --git a/src/commons/meta-merger.test.ts b/src/commons/meta-merger.test.ts new file mode 100644 index 0000000..a636366 --- /dev/null +++ b/src/commons/meta-merger.test.ts @@ -0,0 +1,63 @@ +/* eslint-disable @typescript-eslint/no-explicit-any */ + +import { Meta } from '../interfaces/meta.js'; +import { MetaMerger } from './meta-merger.js'; + +test('can merge two metas', () => { + const meta1 = new Meta(); + meta1.add("key1", "meta1Value"); + meta1.add("key2", 10); + meta1.add("key3", "someOtherValue"); + + const meta2 = new Meta(); + meta2.add("key1", "meta2Value"); + meta2.add("key2", 10); + + const mergedMeta = MetaMerger.mergeMeta([meta1, meta2]); + const entries: ReadonlyMap = mergedMeta.allEntries; + const expectedEntries = new Map([ + ["key1_0", "meta1Value"], + ["key1_1", "meta2Value"], + ["key2", 20], + ["key3", "someOtherValue"] + ]); + + checkMaps(expectedEntries, entries); +}); + +test('can merge three metas', () => { + const meta1 = new Meta(); + meta1.add("key1", "meta1Value"); + meta1.add("key2", 10); + meta1.add("key3", "someOtherValue"); + + const meta2 = new Meta(); + meta2.add("key1", "meta2Value"); + meta2.add("key2", 10); + + const meta3 = new Meta(); + meta3.add("key1", "meta3Value"); + meta3.add("key2", 5); + meta3.add("key4", "additionalValue"); + + const mergedMeta = MetaMerger.mergeMeta([meta1, meta2, meta3]); + const entries: ReadonlyMap = mergedMeta.allEntries; + const expectedEntries = new Map([ + ["key1_0", "meta1Value"], + ["key1_1", "meta2Value"], + ["key1_2", "meta3Value"], + ["key2", 25], + ["key3", "someOtherValue"], + ["key4", "additionalValue"] + ]); + + checkMaps(expectedEntries, entries); +}); + +function checkMaps(expectedMap: ReadonlyMap, actualMap: ReadonlyMap): void { + expect(expectedMap.size).toBe(actualMap.size); + for (const [key, value] of actualMap) { + expect(expectedMap.has(key)).toBe(true); + expect(expectedMap.get(key)).toBe(value); + } +} \ No newline at end of file diff --git a/src/commons/meta-merger.ts b/src/commons/meta-merger.ts index 37fdfa9..b84706e 100644 --- a/src/commons/meta-merger.ts +++ b/src/commons/meta-merger.ts @@ -2,33 +2,56 @@ import { Meta } from '../interfaces/meta.js'; +/** + * Merges {@link Meta} objects. + */ export class MetaMerger { /** - * Merges two {@link Meta} objects into a new {@link Meta} object. - * @param meta1 The first meta object. - * @param meta2 The second meta object. - * @returns The merged meta object. - */ - public static mergeMeta(meta1: Meta, meta2: Meta): Meta { - const newMetaEntries: Map = new Map(); + * Merges {@link Meta} objects into a new {@link Meta} object. Numbers for the same key are summed up, other values + * are stored with an index suffix. + * @param metas The meta objects to merge. + * @returns The merged meta object. + */ + public static mergeMeta(metas: Meta[]): Meta { + if (metas.length === 0) { + return new Meta(); + } + + if (metas.length === 1) { + return metas[0]; + } + + const metaLists: Map = new Map(); - for (const [key, value] of meta1.allEntries) { - newMetaEntries.set(key, value); + for (const meta of metas) { + for (const [key, value] of meta.allEntries) { + const present = metaLists.get(key); + if (present === undefined) { + metaLists.set(key, [value]); + } + else { + present.push(value); + } + } } - for (const [key, value] of meta2.allEntries) { - const presentValue = newMetaEntries.get(key); - if (presentValue === undefined) { - newMetaEntries.set(key, value); + const newMetaEntries: Map = new Map(); + + for (const [key, values] of metaLists) { + if (values.length === 1) { + newMetaEntries.set(key, values[0]); continue; } - if (typeof presentValue === 'number' && typeof value === 'number') { - newMetaEntries.set(key, presentValue + value); + + if (values.every(v => typeof v === 'number')) { + const sum = values.reduce((acc, val) => acc + val, 0); + newMetaEntries.set(key, sum); continue; } - newMetaEntries.delete(key); - newMetaEntries.set(`${key}_0`, presentValue); - newMetaEntries.set(`${key}_1`, value); + + for (let i = 0; i < values.length; i++) { + newMetaEntries.set(`${key}_${i}`, values[i]); + } } return new Meta(newMetaEntries); diff --git a/src/dynamic-searchers/result-merger.ts b/src/dynamic-searchers/result-merger.ts index e80645a..e1d6014 100644 --- a/src/dynamic-searchers/result-merger.ts +++ b/src/dynamic-searchers/result-merger.ts @@ -21,7 +21,7 @@ export class ResultMerger { ): EntityResult { const query: Query = result1.query; const newMatches = this.mergeMatches(result1.matches, result2.matches, query.topN); - const newMeta: Meta = MetaMerger.mergeMeta(result1.meta, result2.meta); + const newMeta: Meta = MetaMerger.mergeMeta([result1.meta, result2.meta]); return new EntityResult(newMatches, query, newMeta); } @@ -47,8 +47,8 @@ export class ResultMerger { const newMatches = [...matches1, ...matches2]; newMatches.sort((m1, m2) => m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : 0 ); return newMatches.length <= topN ? newMatches : newMatches.slice(0, topN); } diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts index 7eaeb7a..8c944c4 100644 --- a/src/performance-test/main.ts +++ b/src/performance-test/main.ts @@ -12,7 +12,7 @@ import { TestRunParameters } from '../performance/test-run-parameters.js'; const outputPath = './src/performance-test/output'; const seed = 0; -const numberOfQueries = 1_000; +const numberOfQueries = 2_000; const topN = 10; const minQuality = 0; diff --git a/src/string-searchers/searcher-switch.ts b/src/string-searchers/searcher-switch.ts index f0ab93e..b201272 100644 --- a/src/string-searchers/searcher-switch.ts +++ b/src/string-searchers/searcher-switch.ts @@ -32,7 +32,7 @@ export class SearcherSwitch implements StringSearcher { index(terms: string[]): Meta { const substringSearcherMeta = this.substringSearcher.index(terms); const fuzzySearcherMeta = this.fuzzySearcher.index(terms); - return MetaMerger.mergeMeta(fuzzySearcherMeta, substringSearcherMeta); + return MetaMerger.mergeMeta([fuzzySearcherMeta, substringSearcherMeta]); } /** From 43bfc81fc4a9495bd0793c7bc749851600d86025 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 16:18:21 +0200 Subject: [PATCH 052/105] feat: new default entity searcher --- .../default-entity-searcher.ts | 79 ++++++++++++++----- 1 file changed, 60 insertions(+), 19 deletions(-) diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 8ae1580..bc22999 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -4,8 +4,10 @@ import { EntityResult } from '../interfaces/entity-result.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; +import { MetaMerger } from '../commons/meta-merger.js'; import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; +import { SearchState } from './search-state.js'; import { SearcherType } from '../interfaces/searcher-type.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; @@ -45,6 +47,15 @@ export class DefaultEntitySearcher implements EntitySearcher implements EntitySearcher { - const stringSearcherQuery: Query = new Query(query.string, Infinity, query.minQuality, [SearcherType.Fuzzy]); - const result = this.stringSearcher.getMatches(stringSearcherQuery); - const matches: EntityMatch[] = this.getMatchesFromResult(result, query.topN); - return new EntityResult(matches, query, result.meta); + const searchState: SearchState = new SearchState(query); + + for (const { searcherType, qualityOffset } of this.searchersAndQualityOffsets) { + if (query.topN == searchState.matches.length) { + break; + } + + if (!query.searcherTypes.includes(searcherType)) { + continue; + } + + this.addMatchesFromSearcher(searchState, searcherType, qualityOffset); + } + + + const mergedMeta = MetaMerger.mergeMeta(searchState.meta); + return new EntityResult(searchState.matches, query, mergedMeta); + } + + /** + * Adds matches from a specific searcher to the search state. + * @param searchState The current search state. + * @param searcherType The type of the searcher. + * @param qualityOffset The quality offset to apply. + */ + private addMatchesFromSearcher( + searchState: SearchState, + searcherType: SearcherType, + qualityOffset: number + ): void { + const stringSearcherQuery: Query = new Query( + searchState.query.string, Infinity, searchState.query.minQuality, [searcherType]); + const result: Result = this.stringSearcher.getMatches(stringSearcherQuery); + this.addMatchesFromResult(searchState, result, qualityOffset); } /** - * Creates entity matches from the string searcher result. + * Add entity matches from the string searcher result to the search state. + * @param searchState The current search state. * @param result The string searcher result. - * @param topN The maximum number of matches to return. - * @returns The entity matches. + * @param qualityOffset The quality offset that is added. */ - private getMatchesFromResult(result: Result, topN: number): EntityMatch[] { - if (topN === 0) { - return []; + private addMatchesFromResult( + searchState: SearchState, + result: Result, + qualityOffset: number + ): void { + if (searchState.query.topN === 0) { + return; } - const matchedIndexes: Set = new Set(); - const matches: EntityMatch[] = []; - for (let i = 0, l = result.matches.length; i < l; i++) { const match = result.matches[i]; const entityIndex: number = this.termIndexToEntityIndex[match.index]; - if (!matchedIndexes.has(entityIndex) && this.entities[entityIndex] !== null) { - matchedIndexes.add(entityIndex); - matches.push(new EntityMatch(this.entities[entityIndex] as TEntity, match.quality, this.terms[match.index])); - if (matches.length === topN) { + if (!searchState.matchedIndexes.has(entityIndex) && this.entities[entityIndex] !== null) { + searchState.matchedIndexes.add(entityIndex); + searchState.matches.push(new EntityMatch( + this.entities[entityIndex] as TEntity, match.quality + qualityOffset, this.terms[match.index])); + if (searchState.matches.length === searchState.query.topN) { break; } } } - - return matches; } /** From ac83e98585b36e3f6f6fc6121e9af0a3b6fbb685 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 16:20:05 +0200 Subject: [PATCH 053/105] run regression test --- src/regression-test/output/_indexing-meta.txt | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 41409b4..855602e 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,11 +1,12 @@ { "numberOfInvalidTerms": 1, - "ngramNormalizationDuration": 272, - "suffixArraySearcherIndexing": 2768, + "indexingDurationFuzzySearcher": 3722, + "ngramNormalizationDuration": 338, + "indexingDurationSuffixArraySearcher": 2870, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1344, + "defaultNormalizationDuration": 1392, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 9585 + "indexingDuration": 9812 } \ No newline at end of file From 73221a87f460b5364bb07e68bc6003a950705cfd Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 16:25:23 +0200 Subject: [PATCH 054/105] fix: restore basic usage --- src/basic-usage.ts | 66 +++++++++++++++++++++++++++++++++++++++------- 1 file changed, 56 insertions(+), 10 deletions(-) diff --git a/src/basic-usage.ts b/src/basic-usage.ts index 515761d..be791d7 100644 --- a/src/basic-usage.ts +++ b/src/basic-usage.ts @@ -1,10 +1,56 @@ -import { DefaultNormalizer } from './normalization/default-normalizer.js'; -import { NormalizerConfig } from './normalization/normalizer-config.js'; - -const config = NormalizerConfig.createDefaultConfig(); -config.allowCharacter = (_c: string) => true; -const normalizer = DefaultNormalizer.create(config); -// const result = normalizer.normalize('Tô'); -const result = 'Tô'.normalize("NFKD") -console.log("Result:") -console.log(result); \ No newline at end of file +// import { Config } from './config'; +import { Query } from './interfaces/query.js'; +import { SearcherFactory } from './searcher-factory.js'; + +class Person { + constructor( + public id: number, + public firstName: string, + public lastName: string + ) { } +} + +const searcher = SearcherFactory.createDefaultSearcher(); + +// If your dataset contains non-latin characters, build the searcher in the following way instead: +/* const config = Config.createDefaultConfig(); +config.normalizerConfig.allowCharacter = (_c) => true; +const searcher = SearcherFactory.createSearcher(config); */ + +const persons = [ + { id: 23501, firstName: 'Alice', lastName: 'King' }, + { id: 99234, firstName: 'Bob', lastName: 'Bishop' }, + { id: 5823, firstName: 'Carol', lastName: 'Queen' }, + { id: 11923, firstName: 'Charlie', lastName: 'Rook' } +]; + +const indexingMeta = searcher.indexEntities( + persons, + (e) => e.id, + (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] +); +console.dir(indexingMeta); + +const result = searcher.getMatches(new Query('alice kign')); +console.dir(result); + +const removalResult = searcher.removeEntities([99234, 5823]); +console.dir(removalResult); + +const persons2 = [ + { id: 723, firstName: 'David', lastName: 'Knight' }, // new + { id: 2634, firstName: 'Eve', lastName: 'Pawn' }, // new + { id: 23501, firstName: 'Allie', lastName: 'King' }, // updated + { id: 11923, firstName: 'Charles', lastName: 'Rook' } // updated +]; + +const upsertMeta = searcher.upsertEntities( + persons2, + (e) => e.id, + (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] +); +console.dir(upsertMeta); + +const result2 = searcher.getMatches(new Query('allie')); +console.dir(result2); +console.log('Finished.'); \ No newline at end of file From 9f3e949e8e958382d7f178e1b5698e7df200fc02 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 16:26:41 +0200 Subject: [PATCH 055/105] chore: make q=0 the default quality threshold --- demo/fuzzy-demo.js | 12 ++---------- src/interfaces/query.ts | 2 +- 2 files changed, 3 insertions(+), 11 deletions(-) diff --git a/demo/fuzzy-demo.js b/demo/fuzzy-demo.js index e53042a..807752e 100644 --- a/demo/fuzzy-demo.js +++ b/demo/fuzzy-demo.js @@ -206,11 +206,7 @@ async function downloadAndIndexOsmData() { return; } - const data = { - entities: entities, - kind: 'osm-places', - latinOnly: false - }; + const data = { entities: entities, kind: 'osm-places', latinOnly: false }; indexingRequest.data = data; indexingRequest.searchDataConfig = self.getSearchDataConfig(data.kind); @@ -272,11 +268,7 @@ function generateAndIndexPersonDataPart2(indexingRequest, numberOfNames, randomS const latinOnly = personData.scripts.size === 1 && personData.scripts.has('Latn'); - const data = { - entities: entities, - kind: 'persons', - latinOnly: latinOnly - }; + const data = { entities: entities, kind: 'persons', latinOnly: latinOnly }; indexingRequest.data = data; indexingRequest.searchDataConfig = self.getSearchDataConfig(data.kind); diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index 1daa811..e43451d 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -36,7 +36,7 @@ export class Query { public constructor( string: string, topN: number = 10, - minQuality: number = 0.3, + minQuality: number = 0, searcherTypes: SearcherType[] = [SearcherType.Prefix, SearcherType.Substring, SearcherType.Fuzzy]) { this.string = string; this.topN = Math.max(0, topN); From dfbccf997b83a895ac9e1708621cb639ce2645ac Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 17:06:39 +0200 Subject: [PATCH 056/105] minor changes --- src/dynamic-searchers/timing-dynamic-searcher.ts | 2 +- src/entity-searchers/entity-searcher-factory.ts | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/src/dynamic-searchers/timing-dynamic-searcher.ts b/src/dynamic-searchers/timing-dynamic-searcher.ts index f16727a..6a19af6 100644 --- a/src/dynamic-searchers/timing-dynamic-searcher.ts +++ b/src/dynamic-searchers/timing-dynamic-searcher.ts @@ -56,7 +56,7 @@ export class TimingDynamicSearcher implements DynamicSearcher = new DefaultEntitySearcher(stringSearcher); entitySearcher = new FastEntitySearcher(entitySearcher); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); @@ -94,7 +94,7 @@ export class EntitySearcherFactory { let fuzzySearcher: StringSearcher = new FuzzySearcher(ngramComputer); fuzzySearcher = new InequalityPenalizingSearcher(fuzzySearcher, config.inequalityPenalty); - fuzzySearcher = new NormalizingSearcher(fuzzySearcher, ngramNormalizer, 'ngramNormalizationDuration'); + fuzzySearcher = new NormalizingSearcher(fuzzySearcher, ngramNormalizer, 'normalizationDurationNgrams'); return fuzzySearcher; } From 79952a1d7c791d709cfeb534b4002f58dbc15e86 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 19:50:39 +0200 Subject: [PATCH 057/105] docs(suffixArray) --- src/suffix-array-searchers/suffix-array.ts | 191 +++++++++++++++------ 1 file changed, 134 insertions(+), 57 deletions(-) diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts index 0ec437c..79a6b4a 100644 --- a/src/suffix-array-searchers/suffix-array.ts +++ b/src/suffix-array-searchers/suffix-array.ts @@ -1,15 +1,52 @@ +/** + * Original: https://github.com/eranmeir/Sufa-Suffix-Array-Csharp + * Copyright (c) 2012 Eran Meir + * SPDX-License-Identifier: MIT + * + * Translation to TypeScript, modifications and refactoring + * (c) 2025 Kevin Schaal + */ + import { StringComparison } from './string-comparison.js'; +/** + * Creates a suffix array for a given string. + */ export class SuffixArray { + /** + * The end-of-chain marker. + */ private readonly eoc: number = 2147483647; - private m_str: string; - private m_sa: Int32Array; - private m_isa: Int32Array; - private m_chainHeadsDict: Map; - private m_chainStack: Chain[] = []; - private m_subChains: Chain[] = []; - private m_nextRank: number = 1; - + /** + * The suffix array. + */ + private sa: Int32Array; + /** + * The inverse suffix array. + */ + private isa: Int32Array; + /** + * The chain heads dictionary. + */ + private chainHeadsDict: Map; + /** + * The chain stack. + */ + private chainStack: Chain[] = []; + /** + * The sub-chains. + */ + private subChains: Chain[] = []; + /** + * The next rank to assign. + */ + private nextRank: number = 1; + + /** + * Creates the suffix array for the given string. + * @param str The input string. + * @returns The suffix array. + */ public static create(str: string): Int32Array { if (str == null) { throw new Error('Input string cannot be null.'); @@ -17,45 +54,57 @@ export class SuffixArray { const suffixArray: SuffixArray = new SuffixArray(str); suffixArray.FormInitialChains(); suffixArray.BuildSufixArray(); - return suffixArray.m_sa; + return suffixArray.sa; } - private constructor(private readonly str: string) { - const l = str.length; - this.m_str = str; - this.m_sa = new Int32Array(l); - this.m_isa = new Int32Array(l); - this.m_chainHeadsDict = new Map(); + /** + * Creates a new instance of the SuffixArray class. + * @param inputString The input string. + */ + private constructor(private readonly inputString: string) { + const length = inputString.length; + this.sa = new Int32Array(length); + this.isa = new Int32Array(length); + this.chainHeadsDict = new Map(); } + /** + * Forms the initial chains. + */ private FormInitialChains(): void { this.FindInitialChains(); this.SortAndPushSubchains(); } + /** + * Finds the initial chains. + */ private FindInitialChains(): void { - for (let i = 0; i < this.m_str.length; i++) { - const char_code = this.m_str.charCodeAt(i); - const chain_head_index = this.m_chainHeadsDict.get(char_code); + for (let i = 0; i < this.inputString.length; i++) { + const char_code = this.inputString.charCodeAt(i); + const chain_head_index = this.chainHeadsDict.get(char_code); if (chain_head_index !== undefined) { - this.m_isa[i] = chain_head_index; + this.isa[i] = chain_head_index; } else { - this.m_isa[i] = this.eoc; + this.isa[i] = this.eoc; } - this.m_chainHeadsDict.set(char_code, i); + this.chainHeadsDict.set(char_code, i); } - for (const headIndex of this.m_chainHeadsDict.values()) { - const newChain = new Chain(this.m_str, headIndex, 1); - this.m_subChains.push(newChain); + for (const headIndex of this.chainHeadsDict.values()) { + const newChain = new Chain(headIndex, 1); + this.subChains.push(newChain); } } + /** + * Builds the suffix array. + */ private BuildSufixArray(): void { - while (this.m_chainStack.length > 0) { - const chain: Chain = this.m_chainStack.pop() as Chain; + while (this.chainStack.length > 0) { + const chain: Chain = this.chainStack.pop() as Chain; - if (this.m_isa[chain.head] === this.eoc) { + if (this.isa[chain.head] === this.eoc) { this.RankSuffix(chain.head); } else { this.RefineChainWithInductionSorting(chain); @@ -63,23 +112,31 @@ export class SuffixArray { } } + /** + * Ranks the suffix at the given index. + * @param index The index of the suffix to rank. + */ private RankSuffix(index: number): void { - this.m_isa[index] = -this.m_nextRank; - this.m_sa[this.m_nextRank - 1] = index; - this.m_nextRank++; + this.isa[index] = -this.nextRank; + this.sa[this.nextRank - 1] = index; + this.nextRank++; } + /** + * Refines the given chain with induction sorting. + * @param chain The chain to refine. + */ private RefineChainWithInductionSorting(chain: Chain): void { const notedSuffixes: SuffixRank[] = []; - this.m_chainHeadsDict.clear(); - this.m_subChains = []; + this.chainHeadsDict.clear(); + this.subChains = []; while (chain.head !== this.eoc) { - const nextIndex: number = this.m_isa[chain.head]; - if (chain.head + chain.length > this.m_str.length - 1) { + const nextIndex: number = this.isa[chain.head]; + if (chain.head + chain.length > this.inputString.length - 1) { this.RankSuffix(chain.head); - } else if (this.m_isa[chain.head + chain.length] < 0) { - const sr: SuffixRank = new SuffixRank(chain.head, -this.m_isa[chain.head + chain.length]); + } else if (this.isa[chain.head + chain.length] < 0) { + const sr: SuffixRank = new SuffixRank(chain.head, -this.isa[chain.head + chain.length]); notedSuffixes.push(sr); } else { this.ExtendChain(chain); @@ -91,20 +148,28 @@ export class SuffixArray { this.SortAndRankNotedSuffixes(notedSuffixes); } + /** + * Extends the given chain. + * @param chain The chain to extend. + */ private ExtendChain(chain: Chain): void { - const sym: number = this.m_str.charCodeAt(chain.head + chain.length); - if (this.m_chainHeadsDict.has(sym)) { - this.m_isa[this.m_chainHeadsDict.get(sym) as number] = chain.head; - this.m_isa[chain.head] = this.eoc; + const sym: number = this.inputString.charCodeAt(chain.head + chain.length); + if (this.chainHeadsDict.has(sym)) { + this.isa[this.chainHeadsDict.get(sym) as number] = chain.head; + this.isa[chain.head] = this.eoc; } else { - this.m_isa[chain.head] = this.eoc; - const newChain: Chain = new Chain(this.m_str, chain.head, chain.length + 1); - this.m_subChains.push(newChain); + this.isa[chain.head] = this.eoc; + const newChain: Chain = new Chain(chain.head, chain.length + 1); + this.subChains.push(newChain); } - this.m_chainHeadsDict.set(sym, chain.head); + this.chainHeadsDict.set(sym, chain.head); } + /** + * Sorts and ranks the noted suffixes. + * @param notedSuffixes The noted suffixes to sort and rank. + */ private SortAndRankNotedSuffixes(notedSuffixes: SuffixRank[]): void { notedSuffixes.sort((a, b) => { return a.rank - b.rank; @@ -115,32 +180,44 @@ export class SuffixArray { } } + /** + * Sorts and pushes the sub-chains onto the chain stack. + */ private SortAndPushSubchains(): void { - this.m_subChains.sort((c1: Chain, c2: Chain): number => { + this.subChains.sort((c1: Chain, c2: Chain): number => { const len = Math.min(c1.length, c2.length); - return StringComparison.compareOrdinal(this.m_str, c1.head, this.m_str, c2.head, len); + return StringComparison.compareOrdinal(this.inputString, c1.head, this.inputString, c2.head, len); }); - for (let i = this.m_subChains.length - 1; i >= 0; i--) { - this.m_chainStack.push(this.m_subChains[i]); + for (let i = this.subChains.length - 1; i >= 0; i--) { + this.chainStack.push(this.subChains[i]); } } } +/** + * Represents a suffix and its rank. + */ class SuffixRank { + /** + * Creates a new instance of the SuffixRank class. + * @param head The head index of the suffix. + * @param rank The rank of the suffix. + */ public constructor( public readonly head: number, public readonly rank: number - ) {} + ) { } } +/** + * Represents a chain of suffixes. + */ class Chain { - public readonly m_str: string; - public head: number; - public length: number; - - public constructor(m_str: string, head: number, length: number) { - this.m_str = m_str; - this.head = head; - this.length = length; + /** + * Creates a new instance of the Chain class. + * @param head The head index of the chain. + * @param length The length of the chain. + */ + public constructor(public head: number, public readonly length: number) { } } From bd40f230e86e92db6f96bf82676bde5a4e0b23df Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Thu, 23 Oct 2025 19:53:29 +0200 Subject: [PATCH 058/105] fix tests --- src/dynamic-searchers/default-dynamic-searcher.test.ts | 6 +++--- src/entity-searchers/default-entity-searcher-latin.test.ts | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/src/dynamic-searchers/default-dynamic-searcher.test.ts b/src/dynamic-searchers/default-dynamic-searcher.test.ts index 63e62aa..8bde130 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.test.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.test.ts @@ -8,7 +8,7 @@ class Person { public id: number, public name: string, public favoriteHobby: string - ) {} + ) { } } function createSearcher(): DynamicSearcher { @@ -57,8 +57,8 @@ test('can remove entities', () => { expect(searcher.tryGetEntity(5823)).toEqual(null); expect(searcher.getEntities()).toEqual(entities.filter((e) => e.id !== 23501 && e.id !== 5823)); expect(searcher.getTerms()).toEqual(['Bob', 'Charlie']); - expect(searcher.getMatches(new Query('Carol')).matches).toEqual([]); - expect(searcher.getMatches(new Query('Bob')).matches[0]).toEqual( + expect(searcher.getMatches(new Query('Carol', 10, 0.3)).matches).toEqual([]); + expect(searcher.getMatches(new Query('Bob', 10, 0.3)).matches[0]).toEqual( new EntityMatch(entities.find((e) => e.name === 'Bob')!, 1, 'Bob') ); }); diff --git a/src/entity-searchers/default-entity-searcher-latin.test.ts b/src/entity-searchers/default-entity-searcher-latin.test.ts index b562c39..3974a80 100644 --- a/src/entity-searchers/default-entity-searcher-latin.test.ts +++ b/src/entity-searchers/default-entity-searcher-latin.test.ts @@ -100,7 +100,7 @@ test('can find persons with approximate match test3', () => { }); test("don't return results below min quality", () => { - const matches = searcher.getMatches(new Query('Sar')).matches; + const matches = searcher.getMatches(new Query('Sar', 10, 0.3)).matches; expect(matches.filter((m) => m.quality < 0.3)).toEqual([]); }); From f97167161b72265ac7ce849a9c517d658461e1f6 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 09:16:26 +0200 Subject: [PATCH 059/105] fixes --- src/entity-searchers/default-entity-searcher.ts | 2 +- src/interfaces/query.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index bc22999..20ff9ec 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -118,7 +118,6 @@ export class DefaultEntitySearcher implements EntitySearcher implements EntitySearcher Date: Fri, 24 Oct 2025 10:34:48 +0200 Subject: [PATCH 060/105] feat: searcher types for the config --- readme.md | 4 +- src/config.ts | 11 ++- .../entity-searcher-factory.ts | 92 +++++++++++++++---- src/string-searchers/searcher-switch.ts | 58 +++++++++--- src/suffix-array-searchers/prefix-searcher.ts | 3 - 5 files changed, 132 insertions(+), 36 deletions(-) diff --git a/readme.md b/readme.md index 1ab3cd0..c5533cf 100644 --- a/readme.md +++ b/readme.md @@ -1,4 +1,4 @@ -# Frontend Fuzzy Search +# Frontend Fuzzy + Substring + Prefix Search @m31coding/fuzzy-search is a frontend library for searching objects with ids (entities) by their names and features (terms). It is @@ -169,7 +169,7 @@ The following parameters are available when creating a query: | --------- | ---- | ------- | ----------- | | string | string | - | The query string. | | topN | number | 10 | The maximum number of matches to return. Provide Infinity to return all matches. | -| minQuality | number | 0.3 | The minimum quality of a match, ranging from 0 to 1. When set to zero, all terms that share at least one common n-gram with the query are considered a match. | +| minQuality | number | 0.0 | The minimum quality of a match, ranging from 0 to 1. When set to zero, all terms that share at least one common n-gram with the query are considered a match. | If the data terms contain characters and strings in non-latin scripts (such as Arabic, Cyrillic, Greek, Han, ... see also [ISO 15924](https://en.wikipedia.org/wiki/ISO_15924)), the default configuration must be adjusted before creating the searcher: diff --git a/src/config.ts b/src/config.ts index 054e364..95b8215 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,5 +1,6 @@ import { FuzzySearchConfig } from './fuzzy-searchers/fuzzy-search-config.js'; import { NormalizerConfig } from './normalization/normalizer-config.js'; +import { SearcherType } from './interfaces/searcher-type.js'; import { SortOrder } from './sort-order.js'; import { SubstringSearchConfig } from './suffix-array-searchers/substring-search-config.js'; @@ -9,6 +10,7 @@ import { SubstringSearchConfig } from './suffix-array-searchers/substring-search export class Config { /** * Creates a new instance of the Config class. + * @param searcherTypes The seπarcher types to use. * @param normalizerConfig The configuration for the default normalizer. * @param maxQueryLength The maximum query length. * @param sortOrder The sort order for the entity matches. @@ -16,11 +18,12 @@ export class Config { * @param substringSearchConfig The substring search configuration. */ public constructor( + public searcherTypes: SearcherType[], public normalizerConfig: NormalizerConfig, public maxQueryLength: number, public sortOrder: SortOrder, - public fuzzySearchConfig: FuzzySearchConfig, - public substringSearchConfig: SubstringSearchConfig + public fuzzySearchConfig?: FuzzySearchConfig, + public substringSearchConfig?: SubstringSearchConfig ) { } /** @@ -28,11 +31,13 @@ export class Config { * @returns The default configuration. */ public static createDefaultConfig(): Config { + const searcherTypes = [SearcherType.Fuzzy, SearcherType.Substring, SearcherType.Prefix]; const normalizerConfig = NormalizerConfig.createDefaultConfig(); const maxQueryLength = 150; const sortOrder = SortOrder.QualityAndMatchedString; const fuzzySearchConfig = FuzzySearchConfig.createDefaultConfig(); const substringSearchConfig = SubstringSearchConfig.createDefaultConfig(); - return new Config(normalizerConfig, maxQueryLength, sortOrder, fuzzySearchConfig, substringSearchConfig); + return new Config( + searcherTypes, normalizerConfig, maxQueryLength, sortOrder, fuzzySearchConfig, substringSearchConfig); } } diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 5fd7f5a..ad355b5 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -14,6 +14,7 @@ import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from '../string-searchers/normalizing-searcher.js'; import { PrefixSearcher } from '../suffix-array-searchers/prefix-searcher.js'; import { SearcherSwitch } from '../string-searchers/searcher-switch.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { SortingEntitySearcher } from './sorting-entity-searcher.js'; import { SortingSearcher } from '../string-searchers/sorting-searcher.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; @@ -34,9 +35,9 @@ export class EntitySearcherFactory { public static createSearcher(config: Config): EntitySearcher { const defaultNormalizer: Normalizer = this.createDefaultNormalizer(config); - const fuzzySearcher: StringSearcher = this.createFuzzySearcher(config.fuzzySearchConfig); - const suffixArraySearcher: SuffixArraySearcher = this.createSubstringSearcher(config.substringSearchConfig); - const prefixSearcher = this.createPrefixSearcher(suffixArraySearcher); + const fuzzySearcher: StringSearcher | null = this.tryCreateFuzzySearcher(config); + const suffixArraySearcher: SuffixArraySearcher | null = this.tryCreateSubstringSearcher(config); + const prefixSearcher: StringSearcher | null = this.tryCreatePrefixSearcher(config, suffixArraySearcher); let stringSearcher: StringSearcher = new SearcherSwitch( prefixSearcher, @@ -59,14 +60,17 @@ export class EntitySearcherFactory { * @returns The default normalizer. */ private static createDefaultNormalizer(config: Config): Normalizer { - const forbiddenCharacters = new Set( - [ - config.fuzzySearchConfig.paddingLeft.split(''), - config.fuzzySearchConfig.paddingRight.split(''), - config.fuzzySearchConfig.paddingMiddle.split(''), - config.substringSearchConfig.suffixArraySeparator.split(''), - ].flat() - ); + const forbiddenCharacters = new Set(); + + if (config.fuzzySearchConfig) { + config.fuzzySearchConfig.paddingLeft.split('').forEach(c => forbiddenCharacters.add(c)); + config.fuzzySearchConfig.paddingRight.split('').forEach(c => forbiddenCharacters.add(c)); + config.fuzzySearchConfig.paddingMiddle.split('').forEach(c => forbiddenCharacters.add(c)); + } + + if (config.substringSearchConfig) { + config.substringSearchConfig.suffixArraySeparator.split('').forEach(c => forbiddenCharacters.add(c)); + } const allowCharacter: (c: string) => boolean = (c) => config.normalizerConfig.allowCharacter(c) && !forbiddenCharacters.has(c); @@ -79,9 +83,24 @@ export class EntitySearcherFactory { return DefaultNormalizer.create(modifiedNormalizerConfig); } + /** + * Creates the fuzzy string searcher if configured. + * @param config The searcher configuration. + * @returns The fuzzy string searcher or null. + */ + private static tryCreateFuzzySearcher(config: Config): StringSearcher | null { + if (!config.searcherTypes.includes(SearcherType.Fuzzy)) { + return null; + } + if (config.fuzzySearchConfig === undefined) { + throw new Error('Unable to create fuzzy searcher: No fuzzy search config provided.'); + } + return this.createFuzzySearcher(config.fuzzySearchConfig); + } + /** * Creates the fuzzy string searcher. - * @param config The fuzzy search configuration. + * @param config The fuzzy searcher configuration. * @returns The fuzzy string searcher. */ private static createFuzzySearcher(config: FuzzySearchConfig): StringSearcher { @@ -98,22 +117,63 @@ export class EntitySearcherFactory { return fuzzySearcher; } + /** + * Creates the substring searcher if configured. + * @param config The searcher configuration. + * @returns The substring searcher or null. + */ + private static tryCreateSubstringSearcher(config: Config): SuffixArraySearcher | null { + if (!config.searcherTypes.includes(SearcherType.Substring)) { + return null; + } + if (config.substringSearchConfig === undefined) { + throw new Error('Unable to create substring searcher: No substring search config provided.'); + } + return this.createSubstringSearcher(config.substringSearchConfig); + } + /** * Creates the substring searcher. - * @param config The substring search configuration. + * @param config The substring search config. * @returns The substring searcher. */ private static createSubstringSearcher(config: SubstringSearchConfig): SuffixArraySearcher { - const substringSearcher = new SuffixArraySearcher(config.suffixArraySeparator); - return substringSearcher; + return new SuffixArraySearcher(config.suffixArraySeparator); + } + + /** + * Creates the prefix searcher if configured. + * @param config The searcher configuration. + * @param suffixArraySearcher The suffix array searcher to use. + * @returns The prefix searcher or null. + */ + private static tryCreatePrefixSearcher( + config: Config, + suffixArraySearcher: SuffixArraySearcher | null + ): StringSearcher | null { + if (!config.searcherTypes.includes(SearcherType.Prefix)) { + return null; + } + if (config.substringSearchConfig === undefined) { + throw new Error('Unable to create prefix searcher: No substring search config provided.'); + } + + return this.createPrefixSearcher(config.substringSearchConfig, suffixArraySearcher); } /** * Creates the prefix searcher. + * @param config The substring search configuration. * @param suffixArraySearcher The suffix array searcher to use. * @returns The prefix searcher. */ - private static createPrefixSearcher(suffixArraySearcher: SuffixArraySearcher): StringSearcher { + private static createPrefixSearcher( + config: SubstringSearchConfig, + suffixArraySearcher: SuffixArraySearcher | null + ): StringSearcher { + if (suffixArraySearcher === null) { + suffixArraySearcher = this.createSubstringSearcher(config); + } return new PrefixSearcher(suffixArraySearcher); } } \ No newline at end of file diff --git a/src/string-searchers/searcher-switch.ts b/src/string-searchers/searcher-switch.ts index b201272..a820a44 100644 --- a/src/string-searchers/searcher-switch.ts +++ b/src/string-searchers/searcher-switch.ts @@ -8,8 +8,8 @@ import { StringSearcher } from '../interfaces/string-searcher.js'; /** * A searcher switch that routes to the prefix searcher, the substring searcher, or the fuzzy searcher. The prefix - * searcher is a simple wrapper around the suffix array searcher and only relevant for getMatches. It will be always up - * to date with the substring searcher. + * searcher is a simple wrapper around the suffix array searcher. It will be always up to date with the substring + * searcher if present. */ export class SearcherSwitch implements StringSearcher { @@ -20,9 +20,9 @@ export class SearcherSwitch implements StringSearcher { * @param fuzzySearcher The fuzzy searcher. */ public constructor( - private readonly prefixSearcher: StringSearcher, - private readonly substringSearcher: StringSearcher, - private readonly fuzzySearcher: StringSearcher, + private readonly prefixSearcher: StringSearcher | null, + private readonly substringSearcher: StringSearcher | null, + private readonly fuzzySearcher: StringSearcher | null, ) { } @@ -30,9 +30,20 @@ export class SearcherSwitch implements StringSearcher { * {@inheritDoc StringSearcher.index} */ index(terms: string[]): Meta { - const substringSearcherMeta = this.substringSearcher.index(terms); - const fuzzySearcherMeta = this.fuzzySearcher.index(terms); - return MetaMerger.mergeMeta([fuzzySearcherMeta, substringSearcherMeta]); + const meta = []; + if (this.prefixSearcher && !this.substringSearcher) { + const prefixSearcherMeta = this.prefixSearcher.index(terms); + meta.push(prefixSearcherMeta); + } + if (this.substringSearcher) { + const substringSearcherMeta = this.substringSearcher.index(terms); + meta.push(substringSearcherMeta); + } + if (this.fuzzySearcher) { + const fuzzySearcherMeta = this.fuzzySearcher.index(terms); + meta.push(fuzzySearcherMeta); + } + return MetaMerger.mergeMeta(meta); } /** @@ -46,10 +57,19 @@ export class SearcherSwitch implements StringSearcher { switch (query.searcherTypes[0]) { case SearcherType.Prefix: + if (!this.prefixSearcher) { + throw new Error('No prefix searcher has been indexed.'); + } return this.prefixSearcher.getMatches(query); case SearcherType.Substring: + if (!this.substringSearcher) { + throw new Error('No substring searcher has been indexed.'); + } return this.substringSearcher.getMatches(query); case SearcherType.Fuzzy: + if (!this.fuzzySearcher) { + throw new Error('No fuzzy searcher has been indexed.'); + } return this.fuzzySearcher.getMatches(query); default: throw new Error(`Unknown searcher type: ${query.searcherTypes[0]}`); @@ -60,15 +80,29 @@ export class SearcherSwitch implements StringSearcher { * {@inheritDoc StringSearcher.save} */ save(memento: Memento): void { - this.substringSearcher.save(memento); - this.fuzzySearcher.save(memento); + if (this.prefixSearcher && !this.substringSearcher) { + this.prefixSearcher.save(memento); + } + if (this.substringSearcher) { + this.substringSearcher.save(memento); + } + if (this.fuzzySearcher) { + this.fuzzySearcher.save(memento); + } } /** * {@inheritDoc StringSearcher.load} */ load(memento: Memento): void { - this.substringSearcher.load(memento); - this.fuzzySearcher.load(memento); + if (this.prefixSearcher && !this.substringSearcher) { + this.prefixSearcher.load(memento); + } + if (this.substringSearcher) { + this.substringSearcher.load(memento); + } + if (this.fuzzySearcher) { + this.fuzzySearcher.load(memento); + } } } \ No newline at end of file diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index 71809e5..61046f0 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -16,7 +16,6 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.index} */ index(terms: string[]): Meta { - console.warn('PrefixSearcher.index was called. Call SuffixArraySearcher.index instead.'); return this.suffixArraySearcher.index(terms); } @@ -40,7 +39,6 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.save} */ save(memento: Memento): void { - console.warn('PrefixSearcher.save was called. Call SuffixArraySearcher.save instead.'); this.suffixArraySearcher.save(memento); } @@ -48,7 +46,6 @@ export class PrefixSearcher implements StringSearcher { * {@inheritDoc StringSearcher.load} */ load(memento: Memento): void { - console.warn('PrefixSearcher.load was called. Call SuffixArraySearcher.load instead.'); this.suffixArraySearcher.load(memento); } From 3daea2ee539d50a67f46e1a4e6f3bfb16840a8a9 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 10:57:41 +0200 Subject: [PATCH 061/105] run performance and regression tests --- .../output/_indexing-meta.txt | 10 +++++----- src/performance-test/output/performance.txt | 12 ++++++------ src/regression-test/output/_indexing-meta.txt | 10 +++++----- .../output/boston-deletion.txt | 8 ++++---- .../output/boulder-creek-substitution.txt | 6 +++--- .../output/carcassonne-prefix.txt | 6 +++--- .../output/carcassonne-substring.txt | 6 +++--- .../output/carcassonne-suffix.txt | 6 +++--- .../output/kuwait-city-prefix.txt | 6 +++--- src/regression-test/output/kuwait-city.txt | 6 +++--- .../output/munich-insertion.txt | 6 +++--- .../output/tbilisi-deletion.txt | 6 +++--- src/regression-test/output/tbilisi.txt | 6 +++--- src/regression-test/output/tokyo-prefix.txt | 6 +++--- src/regression-test/output/tokyo.txt | 19 ++++++++++++++----- .../output/t\303\274bingen-transposition.txt" | 6 +++--- 16 files changed, 67 insertions(+), 58 deletions(-) diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 1816093..5e853b9 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { + "indexingDurationSuffixArraySearcher": 2891, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3828, - "ngramNormalizationDuration": 281, - "indexingDurationSuffixArraySearcher": 2819, + "indexingDurationFuzzySearcher": 3952, + "normalizationDurationNgrams": 286, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1344, + "normalizationDurationDefault": 1432, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 9664 + "indexingDurationTotal": 10036 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index 239ff46..d7f94d4 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -10,19 +10,19 @@ "substringQueries": 997, "transpositionErrors": 299 }, - "totalDuration": 2274.5138099999676, - "averageDuration": 1.1372569049999839, - "standardDeviation": 5.465560304604728, + "totalDuration": 2283.272619000032, + "averageDuration": 1.1416363095000162, + "standardDeviation": 5.4463019800609365, "fastest": { "query": "村", - "duration": 0.012208000000100583 + "duration": 0.012666999999055406 }, "slowest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 220.9579589999994 + "duration": 219.31754200000069 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 220.9579589999994 + "duration": 219.31754200000069 } } \ No newline at end of file diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 855602e..1076516 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { + "indexingDurationSuffixArraySearcher": 2953, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3722, - "ngramNormalizationDuration": 338, - "indexingDurationSuffixArraySearcher": 2870, + "indexingDurationFuzzySearcher": 3742, + "normalizationDurationNgrams": 307, "numberOfDistinctTerms": 1190185, - "defaultNormalizationDuration": 1392, + "normalizationDurationDefault": 1274, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDuration": 9812 + "indexingDurationTotal": 9661 } \ No newline at end of file diff --git a/src/regression-test/output/boston-deletion.txt b/src/regression-test/output/boston-deletion.txt index dfe7936..58d83fd 100644 --- a/src/regression-test/output/boston-deletion.txt +++ b/src/regression-test/output/boston-deletion.txt @@ -1,16 +1,16 @@ { "string": "bostn", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/boulder-creek-substitution.txt b/src/regression-test/output/boulder-creek-substitution.txt index d1b71dc..4f2cb02 100644 --- a/src/regression-test/output/boulder-creek-substitution.txt +++ b/src/regression-test/output/boulder-creek-substitution.txt @@ -1,11 +1,11 @@ { "string": "boulder creak", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/carcassonne-prefix.txt b/src/regression-test/output/carcassonne-prefix.txt index 5fc9b06..ab5efde 100644 --- a/src/regression-test/output/carcassonne-prefix.txt +++ b/src/regression-test/output/carcassonne-prefix.txt @@ -1,11 +1,11 @@ { "string": "carcasso", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/carcassonne-substring.txt b/src/regression-test/output/carcassonne-substring.txt index 4718e56..74a49fa 100644 --- a/src/regression-test/output/carcassonne-substring.txt +++ b/src/regression-test/output/carcassonne-substring.txt @@ -1,11 +1,11 @@ { "string": "cassonn", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/carcassonne-suffix.txt b/src/regression-test/output/carcassonne-suffix.txt index 00d3d67..bf1b3c5 100644 --- a/src/regression-test/output/carcassonne-suffix.txt +++ b/src/regression-test/output/carcassonne-suffix.txt @@ -1,11 +1,11 @@ { "string": "sonne", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/kuwait-city-prefix.txt b/src/regression-test/output/kuwait-city-prefix.txt index 83891bc..bed15f1 100644 --- a/src/regression-test/output/kuwait-city-prefix.txt +++ b/src/regression-test/output/kuwait-city-prefix.txt @@ -1,11 +1,11 @@ { "string": "مدينة الك", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/kuwait-city.txt b/src/regression-test/output/kuwait-city.txt index e56f10c..5b58b65 100644 --- a/src/regression-test/output/kuwait-city.txt +++ b/src/regression-test/output/kuwait-city.txt @@ -1,11 +1,11 @@ { "string": "مدينة الكويت", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/munich-insertion.txt b/src/regression-test/output/munich-insertion.txt index 582b322..1957fca 100644 --- a/src/regression-test/output/munich-insertion.txt +++ b/src/regression-test/output/munich-insertion.txt @@ -1,11 +1,11 @@ { "string": "muniich", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/tbilisi-deletion.txt b/src/regression-test/output/tbilisi-deletion.txt index 766cb55..f003ae4 100644 --- a/src/regression-test/output/tbilisi-deletion.txt +++ b/src/regression-test/output/tbilisi-deletion.txt @@ -1,11 +1,11 @@ { "string": "თბიისი", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/tbilisi.txt b/src/regression-test/output/tbilisi.txt index 34a6d8d..90b0cda 100644 --- a/src/regression-test/output/tbilisi.txt +++ b/src/regression-test/output/tbilisi.txt @@ -1,11 +1,11 @@ { "string": "თბილისი", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/tokyo-prefix.txt b/src/regression-test/output/tokyo-prefix.txt index 17eda29..687f289 100644 --- a/src/regression-test/output/tokyo-prefix.txt +++ b/src/regression-test/output/tokyo-prefix.txt @@ -1,11 +1,11 @@ { "string": "東京", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } diff --git a/src/regression-test/output/tokyo.txt b/src/regression-test/output/tokyo.txt index c882aae..6b859f3 100644 --- a/src/regression-test/output/tokyo.txt +++ b/src/regression-test/output/tokyo.txt @@ -1,18 +1,27 @@ { "string": "東京都", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } { - "queryDuration": 2 + "queryDuration": 4 } Rank Entity Matched String Quality -1 東京都 東京都 3.00 \ No newline at end of file +1 東京都 東京都 3.00 +2 東勢區 東勢區 0.24 +3 東勢鄉 東勢鄉 0.24 +4 東区 東区 0.24 +5 東區 東區 0.24 +6 東山區 東山區 0.24 +7 東河鄉 東河鄉 0.24 +8 東港鎮 東港鎮 0.24 +9 東澳 東澳 0.24 +10 東石鄉 東石鄉 0.24 \ No newline at end of file diff --git "a/src/regression-test/output/t\303\274bingen-transposition.txt" "b/src/regression-test/output/t\303\274bingen-transposition.txt" index a602a63..5c3eebc 100644 --- "a/src/regression-test/output/t\303\274bingen-transposition.txt" +++ "b/src/regression-test/output/t\303\274bingen-transposition.txt" @@ -1,11 +1,11 @@ { "string": "tübignen", "topN": 10, - "minQuality": 0.3, + "minQuality": 0, "searcherTypes": [ - "prefix", + "fuzzy", "substring", - "fuzzy" + "prefix" ] } From eb10a2e046348684ffa1ed75ea6190dbf2587065 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 11:03:24 +0200 Subject: [PATCH 062/105] fix: default dynamic searcher tests --- .../default-dynamic-searcher.test.ts | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/dynamic-searchers/default-dynamic-searcher.test.ts b/src/dynamic-searchers/default-dynamic-searcher.test.ts index 8bde130..de6578d 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.test.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.test.ts @@ -40,7 +40,7 @@ test('can index and query dynamic searcher', () => { expect(searcher.tryGetTerms(11923)).toEqual(['Charlie']); expect(searcher.getTerms()).toEqual(['Alice', 'Bob', 'Carol', 'Charlie']); expect(searcher.getMatches(new Query('Carol')).matches[0]).toEqual( - new EntityMatch(entities.find((e) => e.name === 'Carol')!, 1, 'Carol') + new EntityMatch(entities.find((e) => e.name === 'Carol')!, 3, 'Carol') ); }); @@ -59,7 +59,7 @@ test('can remove entities', () => { expect(searcher.getTerms()).toEqual(['Bob', 'Charlie']); expect(searcher.getMatches(new Query('Carol', 10, 0.3)).matches).toEqual([]); expect(searcher.getMatches(new Query('Bob', 10, 0.3)).matches[0]).toEqual( - new EntityMatch(entities.find((e) => e.name === 'Bob')!, 1, 'Bob') + new EntityMatch(entities.find((e) => e.name === 'Bob')!, 3, 'Bob') ); }); @@ -82,8 +82,8 @@ test('can update non-indexed properties.', () => { expect(searcher.tryGetEntity(23501)).toEqual(alice); expect(searcher.tryGetEntity(bob.id)).toEqual(bob); expect(searcher.getEntities()).toEqual([alice, bob, entities[2], entities[3]]); - expect(searcher.getMatches(new Query('Alice')).matches[0]).toEqual(new EntityMatch(alice, 1, 'Alice')); - expect(searcher.getMatches(new Query('Bob')).matches[0]).toEqual(new EntityMatch(bob, 1, 'Bob')); + expect(searcher.getMatches(new Query('Alice')).matches[0]).toEqual(new EntityMatch(alice, 3, 'Alice')); + expect(searcher.getMatches(new Query('Bob')).matches[0]).toEqual(new EntityMatch(bob, 3, 'Bob')); }); test('can update indexed properties.', () => { @@ -107,6 +107,6 @@ test('can update indexed properties.', () => { expect(searcher.getEntities().sort((e1, e2) => e1.id - e2.id)).toEqual( [alice, bob, entities[2], entities[3]].sort((e1, e2) => e1.id - e2.id) ); - expect(searcher.getMatches(new Query('Alice Queen')).matches[0]).toEqual(new EntityMatch(alice, 1, 'Alice Queen')); - expect(searcher.getMatches(new Query('Bob Bishop')).matches[0]).toEqual(new EntityMatch(bob, 1, 'Bob Bishop')); + expect(searcher.getMatches(new Query('Alice Queen')).matches[0]).toEqual(new EntityMatch(alice, 3, 'Alice Queen')); + expect(searcher.getMatches(new Query('Bob Bishop')).matches[0]).toEqual(new EntityMatch(bob, 3, 'Bob Bishop')); }); From 48cb0baff9074cc5dfb066e7510cdca167c3696f Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 11:05:59 +0200 Subject: [PATCH 063/105] fix: default basic tests --- .../default-entity-searcher-basic.test.ts | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/src/entity-searchers/default-entity-searcher-basic.test.ts b/src/entity-searchers/default-entity-searcher-basic.test.ts index 57bfc55..1208257 100644 --- a/src/entity-searchers/default-entity-searcher-basic.test.ts +++ b/src/entity-searchers/default-entity-searcher-basic.test.ts @@ -21,7 +21,7 @@ entitySearcher.indexEntities( test('can match entity', () => { expect(entitySearcher.getMatches(new Query('Alice')).matches).toEqual([ - new EntityMatch<{ id: number; name: string }>(entities[0], 1.0, 'Alice') + new EntityMatch<{ id: number; name: string }>(entities[0], 3, 'Alice') ]); }); @@ -34,7 +34,7 @@ class Person { public id: number, public name: string, public job: string - ) {} + ) { } } const literalSearcher2: StringSearcher = new LiteralSearcher(); @@ -53,13 +53,13 @@ entitySearcher2.indexEntities( test('can match entity with first term', () => { expect(entitySearcher2.getMatches(new Query('Alice')).matches).toEqual([ - new EntityMatch(entities2[0], 1.0, 'Alice') + new EntityMatch(entities2[0], 3, 'Alice') ]); }); test('can match entity with second term', () => { expect(entitySearcher2.getMatches(new Query('Programmer')).matches).toEqual([ - new EntityMatch<{ id: number; name: string; job: string }>(entities2[0], 1.0, 'Programmer') + new EntityMatch<{ id: number; name: string; job: string }>(entities2[0], 3, 'Programmer') ]); }); From d7dc44e48c506bb2cd1be591947f5e442ca69f61 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 11:20:10 +0200 Subject: [PATCH 064/105] fix: tests --- .../default-entity-searcher-basic.test.ts | 7 +++++-- ... => default-entity-searcher-fuzzy-latin.test.ts} | 10 +++++++--- ...default-entity-searcher-fuzzy-non-latin.test.ts} | 2 ++ src/entity-searchers/default-entity-searcher.ts | 13 ++++++++++++- src/entity-searchers/entity-searcher-factory.ts | 3 ++- 5 files changed, 28 insertions(+), 7 deletions(-) rename src/entity-searchers/{default-entity-searcher-latin.test.ts => default-entity-searcher-fuzzy-latin.test.ts} (94%) rename src/entity-searchers/{default-entity-searcher-non-latin.test.ts => default-entity-searcher-fuzzy-non-latin.test.ts} (98%) diff --git a/src/entity-searchers/default-entity-searcher-basic.test.ts b/src/entity-searchers/default-entity-searcher-basic.test.ts index 1208257..f0132bf 100644 --- a/src/entity-searchers/default-entity-searcher-basic.test.ts +++ b/src/entity-searchers/default-entity-searcher-basic.test.ts @@ -3,10 +3,12 @@ import { EntityMatch } from '../interfaces/entity-match.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { LiteralSearcher } from '../string-searchers/literal-searcher.js'; import { Query } from '../interfaces/query.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; const literalSearcher: StringSearcher = new LiteralSearcher(); -const entitySearcher: EntitySearcher<{ id: number; name: string }, number> = new DefaultEntitySearcher(literalSearcher); +const entitySearcher: EntitySearcher<{ id: number; name: string }, number> = + new DefaultEntitySearcher(literalSearcher, [SearcherType.Prefix]); const entities = [ { id: 23501, name: 'Alice' }, { id: 99234, name: 'Bob' }, @@ -38,7 +40,8 @@ class Person { } const literalSearcher2: StringSearcher = new LiteralSearcher(); -const entitySearcher2: EntitySearcher = new DefaultEntitySearcher(literalSearcher2); +const entitySearcher2: EntitySearcher = + new DefaultEntitySearcher(literalSearcher2, [SearcherType.Prefix]); const entities2 = [ new Person(23501, 'Alice', 'Programmer'), new Person(99234, 'Bob', 'Teacher'), diff --git a/src/entity-searchers/default-entity-searcher-latin.test.ts b/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts similarity index 94% rename from src/entity-searchers/default-entity-searcher-latin.test.ts rename to src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts index 3974a80..19cc9c7 100644 --- a/src/entity-searchers/default-entity-searcher-latin.test.ts +++ b/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts @@ -3,12 +3,16 @@ import { EntityMatch } from '../interfaces/entity-match.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { EntitySearcherFactory } from './entity-searcher-factory.js'; import { Query } from '../interfaces/query.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { TestData } from '../commons/test-data.js'; const persons = TestData.persons.concat(TestData.emptyPersons); +const config = Config.createDefaultConfig(); +config.searcherTypes = [SearcherType.Fuzzy]; + const emptySearcher: EntitySearcher<{ firstName: string; lastName: string }, { firstName: string; lastName: string }> = - EntitySearcherFactory.createSearcher(Config.createDefaultConfig()); + EntitySearcherFactory.createSearcher(config); test('empty searcher returns zero matches for empty query', () => { expect(emptySearcher.getMatches(new Query('')).matches).toEqual([]); @@ -19,7 +23,7 @@ test('empty searcher returns zero matches', () => { }); const searcher: EntitySearcher<{ firstName: string; lastName: string }, { firstName: string; lastName: string }> = - EntitySearcherFactory.createSearcher(Config.createDefaultConfig()); + EntitySearcherFactory.createSearcher(config); searcher.indexEntities( persons, (person) => person, @@ -113,7 +117,7 @@ test('every entity is returned only once', () => { const reindexedSearcher: EntitySearcher< { firstName: string; lastName: string }, { firstName: string; lastName: string } -> = EntitySearcherFactory.createSearcher(Config.createDefaultConfig()); +> = EntitySearcherFactory.createSearcher(config); reindexedSearcher.indexEntities( TestData.personsNonLatin, (person) => person, diff --git a/src/entity-searchers/default-entity-searcher-non-latin.test.ts b/src/entity-searchers/default-entity-searcher-fuzzy-non-latin.test.ts similarity index 98% rename from src/entity-searchers/default-entity-searcher-non-latin.test.ts rename to src/entity-searchers/default-entity-searcher-fuzzy-non-latin.test.ts index 6398735..2bd65b5 100644 --- a/src/entity-searchers/default-entity-searcher-non-latin.test.ts +++ b/src/entity-searchers/default-entity-searcher-fuzzy-non-latin.test.ts @@ -3,9 +3,11 @@ import { EntityMatch } from '../interfaces/entity-match.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { EntitySearcherFactory } from './entity-searcher-factory.js'; import { Query } from '../interfaces/query.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { TestData } from '../commons/test-data.js'; const config: Config = Config.createDefaultConfig(); +config.searcherTypes = [SearcherType.Fuzzy]; config.normalizerConfig.allowCharacter = (_c: string) => true; const searcher: EntitySearcher<{ firstName: string; lastName: string }, { firstName: string; lastName: string }> = EntitySearcherFactory.createSearcher(config); diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 20ff9ec..2a68826 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -22,6 +22,11 @@ export class DefaultEntitySearcher implements EntitySearcher implements EntitySearcher(); this.terms = []; @@ -115,6 +122,10 @@ export class DefaultEntitySearcher implements EntitySearcher = new DefaultEntitySearcher(stringSearcher); + let entitySearcher: EntitySearcher = + new DefaultEntitySearcher(stringSearcher, config.searcherTypes); entitySearcher = new FastEntitySearcher(entitySearcher); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; From 4c107ced9be9b102ddc01c4c5f5358a89a217d0a Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 11:30:50 +0200 Subject: [PATCH 065/105] readme --- readme.md => readm.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) rename readme.md => readm.md (99%) diff --git a/readme.md b/readm.md similarity index 99% rename from readme.md rename to readm.md index c5533cf..5765a7e 100644 --- a/readme.md +++ b/readm.md @@ -169,7 +169,7 @@ The following parameters are available when creating a query: | --------- | ---- | ------- | ----------- | | string | string | - | The query string. | | topN | number | 10 | The maximum number of matches to return. Provide Infinity to return all matches. | -| minQuality | number | 0.0 | The minimum quality of a match, ranging from 0 to 1. When set to zero, all terms that share at least one common n-gram with the query are considered a match. | +| minQuality | number | 0.3 | The minimum quality of a match, ranging from 0 to 1. When set to zero, all terms that share at least one common n-gram with the query are considered a match. | If the data terms contain characters and strings in non-latin scripts (such as Arabic, Cyrillic, Greek, Han, ... see also [ISO 15924](https://en.wikipedia.org/wiki/ISO_15924)), the default configuration must be adjusted before creating the searcher: From 4ebe1bf2ea033fd6d1efe604280d4eea895ed469 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 11:31:00 +0200 Subject: [PATCH 066/105] readme --- readm.md => README.md | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename readm.md => README.md (100%) diff --git a/readm.md b/README.md similarity index 100% rename from readm.md rename to README.md From 0658c0138c665e6e9030806ea6b3c758fc3f41b1 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Fri, 24 Oct 2025 19:25:00 +0200 Subject: [PATCH 067/105] feat: searcher spec config --- .../default-dynamic-searcher.test.ts | 4 +- .../default-dynamic-searcher.ts | 2 +- ...efault-entity-searcher-fuzzy-latin.test.ts | 3 +- .../default-entity-searcher.ts | 32 +-- .../entity-searcher-factory.ts | 4 +- src/entity-searchers/fast-entity-searcher.ts | 182 +++++++++--------- src/entity-searchers/index.ts | 2 +- src/fuzzy-searchers/fuzzy-searcher.test.ts | 16 +- src/fuzzy-searchers/fuzzy-searcher.ts | 4 +- src/fuzzy-searchers/index.ts | 2 +- src/interfaces/index.ts | 2 + src/interfaces/query.ts | 23 +-- src/interfaces/searcher-spec.ts | 75 ++++++++ src/interfaces/string-search-query.ts | 36 ++++ src/interfaces/string-searcher.ts | 4 +- src/performance-test/main.ts | 4 +- .../output/_indexing-meta.txt | 10 +- src/performance-test/output/performance.txt | 31 ++- src/performance/performance-test.ts | 4 +- src/performance/test-run-parameters.ts | 8 +- .../distinct-searcher.test.ts | 8 +- src/string-searchers/distinct-searcher.ts | 8 +- .../inequality-penalizing-searcher.test.ts | 11 +- .../inequality-penalizing-searcher.ts | 4 +- src/string-searchers/literal-searcher.ts | 4 +- .../normalizing-searcher.test.ts | 6 +- src/string-searchers/normalizing-searcher.ts | 8 +- src/string-searchers/result.ts | 6 +- src/string-searchers/searcher-switch.ts | 12 +- src/string-searchers/sorting-searcher.test.ts | 10 +- src/string-searchers/sorting-searcher.ts | 14 +- src/suffix-array-searchers/index.ts | 2 +- .../prefix-searcher.test.ts | 4 +- src/suffix-array-searchers/prefix-searcher.ts | 19 +- .../suffix-array-searcher.test.ts | 4 +- .../suffix-array-searcher.ts | 4 +- 36 files changed, 358 insertions(+), 214 deletions(-) create mode 100644 src/interfaces/searcher-spec.ts create mode 100644 src/interfaces/string-search-query.ts diff --git a/src/dynamic-searchers/default-dynamic-searcher.test.ts b/src/dynamic-searchers/default-dynamic-searcher.test.ts index de6578d..7759fdd 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.test.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.test.ts @@ -57,8 +57,8 @@ test('can remove entities', () => { expect(searcher.tryGetEntity(5823)).toEqual(null); expect(searcher.getEntities()).toEqual(entities.filter((e) => e.id !== 23501 && e.id !== 5823)); expect(searcher.getTerms()).toEqual(['Bob', 'Charlie']); - expect(searcher.getMatches(new Query('Carol', 10, 0.3)).matches).toEqual([]); - expect(searcher.getMatches(new Query('Bob', 10, 0.3)).matches[0]).toEqual( + expect(searcher.getMatches(new Query('Carol', 10)).matches).toEqual([]); + expect(searcher.getMatches(new Query('Bob', 10)).matches[0]).toEqual( new EntityMatch(entities.find((e) => e.name === 'Bob')!, 3, 'Bob') ); }); diff --git a/src/dynamic-searchers/default-dynamic-searcher.ts b/src/dynamic-searchers/default-dynamic-searcher.ts index 1ed82a5..103d185 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.ts @@ -48,7 +48,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher this.maxQueryLength) { query = new Query( - query.string.substring(0, this.maxQueryLength), query.topN, query.minQuality, query.searcherTypes); + query.string.substring(0, this.maxQueryLength), query.topN, query.searchers); } return ResultMerger.mergeResults(this.mainSearcher.getMatches(query), this.secondarySearcher.getMatches(query)); } diff --git a/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts b/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts index 19cc9c7..c8e6de5 100644 --- a/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts +++ b/src/entity-searchers/default-entity-searcher-fuzzy-latin.test.ts @@ -2,6 +2,7 @@ import { Config } from '../config.js'; import { EntityMatch } from '../interfaces/entity-match.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; import { EntitySearcherFactory } from './entity-searcher-factory.js'; +import { FuzzySearcher } from '../interfaces/searcher-spec.js'; import { Query } from '../interfaces/query.js'; import { SearcherType } from '../interfaces/searcher-type.js'; import { TestData } from '../commons/test-data.js'; @@ -104,7 +105,7 @@ test('can find persons with approximate match test3', () => { }); test("don't return results below min quality", () => { - const matches = searcher.getMatches(new Query('Sar', 10, 0.3)).matches; + const matches = searcher.getMatches(new Query('Sar', 10, [new FuzzySearcher(0.3)])).matches; expect(matches.filter((m) => m.quality < 0.3)).toEqual([]); }); diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 2a68826..321a7e3 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -8,7 +8,9 @@ import { MetaMerger } from '../commons/meta-merger.js'; import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; import { SearchState } from './search-state.js'; +import { SearcherSpec } from '../interfaces/searcher-spec.js'; import { SearcherType } from '../interfaces/searcher-type.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -23,9 +25,9 @@ export class DefaultEntitySearcher implements EntitySearcher; /** * The indexed entities. @@ -66,11 +68,12 @@ export class DefaultEntitySearcher implements EntitySearcher(); this.terms = []; @@ -112,21 +115,22 @@ export class DefaultEntitySearcher implements EntitySearcher { const searchState: SearchState = new SearchState(query); + const requestedSearchers = new Map(query.searchers.map(s => [s.type, s])); for (const { searcherType, qualityOffset } of this.searchersAndQualityOffsets) { if (query.topN == searchState.matches.length) { break; } - if (!query.searcherTypes.includes(searcherType)) { + if (!requestedSearchers.has(searcherType)) { continue; } - if (!this.searcherTypes.includes(searcherType)) { + if (!this.searcherTypes.has(searcherType)) { continue; } - this.addMatchesFromSearcher(searchState, searcherType, qualityOffset); + this.addMatchesFromSearcher(searchState, requestedSearchers.get(searcherType)!, qualityOffset); } const mergedMeta = MetaMerger.mergeMeta(searchState.meta); @@ -136,17 +140,23 @@ export class DefaultEntitySearcher implements EntitySearcher, - searcherType: SearcherType, + searcherSpec: SearcherSpec, qualityOffset: number ): void { - const stringSearcherQuery: Query = new Query( - searchState.query.string, Infinity, searchState.query.minQuality, [searcherType]); - const result: Result = this.stringSearcher.getMatches(stringSearcherQuery); + + const minQuality = Math.max(0, searcherSpec.minQuality - qualityOffset); + if (minQuality > 1) { + return; + } + + const stringSearchQuery: StringSearchQuery = new StringSearchQuery( + searchState.query.string, minQuality, searcherSpec.type); + const result: Result = this.stringSearcher.getMatches(stringSearchQuery); this.addMatchesFromResult(searchState, result, qualityOffset); } diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 9ed9add..ed8c5c3 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -3,7 +3,7 @@ import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; -import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; +// import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; // todo import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; @@ -50,7 +50,7 @@ export class EntitySearcherFactory { stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'normalizationDurationDefault'); let entitySearcher: EntitySearcher = new DefaultEntitySearcher(stringSearcher, config.searcherTypes); - entitySearcher = new FastEntitySearcher(entitySearcher); + // entitySearcher = new FastEntitySearcher(entitySearcher); // todo entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; } diff --git a/src/entity-searchers/fast-entity-searcher.ts b/src/entity-searchers/fast-entity-searcher.ts index a050d9c..92da49f 100644 --- a/src/entity-searchers/fast-entity-searcher.ts +++ b/src/entity-searchers/fast-entity-searcher.ts @@ -1,104 +1,106 @@ -import { EntityResult } from "../interfaces/entity-result.js"; -import { EntitySearcher } from "../interfaces/entity-searcher.js"; -import { Memento } from "../interfaces/memento.js"; -import { Meta } from "../interfaces/meta.js"; -import { Query } from "../interfaces/query.js"; +// todo: first shot: prefix searcher with given quality and substring searcher with higher quality if q<0.2 -/** - * A entity searcher that tries to optimize performance by querying with an increased quality threshold at first. - * @typeParam TEntity The type of the entities. - * @typeParam TId The type of the entity ids. - */ -export class FastEntitySearcher implements EntitySearcher { +// import { EntityResult } from "../interfaces/entity-result.js"; +// import { EntitySearcher } from "../interfaces/entity-searcher.js"; +// import { Memento } from "../interfaces/memento.js"; +// import { Meta } from "../interfaces/meta.js"; +// import { Query } from "../interfaces/query.js"; - /** - * Creates a new instance of the FastEntitySearcher class. - * @typeParam TEntity The type of the entities. - * @typeParam TId The type of the entity ids. - * @param entitySearcher - */ - public constructor(private readonly entitySearcher: EntitySearcher) { - } +// /** +// * A entity searcher that tries to optimize performance by querying with an increased quality threshold at first. +// * @typeParam TEntity The type of the entities. +// * @typeParam TId The type of the entity ids. +// */ +// export class FastEntitySearcher implements EntitySearcher { - /** - * {@inheritDoc EntitySearcher.indexEntities} - */ - indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { - return this.entitySearcher.indexEntities(entities, getId, getTerms); - } +// /** +// * Creates a new instance of the FastEntitySearcher class. +// * @typeParam TEntity The type of the entities. +// * @typeParam TId The type of the entity ids. +// * @param entitySearcher +// */ +// public constructor(private readonly entitySearcher: EntitySearcher) { +// } - /** - * {@inheritDoc EntitySearcher.getMatches} - */ - getMatches(query: Query): EntityResult { - if (query.topN === Infinity || query.minQuality >= 0.2) { - return this.entitySearcher.getMatches(query); - } +// /** +// * {@inheritDoc EntitySearcher.indexEntities} +// */ +// indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { +// return this.entitySearcher.indexEntities(entities, getId, getTerms); +// } - const firstQuery = new Query(query.string, query.topN, 0.3, query.searcherTypes); - const result = this.entitySearcher.getMatches(firstQuery); - if (result.matches.length == query.topN) { - return new EntityResult(result.matches, query, result.meta); - } - else { - return this.entitySearcher.getMatches(query); - } - } +// /** +// * {@inheritDoc EntitySearcher.getMatches} +// */ +// getMatches(query: Query): EntityResult { +// if (query.topN === Infinity || query.minQuality >= 0.2) { +// return this.entitySearcher.getMatches(query); +// } - /** - * {@inheritDoc EntitySearcher.tryGetEntity} - */ - tryGetEntity(id: TId): TEntity | null { - return this.entitySearcher.tryGetEntity(id); - } +// const firstQuery = new Query(query.string, query.topN, 0.3, query.searcherTypes); +// const result = this.entitySearcher.getMatches(firstQuery); +// if (result.matches.length == query.topN) { +// return new EntityResult(result.matches, query, result.meta); +// } +// else { +// return this.entitySearcher.getMatches(query); +// } +// } - /** - * {@inheritDoc EntitySearcher.getEntities} - */ - getEntities(): TEntity[] { - return this.entitySearcher.getEntities(); - } +// /** +// * {@inheritDoc EntitySearcher.tryGetEntity} +// */ +// tryGetEntity(id: TId): TEntity | null { +// return this.entitySearcher.tryGetEntity(id); +// } - /** - * {@inheritDoc EntitySearcher.tryGetTerms} - */ - tryGetTerms(id: TId): string[] | null { - return this.entitySearcher.tryGetTerms(id); - } +// /** +// * {@inheritDoc EntitySearcher.getEntities} +// */ +// getEntities(): TEntity[] { +// return this.entitySearcher.getEntities(); +// } - /** - * {@inheritDoc EntitySearcher.getTerms} - */ - getTerms(): string[] { - return this.entitySearcher.getTerms(); - } +// /** +// * {@inheritDoc EntitySearcher.tryGetTerms} +// */ +// tryGetTerms(id: TId): string[] | null { +// return this.entitySearcher.tryGetTerms(id); +// } - /** - * {@inheritDoc EntitySearcher.removeEntity} - */ - removeEntity(id: TId): boolean { - return this.entitySearcher.removeEntity(id); - } +// /** +// * {@inheritDoc EntitySearcher.getTerms} +// */ +// getTerms(): string[] { +// return this.entitySearcher.getTerms(); +// } - /** - * {@inheritDoc EntitySearcher.replaceEntity} - */ - replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { - return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); - } +// /** +// * {@inheritDoc EntitySearcher.removeEntity} +// */ +// removeEntity(id: TId): boolean { +// return this.entitySearcher.removeEntity(id); +// } - /** - * {@inheritDoc EntitySearcher.save} - */ - save(memento: Memento): void { - return this.entitySearcher.save(memento); - } +// /** +// * {@inheritDoc EntitySearcher.replaceEntity} +// */ +// replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { +// return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); +// } - /** - * {@inheritDoc EntitySearcher.load} - */ - load(memento: Memento): void { - return this.entitySearcher.load(memento); - } +// /** +// * {@inheritDoc EntitySearcher.save} +// */ +// save(memento: Memento): void { +// return this.entitySearcher.save(memento); +// } -} \ No newline at end of file +// /** +// * {@inheritDoc EntitySearcher.load} +// */ +// load(memento: Memento): void { +// return this.entitySearcher.load(memento); +// } + +// } \ No newline at end of file diff --git a/src/entity-searchers/index.ts b/src/entity-searchers/index.ts index 03ce00c..c99e9f1 100644 --- a/src/entity-searchers/index.ts +++ b/src/entity-searchers/index.ts @@ -1,5 +1,5 @@ export { DefaultEntitySearcher } from './default-entity-searcher.js'; export { EntitySearcherFactory } from './entity-searcher-factory.js'; -export { FastEntitySearcher } from './fast-entity-searcher.js'; +// export { FastEntitySearcher } from './fast-entity-searcher.js'; // todo export { SearchState } from './search-state.js'; export { SortingEntitySearcher } from './sorting-entity-searcher.js'; diff --git a/src/fuzzy-searchers/fuzzy-searcher.test.ts b/src/fuzzy-searchers/fuzzy-searcher.test.ts index 08c61f1..58b4e02 100644 --- a/src/fuzzy-searchers/fuzzy-searcher.test.ts +++ b/src/fuzzy-searchers/fuzzy-searcher.test.ts @@ -1,7 +1,7 @@ import { FuzzySearcher } from './fuzzy-searcher.js'; import { Match } from '../string-searchers/match.js'; import { NgramComputer } from './ngram-computer.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; const commonNgramComputer = new NgramComputer(3); @@ -9,38 +9,38 @@ const fuzzySearcher: StringSearcher = new FuzzySearcher(commonNgramComputer); fuzzySearcher.index(['Alice', 'Bob', 'Carol', 'Charlie']); test('can find exact match test 1', () => { - expect(fuzzySearcher.getMatches(new Query('Alice')).matches).toEqual([new Match(0, 1)]); + expect(fuzzySearcher.getMatches(new StringSearchQuery('Alice')).matches).toEqual([new Match(0, 1)]); }); test('can find exact match test 2', () => { - expect(fuzzySearcher.getMatches(new Query('Bob')).matches).toEqual([new Match(1, 1)]); + expect(fuzzySearcher.getMatches(new StringSearchQuery('Bob')).matches).toEqual([new Match(1, 1)]); }); test('can find approximate match test 1', () => { - const matches: Match[] = fuzzySearcher.getMatches(new Query('Alic')).matches; + const matches: Match[] = fuzzySearcher.getMatches(new StringSearchQuery('Alic')).matches; expect(matches).toHaveLength(1); expect(matches[0].quality).toBeGreaterThan(0.3); expect(matches[0].quality).toBeLessThan(1); }); test('can find approximate match test 2', () => { - const matches: Match[] = fuzzySearcher.getMatches(new Query('Bobby', 10, 0.0)).matches; + const matches: Match[] = fuzzySearcher.getMatches(new StringSearchQuery('Bobby', 0.0)).matches; expect(matches).toHaveLength(1); expect(matches[0].quality).toBeGreaterThan(0.3); expect(matches[0].quality).toBeLessThan(1); }); test('can find approximate match test 3', () => { - const matches: Match[] = fuzzySearcher.getMatches(new Query('Charlei', 10, 0.0)).matches; + const matches: Match[] = fuzzySearcher.getMatches(new StringSearchQuery('Charlei', 0.0)).matches; expect(matches).toHaveLength(1); expect(matches[0].quality).toBeGreaterThan(0.3); expect(matches[0].quality).toBeLessThan(1); }); test("can't find match for unindexed term", () => { - expect(fuzzySearcher.getMatches(new Query('David')).matches).toEqual([]); + expect(fuzzySearcher.getMatches(new StringSearchQuery('David')).matches).toEqual([]); }); test("can't find match for empty string", () => { - expect(fuzzySearcher.getMatches(new Query('')).matches).toEqual([]); + expect(fuzzySearcher.getMatches(new StringSearchQuery('')).matches).toEqual([]); }); diff --git a/src/fuzzy-searchers/fuzzy-searcher.ts b/src/fuzzy-searchers/fuzzy-searcher.ts index 1c8bf2f..1f1b21b 100644 --- a/src/fuzzy-searchers/fuzzy-searcher.ts +++ b/src/fuzzy-searchers/fuzzy-searcher.ts @@ -4,8 +4,8 @@ import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; import { NgramComputer } from './ngram-computer.js'; import { QualityComputer } from './quality-computer.js'; -import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { TermIds } from './term-ids.js'; @@ -87,7 +87,7 @@ export class FuzzySearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { + public getMatches(query: StringSearchQuery): Result { if (this.invertedIndex.size === 0) { return new Result([], query, new Meta()); } diff --git a/src/fuzzy-searchers/index.ts b/src/fuzzy-searchers/index.ts index 431c7d5..76f3dfd 100644 --- a/src/fuzzy-searchers/index.ts +++ b/src/fuzzy-searchers/index.ts @@ -1,4 +1,4 @@ -export { FuzzySearcher } from './fuzzy-searcher.js'; +export { FuzzySearcher as FuzzySearcherImpl } from './fuzzy-searcher.js'; export { FuzzySearchConfig } from './fuzzy-search-config.js'; export { InvertedIndex } from './inverted-index.js'; export { NgramComputer } from './ngram-computer.js'; diff --git a/src/interfaces/index.ts b/src/interfaces/index.ts index f4daef8..2ff2e7d 100644 --- a/src/interfaces/index.ts +++ b/src/interfaces/index.ts @@ -9,3 +9,5 @@ export { Normalizer } from './normalizer.js'; export { Query } from './query.js'; export { SearcherType } from './searcher-type.js'; export { StringSearcher } from './string-searcher.js'; +export { StringSearchQuery } from './string-search-query.js'; +export { SearcherSpec, FuzzySearcher, SubstringSearcher, PrefixSearcher } from './searcher-spec.js'; diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index a0acca8..008028d 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -1,4 +1,4 @@ -import { SearcherType } from './searcher-type.js' +import { FuzzySearcher, PrefixSearcher, SearcherSpec, SubstringSearcher, } from './searcher-spec.js'; /** * Holds the query string and query parameters. @@ -15,32 +15,23 @@ export class Query { public readonly topN: number; /** - * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the - * number of matches. The value must be between 0 and 1. + * The searchers to use. */ - public readonly minQuality: number; - - /** - * The searcher types to use. - */ - public readonly searcherTypes: SearcherType[]; + public readonly searchers: SearcherSpec[]; /** * Creates a new instance of the Query class. * @param string The query string. * @param topN The maximum number of matches to return. Provide Infinity to return all matches. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. The value must be between 0 and 1; lower or larger values will be clamped. - * @param searcherTypes The searcher types to use. + * @param searchers The searchers to use. */ public constructor( string: string, topN: number = 10, - minQuality: number = 0, - searcherTypes: SearcherType[] = [SearcherType.Fuzzy, SearcherType.Substring, SearcherType.Prefix]) { + searchers = [new FuzzySearcher(0.3), new SubstringSearcher(0), new PrefixSearcher(0)], + ) { this.string = string; this.topN = Math.max(0, topN); - this.minQuality = Math.max(0, Math.min(1, minQuality)); - this.searcherTypes = searcherTypes; + this.searchers = searchers; } } diff --git a/src/interfaces/searcher-spec.ts b/src/interfaces/searcher-spec.ts new file mode 100644 index 0000000..60a56ab --- /dev/null +++ b/src/interfaces/searcher-spec.ts @@ -0,0 +1,75 @@ +import { SearcherType } from '../interfaces/searcher-type.js'; + +export class SearcherSpec { + /** + * The searcher type. + */ + public readonly type: SearcherType; + + /** + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * number of matches. Decreasing this value might retrieve irrelevant matches. The value must be greater than or + * equal to 0. + */ + public readonly minQuality: number; + + /** + * Creates a new instance of the SearcherSpec class. + * @param type The searcher type. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. Values below 0 will be clamped to 0. + */ + public constructor(type: SearcherType, minQuality: number) { + this.type = type; + this.minQuality = Math.max(0, minQuality); + } +} + +/** + * Specification for the fuzzy searcher. + */ +export class FuzzySearcher extends SearcherSpec { + + /** + * Creates a new instance of the FuzzySearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Fuzzy, minQuality); + } +} + +/** + * Specification for the substring searcher. + */ +export class SubstringSearcher extends SearcherSpec { + + /** + * Creates a new instance of the SubstringSearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Substring, minQuality); + } +} + +/** + * Specification for the prefix searcher. + */ +export class PrefixSearcher extends SearcherSpec { + + /** + * Creates a new instance of the PrefixSearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Prefix, minQuality); + } +} \ No newline at end of file diff --git a/src/interfaces/string-search-query.ts b/src/interfaces/string-search-query.ts new file mode 100644 index 0000000..41e8835 --- /dev/null +++ b/src/interfaces/string-search-query.ts @@ -0,0 +1,36 @@ +import { SearcherType } from '../interfaces/searcher-type.js'; + +/** + * Query for the string searchers. + */ +export class StringSearchQuery { + /** + * The query string. + */ + public readonly string: string; + + /** + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * number of matches. Decreasing this value might retrieve irrelevant matches. + */ + public readonly minQuality: number; + + /** + * The type of searcher to use. + */ + public readonly searcherType?: SearcherType; + + /** + * Creates a new instance of the Query class. + * @param string The query string. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * between 0 and 1; lower or larger values will be clamped. + * @param searcherType The type of searcher to use. + */ + public constructor(string: string, minQuality: number = 0, searcherType?: SearcherType) { + this.string = string; + this.minQuality = Math.max(0, Math.min(1, minQuality)); + this.searcherType = searcherType; + } +} \ No newline at end of file diff --git a/src/interfaces/string-searcher.ts b/src/interfaces/string-searcher.ts index eaa9447..8637952 100644 --- a/src/interfaces/string-searcher.ts +++ b/src/interfaces/string-searcher.ts @@ -1,7 +1,7 @@ import { MementoSerializable } from './memento-serializable.js'; import { Meta } from './meta.js'; -import { Query } from './query.js'; import { Result } from '../string-searchers/result.js'; +import { StringSearchQuery } from './string-search-query.js'; /** * A string searcher for indexing strings and retrieving matches. @@ -19,5 +19,5 @@ export interface StringSearcher extends MementoSerializable { * @param query The query. * @returns The matches. */ - getMatches(query: Query): Result; + getMatches(query: StringSearchQuery): Result; } diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts index 8c944c4..ac6fa85 100644 --- a/src/performance-test/main.ts +++ b/src/performance-test/main.ts @@ -6,6 +6,7 @@ import { readFileSync, writeFileSync } from 'fs'; import { Config } from '../config.js'; import { Meta } from '../interfaces/meta.js'; import { PerformanceTest } from '../performance/performance-test.js'; +import { Query } from '../interfaces/query.js'; import { Report } from '../performance/report.js'; import { SearcherFactory } from '../searcher-factory.js'; import { TestRunParameters } from '../performance/test-run-parameters.js'; @@ -14,7 +15,6 @@ const outputPath = './src/performance-test/output'; const seed = 0; const numberOfQueries = 2_000; const topN = 10; -const minQuality = 0; const text = readFileSync('./data/world-ctvs.txt', 'utf8'); const lines = text.split('\n').slice(1); @@ -41,7 +41,7 @@ console.log(metaJson); const performanceTest: PerformanceTest = new PerformanceTest(searcher); const testRunParameters: TestRunParameters = - new TestRunParameters(seed, numberOfQueries, topN, minQuality); + new TestRunParameters(seed, numberOfQueries, topN, new Query('').searchers); console.log('Running performance test...'); const report: Report = performanceTest.run(testRunParameters); diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 5e853b9..2982389 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { - "indexingDurationSuffixArraySearcher": 2891, + "indexingDurationSuffixArraySearcher": 3052, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3952, - "normalizationDurationNgrams": 286, + "indexingDurationFuzzySearcher": 4051, + "normalizationDurationNgrams": 294, "numberOfDistinctTerms": 1190185, - "normalizationDurationDefault": 1432, + "normalizationDurationDefault": 1433, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDurationTotal": 10036 + "indexingDurationTotal": 10389 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index d7f94d4..2b6f2fd 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -3,26 +3,39 @@ "testSeed": 0, "numberOfQueries": 2000, "topN": 10, - "minQuality": 0 + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } + ] }, "queryCounts": { "prefixQueries": 1003, "substringQueries": 997, "transpositionErrors": 299 }, - "totalDuration": 2283.272619000032, - "averageDuration": 1.1416363095000162, - "standardDeviation": 5.4463019800609365, + "totalDuration": 4651.252601999997, + "averageDuration": 2.3256263009999985, + "standardDeviation": 3.839822005245632, "fastest": { - "query": "村", - "duration": 0.012666999999055406 + "query": "tn", + "duration": 0.010874999999941792 }, "slowest": { - "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 219.31754200000069 + "query": "K", + "duration": 28.094750000000204 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 219.31754200000069 + "duration": 3.050999999999476 } } \ No newline at end of file diff --git a/src/performance/performance-test.ts b/src/performance/performance-test.ts index 386e2a6..3498757 100644 --- a/src/performance/performance-test.ts +++ b/src/performance/performance-test.ts @@ -48,7 +48,7 @@ export class PerformanceTest { for (let i = 0, l = numberOfQueries; i < l; i++) { const queryString = this.GetRandomQueryString(); - const query = new Query(queryString, parameters.topN, parameters.minQuality); + const query = new Query(queryString, parameters.topN, [...parameters.searchers]); const start = performance.now(); const _ = this.dynamicSearcher.getMatches(query); const duration = performance.now() - start; @@ -56,7 +56,7 @@ export class PerformanceTest { } return Report.Create( - new TestRunParameters(parameters.testSeed, numberOfQueries, parameters.topN, parameters.minQuality), + new TestRunParameters(parameters.testSeed, numberOfQueries, parameters.topN, parameters.searchers), this.queryCounts, measurements ); diff --git a/src/performance/test-run-parameters.ts b/src/performance/test-run-parameters.ts index 32404d7..23e800a 100644 --- a/src/performance/test-run-parameters.ts +++ b/src/performance/test-run-parameters.ts @@ -1,3 +1,5 @@ +import { SearcherSpec } from "../interfaces/searcher-spec.js"; + /** * Parameters of a performance test run. */ @@ -7,12 +9,12 @@ export class TestRunParameters { * @param testSeed The random seed used for the test. * @param numberOfQueries The number of executed queries. * @param topN The maximum number of matches to return. - * @param minQuality The minimum quality of matches to return. + * @param searchers The searchers used in the test. */ public constructor( public readonly testSeed: number, public readonly numberOfQueries: number, public readonly topN: number, - public readonly minQuality: number - ) {} + public readonly searchers: readonly SearcherSpec[] + ) { } } diff --git a/src/string-searchers/distinct-searcher.test.ts b/src/string-searchers/distinct-searcher.test.ts index b6e8910..12f1b0d 100644 --- a/src/string-searchers/distinct-searcher.test.ts +++ b/src/string-searchers/distinct-searcher.test.ts @@ -1,7 +1,7 @@ import { DistinctSearcher } from './distinct-searcher.js'; import { LiteralSearcher } from './literal-searcher.js'; import { Match } from './match.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; const literalSearcher: StringSearcher = new LiteralSearcher(); @@ -9,15 +9,15 @@ const distinctSearcher = new DistinctSearcher(literalSearcher); distinctSearcher.index(['Bob', 'Carol', 'Alice', 'Bob', 'Charlie', 'Alice', 'Alice']); test('can find matches with distinct searcher test 1', () => { - expect(distinctSearcher.getMatches(new Query('Carol')).matches).toEqual([new Match(1, 1)]); + expect(distinctSearcher.getMatches(new StringSearchQuery('Carol')).matches).toEqual([new Match(1, 1)]); }); test('can find matches with distinct searcher test 2', () => { - expect(distinctSearcher.getMatches(new Query('Bob')).matches).toEqual([new Match(0, 1), new Match(3, 1)]); + expect(distinctSearcher.getMatches(new StringSearchQuery('Bob')).matches).toEqual([new Match(0, 1), new Match(3, 1)]); }); test('can find matches with distinct searcher test 3', () => { - expect(distinctSearcher.getMatches(new Query('Alice')).matches).toEqual([ + expect(distinctSearcher.getMatches(new StringSearchQuery('Alice')).matches).toEqual([ new Match(2, 1), new Match(5, 1), new Match(6, 1) diff --git a/src/string-searchers/distinct-searcher.ts b/src/string-searchers/distinct-searcher.ts index d3a4256..93f27d1 100644 --- a/src/string-searchers/distinct-searcher.ts +++ b/src/string-searchers/distinct-searcher.ts @@ -2,8 +2,8 @@ import { ArrayUtilities } from '../commons/array-utilities.js'; import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -39,8 +39,8 @@ export class DistinctSearcher implements StringSearcher { const termsSorted = terms.map((term, index) => ({ term, index })); termsSorted.sort((t1, t2) => t1.term < t2.term ? -1 - : t1.term > t2.term ? 1 - : 0 + : t1.term > t2.term ? 1 + : 0 ); this.sortMapping = new Int32Array(termsSorted.length); for (let i = 0, l = termsSorted.length; i < l; i++) { @@ -75,7 +75,7 @@ export class DistinctSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { + public getMatches(query: StringSearchQuery): Result { const result: Result = this.stringSearcher.getMatches(query); const newMatches: Match[] = new Array(result.matches.length); let j = 0; diff --git a/src/string-searchers/inequality-penalizing-searcher.test.ts b/src/string-searchers/inequality-penalizing-searcher.test.ts index 512456e..87a96a9 100644 --- a/src/string-searchers/inequality-penalizing-searcher.test.ts +++ b/src/string-searchers/inequality-penalizing-searcher.test.ts @@ -2,21 +2,22 @@ import { InequalityPenalizingSearcher } from './inequality-penalizing-searcher.j import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; const stringSearcher = { index: (_terms: string[]): Meta => new Meta(), - getMatches: (query: Query): Result => new Result([new Match(0, 1.0), new Match(1, 0.6)], query, new Meta()), - save: (_memento: Memento): void => {}, - load: (_memento: Memento): void => {} + getMatches: (query: StringSearchQuery): Result => new Result( + [new Match(0, 1.0), new Match(1, 0.6)], query, new Meta()), + save: (_memento: Memento): void => { }, + load: (_memento: Memento): void => { } }; const inequalityPenalizingSearcher = new InequalityPenalizingSearcher(stringSearcher, 0.05); inequalityPenalizingSearcher.index(['hello', 'yellow']); test('can penalize unequal match', () => { - expect(inequalityPenalizingSearcher.getMatches(new Query('hello')).matches).toEqual([ + expect(inequalityPenalizingSearcher.getMatches(new StringSearchQuery('hello')).matches).toEqual([ new Match(0, 1), new Match(1, 0.6 * (1 - 0.05)) ]); diff --git a/src/string-searchers/inequality-penalizing-searcher.ts b/src/string-searchers/inequality-penalizing-searcher.ts index 72e2178..92ba85a 100644 --- a/src/string-searchers/inequality-penalizing-searcher.ts +++ b/src/string-searchers/inequality-penalizing-searcher.ts @@ -2,8 +2,8 @@ import { HashUtilities } from '../commons/hash-utilities.js'; import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -50,7 +50,7 @@ export class InequalityPenalizingSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { + public getMatches(query: StringSearchQuery): Result { const hashCodeQuery = HashUtilities.getHashCode(query.string); const result = this.stringSearcher.getMatches(query); const newMatches = result.matches diff --git a/src/string-searchers/literal-searcher.ts b/src/string-searchers/literal-searcher.ts index d1ccad1..d70180b 100644 --- a/src/string-searchers/literal-searcher.ts +++ b/src/string-searchers/literal-searcher.ts @@ -1,8 +1,8 @@ import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -32,7 +32,7 @@ export class LiteralSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { + public getMatches(query: StringSearchQuery): Result { const matches: Match[] = []; for (let i = 0, l = this.terms.length; i < l; i++) { const term = this.terms[i]; diff --git a/src/string-searchers/normalizing-searcher.test.ts b/src/string-searchers/normalizing-searcher.test.ts index e7ec209..5aa5099 100644 --- a/src/string-searchers/normalizing-searcher.test.ts +++ b/src/string-searchers/normalizing-searcher.test.ts @@ -5,7 +5,7 @@ import { MultiNormalizer } from '../normalization/multi-normalizer.js'; import { NgramNormalizer } from '../normalization/ngram-normalizer.js'; import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from './normalizing-searcher.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; const defaultNormalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); @@ -16,9 +16,9 @@ const normalizingSearcher: StringSearcher = new NormalizingSearcher(literalSearc normalizingSearcher.index(['Hello world!']); test('can find matches normalized test 1', () => { - expect(normalizingSearcher.getMatches(new Query('HELLO-WORLD')).matches).toEqual([new Match(0, 1)]); + expect(normalizingSearcher.getMatches(new StringSearchQuery('HELLO-WORLD')).matches).toEqual([new Match(0, 1)]); }); test('can find matches normalized test 2', () => { - expect(normalizingSearcher.getMatches(new Query('!hello! world!!!')).matches).toEqual([new Match(0, 1)]); + expect(normalizingSearcher.getMatches(new StringSearchQuery('!hello! world!!!')).matches).toEqual([new Match(0, 1)]); }); diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index 47a50d3..068d539 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -1,8 +1,8 @@ import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; import { Normalizer } from '../interfaces/normalizer.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -39,9 +39,9 @@ export class NormalizingSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { - const normalizedQuery = new Query( - this.normalizer.normalize(query.string), query.topN, query.minQuality, query.searcherTypes); + public getMatches(query: StringSearchQuery): Result { + const normalizedQuery = new StringSearchQuery( + this.normalizer.normalize(query.string), query.minQuality, query.searcherType); const result = this.stringSearcher.getMatches(normalizedQuery); return new Result(result.matches, query, result.meta); } diff --git a/src/string-searchers/result.ts b/src/string-searchers/result.ts index 8032348..1eb4880 100644 --- a/src/string-searchers/result.ts +++ b/src/string-searchers/result.ts @@ -1,6 +1,6 @@ import { Match } from './match.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; /** * A string searcher result. @@ -14,7 +14,7 @@ export class Result { */ public constructor( public readonly matches: Match[], - public readonly query: Query, + public readonly query: StringSearchQuery, public readonly meta: Meta - ) {} + ) { } } diff --git a/src/string-searchers/searcher-switch.ts b/src/string-searchers/searcher-switch.ts index a820a44..493416a 100644 --- a/src/string-searchers/searcher-switch.ts +++ b/src/string-searchers/searcher-switch.ts @@ -1,9 +1,9 @@ import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; import { MetaMerger } from '../commons/meta-merger.js'; -import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; import { SearcherType } from '../interfaces/searcher-type.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -49,13 +49,13 @@ export class SearcherSwitch implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - getMatches(query: Query): Result { + getMatches(query: StringSearchQuery): Result { - if (query.searcherTypes.length != 1) { - throw new Error('SearcherSwitch.getMatches only supports queries with a single searcher type.'); + if (!query.searcherType) { + throw new Error('SearcherSwitch requires a searcher type.'); } - switch (query.searcherTypes[0]) { + switch (query.searcherType) { case SearcherType.Prefix: if (!this.prefixSearcher) { throw new Error('No prefix searcher has been indexed.'); @@ -72,7 +72,7 @@ export class SearcherSwitch implements StringSearcher { } return this.fuzzySearcher.getMatches(query); default: - throw new Error(`Unknown searcher type: ${query.searcherTypes[0]}`); + throw new Error(`Unknown searcher type: ${query.searcherType}`); } } diff --git a/src/string-searchers/sorting-searcher.test.ts b/src/string-searchers/sorting-searcher.test.ts index aec1d86..8e5c774 100644 --- a/src/string-searchers/sorting-searcher.test.ts +++ b/src/string-searchers/sorting-searcher.test.ts @@ -1,26 +1,26 @@ import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; import { SortingSearcher } from './sorting-searcher.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; const stringSearcher = { index: (_terms: string[]): Meta => new Meta(), - getMatches: (query: Query): Result => + getMatches: (query: StringSearchQuery): Result => new Result( [new Match(7, 0.2), new Match(16, 1.0), new Match(10, 0.5), new Match(4, 0.5), new Match(8, 0.6)], query, new Meta() ), - save: (_memento: Memento): void => {}, - load: (_memento: Memento): void => {} + save: (_memento: Memento): void => { }, + load: (_memento: Memento): void => { } }; const sortingSearcher = new SortingSearcher(stringSearcher); test('can sort matches by quality and index', () => { - expect(sortingSearcher.getMatches(new Query('some query')).matches).toEqual([ + expect(sortingSearcher.getMatches(new StringSearchQuery('some query')).matches).toEqual([ new Match(16, 1.0), new Match(8, 0.6), new Match(4, 0.5), diff --git a/src/string-searchers/sorting-searcher.ts b/src/string-searchers/sorting-searcher.ts index 4fd98b9..d45eddf 100644 --- a/src/string-searchers/sorting-searcher.ts +++ b/src/string-searchers/sorting-searcher.ts @@ -1,8 +1,8 @@ import { Match } from './match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from './result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; /** @@ -14,7 +14,7 @@ export class SortingSearcher implements StringSearcher { * Creates a new instance of the SortingSearcher class. * @param stringSearcher The string searcher to use. */ - public constructor(private readonly stringSearcher: StringSearcher) {} + public constructor(private readonly stringSearcher: StringSearcher) { } /** * {@inheritDoc StringSearcher.index} @@ -26,7 +26,7 @@ export class SortingSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - public getMatches(query: Query): Result { + public getMatches(query: StringSearchQuery): Result { const result: Result = this.stringSearcher.getMatches(query); result.matches.sort(this.compareMatchesByQualityAndIndex); return result; @@ -43,10 +43,10 @@ export class SortingSearcher implements StringSearcher { private compareMatchesByQualityAndIndex(m1: Match, m2: Match): number { return ( m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : m1.index < m2.index ? -1 - : m1.index > m2.index ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : m1.index < m2.index ? -1 + : m1.index > m2.index ? 1 + : 0 ); } diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index bcdacbb..fbb71bf 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -2,4 +2,4 @@ export { StringComparison } from './string-comparison.js'; export { SuffixArray } from './suffix-array.js'; export { SuffixArraySearcher } from './suffix-array-searcher.js'; export { SubstringSearchConfig } from './substring-search-config.js'; -export { PrefixSearcher } from './prefix-searcher.js'; +export { PrefixSearcher as PrefixSearcherImpl } from './prefix-searcher.js'; diff --git a/src/suffix-array-searchers/prefix-searcher.test.ts b/src/suffix-array-searchers/prefix-searcher.test.ts index 2161982..7d142af 100644 --- a/src/suffix-array-searchers/prefix-searcher.test.ts +++ b/src/suffix-array-searchers/prefix-searcher.test.ts @@ -1,6 +1,6 @@ import { Match } from '../string-searchers/match.js'; import { PrefixSearcher } from './prefix-searcher.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { SuffixArraySearcher } from './suffix-array-searcher.js'; const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher('$'); @@ -8,7 +8,7 @@ suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); const prefixSearcher: PrefixSearcher = new PrefixSearcher(suffixArraySearcher); function getMatches(queryString: string): Match[] { - const query = new Query(queryString, 10, 0); + const query = new StringSearchQuery(queryString, 0); const matches = prefixSearcher.getMatches(query).matches; matches.sort((m1, m2) => m1.index - m2.index); return matches; diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index 61046f0..05df96d 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -1,12 +1,19 @@ import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SuffixArraySearcher } from './suffix-array-searcher.js'; +/** + * A prefix searcher that is a simple wrapper around the suffix array searcher. + */ export class PrefixSearcher implements StringSearcher { + /** + * Creates a new instance of the PrefixSearcher class. + * @param suffixArraySearcher The suffix array searcher. + */ public constructor( private readonly suffixArraySearcher: SuffixArraySearcher, ) { @@ -22,15 +29,20 @@ export class PrefixSearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - getMatches(query: Query): Result { + getMatches(query: StringSearchQuery): Result { if (!query.string) { return new Result([], query, new Meta()); } const modifiedQueryString = this.modifyQueryString(query.string); - const modifiedQuery = new Query(modifiedQueryString, query.topN, query.minQuality, query.searcherTypes); + const modifiedQuery = new StringSearchQuery(modifiedQueryString, query.minQuality, query.searcherType); return this.suffixArraySearcher.getMatches(modifiedQuery, query.string.length); } + /** + * Modifies the original query string by prepending the suffix array searcher's separator. + * @param original The original query string. + * @returns The modified query string. + */ private modifyQueryString(original: string): string { return `${this.suffixArraySearcher.separator}${original}`; } @@ -48,5 +60,4 @@ export class PrefixSearcher implements StringSearcher { load(memento: Memento): void { this.suffixArraySearcher.load(memento); } - } \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array-searcher.test.ts b/src/suffix-array-searchers/suffix-array-searcher.test.ts index e9b246c..c8908d0 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.test.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.test.ts @@ -1,12 +1,12 @@ import { Match } from '../string-searchers/match.js'; -import { Query } from '../interfaces/query.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { SuffixArraySearcher } from './suffix-array-searcher.js'; const suffixArraySearcher: SuffixArraySearcher = new SuffixArraySearcher('$'); suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); function getMatches(queryString: string): Match[] { - const query = new Query(queryString, 10, 0); + const query = new StringSearchQuery(queryString, 0); const matches = suffixArraySearcher.getMatches(query).matches; matches.sort((m1, m2) => m1.index - m2.index); return matches; diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index e00808b..996aca2 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -1,9 +1,9 @@ import { Match } from '../string-searchers/match.js'; import { Memento } from '../interfaces/memento.js'; import { Meta } from '../interfaces/meta.js'; -import { Query } from '../interfaces/query.js'; import { Result } from '../string-searchers/result.js'; import { StringComparison } from './string-comparison.js'; +import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SuffixArray } from './suffix-array.js'; @@ -53,7 +53,7 @@ export class SuffixArraySearcher implements StringSearcher { /** * {@inheritDoc StringSearcher.getMatches} */ - getMatches(query: Query, queryLength?: number): Result { + getMatches(query: StringSearchQuery, queryLength?: number): Result { if (!query.string) { return new Result([], query, new Meta()); } From 5e26a950a9d38c7832bb00dbc4e0e5251d2d0d62 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 10:21:40 +0200 Subject: [PATCH 068/105] feat: fast entity searcher --- .../entity-searcher-factory.ts | 4 +- src/entity-searchers/fast-entity-searcher.ts | 220 +++++++++--------- src/entity-searchers/index.ts | 2 +- .../output/_indexing-meta.txt | 10 +- src/performance-test/output/performance.txt | 16 +- src/regression-test/output/_indexing-meta.txt | 10 +- .../suffix-array-searcher.ts | 6 +- 7 files changed, 136 insertions(+), 132 deletions(-) diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index ed8c5c3..35a806d 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -3,7 +3,7 @@ import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; -// import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; // todo +import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; // todo import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; @@ -50,7 +50,7 @@ export class EntitySearcherFactory { stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'normalizationDurationDefault'); let entitySearcher: EntitySearcher = new DefaultEntitySearcher(stringSearcher, config.searcherTypes); - // entitySearcher = new FastEntitySearcher(entitySearcher); // todo + entitySearcher = new FastEntitySearcher(entitySearcher); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; } diff --git a/src/entity-searchers/fast-entity-searcher.ts b/src/entity-searchers/fast-entity-searcher.ts index 92da49f..cf72cee 100644 --- a/src/entity-searchers/fast-entity-searcher.ts +++ b/src/entity-searchers/fast-entity-searcher.ts @@ -1,106 +1,114 @@ -// todo: first shot: prefix searcher with given quality and substring searcher with higher quality if q<0.2 - -// import { EntityResult } from "../interfaces/entity-result.js"; -// import { EntitySearcher } from "../interfaces/entity-searcher.js"; -// import { Memento } from "../interfaces/memento.js"; -// import { Meta } from "../interfaces/meta.js"; -// import { Query } from "../interfaces/query.js"; - -// /** -// * A entity searcher that tries to optimize performance by querying with an increased quality threshold at first. -// * @typeParam TEntity The type of the entities. -// * @typeParam TId The type of the entity ids. -// */ -// export class FastEntitySearcher implements EntitySearcher { - -// /** -// * Creates a new instance of the FastEntitySearcher class. -// * @typeParam TEntity The type of the entities. -// * @typeParam TId The type of the entity ids. -// * @param entitySearcher -// */ -// public constructor(private readonly entitySearcher: EntitySearcher) { -// } - -// /** -// * {@inheritDoc EntitySearcher.indexEntities} -// */ -// indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { -// return this.entitySearcher.indexEntities(entities, getId, getTerms); -// } - -// /** -// * {@inheritDoc EntitySearcher.getMatches} -// */ -// getMatches(query: Query): EntityResult { -// if (query.topN === Infinity || query.minQuality >= 0.2) { -// return this.entitySearcher.getMatches(query); -// } - -// const firstQuery = new Query(query.string, query.topN, 0.3, query.searcherTypes); -// const result = this.entitySearcher.getMatches(firstQuery); -// if (result.matches.length == query.topN) { -// return new EntityResult(result.matches, query, result.meta); -// } -// else { -// return this.entitySearcher.getMatches(query); -// } -// } - -// /** -// * {@inheritDoc EntitySearcher.tryGetEntity} -// */ -// tryGetEntity(id: TId): TEntity | null { -// return this.entitySearcher.tryGetEntity(id); -// } - -// /** -// * {@inheritDoc EntitySearcher.getEntities} -// */ -// getEntities(): TEntity[] { -// return this.entitySearcher.getEntities(); -// } - -// /** -// * {@inheritDoc EntitySearcher.tryGetTerms} -// */ -// tryGetTerms(id: TId): string[] | null { -// return this.entitySearcher.tryGetTerms(id); -// } - -// /** -// * {@inheritDoc EntitySearcher.getTerms} -// */ -// getTerms(): string[] { -// return this.entitySearcher.getTerms(); -// } - -// /** -// * {@inheritDoc EntitySearcher.removeEntity} -// */ -// removeEntity(id: TId): boolean { -// return this.entitySearcher.removeEntity(id); -// } - -// /** -// * {@inheritDoc EntitySearcher.replaceEntity} -// */ -// replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { -// return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); -// } - -// /** -// * {@inheritDoc EntitySearcher.save} -// */ -// save(memento: Memento): void { -// return this.entitySearcher.save(memento); -// } - -// /** -// * {@inheritDoc EntitySearcher.load} -// */ -// load(memento: Memento): void { -// return this.entitySearcher.load(memento); -// } - -// } \ No newline at end of file +// todo: remove: first shot: prefix searcher with given quality and substring searcher with higher quality if q < 0.2 + +import { EntityResult } from "../interfaces/entity-result.js"; +import { EntitySearcher } from "../interfaces/entity-searcher.js"; +import { Memento } from "../interfaces/memento.js"; +import { Meta } from "../interfaces/meta.js"; +import { Query } from "../interfaces/query.js"; +import { SearcherType } from "../interfaces/searcher-type.js"; +import { FuzzySearcher, PrefixSearcher, SearcherSpec, SubstringSearcher } from "../interfaces/searcher-spec.js"; + +/** + * A entity searcher that tries to optimize performance by querying with limited searchers and an increased quality + * threshold at first. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + */ +export class FastEntitySearcher implements EntitySearcher { + + /** + * Creates a new instance of the FastEntitySearcher class. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + * @param entitySearcher + */ + public constructor(private readonly entitySearcher: EntitySearcher) { + } + + /** + * {@inheritDoc EntitySearcher.indexEntities} + */ + indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { + return this.entitySearcher.indexEntities(entities, getId, getTerms); + } + + /** + * {@inheritDoc EntitySearcher.getMatches} + */ + getMatches(query: Query): EntityResult { + const searchers = new Map(query.searchers.map(s => [s.type, s])); + + if (query.topN === Infinity) { + return this.entitySearcher.getMatches(query); + } + + if (query.string.length <= 3 && + searchers.has(SearcherType.Prefix) && + searchers.get(SearcherType.Prefix)!.minQuality < 2.2) { + const newQuery = new Query(query.string, query.topN, [new PrefixSearcher(2.3)]); + const result = this.entitySearcher.getMatches(newQuery); + if (result.matches.length == query.topN) { + return new EntityResult(result.matches, query, result.meta); + } + } + + return this.entitySearcher.getMatches(query); + } + + /** + * {@inheritDoc EntitySearcher.tryGetEntity} + */ + tryGetEntity(id: TId): TEntity | null { + return this.entitySearcher.tryGetEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.getEntities} + */ + getEntities(): TEntity[] { + return this.entitySearcher.getEntities(); + } + + /** + * {@inheritDoc EntitySearcher.tryGetTerms} + */ + tryGetTerms(id: TId): string[] | null { + return this.entitySearcher.tryGetTerms(id); + } + + /** + * {@inheritDoc EntitySearcher.getTerms} + */ + getTerms(): string[] { + return this.entitySearcher.getTerms(); + } + + /** + * {@inheritDoc EntitySearcher.removeEntity} + */ + removeEntity(id: TId): boolean { + return this.entitySearcher.removeEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.replaceEntity} + */ + replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { + return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + } + + /** + * {@inheritDoc EntitySearcher.save} + */ + save(memento: Memento): void { + return this.entitySearcher.save(memento); + } + + /** + * {@inheritDoc EntitySearcher.load} + */ + load(memento: Memento): void { + return this.entitySearcher.load(memento); + } + +} \ No newline at end of file diff --git a/src/entity-searchers/index.ts b/src/entity-searchers/index.ts index c99e9f1..03ce00c 100644 --- a/src/entity-searchers/index.ts +++ b/src/entity-searchers/index.ts @@ -1,5 +1,5 @@ export { DefaultEntitySearcher } from './default-entity-searcher.js'; export { EntitySearcherFactory } from './entity-searcher-factory.js'; -// export { FastEntitySearcher } from './fast-entity-searcher.js'; // todo +export { FastEntitySearcher } from './fast-entity-searcher.js'; export { SearchState } from './search-state.js'; export { SortingEntitySearcher } from './sorting-entity-searcher.js'; diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 2982389..4df9c79 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { - "indexingDurationSuffixArraySearcher": 3052, + "indexingDurationSuffixArraySearcher": 3024, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 4051, - "normalizationDurationNgrams": 294, + "indexingDurationFuzzySearcher": 4077, + "normalizationDurationNgrams": 295, "numberOfDistinctTerms": 1190185, - "normalizationDurationDefault": 1433, + "normalizationDurationDefault": 1419, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDurationTotal": 10389 + "indexingDurationTotal": 10369 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index 2b6f2fd..e1bd224 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -23,19 +23,19 @@ "substringQueries": 997, "transpositionErrors": 299 }, - "totalDuration": 4651.252601999997, - "averageDuration": 2.3256263009999985, - "standardDeviation": 3.839822005245632, + "totalDuration": 1675.3076380000512, + "averageDuration": 0.8376538190000256, + "standardDeviation": 1.0061714290337511, "fastest": { - "query": "tn", - "duration": 0.010874999999941792 + "query": "iji", + "duration": 0.010958000000755419 }, "slowest": { - "query": "K", - "duration": 28.094750000000204 + "query": "浩庄", + "duration": 9.829207999999198 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 3.050999999999476 + "duration": 3.016749999998865 } } \ No newline at end of file diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index 1076516..afe827b 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { - "indexingDurationSuffixArraySearcher": 2953, + "indexingDurationSuffixArraySearcher": 2851, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3742, - "normalizationDurationNgrams": 307, + "indexingDurationFuzzySearcher": 3770, + "normalizationDurationNgrams": 271, "numberOfDistinctTerms": 1190185, - "normalizationDurationDefault": 1274, + "normalizationDurationDefault": 1369, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDurationTotal": 9661 + "indexingDurationTotal": 9638 } \ No newline at end of file diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 996aca2..aee36e4 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -7,8 +7,7 @@ import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SuffixArray } from './suffix-array.js'; -// todo: make sure the terms don't have the separating character. Move to a suffix array config. - +// todo: docs. export class SuffixArraySearcher implements StringSearcher { public readonly separator: string; private str: string; @@ -58,9 +57,7 @@ export class SuffixArraySearcher implements StringSearcher { return new Result([], query, new Meta()); } - // todo prefix: pass query.string modified const [start, end] = this.GetPositionsInSuffixArray(query.string); - // todo: refactor such that end is included. const matchedTermIds = new Int32Array(end - start); let i = 0; @@ -80,7 +77,6 @@ export class SuffixArraySearcher implements StringSearcher { } } - // todo: remove duplicate matches, measure performance return new Result(matches, query, new Meta()); } From bced12be607c3fa0a256dc57a86b74dc70e36965 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 14:16:31 +0200 Subject: [PATCH 069/105] feat: filter duplicate indexes --- src/suffix-array-searchers/suffix-array-searcher.ts | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index aee36e4..787c77e 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -59,18 +59,22 @@ export class SuffixArraySearcher implements StringSearcher { const [start, end] = this.GetPositionsInSuffixArray(query.string); const matchedTermIds = new Int32Array(end - start); + const seen: Set = new Set(); + let uniqueCount = 0; - let i = 0; for (let j = start; j < end; j++) { const termIndex = this.indexToTermIndex[this.suffixArray[j]]; - matchedTermIds[i++] = termIndex; + if (!seen.has(termIndex)) { + seen.add(termIndex); + matchedTermIds[uniqueCount++] = termIndex; + } } const matches: Match[] = []; queryLength = queryLength ?? query.string.length; let quality = 0; - for (let k = 0; k < matchedTermIds.length; k++) { + for (let k = 0; k < uniqueCount; k++) { quality = this.computeQuality(queryLength, this.termLengths[matchedTermIds[k]]); if (quality > query.minQuality) { matches.push(new Match(matchedTermIds[k], quality)); From cd9e0a33d8a57943a1a2827559c709be43482d2f Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 14:20:20 +0200 Subject: [PATCH 070/105] work work --- src/commons/index.ts | 1 + src/commons/usable-searchers.ts | 28 +++++++++++++++ src/config.ts | 2 +- .../entity-searcher-factory.ts | 4 +-- src/entity-searchers/fast-entity-searcher.ts | 34 ++++++++++++++----- src/performance-test/main.ts | 2 ++ 6 files changed, 59 insertions(+), 12 deletions(-) create mode 100644 src/commons/usable-searchers.ts diff --git a/src/commons/index.ts b/src/commons/index.ts index 6e1b7aa..17011de 100644 --- a/src/commons/index.ts +++ b/src/commons/index.ts @@ -2,3 +2,4 @@ export { ArrayUtilities } from './array-utilities.js'; export { HashUtilities } from './hash-utilities.js'; export { MetaMerger } from './meta-merger.js'; export { StringUtilities } from './string-utilities.js'; +export { UsableSearchers } from './usable-searchers.js'; diff --git a/src/commons/usable-searchers.ts b/src/commons/usable-searchers.ts new file mode 100644 index 0000000..001fdc3 --- /dev/null +++ b/src/commons/usable-searchers.ts @@ -0,0 +1,28 @@ +import { SearcherSpec } from '../interfaces/searcher-spec.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; + +// todo: docs +export class UsableSearchers { + + private readonly usableSearchers: Map; + + public constructor( + availableSearchers: SearcherType[], + requestedSearchers: SearcherSpec[]) { + this.usableSearchers = new Map(); + + for (const requestedSearcher of requestedSearchers) { + if (availableSearchers.includes(requestedSearcher.type)) { + this.usableSearchers.set(requestedSearcher.type, requestedSearcher); + } + } + } + + public has(type: SearcherType): boolean { + return this.usableSearchers.has(type); + } + + public minQuality(type: SearcherType): number { + return this.usableSearchers.get(type)!.minQuality; + } +} \ No newline at end of file diff --git a/src/config.ts b/src/config.ts index 95b8215..223ea84 100644 --- a/src/config.ts +++ b/src/config.ts @@ -10,7 +10,7 @@ import { SubstringSearchConfig } from './suffix-array-searchers/substring-search export class Config { /** * Creates a new instance of the Config class. - * @param searcherTypes The seπarcher types to use. + * @param searcherTypes The searcher types to use. * @param normalizerConfig The configuration for the default normalizer. * @param maxQueryLength The maximum query length. * @param sortOrder The sort order for the entity matches. diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 35a806d..8b948a0 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -3,7 +3,7 @@ import { DefaultEntitySearcher } from './default-entity-searcher.js'; import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { DistinctSearcher } from '../string-searchers/distinct-searcher.js'; import { EntitySearcher } from '../interfaces/entity-searcher.js'; -import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; // todo +import { FastEntitySearcher } from '../entity-searchers/fast-entity-searcher.js'; import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; @@ -50,7 +50,7 @@ export class EntitySearcherFactory { stringSearcher = new NormalizingSearcher(stringSearcher, defaultNormalizer, 'normalizationDurationDefault'); let entitySearcher: EntitySearcher = new DefaultEntitySearcher(stringSearcher, config.searcherTypes); - entitySearcher = new FastEntitySearcher(entitySearcher); + entitySearcher = new FastEntitySearcher(entitySearcher, config.searcherTypes); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; } diff --git a/src/entity-searchers/fast-entity-searcher.ts b/src/entity-searchers/fast-entity-searcher.ts index cf72cee..9e60d5f 100644 --- a/src/entity-searchers/fast-entity-searcher.ts +++ b/src/entity-searchers/fast-entity-searcher.ts @@ -1,12 +1,11 @@ -// todo: remove: first shot: prefix searcher with given quality and substring searcher with higher quality if q < 0.2 - +import { PrefixSearcher, SubstringSearcher } from "../interfaces/searcher-spec.js"; import { EntityResult } from "../interfaces/entity-result.js"; import { EntitySearcher } from "../interfaces/entity-searcher.js"; import { Memento } from "../interfaces/memento.js"; import { Meta } from "../interfaces/meta.js"; import { Query } from "../interfaces/query.js"; import { SearcherType } from "../interfaces/searcher-type.js"; -import { FuzzySearcher, PrefixSearcher, SearcherSpec, SubstringSearcher } from "../interfaces/searcher-spec.js"; +import { UsableSearchers } from "../commons/usable-searchers.js"; /** * A entity searcher that tries to optimize performance by querying with limited searchers and an increased quality @@ -20,9 +19,12 @@ export class FastEntitySearcher implements EntitySearcher) { + public constructor( + private readonly entitySearcher: EntitySearcher, + private readonly searcherTypes: SearcherType[]) { } /** @@ -36,15 +38,17 @@ export class FastEntitySearcher implements EntitySearcher { - const searchers = new Map(query.searchers.map(s => [s.type, s])); - if (query.topN === Infinity) { + const usableSearchers = new UsableSearchers(this.searcherTypes, query.searchers); + + if (query.topN > 200) { return this.entitySearcher.getMatches(query); } + // Make the prefix searcher faster for short queries. if (query.string.length <= 3 && - searchers.has(SearcherType.Prefix) && - searchers.get(SearcherType.Prefix)!.minQuality < 2.2) { + usableSearchers.has(SearcherType.Prefix) && + usableSearchers.minQuality(SearcherType.Prefix) < 2.2) { const newQuery = new Query(query.string, query.topN, [new PrefixSearcher(2.3)]); const result = this.entitySearcher.getMatches(newQuery); if (result.matches.length == query.topN) { @@ -52,6 +56,18 @@ export class FastEntitySearcher implements EntitySearcher(result.matches, query, result.meta); + } + } + return this.entitySearcher.getMatches(query); } diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts index ac6fa85..6bb2b62 100644 --- a/src/performance-test/main.ts +++ b/src/performance-test/main.ts @@ -9,6 +9,7 @@ import { PerformanceTest } from '../performance/performance-test.js'; import { Query } from '../interfaces/query.js'; import { Report } from '../performance/report.js'; import { SearcherFactory } from '../searcher-factory.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; import { TestRunParameters } from '../performance/test-run-parameters.js'; const outputPath = './src/performance-test/output'; @@ -26,6 +27,7 @@ interface GeoEntity { } const config = Config.createDefaultConfig(); +config.searcherTypes = [SearcherType.Fuzzy, SearcherType.Substring, SearcherType.Prefix]; config.normalizerConfig.allowCharacter = (_) => true; const searcher = SearcherFactory.createSearcher(config); From 224b6aedf5649a58be21dad4f8ed5b7fcb276625 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 14:23:31 +0200 Subject: [PATCH 071/105] run performance test --- src/performance-test/output/_indexing-meta.txt | 10 +++++----- src/performance-test/output/performance.txt | 14 +++++++------- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/src/performance-test/output/_indexing-meta.txt b/src/performance-test/output/_indexing-meta.txt index 4df9c79..228c509 100644 --- a/src/performance-test/output/_indexing-meta.txt +++ b/src/performance-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { - "indexingDurationSuffixArraySearcher": 3024, + "indexingDurationSuffixArraySearcher": 2920, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 4077, - "normalizationDurationNgrams": 295, + "indexingDurationFuzzySearcher": 4045, + "normalizationDurationNgrams": 292, "numberOfDistinctTerms": 1190185, - "normalizationDurationDefault": 1419, + "normalizationDurationDefault": 1410, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDurationTotal": 10369 + "indexingDurationTotal": 10177 } \ No newline at end of file diff --git a/src/performance-test/output/performance.txt b/src/performance-test/output/performance.txt index e1bd224..93aab57 100644 --- a/src/performance-test/output/performance.txt +++ b/src/performance-test/output/performance.txt @@ -23,19 +23,19 @@ "substringQueries": 997, "transpositionErrors": 299 }, - "totalDuration": 1675.3076380000512, - "averageDuration": 0.8376538190000256, - "standardDeviation": 1.0061714290337511, + "totalDuration": 2292.7832040000285, + "averageDuration": 1.1463916020000142, + "standardDeviation": 1.143674511472282, "fastest": { - "query": "iji", - "duration": 0.010958000000755419 + "query": "gaar", + "duration": 0.012042000000292319 }, "slowest": { "query": "浩庄", - "duration": 9.829207999999198 + "duration": 9.896040999999968 }, "longest": { "query": "Şehit Jandarma Binbaşı Kıvanç Cesur M", - "duration": 3.016749999998865 + "duration": 2.9509589999997843 } } \ No newline at end of file From d1289f42688e4a38a0e8e4ea5f120fd289c1f622 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:27:39 +0200 Subject: [PATCH 072/105] use usable searchers --- src/commons/usable-searchers.ts | 4 ++++ src/entity-searchers/default-entity-searcher.ts | 15 ++++++--------- 2 files changed, 10 insertions(+), 9 deletions(-) diff --git a/src/commons/usable-searchers.ts b/src/commons/usable-searchers.ts index 001fdc3..81d12f7 100644 --- a/src/commons/usable-searchers.ts +++ b/src/commons/usable-searchers.ts @@ -25,4 +25,8 @@ export class UsableSearchers { public minQuality(type: SearcherType): number { return this.usableSearchers.get(type)!.minQuality; } + + public spec(type: SearcherType): SearcherSpec { + return this.usableSearchers.get(type)!; + } } \ No newline at end of file diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 321a7e3..05bec42 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -12,6 +12,7 @@ import { SearcherSpec } from '../interfaces/searcher-spec.js'; import { SearcherType } from '../interfaces/searcher-type.js'; import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; +import { UsableSearchers } from '../commons/usable-searchers.js'; /** * A searcher that indexes and retrieves entities. @@ -27,7 +28,7 @@ export class DefaultEntitySearcher implements EntitySearcher; + private readonly searcherTypes: SearcherType[]; /** * The indexed entities. @@ -73,7 +74,7 @@ export class DefaultEntitySearcher implements EntitySearcher(); this.terms = []; @@ -115,22 +116,18 @@ export class DefaultEntitySearcher implements EntitySearcher { const searchState: SearchState = new SearchState(query); - const requestedSearchers = new Map(query.searchers.map(s => [s.type, s])); + const usableSearchers = new UsableSearchers(this.searcherTypes, query.searchers); for (const { searcherType, qualityOffset } of this.searchersAndQualityOffsets) { if (query.topN == searchState.matches.length) { break; } - if (!requestedSearchers.has(searcherType)) { + if (!usableSearchers.has(searcherType)) { continue; } - if (!this.searcherTypes.has(searcherType)) { - continue; - } - - this.addMatchesFromSearcher(searchState, requestedSearchers.get(searcherType)!, qualityOffset); + this.addMatchesFromSearcher(searchState, usableSearchers.spec(searcherType)!, qualityOffset); } const mergedMeta = MetaMerger.mergeMeta(searchState.meta); From 3ef681b5b35743aa25b13e2318e962adcd429fc8 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:29:20 +0200 Subject: [PATCH 073/105] run regression test --- src/regression-test/output/_indexing-meta.txt | 10 +++--- .../output/boston-deletion.txt | 18 ++++++++--- .../output/boulder-creek-substitution.txt | 20 ++++++++---- .../output/carcassonne-prefix.txt | 20 ++++++++---- .../output/carcassonne-substring.txt | 20 ++++++++---- .../output/carcassonne-suffix.txt | 18 ++++++++--- .../output/kuwait-city-prefix.txt | 18 ++++++++--- src/regression-test/output/kuwait-city.txt | 18 ++++++++--- .../output/munich-insertion.txt | 18 ++++++++--- .../output/tbilisi-deletion.txt | 18 ++++++++--- src/regression-test/output/tbilisi.txt | 18 ++++++++--- src/regression-test/output/tokyo-prefix.txt | 18 ++++++++--- src/regression-test/output/tokyo.txt | 31 +++++++++---------- .../output/t\303\274bingen-transposition.txt" | 20 ++++++++---- 14 files changed, 180 insertions(+), 85 deletions(-) diff --git a/src/regression-test/output/_indexing-meta.txt b/src/regression-test/output/_indexing-meta.txt index afe827b..70bb6ee 100644 --- a/src/regression-test/output/_indexing-meta.txt +++ b/src/regression-test/output/_indexing-meta.txt @@ -1,12 +1,12 @@ { - "indexingDurationSuffixArraySearcher": 2851, + "indexingDurationSuffixArraySearcher": 2829, "numberOfInvalidTerms": 1, - "indexingDurationFuzzySearcher": 3770, - "normalizationDurationNgrams": 271, + "indexingDurationFuzzySearcher": 3951, + "normalizationDurationNgrams": 292, "numberOfDistinctTerms": 1190185, - "normalizationDurationDefault": 1369, + "normalizationDurationDefault": 1302, "numberOfSurrogateCharacters": 46, "numberOfEntities": 1237154, "numberOfTerms": 1237486, - "indexingDurationTotal": 9638 + "indexingDurationTotal": 9790 } \ No newline at end of file diff --git a/src/regression-test/output/boston-deletion.txt b/src/regression-test/output/boston-deletion.txt index 58d83fd..ef72460 100644 --- a/src/regression-test/output/boston-deletion.txt +++ b/src/regression-test/output/boston-deletion.txt @@ -1,11 +1,19 @@ { "string": "bostn", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/boulder-creek-substitution.txt b/src/regression-test/output/boulder-creek-substitution.txt index 4f2cb02..2e01f4b 100644 --- a/src/regression-test/output/boulder-creek-substitution.txt +++ b/src/regression-test/output/boulder-creek-substitution.txt @@ -1,16 +1,24 @@ { "string": "boulder creak", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/carcassonne-prefix.txt b/src/regression-test/output/carcassonne-prefix.txt index ab5efde..00399a2 100644 --- a/src/regression-test/output/carcassonne-prefix.txt +++ b/src/regression-test/output/carcassonne-prefix.txt @@ -1,16 +1,24 @@ { "string": "carcasso", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } { - "queryDuration": 9 + "queryDuration": 10 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/carcassonne-substring.txt b/src/regression-test/output/carcassonne-substring.txt index 74a49fa..8e8a08a 100644 --- a/src/regression-test/output/carcassonne-substring.txt +++ b/src/regression-test/output/carcassonne-substring.txt @@ -1,16 +1,24 @@ { "string": "cassonn", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } { - "queryDuration": 4 + "queryDuration": 5 } Rank Entity Matched String Quality diff --git a/src/regression-test/output/carcassonne-suffix.txt b/src/regression-test/output/carcassonne-suffix.txt index bf1b3c5..2bd27d5 100644 --- a/src/regression-test/output/carcassonne-suffix.txt +++ b/src/regression-test/output/carcassonne-suffix.txt @@ -1,11 +1,19 @@ { "string": "sonne", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/kuwait-city-prefix.txt b/src/regression-test/output/kuwait-city-prefix.txt index bed15f1..57c7370 100644 --- a/src/regression-test/output/kuwait-city-prefix.txt +++ b/src/regression-test/output/kuwait-city-prefix.txt @@ -1,11 +1,19 @@ { "string": "مدينة الك", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/kuwait-city.txt b/src/regression-test/output/kuwait-city.txt index 5b58b65..1f4317e 100644 --- a/src/regression-test/output/kuwait-city.txt +++ b/src/regression-test/output/kuwait-city.txt @@ -1,11 +1,19 @@ { "string": "مدينة الكويت", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/munich-insertion.txt b/src/regression-test/output/munich-insertion.txt index 1957fca..e842003 100644 --- a/src/regression-test/output/munich-insertion.txt +++ b/src/regression-test/output/munich-insertion.txt @@ -1,11 +1,19 @@ { "string": "muniich", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/tbilisi-deletion.txt b/src/regression-test/output/tbilisi-deletion.txt index f003ae4..ee7805a 100644 --- a/src/regression-test/output/tbilisi-deletion.txt +++ b/src/regression-test/output/tbilisi-deletion.txt @@ -1,11 +1,19 @@ { "string": "თბიისი", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/tbilisi.txt b/src/regression-test/output/tbilisi.txt index 90b0cda..c712199 100644 --- a/src/regression-test/output/tbilisi.txt +++ b/src/regression-test/output/tbilisi.txt @@ -1,11 +1,19 @@ { "string": "თბილისი", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/tokyo-prefix.txt b/src/regression-test/output/tokyo-prefix.txt index 687f289..d41c226 100644 --- a/src/regression-test/output/tokyo-prefix.txt +++ b/src/regression-test/output/tokyo-prefix.txt @@ -1,11 +1,19 @@ { "string": "東京", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } diff --git a/src/regression-test/output/tokyo.txt b/src/regression-test/output/tokyo.txt index 6b859f3..a0858ed 100644 --- a/src/regression-test/output/tokyo.txt +++ b/src/regression-test/output/tokyo.txt @@ -1,27 +1,26 @@ { "string": "東京都", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } { - "queryDuration": 4 + "queryDuration": 2 } Rank Entity Matched String Quality -1 東京都 東京都 3.00 -2 東勢區 東勢區 0.24 -3 東勢鄉 東勢鄉 0.24 -4 東区 東区 0.24 -5 東區 東區 0.24 -6 東山區 東山區 0.24 -7 東河鄉 東河鄉 0.24 -8 東港鎮 東港鎮 0.24 -9 東澳 東澳 0.24 -10 東石鄉 東石鄉 0.24 \ No newline at end of file +1 東京都 東京都 3.00 \ No newline at end of file diff --git "a/src/regression-test/output/t\303\274bingen-transposition.txt" "b/src/regression-test/output/t\303\274bingen-transposition.txt" index 5c3eebc..df2b322 100644 --- "a/src/regression-test/output/t\303\274bingen-transposition.txt" +++ "b/src/regression-test/output/t\303\274bingen-transposition.txt" @@ -1,16 +1,24 @@ { "string": "tübignen", "topN": 10, - "minQuality": 0, - "searcherTypes": [ - "fuzzy", - "substring", - "prefix" + "searchers": [ + { + "type": "fuzzy", + "minQuality": 0.3 + }, + { + "type": "substring", + "minQuality": 0 + }, + { + "type": "prefix", + "minQuality": 0 + } ] } { - "queryDuration": 2 + "queryDuration": 3 } Rank Entity Matched String Quality From fe0a4d92fd153013774d397b6de09af9952d2513 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:30:38 +0200 Subject: [PATCH 074/105] docs --- src/commons/usable-searchers.ts | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/src/commons/usable-searchers.ts b/src/commons/usable-searchers.ts index 81d12f7..f33df6f 100644 --- a/src/commons/usable-searchers.ts +++ b/src/commons/usable-searchers.ts @@ -1,11 +1,21 @@ import { SearcherSpec } from '../interfaces/searcher-spec.js'; import { SearcherType } from '../interfaces/searcher-type.js'; -// todo: docs +/** + * Holds the searchers that are usable based on the available and requested searchers. + */ export class UsableSearchers { + /** + * The usable searchers mapped by their type. + */ private readonly usableSearchers: Map; + /** + * Creates a new instance of the UsableSearchers class. + * @param availableSearchers The types of searchers that are available. + * @param requestedSearchers The searchers requested for the query. + */ public constructor( availableSearchers: SearcherType[], requestedSearchers: SearcherSpec[]) { @@ -18,14 +28,29 @@ export class UsableSearchers { } } + /** + * Checks whether a searcher of the given type is usable. + * @param type The type of the searcher. + * @returns Whether the searcher is usable. + */ public has(type: SearcherType): boolean { return this.usableSearchers.has(type); } + /** + * Gets the minimum quality for the given searcher type. + * @param type The type of the searcher. + * @returns The minimum quality for the searcher type. + */ public minQuality(type: SearcherType): number { return this.usableSearchers.get(type)!.minQuality; } + /** + * Gets the searcher specification for the given type. + * @param type The type of the searcher. + * @returns The searcher specification for the searcher type. + */ public spec(type: SearcherType): SearcherSpec { return this.usableSearchers.get(type)!; } From 91ec2172e1b8a1a772d9217f80d3e39a5a87c15c Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:33:03 +0200 Subject: [PATCH 075/105] move ngram normalizer --- src/entity-searchers/entity-searcher-factory.ts | 2 +- src/fuzzy-searchers/index.ts | 1 + src/{normalization => fuzzy-searchers}/ngram-normalizer.ts | 5 ++--- src/normalization/index.ts | 1 - src/normalization/ngram-full-normalizer.test.ts | 2 +- src/normalization/ngram-normalizer.test.ts | 2 +- src/string-searchers/normalizing-searcher.test.ts | 2 +- 7 files changed, 7 insertions(+), 8 deletions(-) rename src/{normalization => fuzzy-searchers}/ngram-normalizer.ts (91%) diff --git a/src/entity-searchers/entity-searcher-factory.ts b/src/entity-searchers/entity-searcher-factory.ts index 8b948a0..30eff95 100644 --- a/src/entity-searchers/entity-searcher-factory.ts +++ b/src/entity-searchers/entity-searcher-factory.ts @@ -8,7 +8,7 @@ import { FuzzySearchConfig } from '../fuzzy-search.js'; import { FuzzySearcher } from '../fuzzy-searchers/fuzzy-searcher.js'; import { InequalityPenalizingSearcher } from '../string-searchers/inequality-penalizing-searcher.js'; import { NgramComputer } from '../fuzzy-searchers/ngram-computer.js'; -import { NgramNormalizer } from '../normalization/ngram-normalizer.js'; +import { NgramNormalizer } from '../fuzzy-searchers/ngram-normalizer.js'; import { Normalizer } from '../interfaces/normalizer.js'; import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from '../string-searchers/normalizing-searcher.js'; diff --git a/src/fuzzy-searchers/index.ts b/src/fuzzy-searchers/index.ts index 76f3dfd..f78f73e 100644 --- a/src/fuzzy-searchers/index.ts +++ b/src/fuzzy-searchers/index.ts @@ -1,6 +1,7 @@ export { FuzzySearcher as FuzzySearcherImpl } from './fuzzy-searcher.js'; export { FuzzySearchConfig } from './fuzzy-search-config.js'; export { InvertedIndex } from './inverted-index.js'; +export { NgramNormalizer } from './ngram-normalizer.js'; export { NgramComputer } from './ngram-computer.js'; export { QualityComputer } from './quality-computer.js'; export { TermIds } from './term-ids.js'; diff --git a/src/normalization/ngram-normalizer.ts b/src/fuzzy-searchers/ngram-normalizer.ts similarity index 91% rename from src/normalization/ngram-normalizer.ts rename to src/fuzzy-searchers/ngram-normalizer.ts index b5efa64..5a7738d 100644 --- a/src/normalization/ngram-normalizer.ts +++ b/src/fuzzy-searchers/ngram-normalizer.ts @@ -1,7 +1,6 @@ import { Meta } from '../interfaces/meta.js'; -import { NormalizationResult } from './normalization-result.js'; +import { NormalizationResult } from '../normalization/normalization-result.js'; import { Normalizer } from '../interfaces/normalizer.js'; -// todo: move to fuzzy-searchers /** * Normalization for creating proper n-grams. */ @@ -16,7 +15,7 @@ export class NgramNormalizer implements Normalizer { public readonly paddingLeft: string, public readonly paddingRight: string, public readonly paddingMiddle: string - ) {} + ) { } /** * {@inheritDoc Normalizer.normalize} diff --git a/src/normalization/index.ts b/src/normalization/index.ts index 7be12ef..a9f92fe 100644 --- a/src/normalization/index.ts +++ b/src/normalization/index.ts @@ -3,7 +3,6 @@ export { DefaultNormalizer } from './default-normalizer.js'; export { GenericNormalizer } from './generic-normalizer.js'; export { LatinReplacements } from './latin-replacements.js'; export { MultiNormalizer } from './multi-normalizer.js'; -export { NgramNormalizer } from './ngram-normalizer.js'; export { NormalizationResult } from './normalization-result.js'; export { NormalizerConfig } from './normalizer-config.js'; export { SanitizingNormalizer } from './sanitizing-normalizer.js'; diff --git a/src/normalization/ngram-full-normalizer.test.ts b/src/normalization/ngram-full-normalizer.test.ts index 3725025..af14c22 100644 --- a/src/normalization/ngram-full-normalizer.test.ts +++ b/src/normalization/ngram-full-normalizer.test.ts @@ -1,6 +1,6 @@ import { DefaultNormalizer } from './default-normalizer.js'; import { MultiNormalizer } from './multi-normalizer.js'; -import { NgramNormalizer } from './ngram-normalizer.js'; +import { NgramNormalizer } from '../fuzzy-searchers/ngram-normalizer.js'; import { NormalizerConfig } from './normalizer-config.js'; const defaultNormalizer = DefaultNormalizer.create(NormalizerConfig.createDefaultConfig()); diff --git a/src/normalization/ngram-normalizer.test.ts b/src/normalization/ngram-normalizer.test.ts index c51aad9..a6f6054 100644 --- a/src/normalization/ngram-normalizer.test.ts +++ b/src/normalization/ngram-normalizer.test.ts @@ -1,4 +1,4 @@ -import { NgramNormalizer } from './ngram-normalizer.js'; +import { NgramNormalizer } from '../fuzzy-searchers/ngram-normalizer.js'; const normalizer = new NgramNormalizer('$$', '!!', '%%'); diff --git a/src/string-searchers/normalizing-searcher.test.ts b/src/string-searchers/normalizing-searcher.test.ts index 5aa5099..86d3dde 100644 --- a/src/string-searchers/normalizing-searcher.test.ts +++ b/src/string-searchers/normalizing-searcher.test.ts @@ -2,7 +2,7 @@ import { DefaultNormalizer } from '../normalization/default-normalizer.js'; import { LiteralSearcher } from './literal-searcher.js'; import { Match } from './match.js'; import { MultiNormalizer } from '../normalization/multi-normalizer.js'; -import { NgramNormalizer } from '../normalization/ngram-normalizer.js'; +import { NgramNormalizer } from '../fuzzy-searchers/ngram-normalizer.js'; import { NormalizerConfig } from '../normalization/normalizer-config.js'; import { NormalizingSearcher } from './normalizing-searcher.js'; import { StringSearchQuery } from '../interfaces/string-search-query.js'; From 9d29da8b6a2082d981fb7c6621f3216f2734ca2e Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:34:48 +0200 Subject: [PATCH 076/105] docs --- .../suffix-array-searcher.ts | 20 ++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 787c77e..86bdba7 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -7,7 +7,9 @@ import { StringSearchQuery } from '../interfaces/string-search-query.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; import { SuffixArray } from './suffix-array.js'; -// todo: docs. +/** + * A suffix array-based string searcher. + */ export class SuffixArraySearcher implements StringSearcher { public readonly separator: string; private str: string; @@ -15,6 +17,10 @@ export class SuffixArraySearcher implements StringSearcher { private indexToTermIndex: Int32Array; private termLengths: Int32Array; + /** + * Creates a new instance of the SuffixArraySearcher class. + * @param separator The separator used between terms. + */ public constructor(separator: string) { this.separator = separator; this.str = ''; @@ -84,10 +90,22 @@ export class SuffixArraySearcher implements StringSearcher { return new Result(matches, query, new Meta()); } + /** + * Computes the quality of a match based on the lengths of the query and the term. + * @param queryLength The length of the query. + * @param termLength The length of the term. + * @returns The quality of the match. + */ private computeQuality(queryLength: number, termLength: number): number { return queryLength / termLength; } + /** + * Gets the positions in the suffix array where the given substring matches. + * @param substring The substring to search for. + * @returns The start and end positions of the substring in the suffix array. The start is inclusive, the end is + * exclusive. + */ private GetPositionsInSuffixArray(substring: string): number[] { let l = 0; let r = this.suffixArray.length; From 0cf47b0f1cfbf272390bfde125a23f7abe87ee85 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:35:56 +0200 Subject: [PATCH 077/105] prettier --- .vscode/launch.json | 18 +- .vscode/tasks.json | 6 +- src/basic-usage.ts | 40 +-- src/commons/meta-merger.test.ts | 102 ++++---- src/commons/meta-merger.ts | 15 +- src/commons/usable-searchers.ts | 85 +++---- src/config.ts | 10 +- .../default-dynamic-searcher.test.ts | 2 +- .../default-dynamic-searcher.ts | 11 +- src/dynamic-searchers/result-merger.ts | 4 +- .../timing-dynamic-searcher.ts | 2 +- .../default-entity-searcher-basic.test.ts | 13 +- .../default-entity-searcher.ts | 18 +- .../entity-searcher-factory.ts | 22 +- src/entity-searchers/fast-entity-searcher.ts | 235 +++++++++--------- src/entity-searchers/search-state.ts | 46 ++-- .../sorting-entity-searcher.ts | 229 +++++++++-------- src/fuzzy-searchers/ngram-normalizer.ts | 2 +- src/interfaces/query.ts | 4 +- src/interfaces/searcher-spec.ts | 101 ++++---- src/interfaces/searcher-type.ts | 27 +- src/interfaces/string-search-query.ts | 54 ++-- src/performance-test/main.ts | 10 +- src/performance/performance-test.ts | 13 +- src/performance/query-counts.ts | 35 ++- src/performance/report.ts | 15 +- src/performance/test-run-parameters.ts | 4 +- src/sort-order.ts | 18 +- src/string-searchers/distinct-searcher.ts | 4 +- .../inequality-penalizing-searcher.test.ts | 8 +- src/string-searchers/normalizing-searcher.ts | 7 +- src/string-searchers/result.ts | 2 +- src/string-searchers/searcher-switch.ts | 163 ++++++------ src/string-searchers/sorting-searcher.test.ts | 4 +- src/string-searchers/sorting-searcher.ts | 10 +- .../prefix-searcher.test.ts | 28 +-- src/suffix-array-searchers/prefix-searcher.ts | 88 ++++--- .../substring-search-config.ts | 4 +- .../suffix-array-searcher.ts | 2 +- src/suffix-array-searchers/suffix-array.ts | 10 +- 40 files changed, 734 insertions(+), 737 deletions(-) diff --git a/.vscode/launch.json b/.vscode/launch.json index e796d7a..82d9d50 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -5,25 +5,17 @@ "type": "node", "request": "launch", "name": "Launch Basic Usage", - "skipFiles": [ - "/**" - ], + "skipFiles": ["/**"], "program": "${workspaceFolder}/dist/basic-usage.js", - "outFiles": [ - "${workspaceFolder}/**/*.js" - ] + "outFiles": ["${workspaceFolder}/**/*.js"] }, { "type": "node", "request": "launch", "name": "Run Regression Test", - "skipFiles": [ - "/**" - ], + "skipFiles": ["/**"], "program": "${workspaceFolder}/dist/regression-test/main.js", - "outFiles": [ - "${workspaceFolder}/**/*.js" - ] + "outFiles": ["${workspaceFolder}/**/*.js"] } ] -} \ No newline at end of file +} diff --git a/.vscode/tasks.json b/.vscode/tasks.json index 6d68c89..5520ab9 100644 --- a/.vscode/tasks.json +++ b/.vscode/tasks.json @@ -5,9 +5,7 @@ "type": "typescript", "tsconfig": "tsconfig.json", "option": "watch", - "problemMatcher": [ - "$tsc-watch" - ], + "problemMatcher": ["$tsc-watch"], "group": { "kind": "build", "isDefault": true @@ -22,4 +20,4 @@ "group": "test" } ] -} \ No newline at end of file +} diff --git a/src/basic-usage.ts b/src/basic-usage.ts index be791d7..c22418f 100644 --- a/src/basic-usage.ts +++ b/src/basic-usage.ts @@ -3,11 +3,11 @@ import { Query } from './interfaces/query.js'; import { SearcherFactory } from './searcher-factory.js'; class Person { - constructor( - public id: number, - public firstName: string, - public lastName: string - ) { } + constructor( + public id: number, + public firstName: string, + public lastName: string + ) {} } const searcher = SearcherFactory.createDefaultSearcher(); @@ -18,16 +18,16 @@ config.normalizerConfig.allowCharacter = (_c) => true; const searcher = SearcherFactory.createSearcher(config); */ const persons = [ - { id: 23501, firstName: 'Alice', lastName: 'King' }, - { id: 99234, firstName: 'Bob', lastName: 'Bishop' }, - { id: 5823, firstName: 'Carol', lastName: 'Queen' }, - { id: 11923, firstName: 'Charlie', lastName: 'Rook' } + { id: 23501, firstName: 'Alice', lastName: 'King' }, + { id: 99234, firstName: 'Bob', lastName: 'Bishop' }, + { id: 5823, firstName: 'Carol', lastName: 'Queen' }, + { id: 11923, firstName: 'Charlie', lastName: 'Rook' } ]; const indexingMeta = searcher.indexEntities( - persons, - (e) => e.id, - (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] + persons, + (e) => e.id, + (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] ); console.dir(indexingMeta); @@ -38,19 +38,19 @@ const removalResult = searcher.removeEntities([99234, 5823]); console.dir(removalResult); const persons2 = [ - { id: 723, firstName: 'David', lastName: 'Knight' }, // new - { id: 2634, firstName: 'Eve', lastName: 'Pawn' }, // new - { id: 23501, firstName: 'Allie', lastName: 'King' }, // updated - { id: 11923, firstName: 'Charles', lastName: 'Rook' } // updated + { id: 723, firstName: 'David', lastName: 'Knight' }, // new + { id: 2634, firstName: 'Eve', lastName: 'Pawn' }, // new + { id: 23501, firstName: 'Allie', lastName: 'King' }, // updated + { id: 11923, firstName: 'Charles', lastName: 'Rook' } // updated ]; const upsertMeta = searcher.upsertEntities( - persons2, - (e) => e.id, - (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] + persons2, + (e) => e.id, + (e) => [e.firstName, e.lastName, `${e.firstName} ${e.lastName}`] ); console.dir(upsertMeta); const result2 = searcher.getMatches(new Query('allie')); console.dir(result2); -console.log('Finished.'); \ No newline at end of file +console.log('Finished.'); diff --git a/src/commons/meta-merger.test.ts b/src/commons/meta-merger.test.ts index a636366..bf1f032 100644 --- a/src/commons/meta-merger.test.ts +++ b/src/commons/meta-merger.test.ts @@ -4,60 +4,60 @@ import { Meta } from '../interfaces/meta.js'; import { MetaMerger } from './meta-merger.js'; test('can merge two metas', () => { - const meta1 = new Meta(); - meta1.add("key1", "meta1Value"); - meta1.add("key2", 10); - meta1.add("key3", "someOtherValue"); - - const meta2 = new Meta(); - meta2.add("key1", "meta2Value"); - meta2.add("key2", 10); - - const mergedMeta = MetaMerger.mergeMeta([meta1, meta2]); - const entries: ReadonlyMap = mergedMeta.allEntries; - const expectedEntries = new Map([ - ["key1_0", "meta1Value"], - ["key1_1", "meta2Value"], - ["key2", 20], - ["key3", "someOtherValue"] - ]); - - checkMaps(expectedEntries, entries); + const meta1 = new Meta(); + meta1.add('key1', 'meta1Value'); + meta1.add('key2', 10); + meta1.add('key3', 'someOtherValue'); + + const meta2 = new Meta(); + meta2.add('key1', 'meta2Value'); + meta2.add('key2', 10); + + const mergedMeta = MetaMerger.mergeMeta([meta1, meta2]); + const entries: ReadonlyMap = mergedMeta.allEntries; + const expectedEntries = new Map([ + ['key1_0', 'meta1Value'], + ['key1_1', 'meta2Value'], + ['key2', 20], + ['key3', 'someOtherValue'] + ]); + + checkMaps(expectedEntries, entries); }); test('can merge three metas', () => { - const meta1 = new Meta(); - meta1.add("key1", "meta1Value"); - meta1.add("key2", 10); - meta1.add("key3", "someOtherValue"); - - const meta2 = new Meta(); - meta2.add("key1", "meta2Value"); - meta2.add("key2", 10); - - const meta3 = new Meta(); - meta3.add("key1", "meta3Value"); - meta3.add("key2", 5); - meta3.add("key4", "additionalValue"); - - const mergedMeta = MetaMerger.mergeMeta([meta1, meta2, meta3]); - const entries: ReadonlyMap = mergedMeta.allEntries; - const expectedEntries = new Map([ - ["key1_0", "meta1Value"], - ["key1_1", "meta2Value"], - ["key1_2", "meta3Value"], - ["key2", 25], - ["key3", "someOtherValue"], - ["key4", "additionalValue"] - ]); - - checkMaps(expectedEntries, entries); + const meta1 = new Meta(); + meta1.add('key1', 'meta1Value'); + meta1.add('key2', 10); + meta1.add('key3', 'someOtherValue'); + + const meta2 = new Meta(); + meta2.add('key1', 'meta2Value'); + meta2.add('key2', 10); + + const meta3 = new Meta(); + meta3.add('key1', 'meta3Value'); + meta3.add('key2', 5); + meta3.add('key4', 'additionalValue'); + + const mergedMeta = MetaMerger.mergeMeta([meta1, meta2, meta3]); + const entries: ReadonlyMap = mergedMeta.allEntries; + const expectedEntries = new Map([ + ['key1_0', 'meta1Value'], + ['key1_1', 'meta2Value'], + ['key1_2', 'meta3Value'], + ['key2', 25], + ['key3', 'someOtherValue'], + ['key4', 'additionalValue'] + ]); + + checkMaps(expectedEntries, entries); }); function checkMaps(expectedMap: ReadonlyMap, actualMap: ReadonlyMap): void { - expect(expectedMap.size).toBe(actualMap.size); - for (const [key, value] of actualMap) { - expect(expectedMap.has(key)).toBe(true); - expect(expectedMap.get(key)).toBe(value); - } -} \ No newline at end of file + expect(expectedMap.size).toBe(actualMap.size); + for (const [key, value] of actualMap) { + expect(expectedMap.has(key)).toBe(true); + expect(expectedMap.get(key)).toBe(value); + } +} diff --git a/src/commons/meta-merger.ts b/src/commons/meta-merger.ts index b84706e..cc4f409 100644 --- a/src/commons/meta-merger.ts +++ b/src/commons/meta-merger.ts @@ -7,11 +7,11 @@ import { Meta } from '../interfaces/meta.js'; */ export class MetaMerger { /** - * Merges {@link Meta} objects into a new {@link Meta} object. Numbers for the same key are summed up, other values - * are stored with an index suffix. - * @param metas The meta objects to merge. - * @returns The merged meta object. - */ + * Merges {@link Meta} objects into a new {@link Meta} object. Numbers for the same key are summed up, other values + * are stored with an index suffix. + * @param metas The meta objects to merge. + * @returns The merged meta object. + */ public static mergeMeta(metas: Meta[]): Meta { if (metas.length === 0) { return new Meta(); @@ -28,8 +28,7 @@ export class MetaMerger { const present = metaLists.get(key); if (present === undefined) { metaLists.set(key, [value]); - } - else { + } else { present.push(value); } } @@ -43,7 +42,7 @@ export class MetaMerger { continue; } - if (values.every(v => typeof v === 'number')) { + if (values.every((v) => typeof v === 'number')) { const sum = values.reduce((acc, val) => acc + val, 0); newMetaEntries.set(key, sum); continue; diff --git a/src/commons/usable-searchers.ts b/src/commons/usable-searchers.ts index f33df6f..54c6794 100644 --- a/src/commons/usable-searchers.ts +++ b/src/commons/usable-searchers.ts @@ -5,53 +5,50 @@ import { SearcherType } from '../interfaces/searcher-type.js'; * Holds the searchers that are usable based on the available and requested searchers. */ export class UsableSearchers { + /** + * The usable searchers mapped by their type. + */ + private readonly usableSearchers: Map; - /** - * The usable searchers mapped by their type. - */ - private readonly usableSearchers: Map; + /** + * Creates a new instance of the UsableSearchers class. + * @param availableSearchers The types of searchers that are available. + * @param requestedSearchers The searchers requested for the query. + */ + public constructor(availableSearchers: SearcherType[], requestedSearchers: SearcherSpec[]) { + this.usableSearchers = new Map(); - /** - * Creates a new instance of the UsableSearchers class. - * @param availableSearchers The types of searchers that are available. - * @param requestedSearchers The searchers requested for the query. - */ - public constructor( - availableSearchers: SearcherType[], - requestedSearchers: SearcherSpec[]) { - this.usableSearchers = new Map(); - - for (const requestedSearcher of requestedSearchers) { - if (availableSearchers.includes(requestedSearcher.type)) { - this.usableSearchers.set(requestedSearcher.type, requestedSearcher); - } - } + for (const requestedSearcher of requestedSearchers) { + if (availableSearchers.includes(requestedSearcher.type)) { + this.usableSearchers.set(requestedSearcher.type, requestedSearcher); + } } + } - /** - * Checks whether a searcher of the given type is usable. - * @param type The type of the searcher. - * @returns Whether the searcher is usable. - */ - public has(type: SearcherType): boolean { - return this.usableSearchers.has(type); - } + /** + * Checks whether a searcher of the given type is usable. + * @param type The type of the searcher. + * @returns Whether the searcher is usable. + */ + public has(type: SearcherType): boolean { + return this.usableSearchers.has(type); + } - /** - * Gets the minimum quality for the given searcher type. - * @param type The type of the searcher. - * @returns The minimum quality for the searcher type. - */ - public minQuality(type: SearcherType): number { - return this.usableSearchers.get(type)!.minQuality; - } + /** + * Gets the minimum quality for the given searcher type. + * @param type The type of the searcher. + * @returns The minimum quality for the searcher type. + */ + public minQuality(type: SearcherType): number { + return this.usableSearchers.get(type)!.minQuality; + } - /** - * Gets the searcher specification for the given type. - * @param type The type of the searcher. - * @returns The searcher specification for the searcher type. - */ - public spec(type: SearcherType): SearcherSpec { - return this.usableSearchers.get(type)!; - } -} \ No newline at end of file + /** + * Gets the searcher specification for the given type. + * @param type The type of the searcher. + * @returns The searcher specification for the searcher type. + */ + public spec(type: SearcherType): SearcherSpec { + return this.usableSearchers.get(type)!; + } +} diff --git a/src/config.ts b/src/config.ts index 223ea84..2d3f039 100644 --- a/src/config.ts +++ b/src/config.ts @@ -24,7 +24,7 @@ export class Config { public sortOrder: SortOrder, public fuzzySearchConfig?: FuzzySearchConfig, public substringSearchConfig?: SubstringSearchConfig - ) { } + ) {} /** * Creates an opinionated default configuration. @@ -38,6 +38,12 @@ export class Config { const fuzzySearchConfig = FuzzySearchConfig.createDefaultConfig(); const substringSearchConfig = SubstringSearchConfig.createDefaultConfig(); return new Config( - searcherTypes, normalizerConfig, maxQueryLength, sortOrder, fuzzySearchConfig, substringSearchConfig); + searcherTypes, + normalizerConfig, + maxQueryLength, + sortOrder, + fuzzySearchConfig, + substringSearchConfig + ); } } diff --git a/src/dynamic-searchers/default-dynamic-searcher.test.ts b/src/dynamic-searchers/default-dynamic-searcher.test.ts index 7759fdd..07997a8 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.test.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.test.ts @@ -8,7 +8,7 @@ class Person { public id: number, public name: string, public favoriteHobby: string - ) { } + ) {} } function createSearcher(): DynamicSearcher { diff --git a/src/dynamic-searchers/default-dynamic-searcher.ts b/src/dynamic-searchers/default-dynamic-searcher.ts index 103d185..0053296 100644 --- a/src/dynamic-searchers/default-dynamic-searcher.ts +++ b/src/dynamic-searchers/default-dynamic-searcher.ts @@ -25,7 +25,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher, private readonly secondarySearcher: EntitySearcher - ) { } + ) {} /** * {@inheritDoc DynamicSearcher.indexEntities} @@ -47,8 +47,7 @@ export class DefaultDynamicSearcher implements DynamicSearcher this.maxQueryLength) { - query = new Query( - query.string.substring(0, this.maxQueryLength), query.topN, query.searchers); + query = new Query(query.string.substring(0, this.maxQueryLength), query.topN, query.searchers); } return ResultMerger.mergeResults(this.mainSearcher.getMatches(query), this.secondarySearcher.getMatches(query)); } @@ -203,8 +202,10 @@ export class DefaultDynamicSearcher implements DynamicSearcher m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : 0 ); return newMatches.length <= topN ? newMatches : newMatches.slice(0, topN); } diff --git a/src/dynamic-searchers/timing-dynamic-searcher.ts b/src/dynamic-searchers/timing-dynamic-searcher.ts index 6a19af6..2421675 100644 --- a/src/dynamic-searchers/timing-dynamic-searcher.ts +++ b/src/dynamic-searchers/timing-dynamic-searcher.ts @@ -17,7 +17,7 @@ export class TimingDynamicSearcher implements DynamicSearcher) { } + public constructor(private readonly dynamicSearcher: DynamicSearcher) {} /** * {@inheritDoc DynamicSearcher.removeEntities} diff --git a/src/entity-searchers/default-entity-searcher-basic.test.ts b/src/entity-searchers/default-entity-searcher-basic.test.ts index f0132bf..d5b4f84 100644 --- a/src/entity-searchers/default-entity-searcher-basic.test.ts +++ b/src/entity-searchers/default-entity-searcher-basic.test.ts @@ -7,8 +7,10 @@ import { SearcherType } from '../interfaces/searcher-type.js'; import { StringSearcher } from '../interfaces/string-searcher.js'; const literalSearcher: StringSearcher = new LiteralSearcher(); -const entitySearcher: EntitySearcher<{ id: number; name: string }, number> = - new DefaultEntitySearcher(literalSearcher, [SearcherType.Prefix]); +const entitySearcher: EntitySearcher<{ id: number; name: string }, number> = new DefaultEntitySearcher( + literalSearcher, + [SearcherType.Prefix] +); const entities = [ { id: 23501, name: 'Alice' }, { id: 99234, name: 'Bob' }, @@ -36,12 +38,13 @@ class Person { public id: number, public name: string, public job: string - ) { } + ) {} } const literalSearcher2: StringSearcher = new LiteralSearcher(); -const entitySearcher2: EntitySearcher = - new DefaultEntitySearcher(literalSearcher2, [SearcherType.Prefix]); +const entitySearcher2: EntitySearcher = new DefaultEntitySearcher(literalSearcher2, [ + SearcherType.Prefix +]); const entities2 = [ new Person(23501, 'Alice', 'Programmer'), new Person(99234, 'Bob', 'Teacher'), diff --git a/src/entity-searchers/default-entity-searcher.ts b/src/entity-searchers/default-entity-searcher.ts index 05bec42..cce391d 100644 --- a/src/entity-searchers/default-entity-searcher.ts +++ b/src/entity-searchers/default-entity-searcher.ts @@ -70,7 +70,6 @@ export class DefaultEntitySearcher implements EntitySearcher implements EntitySearcher 1) { return; } const stringSearchQuery: StringSearchQuery = new StringSearchQuery( - searchState.query.string, minQuality, searcherSpec.type); + searchState.query.string, + minQuality, + searcherSpec.type + ); const result: Result = this.stringSearcher.getMatches(stringSearchQuery); this.addMatchesFromResult(searchState, result, qualityOffset); } @@ -163,11 +164,7 @@ export class DefaultEntitySearcher implements EntitySearcher, - result: Result, - qualityOffset: number - ): void { + private addMatchesFromResult(searchState: SearchState, result: Result, qualityOffset: number): void { searchState.meta.push(result.meta); if (searchState.query.topN === 0) { return; @@ -177,8 +174,9 @@ export class DefaultEntitySearcher implements EntitySearcher = - new DefaultEntitySearcher(stringSearcher, config.searcherTypes); + let entitySearcher: EntitySearcher = new DefaultEntitySearcher( + stringSearcher, + config.searcherTypes + ); entitySearcher = new FastEntitySearcher(entitySearcher, config.searcherTypes); entitySearcher = new SortingEntitySearcher(config.sortOrder, entitySearcher); return entitySearcher; @@ -64,13 +62,13 @@ export class EntitySearcherFactory { const forbiddenCharacters = new Set(); if (config.fuzzySearchConfig) { - config.fuzzySearchConfig.paddingLeft.split('').forEach(c => forbiddenCharacters.add(c)); - config.fuzzySearchConfig.paddingRight.split('').forEach(c => forbiddenCharacters.add(c)); - config.fuzzySearchConfig.paddingMiddle.split('').forEach(c => forbiddenCharacters.add(c)); + config.fuzzySearchConfig.paddingLeft.split('').forEach((c) => forbiddenCharacters.add(c)); + config.fuzzySearchConfig.paddingRight.split('').forEach((c) => forbiddenCharacters.add(c)); + config.fuzzySearchConfig.paddingMiddle.split('').forEach((c) => forbiddenCharacters.add(c)); } if (config.substringSearchConfig) { - config.substringSearchConfig.suffixArraySeparator.split('').forEach(c => forbiddenCharacters.add(c)); + config.substringSearchConfig.suffixArraySeparator.split('').forEach((c) => forbiddenCharacters.add(c)); } const allowCharacter: (c: string) => boolean = (c) => @@ -177,4 +175,4 @@ export class EntitySearcherFactory { } return new PrefixSearcher(suffixArraySearcher); } -} \ No newline at end of file +} diff --git a/src/entity-searchers/fast-entity-searcher.ts b/src/entity-searchers/fast-entity-searcher.ts index 9e60d5f..9a5d25c 100644 --- a/src/entity-searchers/fast-entity-searcher.ts +++ b/src/entity-searchers/fast-entity-searcher.ts @@ -1,130 +1,131 @@ -import { PrefixSearcher, SubstringSearcher } from "../interfaces/searcher-spec.js"; -import { EntityResult } from "../interfaces/entity-result.js"; -import { EntitySearcher } from "../interfaces/entity-searcher.js"; -import { Memento } from "../interfaces/memento.js"; -import { Meta } from "../interfaces/meta.js"; -import { Query } from "../interfaces/query.js"; -import { SearcherType } from "../interfaces/searcher-type.js"; -import { UsableSearchers } from "../commons/usable-searchers.js"; +import { PrefixSearcher, SubstringSearcher } from '../interfaces/searcher-spec.js'; +import { EntityResult } from '../interfaces/entity-result.js'; +import { EntitySearcher } from '../interfaces/entity-searcher.js'; +import { Memento } from '../interfaces/memento.js'; +import { Meta } from '../interfaces/meta.js'; +import { Query } from '../interfaces/query.js'; +import { SearcherType } from '../interfaces/searcher-type.js'; +import { UsableSearchers } from '../commons/usable-searchers.js'; /** - * A entity searcher that tries to optimize performance by querying with limited searchers and an increased quality + * A entity searcher that tries to optimize performance by querying with limited searchers and an increased quality * threshold at first. * @typeParam TEntity The type of the entities. * @typeParam TId The type of the entity ids. */ export class FastEntitySearcher implements EntitySearcher { - - /** - * Creates a new instance of the FastEntitySearcher class. - * @typeParam TEntity The type of the entities. - * @typeParam TId The type of the entity ids. - * @param entitySearcher The entity searcher. - * @param searcherTypes The available searcher types. - */ - public constructor( - private readonly entitySearcher: EntitySearcher, - private readonly searcherTypes: SearcherType[]) { - } - - /** - * {@inheritDoc EntitySearcher.indexEntities} - */ - indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { - return this.entitySearcher.indexEntities(entities, getId, getTerms); - } - - /** - * {@inheritDoc EntitySearcher.getMatches} - */ - getMatches(query: Query): EntityResult { - - const usableSearchers = new UsableSearchers(this.searcherTypes, query.searchers); - - if (query.topN > 200) { - return this.entitySearcher.getMatches(query); - } - - // Make the prefix searcher faster for short queries. - if (query.string.length <= 3 && - usableSearchers.has(SearcherType.Prefix) && - usableSearchers.minQuality(SearcherType.Prefix) < 2.2) { - const newQuery = new Query(query.string, query.topN, [new PrefixSearcher(2.3)]); - const result = this.entitySearcher.getMatches(newQuery); - if (result.matches.length == query.topN) { - return new EntityResult(result.matches, query, result.meta); - } - } - - // If there is no prefix searcher, make the substring searcher faster for short queries. - if (query.string.length <= 3 && - !usableSearchers.has(SearcherType.Prefix) && - usableSearchers.has(SearcherType.Substring) && - usableSearchers.minQuality(SearcherType.Substring) < 1.2) { - const newQuery = new Query(query.string, query.topN, [new SubstringSearcher(1.3)]); - const result = this.entitySearcher.getMatches(newQuery); - if (result.matches.length == query.topN) { - return new EntityResult(result.matches, query, result.meta); - } - } - - return this.entitySearcher.getMatches(query); - } - - /** - * {@inheritDoc EntitySearcher.tryGetEntity} - */ - tryGetEntity(id: TId): TEntity | null { - return this.entitySearcher.tryGetEntity(id); - } - - /** - * {@inheritDoc EntitySearcher.getEntities} - */ - getEntities(): TEntity[] { - return this.entitySearcher.getEntities(); - } - - /** - * {@inheritDoc EntitySearcher.tryGetTerms} - */ - tryGetTerms(id: TId): string[] | null { - return this.entitySearcher.tryGetTerms(id); - } - - /** - * {@inheritDoc EntitySearcher.getTerms} - */ - getTerms(): string[] { - return this.entitySearcher.getTerms(); - } - - /** - * {@inheritDoc EntitySearcher.removeEntity} - */ - removeEntity(id: TId): boolean { - return this.entitySearcher.removeEntity(id); - } - - /** - * {@inheritDoc EntitySearcher.replaceEntity} - */ - replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { - return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + /** + * Creates a new instance of the FastEntitySearcher class. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + * @param entitySearcher The entity searcher. + * @param searcherTypes The available searcher types. + */ + public constructor( + private readonly entitySearcher: EntitySearcher, + private readonly searcherTypes: SearcherType[] + ) {} + + /** + * {@inheritDoc EntitySearcher.indexEntities} + */ + indexEntities(entities: TEntity[], getId: (entity: TEntity) => TId, getTerms: (entity: TEntity) => string[]): Meta { + return this.entitySearcher.indexEntities(entities, getId, getTerms); + } + + /** + * {@inheritDoc EntitySearcher.getMatches} + */ + getMatches(query: Query): EntityResult { + const usableSearchers = new UsableSearchers(this.searcherTypes, query.searchers); + + if (query.topN > 200) { + return this.entitySearcher.getMatches(query); } - /** - * {@inheritDoc EntitySearcher.save} - */ - save(memento: Memento): void { - return this.entitySearcher.save(memento); + // Make the prefix searcher faster for short queries. + if ( + query.string.length <= 3 && + usableSearchers.has(SearcherType.Prefix) && + usableSearchers.minQuality(SearcherType.Prefix) < 2.2 + ) { + const newQuery = new Query(query.string, query.topN, [new PrefixSearcher(2.3)]); + const result = this.entitySearcher.getMatches(newQuery); + if (result.matches.length == query.topN) { + return new EntityResult(result.matches, query, result.meta); + } } - /** - * {@inheritDoc EntitySearcher.load} - */ - load(memento: Memento): void { - return this.entitySearcher.load(memento); + // If there is no prefix searcher, make the substring searcher faster for short queries. + if ( + query.string.length <= 3 && + !usableSearchers.has(SearcherType.Prefix) && + usableSearchers.has(SearcherType.Substring) && + usableSearchers.minQuality(SearcherType.Substring) < 1.2 + ) { + const newQuery = new Query(query.string, query.topN, [new SubstringSearcher(1.3)]); + const result = this.entitySearcher.getMatches(newQuery); + if (result.matches.length == query.topN) { + return new EntityResult(result.matches, query, result.meta); + } } -} \ No newline at end of file + return this.entitySearcher.getMatches(query); + } + + /** + * {@inheritDoc EntitySearcher.tryGetEntity} + */ + tryGetEntity(id: TId): TEntity | null { + return this.entitySearcher.tryGetEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.getEntities} + */ + getEntities(): TEntity[] { + return this.entitySearcher.getEntities(); + } + + /** + * {@inheritDoc EntitySearcher.tryGetTerms} + */ + tryGetTerms(id: TId): string[] | null { + return this.entitySearcher.tryGetTerms(id); + } + + /** + * {@inheritDoc EntitySearcher.getTerms} + */ + getTerms(): string[] { + return this.entitySearcher.getTerms(); + } + + /** + * {@inheritDoc EntitySearcher.removeEntity} + */ + removeEntity(id: TId): boolean { + return this.entitySearcher.removeEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.replaceEntity} + */ + replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { + return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + } + + /** + * {@inheritDoc EntitySearcher.save} + */ + save(memento: Memento): void { + return this.entitySearcher.save(memento); + } + + /** + * {@inheritDoc EntitySearcher.load} + */ + load(memento: Memento): void { + return this.entitySearcher.load(memento); + } +} diff --git a/src/entity-searchers/search-state.ts b/src/entity-searchers/search-state.ts index aca98e7..1b5e995 100644 --- a/src/entity-searchers/search-state.ts +++ b/src/entity-searchers/search-state.ts @@ -7,27 +7,27 @@ import { Query } from '../interfaces/query.js'; * @typeParam TEntity The type of the entities. */ export class SearchState { - /** - * The search query. - */ - public readonly query: Query; - /** - * The indexes of the entities that have already been matched. - */ - public readonly matchedIndexes: Set = new Set(); - /** - * The matched entities. - */ - public readonly matches: EntityMatch[] = []; - /** - * The meta data for each searcher type. - */ - public readonly meta: Meta[] = []; - /** - * Creates a new instance of the SearchState class. - * @param query The search query. - */ - public constructor(query: Query) { - this.query = query; - } + /** + * The search query. + */ + public readonly query: Query; + /** + * The indexes of the entities that have already been matched. + */ + public readonly matchedIndexes: Set = new Set(); + /** + * The matched entities. + */ + public readonly matches: EntityMatch[] = []; + /** + * The meta data for each searcher type. + */ + public readonly meta: Meta[] = []; + /** + * Creates a new instance of the SearchState class. + * @param query The search query. + */ + public constructor(query: Query) { + this.query = query; + } } diff --git a/src/entity-searchers/sorting-entity-searcher.ts b/src/entity-searchers/sorting-entity-searcher.ts index 0341294..80c1a5a 100644 --- a/src/entity-searchers/sorting-entity-searcher.ts +++ b/src/entity-searchers/sorting-entity-searcher.ts @@ -12,120 +12,119 @@ import { SortOrder } from '../sort-order.js'; * @typeParam TId The type of the entity ids. */ export class SortingEntitySearcher implements EntitySearcher { - - /** - * The collator for string comparisons. - */ - private readonly collator = new Intl.Collator(undefined, { numeric: true }); - - /** - * Creates a new instance of the SortingEntitySearcher class. - * @typeParam TEntity The type of the entities. - * @typeParam TId The type of the entity ids. - * @param sortOrder The sort order to use for the entity matches. - * @param entitySearcher The entity searcher to wrap and sort results from. - */ - public constructor( - private readonly sortOrder: SortOrder, - private readonly entitySearcher: EntitySearcher) { } - - /** - * {@inheritDoc EntitySearcher.indexEntities} - */ - public indexEntities( - entities: TEntity[], - getId: (entity: TEntity) => TId, - getTerms: (entity: TEntity) => string[] - ): Meta { - return this.entitySearcher.indexEntities(entities, getId, getTerms); - } - - /** - * {@inheritDoc EntitySearcher.getMatches} - */ - public getMatches(query: Query): EntityResult { - const result = this.entitySearcher.getMatches(query); - - switch (this.sortOrder) { - case SortOrder.QualityAndIndex: - // The entity matches are already sorted by quality and index. - return result; - case SortOrder.QualityAndMatchedString: - this.sortMatchesByQualityAndMatchedString(result.matches); - return result; - default: - throw new Error(`Unsupported sort order: ${this.sortOrder}`); - } - } - - /** - * Sorts the entity matches in place by quality and matched string. - * @param matches The entity matches to sort. - */ - private sortMatchesByQualityAndMatchedString(matches: EntityMatch[]): void { - - matches.sort((m1, m2) => { - return ( - m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : this.collator.compare(m1.matchedString, m2.matchedString) - ); - }); - } - - /** - * {@inheritDoc EntitySearcher.tryGetEntity} - */ - public tryGetEntity(id: TId): TEntity | null { - return this.entitySearcher.tryGetEntity(id); - } - - /** - * {@inheritDoc EntitySearcher.getEntities} - */ - public getEntities(): TEntity[] { - return this.entitySearcher.getEntities(); - } - - /** - * {@inheritDoc EntitySearcher.tryGetTerms} - */ - public tryGetTerms(id: TId): string[] | null { - return this.entitySearcher.tryGetTerms(id); - } - - /** - * {@inheritDoc EntitySearcher.getTerms} - */ - public getTerms(): string[] { - return this.entitySearcher.getTerms(); - } - - /** - * {@inheritDoc EntitySearcher.save} - */ - public save(memento: Memento): void { - this.entitySearcher.save(memento); - } - - /** - * {@inheritDoc EntitySearcher.load} - */ - public load(memento: Memento): void { - this.entitySearcher.load(memento); - } - - /** - * {@inheritDoc EntitySearcher.removeEntity} - */ - removeEntity(id: TId): boolean { - return this.entitySearcher.removeEntity(id); - } - - /** - * {@inheritDoc EntitySearcher.replaceEntity} - */ - replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { - return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + /** + * The collator for string comparisons. + */ + private readonly collator = new Intl.Collator(undefined, { numeric: true }); + + /** + * Creates a new instance of the SortingEntitySearcher class. + * @typeParam TEntity The type of the entities. + * @typeParam TId The type of the entity ids. + * @param sortOrder The sort order to use for the entity matches. + * @param entitySearcher The entity searcher to wrap and sort results from. + */ + public constructor( + private readonly sortOrder: SortOrder, + private readonly entitySearcher: EntitySearcher + ) {} + + /** + * {@inheritDoc EntitySearcher.indexEntities} + */ + public indexEntities( + entities: TEntity[], + getId: (entity: TEntity) => TId, + getTerms: (entity: TEntity) => string[] + ): Meta { + return this.entitySearcher.indexEntities(entities, getId, getTerms); + } + + /** + * {@inheritDoc EntitySearcher.getMatches} + */ + public getMatches(query: Query): EntityResult { + const result = this.entitySearcher.getMatches(query); + + switch (this.sortOrder) { + case SortOrder.QualityAndIndex: + // The entity matches are already sorted by quality and index. + return result; + case SortOrder.QualityAndMatchedString: + this.sortMatchesByQualityAndMatchedString(result.matches); + return result; + default: + throw new Error(`Unsupported sort order: ${this.sortOrder}`); } + } + + /** + * Sorts the entity matches in place by quality and matched string. + * @param matches The entity matches to sort. + */ + private sortMatchesByQualityAndMatchedString(matches: EntityMatch[]): void { + matches.sort((m1, m2) => { + return ( + m1.quality > m2.quality ? -1 + : m1.quality < m2.quality ? 1 + : this.collator.compare(m1.matchedString, m2.matchedString) + ); + }); + } + + /** + * {@inheritDoc EntitySearcher.tryGetEntity} + */ + public tryGetEntity(id: TId): TEntity | null { + return this.entitySearcher.tryGetEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.getEntities} + */ + public getEntities(): TEntity[] { + return this.entitySearcher.getEntities(); + } + + /** + * {@inheritDoc EntitySearcher.tryGetTerms} + */ + public tryGetTerms(id: TId): string[] | null { + return this.entitySearcher.tryGetTerms(id); + } + + /** + * {@inheritDoc EntitySearcher.getTerms} + */ + public getTerms(): string[] { + return this.entitySearcher.getTerms(); + } + + /** + * {@inheritDoc EntitySearcher.save} + */ + public save(memento: Memento): void { + this.entitySearcher.save(memento); + } + + /** + * {@inheritDoc EntitySearcher.load} + */ + public load(memento: Memento): void { + this.entitySearcher.load(memento); + } + + /** + * {@inheritDoc EntitySearcher.removeEntity} + */ + removeEntity(id: TId): boolean { + return this.entitySearcher.removeEntity(id); + } + + /** + * {@inheritDoc EntitySearcher.replaceEntity} + */ + replaceEntity(id: TId, newEntity: TEntity, newEntityId: TId): boolean { + return this.entitySearcher.replaceEntity(id, newEntity, newEntityId); + } } diff --git a/src/fuzzy-searchers/ngram-normalizer.ts b/src/fuzzy-searchers/ngram-normalizer.ts index 5a7738d..6b43a47 100644 --- a/src/fuzzy-searchers/ngram-normalizer.ts +++ b/src/fuzzy-searchers/ngram-normalizer.ts @@ -15,7 +15,7 @@ export class NgramNormalizer implements Normalizer { public readonly paddingLeft: string, public readonly paddingRight: string, public readonly paddingMiddle: string - ) { } + ) {} /** * {@inheritDoc Normalizer.normalize} diff --git a/src/interfaces/query.ts b/src/interfaces/query.ts index 008028d..381ed26 100644 --- a/src/interfaces/query.ts +++ b/src/interfaces/query.ts @@ -1,4 +1,4 @@ -import { FuzzySearcher, PrefixSearcher, SearcherSpec, SubstringSearcher, } from './searcher-spec.js'; +import { FuzzySearcher, PrefixSearcher, SearcherSpec, SubstringSearcher } from './searcher-spec.js'; /** * Holds the query string and query parameters. @@ -28,7 +28,7 @@ export class Query { public constructor( string: string, topN: number = 10, - searchers = [new FuzzySearcher(0.3), new SubstringSearcher(0), new PrefixSearcher(0)], + searchers = [new FuzzySearcher(0.3), new SubstringSearcher(0), new PrefixSearcher(0)] ) { this.string = string; this.topN = Math.max(0, topN); diff --git a/src/interfaces/searcher-spec.ts b/src/interfaces/searcher-spec.ts index 60a56ab..df02bc5 100644 --- a/src/interfaces/searcher-spec.ts +++ b/src/interfaces/searcher-spec.ts @@ -1,75 +1,72 @@ import { SearcherType } from '../interfaces/searcher-type.js'; export class SearcherSpec { - /** - * The searcher type. - */ - public readonly type: SearcherType; + /** + * The searcher type. + */ + public readonly type: SearcherType; - /** - * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the - * number of matches. Decreasing this value might retrieve irrelevant matches. The value must be greater than or - * equal to 0. - */ - public readonly minQuality: number; + /** + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * number of matches. Decreasing this value might retrieve irrelevant matches. The value must be greater than or + * equal to 0. + */ + public readonly minQuality: number; - /** - * Creates a new instance of the SearcherSpec class. - * @param type The searcher type. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be - * greater than or equal to 0. Values below 0 will be clamped to 0. - */ - public constructor(type: SearcherType, minQuality: number) { - this.type = type; - this.minQuality = Math.max(0, minQuality); - } + /** + * Creates a new instance of the SearcherSpec class. + * @param type The searcher type. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. Values below 0 will be clamped to 0. + */ + public constructor(type: SearcherType, minQuality: number) { + this.type = type; + this.minQuality = Math.max(0, minQuality); + } } /** * Specification for the fuzzy searcher. */ export class FuzzySearcher extends SearcherSpec { - - /** - * Creates a new instance of the FuzzySearcher class. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be - * greater than or equal to 0. - */ - public constructor(minQuality: number) { - super(SearcherType.Fuzzy, minQuality); - } + /** + * Creates a new instance of the FuzzySearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Fuzzy, minQuality); + } } /** * Specification for the substring searcher. */ export class SubstringSearcher extends SearcherSpec { - - /** - * Creates a new instance of the SubstringSearcher class. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be - * greater than or equal to 0. - */ - public constructor(minQuality: number) { - super(SearcherType.Substring, minQuality); - } + /** + * Creates a new instance of the SubstringSearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Substring, minQuality); + } } /** * Specification for the prefix searcher. */ export class PrefixSearcher extends SearcherSpec { - - /** - * Creates a new instance of the PrefixSearcher class. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be - * greater than or equal to 0. - */ - public constructor(minQuality: number) { - super(SearcherType.Prefix, minQuality); - } -} \ No newline at end of file + /** + * Creates a new instance of the PrefixSearcher class. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * greater than or equal to 0. + */ + public constructor(minQuality: number) { + super(SearcherType.Prefix, minQuality); + } +} diff --git a/src/interfaces/searcher-type.ts b/src/interfaces/searcher-type.ts index 3c3a515..3b899d6 100644 --- a/src/interfaces/searcher-type.ts +++ b/src/interfaces/searcher-type.ts @@ -2,19 +2,18 @@ * Searcher types. */ export enum SearcherType { + /** + * Fuzzy searcher type. + */ + Fuzzy = 'fuzzy', - /** - * Fuzzy searcher type. - */ - Fuzzy = "fuzzy", + /** + * Substring searcher type. + */ + Substring = 'substring', - /** - * Substring searcher type. - */ - Substring = "substring", - - /** - * Prefix searcher type. - */ - Prefix = "prefix" -} \ No newline at end of file + /** + * Prefix searcher type. + */ + Prefix = 'prefix' +} diff --git a/src/interfaces/string-search-query.ts b/src/interfaces/string-search-query.ts index 41e8835..a7c20d7 100644 --- a/src/interfaces/string-search-query.ts +++ b/src/interfaces/string-search-query.ts @@ -4,33 +4,33 @@ import { SearcherType } from '../interfaces/searcher-type.js'; * Query for the string searchers. */ export class StringSearchQuery { - /** - * The query string. - */ - public readonly string: string; + /** + * The query string. + */ + public readonly string: string; - /** - * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the - * number of matches. Decreasing this value might retrieve irrelevant matches. - */ - public readonly minQuality: number; + /** + * The minimum quality of matches to return. Increasing this value will increase the performance but reduce the + * number of matches. Decreasing this value might retrieve irrelevant matches. + */ + public readonly minQuality: number; - /** - * The type of searcher to use. - */ - public readonly searcherType?: SearcherType; + /** + * The type of searcher to use. + */ + public readonly searcherType?: SearcherType; - /** - * Creates a new instance of the Query class. - * @param string The query string. - * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance - * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be - * between 0 and 1; lower or larger values will be clamped. - * @param searcherType The type of searcher to use. - */ - public constructor(string: string, minQuality: number = 0, searcherType?: SearcherType) { - this.string = string; - this.minQuality = Math.max(0, Math.min(1, minQuality)); - this.searcherType = searcherType; - } -} \ No newline at end of file + /** + * Creates a new instance of the Query class. + * @param string The query string. + * @param minQuality The minimum quality of matches to return. Increasing this value will increase the performance + * but reduce the number of matches. Decreasing this value might retrieve irrelevant matches. The value must be + * between 0 and 1; lower or larger values will be clamped. + * @param searcherType The type of searcher to use. + */ + public constructor(string: string, minQuality: number = 0, searcherType?: SearcherType) { + this.string = string; + this.minQuality = Math.max(0, Math.min(1, minQuality)); + this.searcherType = searcherType; + } +} diff --git a/src/performance-test/main.ts b/src/performance-test/main.ts index 6bb2b62..2f8b844 100644 --- a/src/performance-test/main.ts +++ b/src/performance-test/main.ts @@ -1,6 +1,6 @@ /** * Performance test. Run this file and check the git diff of the output files to see how the performance changed. -*/ + */ import { readFileSync, writeFileSync } from 'fs'; import { Config } from '../config.js'; @@ -42,8 +42,12 @@ writeFileSync(`${outputPath}/_indexing-meta.txt`, metaJson, { encoding: 'utf8' } console.log(metaJson); const performanceTest: PerformanceTest = new PerformanceTest(searcher); -const testRunParameters: TestRunParameters = - new TestRunParameters(seed, numberOfQueries, topN, new Query('').searchers); +const testRunParameters: TestRunParameters = new TestRunParameters( + seed, + numberOfQueries, + topN, + new Query('').searchers +); console.log('Running performance test...'); const report: Report = performanceTest.run(testRunParameters); diff --git a/src/performance/performance-test.ts b/src/performance/performance-test.ts index 3498757..eade785 100644 --- a/src/performance/performance-test.ts +++ b/src/performance/performance-test.ts @@ -74,8 +74,7 @@ export class PerformanceTest { if (this.RollPercentage(0.5)) { start = 0; this.queryCounts.prefixQueries++; - } - else { + } else { start = this.getRandomInteger(0, term.length); this.queryCounts.substringQueries++; } @@ -89,10 +88,12 @@ export class PerformanceTest { const errorPosition = this.getRandomInteger(0, substring.length - 1); this.queryCounts.transpositionErrors++; - return substring.substring(0, errorPosition) - + substring[errorPosition + 1] - + substring[errorPosition] - + substring.substring(errorPosition + 2); + return ( + substring.substring(0, errorPosition) + + substring[errorPosition + 1] + + substring[errorPosition] + + substring.substring(errorPosition + 2) + ); } private RollPercentage(chance: number): boolean { diff --git a/src/performance/query-counts.ts b/src/performance/query-counts.ts index 68014ce..aa7be18 100644 --- a/src/performance/query-counts.ts +++ b/src/performance/query-counts.ts @@ -2,24 +2,23 @@ * Counts of the generated queries. */ export class QueryCounts { + /** + * The number of generated prefix queries. + */ + public prefixQueries: number = 0; - /** - * The number of generated prefix queries. - */ - public prefixQueries: number = 0; + /** + * The number of generated substring queries. + */ + public substringQueries: number = 0; - /** - * The number of generated substring queries. - */ - public substringQueries: number = 0; + /** + * The number of queries that included a transposition error. + */ + public transpositionErrors: number = 0; - /** - * The number of queries that included a transposition error. - */ - public transpositionErrors: number = 0; - - /** - * Creates a new instance of the QueryCounts class. - */ - public constructor() { } -} \ No newline at end of file + /** + * Creates a new instance of the QueryCounts class. + */ + public constructor() {} +} diff --git a/src/performance/report.ts b/src/performance/report.ts index 5ec5d1c..1b7257a 100644 --- a/src/performance/report.ts +++ b/src/performance/report.ts @@ -26,7 +26,7 @@ export class Report { public readonly fastest: TimedQuery, public readonly slowest: TimedQuery, public readonly longest: TimedQuery - ) { } + ) {} /** * Creates a report from the timed queries. @@ -38,7 +38,8 @@ export class Report { public static Create( testRunParameters: TestRunParameters, queryCounts: QueryCounts, - measurements: TimedQuery[]): Report { + measurements: TimedQuery[] + ): Report { if (measurements.length === 0) { return new Report( testRunParameters, @@ -61,7 +62,15 @@ export class Report { const slowest: TimedQuery = measurements.reduce((prev, cur) => (prev.duration > cur.duration ? prev : cur)); const longest: TimedQuery = measurements.reduce((prev, cur) => (prev.query.length > cur.query.length ? prev : cur)); return new Report( - testRunParameters, queryCounts, totalDuration, averageDuration, standardDeviation, fastest, slowest, longest); + testRunParameters, + queryCounts, + totalDuration, + averageDuration, + standardDeviation, + fastest, + slowest, + longest + ); } /** diff --git a/src/performance/test-run-parameters.ts b/src/performance/test-run-parameters.ts index 23e800a..591407c 100644 --- a/src/performance/test-run-parameters.ts +++ b/src/performance/test-run-parameters.ts @@ -1,4 +1,4 @@ -import { SearcherSpec } from "../interfaces/searcher-spec.js"; +import { SearcherSpec } from '../interfaces/searcher-spec.js'; /** * Parameters of a performance test run. @@ -16,5 +16,5 @@ export class TestRunParameters { public readonly numberOfQueries: number, public readonly topN: number, public readonly searchers: readonly SearcherSpec[] - ) { } + ) {} } diff --git a/src/sort-order.ts b/src/sort-order.ts index 33165c7..c70122a 100644 --- a/src/sort-order.ts +++ b/src/sort-order.ts @@ -2,13 +2,13 @@ * Specifies the sort order for the entity matches. */ export enum SortOrder { - /** - * Sort by quality first, then by index. - */ - QualityAndIndex = 'qualityAndIndex', + /** + * Sort by quality first, then by index. + */ + QualityAndIndex = 'qualityAndIndex', - /** - * Sort by quality first, then by matched string. - */ - QualityAndMatchedString = 'qualityAndMatchedString', -} \ No newline at end of file + /** + * Sort by quality first, then by matched string. + */ + QualityAndMatchedString = 'qualityAndMatchedString' +} diff --git a/src/string-searchers/distinct-searcher.ts b/src/string-searchers/distinct-searcher.ts index 93f27d1..e0e62c2 100644 --- a/src/string-searchers/distinct-searcher.ts +++ b/src/string-searchers/distinct-searcher.ts @@ -39,8 +39,8 @@ export class DistinctSearcher implements StringSearcher { const termsSorted = terms.map((term, index) => ({ term, index })); termsSorted.sort((t1, t2) => t1.term < t2.term ? -1 - : t1.term > t2.term ? 1 - : 0 + : t1.term > t2.term ? 1 + : 0 ); this.sortMapping = new Int32Array(termsSorted.length); for (let i = 0, l = termsSorted.length; i < l; i++) { diff --git a/src/string-searchers/inequality-penalizing-searcher.test.ts b/src/string-searchers/inequality-penalizing-searcher.test.ts index 87a96a9..634114a 100644 --- a/src/string-searchers/inequality-penalizing-searcher.test.ts +++ b/src/string-searchers/inequality-penalizing-searcher.test.ts @@ -7,10 +7,10 @@ import { StringSearchQuery } from '../interfaces/string-search-query.js'; const stringSearcher = { index: (_terms: string[]): Meta => new Meta(), - getMatches: (query: StringSearchQuery): Result => new Result( - [new Match(0, 1.0), new Match(1, 0.6)], query, new Meta()), - save: (_memento: Memento): void => { }, - load: (_memento: Memento): void => { } + getMatches: (query: StringSearchQuery): Result => + new Result([new Match(0, 1.0), new Match(1, 0.6)], query, new Meta()), + save: (_memento: Memento): void => {}, + load: (_memento: Memento): void => {} }; const inequalityPenalizingSearcher = new InequalityPenalizingSearcher(stringSearcher, 0.05); diff --git a/src/string-searchers/normalizing-searcher.ts b/src/string-searchers/normalizing-searcher.ts index 068d539..042305d 100644 --- a/src/string-searchers/normalizing-searcher.ts +++ b/src/string-searchers/normalizing-searcher.ts @@ -19,7 +19,7 @@ export class NormalizingSearcher implements StringSearcher { private readonly stringSearcher: StringSearcher, private readonly normalizer: Normalizer, private readonly normalizationDurationMetaKey: string = 'normalizationDuration' - ) { } + ) {} /** * {@inheritDoc StringSearcher.index} @@ -41,7 +41,10 @@ export class NormalizingSearcher implements StringSearcher { */ public getMatches(query: StringSearchQuery): Result { const normalizedQuery = new StringSearchQuery( - this.normalizer.normalize(query.string), query.minQuality, query.searcherType); + this.normalizer.normalize(query.string), + query.minQuality, + query.searcherType + ); const result = this.stringSearcher.getMatches(normalizedQuery); return new Result(result.matches, query, result.meta); } diff --git a/src/string-searchers/result.ts b/src/string-searchers/result.ts index 1eb4880..f3882bb 100644 --- a/src/string-searchers/result.ts +++ b/src/string-searchers/result.ts @@ -16,5 +16,5 @@ export class Result { public readonly matches: Match[], public readonly query: StringSearchQuery, public readonly meta: Meta - ) { } + ) {} } diff --git a/src/string-searchers/searcher-switch.ts b/src/string-searchers/searcher-switch.ts index 493416a..c8869f9 100644 --- a/src/string-searchers/searcher-switch.ts +++ b/src/string-searchers/searcher-switch.ts @@ -8,101 +8,98 @@ import { StringSearcher } from '../interfaces/string-searcher.js'; /** * A searcher switch that routes to the prefix searcher, the substring searcher, or the fuzzy searcher. The prefix - * searcher is a simple wrapper around the suffix array searcher. It will be always up to date with the substring + * searcher is a simple wrapper around the suffix array searcher. It will be always up to date with the substring * searcher if present. */ export class SearcherSwitch implements StringSearcher { + /** + * Creates a new instance of the SearcherSwitch class. + * @param prefixSearcher The prefix searcher. + * @param substringSearcher The substring searcher. + * @param fuzzySearcher The fuzzy searcher. + */ + public constructor( + private readonly prefixSearcher: StringSearcher | null, + private readonly substringSearcher: StringSearcher | null, + private readonly fuzzySearcher: StringSearcher | null + ) {} - /** - * Creates a new instance of the SearcherSwitch class. - * @param prefixSearcher The prefix searcher. - * @param substringSearcher The substring searcher. - * @param fuzzySearcher The fuzzy searcher. - */ - public constructor( - private readonly prefixSearcher: StringSearcher | null, - private readonly substringSearcher: StringSearcher | null, - private readonly fuzzySearcher: StringSearcher | null, - ) { + /** + * {@inheritDoc StringSearcher.index} + */ + index(terms: string[]): Meta { + const meta = []; + if (this.prefixSearcher && !this.substringSearcher) { + const prefixSearcherMeta = this.prefixSearcher.index(terms); + meta.push(prefixSearcherMeta); } - - /** - * {@inheritDoc StringSearcher.index} - */ - index(terms: string[]): Meta { - const meta = []; - if (this.prefixSearcher && !this.substringSearcher) { - const prefixSearcherMeta = this.prefixSearcher.index(terms); - meta.push(prefixSearcherMeta); - } - if (this.substringSearcher) { - const substringSearcherMeta = this.substringSearcher.index(terms); - meta.push(substringSearcherMeta); - } - if (this.fuzzySearcher) { - const fuzzySearcherMeta = this.fuzzySearcher.index(terms); - meta.push(fuzzySearcherMeta); - } - return MetaMerger.mergeMeta(meta); + if (this.substringSearcher) { + const substringSearcherMeta = this.substringSearcher.index(terms); + meta.push(substringSearcherMeta); } + if (this.fuzzySearcher) { + const fuzzySearcherMeta = this.fuzzySearcher.index(terms); + meta.push(fuzzySearcherMeta); + } + return MetaMerger.mergeMeta(meta); + } - /** - * {@inheritDoc StringSearcher.getMatches} - */ - getMatches(query: StringSearchQuery): Result { - - if (!query.searcherType) { - throw new Error('SearcherSwitch requires a searcher type.'); - } - - switch (query.searcherType) { - case SearcherType.Prefix: - if (!this.prefixSearcher) { - throw new Error('No prefix searcher has been indexed.'); - } - return this.prefixSearcher.getMatches(query); - case SearcherType.Substring: - if (!this.substringSearcher) { - throw new Error('No substring searcher has been indexed.'); - } - return this.substringSearcher.getMatches(query); - case SearcherType.Fuzzy: - if (!this.fuzzySearcher) { - throw new Error('No fuzzy searcher has been indexed.'); - } - return this.fuzzySearcher.getMatches(query); - default: - throw new Error(`Unknown searcher type: ${query.searcherType}`); - } + /** + * {@inheritDoc StringSearcher.getMatches} + */ + getMatches(query: StringSearchQuery): Result { + if (!query.searcherType) { + throw new Error('SearcherSwitch requires a searcher type.'); } - /** - * {@inheritDoc StringSearcher.save} - */ - save(memento: Memento): void { - if (this.prefixSearcher && !this.substringSearcher) { - this.prefixSearcher.save(memento); + switch (query.searcherType) { + case SearcherType.Prefix: + if (!this.prefixSearcher) { + throw new Error('No prefix searcher has been indexed.'); } - if (this.substringSearcher) { - this.substringSearcher.save(memento); + return this.prefixSearcher.getMatches(query); + case SearcherType.Substring: + if (!this.substringSearcher) { + throw new Error('No substring searcher has been indexed.'); } - if (this.fuzzySearcher) { - this.fuzzySearcher.save(memento); + return this.substringSearcher.getMatches(query); + case SearcherType.Fuzzy: + if (!this.fuzzySearcher) { + throw new Error('No fuzzy searcher has been indexed.'); } + return this.fuzzySearcher.getMatches(query); + default: + throw new Error(`Unknown searcher type: ${query.searcherType}`); } + } - /** - * {@inheritDoc StringSearcher.load} - */ - load(memento: Memento): void { - if (this.prefixSearcher && !this.substringSearcher) { - this.prefixSearcher.load(memento); - } - if (this.substringSearcher) { - this.substringSearcher.load(memento); - } - if (this.fuzzySearcher) { - this.fuzzySearcher.load(memento); - } + /** + * {@inheritDoc StringSearcher.save} + */ + save(memento: Memento): void { + if (this.prefixSearcher && !this.substringSearcher) { + this.prefixSearcher.save(memento); + } + if (this.substringSearcher) { + this.substringSearcher.save(memento); + } + if (this.fuzzySearcher) { + this.fuzzySearcher.save(memento); + } + } + + /** + * {@inheritDoc StringSearcher.load} + */ + load(memento: Memento): void { + if (this.prefixSearcher && !this.substringSearcher) { + this.prefixSearcher.load(memento); + } + if (this.substringSearcher) { + this.substringSearcher.load(memento); + } + if (this.fuzzySearcher) { + this.fuzzySearcher.load(memento); } -} \ No newline at end of file + } +} diff --git a/src/string-searchers/sorting-searcher.test.ts b/src/string-searchers/sorting-searcher.test.ts index 8e5c774..2d0e648 100644 --- a/src/string-searchers/sorting-searcher.test.ts +++ b/src/string-searchers/sorting-searcher.test.ts @@ -13,8 +13,8 @@ const stringSearcher = { query, new Meta() ), - save: (_memento: Memento): void => { }, - load: (_memento: Memento): void => { } + save: (_memento: Memento): void => {}, + load: (_memento: Memento): void => {} }; const sortingSearcher = new SortingSearcher(stringSearcher); diff --git a/src/string-searchers/sorting-searcher.ts b/src/string-searchers/sorting-searcher.ts index d45eddf..f15d9a2 100644 --- a/src/string-searchers/sorting-searcher.ts +++ b/src/string-searchers/sorting-searcher.ts @@ -14,7 +14,7 @@ export class SortingSearcher implements StringSearcher { * Creates a new instance of the SortingSearcher class. * @param stringSearcher The string searcher to use. */ - public constructor(private readonly stringSearcher: StringSearcher) { } + public constructor(private readonly stringSearcher: StringSearcher) {} /** * {@inheritDoc StringSearcher.index} @@ -43,10 +43,10 @@ export class SortingSearcher implements StringSearcher { private compareMatchesByQualityAndIndex(m1: Match, m2: Match): number { return ( m1.quality > m2.quality ? -1 - : m1.quality < m2.quality ? 1 - : m1.index < m2.index ? -1 - : m1.index > m2.index ? 1 - : 0 + : m1.quality < m2.quality ? 1 + : m1.index < m2.index ? -1 + : m1.index > m2.index ? 1 + : 0 ); } diff --git a/src/suffix-array-searchers/prefix-searcher.test.ts b/src/suffix-array-searchers/prefix-searcher.test.ts index 7d142af..7e794f5 100644 --- a/src/suffix-array-searchers/prefix-searcher.test.ts +++ b/src/suffix-array-searchers/prefix-searcher.test.ts @@ -8,40 +8,40 @@ suffixArraySearcher.index(['Alice', 'Bob', 'Carlos', 'Carol', 'Charlie']); const prefixSearcher: PrefixSearcher = new PrefixSearcher(suffixArraySearcher); function getMatches(queryString: string): Match[] { - const query = new StringSearchQuery(queryString, 0); - const matches = prefixSearcher.getMatches(query).matches; - matches.sort((m1, m2) => m1.index - m2.index); - return matches; + const query = new StringSearchQuery(queryString, 0); + const matches = prefixSearcher.getMatches(query).matches; + matches.sort((m1, m2) => m1.index - m2.index); + return matches; } test('can find exact match test 1', () => { - expect(getMatches('Alice')).toEqual([new Match(0, 1)]); + expect(getMatches('Alice')).toEqual([new Match(0, 1)]); }); test('can find prefix match test 1', () => { - expect(getMatches('B')).toEqual([new Match(1, 1 / 3)]); + expect(getMatches('B')).toEqual([new Match(1, 1 / 3)]); }); test('can find prefix match test 2', () => { - expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); + expect(getMatches('Bo')).toEqual([new Match(1, 2 / 3)]); }); test('can find prefix match test 2', () => { - expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); + expect(getMatches('Car')).toEqual([new Match(2, 3 / 6), new Match(3, 3 / 5)]); }); test('can not find substring matches', () => { - expect(getMatches('lice')).toEqual([]); -}) + expect(getMatches('lice')).toEqual([]); +}); test('empty query returns no matches', () => { - expect(getMatches('')).toEqual([]); + expect(getMatches('')).toEqual([]); }); test('null query returns no matches', () => { - expect(getMatches(null!)).toEqual([]); + expect(getMatches(null!)).toEqual([]); }); test('undefined query returns no matches', () => { - expect(getMatches(undefined!)).toEqual([]); -}); \ No newline at end of file + expect(getMatches(undefined!)).toEqual([]); +}); diff --git a/src/suffix-array-searchers/prefix-searcher.ts b/src/suffix-array-searchers/prefix-searcher.ts index 05df96d..c332058 100644 --- a/src/suffix-array-searchers/prefix-searcher.ts +++ b/src/suffix-array-searchers/prefix-searcher.ts @@ -9,55 +9,51 @@ import { SuffixArraySearcher } from './suffix-array-searcher.js'; * A prefix searcher that is a simple wrapper around the suffix array searcher. */ export class PrefixSearcher implements StringSearcher { + /** + * Creates a new instance of the PrefixSearcher class. + * @param suffixArraySearcher The suffix array searcher. + */ + public constructor(private readonly suffixArraySearcher: SuffixArraySearcher) {} - /** - * Creates a new instance of the PrefixSearcher class. - * @param suffixArraySearcher The suffix array searcher. - */ - public constructor( - private readonly suffixArraySearcher: SuffixArraySearcher, - ) { - } - - /** - * {@inheritDoc StringSearcher.index} - */ - index(terms: string[]): Meta { - return this.suffixArraySearcher.index(terms); - } + /** + * {@inheritDoc StringSearcher.index} + */ + index(terms: string[]): Meta { + return this.suffixArraySearcher.index(terms); + } - /** - * {@inheritDoc StringSearcher.getMatches} - */ - getMatches(query: StringSearchQuery): Result { - if (!query.string) { - return new Result([], query, new Meta()); - } - const modifiedQueryString = this.modifyQueryString(query.string); - const modifiedQuery = new StringSearchQuery(modifiedQueryString, query.minQuality, query.searcherType); - return this.suffixArraySearcher.getMatches(modifiedQuery, query.string.length); + /** + * {@inheritDoc StringSearcher.getMatches} + */ + getMatches(query: StringSearchQuery): Result { + if (!query.string) { + return new Result([], query, new Meta()); } + const modifiedQueryString = this.modifyQueryString(query.string); + const modifiedQuery = new StringSearchQuery(modifiedQueryString, query.minQuality, query.searcherType); + return this.suffixArraySearcher.getMatches(modifiedQuery, query.string.length); + } - /** - * Modifies the original query string by prepending the suffix array searcher's separator. - * @param original The original query string. - * @returns The modified query string. - */ - private modifyQueryString(original: string): string { - return `${this.suffixArraySearcher.separator}${original}`; - } + /** + * Modifies the original query string by prepending the suffix array searcher's separator. + * @param original The original query string. + * @returns The modified query string. + */ + private modifyQueryString(original: string): string { + return `${this.suffixArraySearcher.separator}${original}`; + } - /** - * {@inheritDoc StringSearcher.save} - */ - save(memento: Memento): void { - this.suffixArraySearcher.save(memento); - } + /** + * {@inheritDoc StringSearcher.save} + */ + save(memento: Memento): void { + this.suffixArraySearcher.save(memento); + } - /** - * {@inheritDoc StringSearcher.load} - */ - load(memento: Memento): void { - this.suffixArraySearcher.load(memento); - } -} \ No newline at end of file + /** + * {@inheritDoc StringSearcher.load} + */ + load(memento: Memento): void { + this.suffixArraySearcher.load(memento); + } +} diff --git a/src/suffix-array-searchers/substring-search-config.ts b/src/suffix-array-searchers/substring-search-config.ts index 67d4f4a..55097b2 100644 --- a/src/suffix-array-searchers/substring-search-config.ts +++ b/src/suffix-array-searchers/substring-search-config.ts @@ -6,9 +6,7 @@ export class SubstringSearchConfig { * Creates a new instance of the SubstringSearchConfig class. * @param suffixArraySeparator The suffix array separator character. */ - public constructor( - public suffixArraySeparator: string - ) { } + public constructor(public suffixArraySeparator: string) {} /** * Creates a default configuration with '$' as the suffix array separator. diff --git a/src/suffix-array-searchers/suffix-array-searcher.ts b/src/suffix-array-searchers/suffix-array-searcher.ts index 86bdba7..857f0ad 100644 --- a/src/suffix-array-searchers/suffix-array-searcher.ts +++ b/src/suffix-array-searchers/suffix-array-searcher.ts @@ -103,7 +103,7 @@ export class SuffixArraySearcher implements StringSearcher { /** * Gets the positions in the suffix array where the given substring matches. * @param substring The substring to search for. - * @returns The start and end positions of the substring in the suffix array. The start is inclusive, the end is + * @returns The start and end positions of the substring in the suffix array. The start is inclusive, the end is * exclusive. */ private GetPositionsInSuffixArray(substring: string): number[] { diff --git a/src/suffix-array-searchers/suffix-array.ts b/src/suffix-array-searchers/suffix-array.ts index 79a6b4a..b9f9c37 100644 --- a/src/suffix-array-searchers/suffix-array.ts +++ b/src/suffix-array-searchers/suffix-array.ts @@ -206,7 +206,7 @@ class SuffixRank { public constructor( public readonly head: number, public readonly rank: number - ) { } + ) {} } /** @@ -214,10 +214,12 @@ class SuffixRank { */ class Chain { /** - * Creates a new instance of the Chain class. + * Creates a new instance of the Chain class. * @param head The head index of the chain. * @param length The length of the chain. */ - public constructor(public head: number, public readonly length: number) { - } + public constructor( + public head: number, + public readonly length: number + ) {} } From c9f92077238f583dfa4a5c15cc1d6170d83ff416 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:37:04 +0200 Subject: [PATCH 078/105] export sort order --- src/suffix-array-searchers/index.ts | 1 + 1 file changed, 1 insertion(+) diff --git a/src/suffix-array-searchers/index.ts b/src/suffix-array-searchers/index.ts index fbb71bf..cdb7c1d 100644 --- a/src/suffix-array-searchers/index.ts +++ b/src/suffix-array-searchers/index.ts @@ -3,3 +3,4 @@ export { SuffixArray } from './suffix-array.js'; export { SuffixArraySearcher } from './suffix-array-searcher.js'; export { SubstringSearchConfig } from './substring-search-config.js'; export { PrefixSearcher as PrefixSearcherImpl } from './prefix-searcher.js'; +export { SortOrder } from '../sort-order.js'; From 6921295fb3eb9249b9719f7ec6dbf55e978833d8 Mon Sep 17 00:00:00 2001 From: Kevin Schaal Date: Sat, 25 Oct 2025 16:51:42 +0200 Subject: [PATCH 079/105] fix demo --- demo/fuzzy-demo.js | 27 +- demo/fuzzy-search-demo.html | 386 ++++++++++++++-------------- src/performance/performance-test.ts | 1 + 3 files changed, 221 insertions(+), 193 deletions(-) diff --git a/demo/fuzzy-demo.js b/demo/fuzzy-demo.js index 807752e..3093602 100644 --- a/demo/fuzzy-demo.js +++ b/demo/fuzzy-demo.js @@ -23,7 +23,13 @@ const dataPreview = document.getElementById('data-preview'); const performanceTestRandomSeedInput = document.querySelector('#performance-test input[name="random-seed"]'); const performanceTestNumberOfQueriesInput = document.querySelector('#performance-test input[name="number-of-queries"]'); const performanceTestMaxMatchesInput = document.querySelector('#performance-test input[name="max-matches"]'); -const performanceTestMinQualityInput = document.querySelector('#performance-test input[name="min-quality"]'); +const performanceTestMinQualityFuzzyInput = document.querySelector('#performance-test input[name="min-quality-fuzzy"]'); +const performanceTestMinQualitySubstringInput = document.querySelector( + '#performance-test input[name="min-quality-substring"]' +); +const performanceTestMinQualityPrefixInput = document.querySelector( + '#performance-test input[name="min-quality-prefix"]' +); document.getElementById('osm-data-card').addEventListener('click', downloadAndIndexOsmData); document .getElementById('person-data-card') @@ -52,7 +58,9 @@ initializeParameterInput(personRandomSeedInput, parseIntInput); initializeParameterInput(performanceTestRandomSeedInput, parseIntInput); initializeParameterInput(performanceTestNumberOfQueriesInput, parsePositiveIntInput); initializeParameterInput(performanceTestMaxMatchesInput, parsePositiveIntInput); -initializeParameterInput(performanceTestMinQualityInput, parsePositiveFloatInput); +initializeParameterInput(performanceTestMinQualityFuzzyInput, parsePositiveFloatInput); +initializeParameterInput(performanceTestMinQualitySubstringInput, parsePositiveFloatInput); +initializeParameterInput(performanceTestMinQualityPrefixInput, parsePositiveFloatInput); wireTableRows(); @@ -807,19 +815,28 @@ function runPerformanceTest() { const randomSeed = performanceTestRandomSeedInput.dataValue; const numberOfQueries = performanceTestNumberOfQueriesInput.dataValue; const maxMatches = performanceTestMaxMatchesInput.dataValue; - const minQuality = performanceTestMinQualityInput.dataValue; + const minQualityFuzzy = performanceTestMinQualityFuzzyInput.dataValue; + const minQualitySubstring = performanceTestMinQualitySubstringInput.dataValue; + const minQualityPrefix = performanceTestMinQualityPrefixInput.dataValue; if ( nullOrUndefined(randomSeed) || nullOrUndefined(numberOfQueries) || nullOrUndefined(maxMatches) || - nullOrUndefined(minQuality) + nullOrUndefined(minQualityFuzzy) || + nullOrUndefined(minQualitySubstring) || + nullOrUndefined(minQualityPrefix) ) { renderPerformanceTestResult(); return; } - const testRunParameters = new fuzzySearch.TestRunParameters(randomSeed, numberOfQueries, maxMatches, minQuality); + const searchers = [ + new fuzzySearch.FuzzySearcher(minQualityFuzzy), + new fuzzySearch.SubstringSearcher(minQualitySubstring), + new fuzzySearch.PrefixSearcher(minQualityPrefix) + ]; + const testRunParameters = new fuzzySearch.TestRunParameters(randomSeed, numberOfQueries, maxMatches, searchers); const performanceTest = new fuzzySearch.PerformanceTest(currentInstance.searcher); const report = performanceTest.run(testRunParameters); renderPerformanceTestResult(report); diff --git a/demo/fuzzy-search-demo.html b/demo/fuzzy-search-demo.html index 95dfff3..24a3a72 100644 --- a/demo/fuzzy-search-demo.html +++ b/demo/fuzzy-search-demo.html @@ -1,211 +1,221 @@ - - - - Fuzzy Search Demo - - - - -
-

Fuzzy Search Demo

- -

Select a dataset

- -
-
-

Open Street Map Data (~16 Mb)

-

- All names of cities, towns, villages and suburbs worldwide in their local language. More than one million - terms. See also openstreetmap.org/copyright. -

-
-
-

Person names (~5 Mb)

-

- First names and last names generated with faker-js for all available languages. Demonstrates how entities - can be found by different terms. -

-
+ + + + Fuzzy Search Demo + + + + +
+

Fuzzy Search Demo

+ +

Select a dataset

+ +
+
+

Open Street Map Data (~16 Mb)

+

+ All names of cities, towns, villages and suburbs worldwide in their local language. More than one million + terms. See also openstreetmap.org/copyright. +

- + + - - + - + - + -