Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 5 additions & 2 deletions src/typeagent/knowpro/add_messages.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@

from ..aitools.embeddings import IEmbeddingModel, NormalizedEmbedding
from ..storage.memory import semrefindex
from .common import normalize_term
from .interfaces import (
AddMessagesResult,
IKnowledgeExtractor,
Expand Down Expand Up @@ -244,13 +245,15 @@ def _collect_related_terms_for_fuzzy_index(
"""Collect canonical related-term texts for the fuzzy related-terms index.

These terms are derived from the same knowledge that feeds semantic refs.
We lowercase and deduplicate while preserving order to match index behavior.
We normalize each term with `normalize_term` (strip, NFC, collapse
whitespace, lowercase), drop empty results, and deduplicate while
preserving order to match index behavior.
"""
seen: set[str] = set()
related_terms: list[str] = []

def _add_term(term: str) -> None:
canonical = term.strip().lower()
canonical = normalize_term(term)
Comment thread
bmerkle marked this conversation as resolved.
if canonical and canonical not in seen:
seen.add(canonical)
related_terms.append(canonical)
Expand Down
14 changes: 14 additions & 0 deletions src/typeagent/knowpro/common.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,23 @@
# Copyright (c) Microsoft Corporation.
# Licensed under the MIT License.

import re
import unicodedata

from .interfaces import SearchTerm


def normalize_term(term: str) -> str:
"""Canonical form of an index term, shared by all storage backends.

Strips surrounding whitespace, applies NFC normalization, collapses runs
of whitespace to a single space, and lowercases. Every term index must
apply this on both write and lookup so backends agree on term identity.
"""
term = unicodedata.normalize("NFC", term.strip())
return re.sub(r"\s+", " ", term).lower()


def is_search_term_wildcard(search_term: SearchTerm) -> bool:
"""Check if a search term is a wildcard."""
return search_term.term.text == "*"
10 changes: 6 additions & 4 deletions src/typeagent/knowpro/conversation_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@
from ..aitools import model_adapters, utils
from ..aitools.embeddings import NormalizedEmbedding
from ..storage.memory import semrefindex
from .common import normalize_term
from .convsettings import ConversationSettings
from .interfaces import (
AddMessagesResult,
Expand Down Expand Up @@ -490,16 +491,17 @@ async def _update_related_terms_incremental(

fuzzy_index = self.secondary_indexes.term_to_related_terms_index.fuzzy_index
if fuzzy_index is not None and new_semrefs:
new_terms = set()
new_terms: set[str] = set()
for semref in new_semrefs:
knowledge = semref.knowledge
if isinstance(knowledge, ConcreteEntity):
new_terms.add(knowledge.name.lower())
new_terms.add(normalize_term(knowledge.name))
elif isinstance(knowledge, Topic):
new_terms.add(knowledge.text.lower())
new_terms.add(normalize_term(knowledge.text))
elif isinstance(knowledge, Action):
for verb in knowledge.verbs:
new_terms.add(verb.lower())
new_terms.add(normalize_term(verb))
new_terms.discard("") # Whitespace-only terms must not be embedded

if new_terms:
await fuzzy_index.add_terms(list(new_terms))
Expand Down
16 changes: 11 additions & 5 deletions src/typeagent/storage/memory/propindex.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
from typing import assert_never

from ...knowpro.collections import TextRangesInScope
from ...knowpro.common import normalize_term
from ...knowpro.interfaces import (
IConversation,
IPropertyToSemanticRefIndex,
Expand Down Expand Up @@ -240,13 +241,12 @@ async def add_property(
value: str,
semantic_ref_ordinal: SemanticRefOrdinal | ScoredSemanticRefOrdinal,
) -> None:
term_text = make_property_term_text(property_name, value)
term_text = self._make_term_text(property_name, value)
if isinstance(semantic_ref_ordinal, int):
semantic_ref_ordinal = ScoredSemanticRefOrdinal(
semantic_ref_ordinal,
1.0,
)
term_text = self._prepare_term_text(term_text)
if term_text in self._map:
self._map[term_text].append(semantic_ref_ordinal)
else:
Expand All @@ -269,12 +269,12 @@ async def lookup_property(
property_name: str,
value: str,
) -> list[ScoredSemanticRefOrdinal] | None:
term_text = make_property_term_text(property_name, value)
return self._map.get(self._prepare_term_text(term_text))
return self._map.get(self._make_term_text(property_name, value))
Comment thread
bmerkle marked this conversation as resolved.

async def remove_property(self, prop_name: str, semref_id: int) -> None:
"""Remove all properties for a specific property name and semantic ref."""
# Find and remove entries matching both property name and semref_id
prop_name = normalize_term(prop_name)
keys_to_remove = []
for term_text, scored_refs in self._map.items():
prop_name_from_term, _ = split_property_term_text(term_text)
Expand Down Expand Up @@ -315,7 +315,13 @@ async def remove_all_for_semref(self, semref_id: int) -> None:

def _prepare_term_text(self, term_text: str) -> str:
"""Do any pre-processing of the term."""
return term_text.lower()
return normalize_term(term_text)

def _make_term_text(self, property_name: str, value: str) -> str:
"""Build the normalized key; must match the SQLite property index."""
return make_property_term_text(
normalize_term(property_name), normalize_term(value)
)


async def lookup_property_in_property_index(
Expand Down
5 changes: 3 additions & 2 deletions src/typeagent/storage/memory/semrefindex.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
from typechat import Failure

from ...knowpro import convknowledge, secindex
from ...knowpro.common import normalize_term
from ...knowpro.convsettings import ConversationSettings, SemanticRefIndexSettings
from ...knowpro.interfaces import ( # Interfaces.; Other imports.
IConversation,
Expand Down Expand Up @@ -735,10 +736,10 @@ async def deserialize(self, data: TermToSemanticRefIndexData) -> None:
scored_refs = [
ScoredSemanticRefOrdinal.deserialize(s) for s in scored_refs_data
]
self._map[term] = scored_refs
self._map.setdefault(term, []).extend(scored_refs)

def _prepare_term(self, term: str) -> str:
return term.lower()
return normalize_term(term)


# ...
Expand Down
29 changes: 9 additions & 20 deletions src/typeagent/storage/sqlite/propindex.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,12 +6,12 @@
from collections.abc import Sequence
import sqlite3

from ...knowpro.common import normalize_term
from ...knowpro.interfaces import (
IPropertyToSemanticRefIndex,
ScoredSemanticRefOrdinal,
SemanticRefOrdinal,
)
from ...storage.memory import propindex


class SqlitePropertyIndex(IPropertyToSemanticRefIndex):
Expand Down Expand Up @@ -49,12 +49,8 @@ async def add_property(
score = 1.0

# Normalize property name and value (to match in-memory implementation)
term_text = propindex.make_property_term_text(property_name, value)
term_text = term_text.lower() # Matches PropertyIndex._prepare_term_text
property_name, value = propindex.split_property_term_text(term_text)
# Remove "prop." prefix that was added by make_property_term_text
if property_name.startswith("prop."):
property_name = property_name[5:]
property_name = normalize_term(property_name)
value = normalize_term(value)

cursor = self.db.cursor()
cursor.execute(
Expand Down Expand Up @@ -85,12 +81,9 @@ async def add_properties_batch(
else:
semref_id = ordinal
score = 1.0
term_text = propindex.make_property_term_text(property_name, value)
term_text = term_text.lower()
property_name, value = propindex.split_property_term_text(term_text)
if property_name.startswith("prop."):
property_name = property_name[5:]
rows.append((property_name, value, score, semref_id))
rows.append(
(normalize_term(property_name), normalize_term(value), score, semref_id)
)
cursor = self.db.cursor()
cursor.executemany(
"INSERT INTO PropertyIndex (prop_name, value_str, score, semref_id) VALUES (?, ?, ?, ?)",
Expand All @@ -107,12 +100,8 @@ async def lookup_property(
value: str,
) -> list[ScoredSemanticRefOrdinal] | None:
# Normalize property name and value (to match in-memory implementation)
term_text = propindex.make_property_term_text(property_name, value)
term_text = term_text.lower() # Matches PropertyIndex._prepare_term_text
property_name, value = propindex.split_property_term_text(term_text)
# Remove "prop." prefix that was added by make_property_term_text
if property_name.startswith("prop."):
property_name = property_name[5:]
property_name = normalize_term(property_name)
value = normalize_term(value)
Comment thread
bmerkle marked this conversation as resolved.

cursor = self.db.cursor()
cursor.execute(
Expand All @@ -132,7 +121,7 @@ async def remove_property(self, prop_name: str, semref_id: int) -> None:
cursor = self.db.cursor()
cursor.execute(
"DELETE FROM PropertyIndex WHERE prop_name = ? AND semref_id = ?",
(prop_name, semref_id),
(normalize_term(prop_name), semref_id),
)

async def remove_all_for_semref(self, semref_id: int) -> None:
Expand Down
18 changes: 3 additions & 15 deletions src/typeagent/storage/sqlite/semrefindex.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,10 +4,9 @@
"""SQLite-based semantic reference index implementation."""

from collections.abc import Sequence
import re
import sqlite3
import unicodedata

from ...knowpro.common import normalize_term
from ...knowpro.interfaces import (
ITermToSemanticRefIndex,
ScoredSemanticRefOrdinal,
Expand Down Expand Up @@ -150,7 +149,7 @@ async def deserialize(self, data: TermToSemanticRefIndexData) -> None:
# Prepare all insertion data for bulk operation
insertion_data = []
for item in data["items"]:
if item and item["term"]:
if item and item.get("term") is not None:
term = self._prepare_term(item["term"])
for semref_ordinal_data in item["semanticRefOrdinals"]:
if isinstance(semref_ordinal_data, dict):
Expand All @@ -168,15 +167,4 @@ async def deserialize(self, data: TermToSemanticRefIndexData) -> None:
)

def _prepare_term(self, term: str) -> str:
"""Normalize term by converting to lowercase, stripping whitespace, and normalizing Unicode."""
# Strip leading/trailing whitespace
term = term.strip()

# Normalize Unicode to NFC form (canonical composition)
term = unicodedata.normalize("NFC", term)

# Collapse multiple whitespace characters to single space
term = re.sub(r"\s+", " ", term)

# Convert to lowercase
return term.lower()
return normalize_term(term)
Comment thread
bmerkle marked this conversation as resolved.
Loading
Loading