From bc0b652d0ddf9e783de09626127a36c5ab55427c Mon Sep 17 00:00:00 2001 From: Bernhard Merkle Date: Mon, 28 Sep 2026 23:57:57 +0200 Subject: [PATCH 1/2] Normalize timestamps to UTC in the timestamp index (#320) TimestampToTextRangeIndex sorted and bisected ISO strings, which only works if every timestamp has the same UTC offset. Email ingestion kept the sender's offset, so range lookups silently dropped in-range messages. - Normalize stored timestamps and query bounds to fixed-format UTC (microseconds, "Z" suffix); naive timestamps are treated as UTC. - Normalize email Date headers to UTC on import so the SQLite index, which compares strings too, gets consistent values. - Add tests for mixed offsets, incremental inserts, naive/aware boundaries, fractional seconds and email date normalization. Co-Authored-By: Claude Sonnet 5.5 --- src/typeagent/emails/email_import.py | 10 ++- .../storage/memory/timestampindex.py | 25 ++++-- tests/test_email_import.py | 28 +++++++ tests/test_timestampindex.py | 78 +++++++++++++++++-- 4 files changed, 127 insertions(+), 14 deletions(-) diff --git a/src/typeagent/emails/email_import.py b/src/typeagent/emails/email_import.py index 88f13b94..4f70640a 100644 --- a/src/typeagent/emails/email_import.py +++ b/src/typeagent/emails/email_import.py @@ -1,7 +1,7 @@ # Copyright (c) Microsoft Corporation. # Licensed under the MIT License. -from datetime import datetime +from datetime import datetime, timezone from email import message_from_string from email.header import decode_header, Header, make_header from email.message import Message @@ -10,6 +10,7 @@ import re from typing import Iterable, overload +from ..knowpro.universal_message import format_timestamp_utc from .email_message import EmailMessage, EmailMessageMeta @@ -100,7 +101,12 @@ def import_email_message(msg: Message, max_chunk_length: int) -> EmailMessage: timestamp: str | None = None timestamp_date = msg.get("Date", None) if timestamp_date is not None: - timestamp = parsedate_to_datetime(timestamp_date).isoformat() + parsed_date = parsedate_to_datetime(timestamp_date) + if parsed_date.tzinfo is None: + # RFC 5322 "-0000": time is UTC but the origin zone is unknown. + parsed_date = parsed_date.replace(tzinfo=timezone.utc) + # Normalize to UTC so timestamps sort lexicographically across senders. + timestamp = format_timestamp_utc(parsed_date) # Get email body. # If the email was a reply, then ensure we only pick up the latest response diff --git a/src/typeagent/storage/memory/timestampindex.py b/src/typeagent/storage/memory/timestampindex.py index 59bc404e..c62092e1 100644 --- a/src/typeagent/storage/memory/timestampindex.py +++ b/src/typeagent/storage/memory/timestampindex.py @@ -4,8 +4,9 @@ # Timestamp-to-text-range in-memory index (pre-SQLite prep). # # Contract (stable regardless of backing store): -# - add_timestamp(s) accepts ISO 8601 timestamps that are lexicographically sortable -# (Datetime.isoformat). Missing/None timestamps are ignored. +# - add_timestamp(s) accepts ISO 8601 timestamps with any UTC offset (naive +# timestamps are assumed to be UTC). They are normalized to UTC ("...Z") so that +# lexicographic order equals chronological order. Missing/None timestamps are ignored. # - lookup_range(DateRange) returns items whose ISO timestamp t satisfies # start <= t < end (end is exclusive). If end is None, treat as a point # query with end = start + epsilon. @@ -20,6 +21,7 @@ import bisect from collections.abc import AsyncIterable, Callable +from datetime import timezone from typing import Any from ...knowpro.interfaces import ( @@ -56,8 +58,10 @@ async def lookup_range(self, date_range: DateRange) -> list[TimestampedTextRange return self._lookup_range(date_range) def _lookup_range(self, date_range: DateRange) -> list[TimestampedTextRange]: - start_at = date_range.start.isoformat() - stop_at = None if date_range.end is None else date_range.end.isoformat() + start_at = _normalize_datetime(date_range.start) + stop_at = ( + None if date_range.end is None else _normalize_datetime(date_range.end) + ) return get_in_range( self._ranges, start_at, @@ -101,11 +105,10 @@ def _insert_timestamp( ) -> bool: if not timestamp: return False - timestamp_datetime = Datetime.fromisoformat(timestamp) entry: TimestampedTextRange = TimestampedTextRange( range=text_range_from_message_chunk(message_ordinal), - # This string is formatted to be lexically sortable. - timestamp=timestamp_datetime.isoformat(), + # Normalized to UTC so that the string is lexically sortable. + timestamp=_normalize_datetime(Datetime.fromisoformat(timestamp)), ) if in_order: where = bisect.bisect_left( @@ -117,6 +120,14 @@ def _insert_timestamp( return True +def _normalize_datetime(dt: Datetime) -> str: + # Render as UTC with a fixed format ("Z" suffix, always microseconds) so that + # lexicographic order equals chronological order. Naive datetimes are UTC. + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.astimezone(timezone.utc).isoformat(timespec="microseconds")[:-6] + "Z" + + def get_in_range[T, S: Any]( values: list[T], start_at: S, diff --git a/tests/test_email_import.py b/tests/test_email_import.py index 371136bc..78650799 100644 --- a/tests/test_email_import.py +++ b/tests/test_email_import.py @@ -5,6 +5,7 @@ _merge_chunks, _split_into_paragraphs, _text_to_chunks, + import_email_string, ) @@ -100,3 +101,30 @@ def test_no_leading_separator_in_any_chunk(self) -> None: assert not chunk.startswith( "\n\n" ), f"chunk {chunk!r} has leading separator" + + +class TestEmailTimestampNormalization: + """Email Date headers with different offsets must yield comparable timestamps.""" + + @staticmethod + def _timestamp(date_header: str) -> str | None: + raw = f"From: a@example.com\nTo: b@example.com\nDate: {date_header}\nSubject: s\n\nbody\n" + return import_email_string(raw, 1000).timestamp + + def test_offsets_normalized_to_utc(self) -> None: + assert ( + self._timestamp("Mon, 1 Jan 2024 12:00:00 +0500") == "2024-01-01T07:00:00Z" + ) + assert ( + self._timestamp("Mon, 1 Jan 2024 10:00:00 -0800") == "2024-01-01T18:00:00Z" + ) + + def test_lexicographic_order_is_chronological(self) -> None: + a = self._timestamp("Mon, 1 Jan 2024 12:00:00 +0500") # 07:00Z + b = self._timestamp("Mon, 1 Jan 2024 08:00:00 +0000") # 08:00Z + assert a is not None and b is not None and a < b + + def test_unknown_zone_treated_as_utc(self) -> None: + assert ( + self._timestamp("Mon, 1 Jan 2024 12:00:00 -0000") == "2024-01-01T12:00:00Z" + ) diff --git a/tests/test_timestampindex.py b/tests/test_timestampindex.py index 91803780..9b7324a7 100644 --- a/tests/test_timestampindex.py +++ b/tests/test_timestampindex.py @@ -1,6 +1,8 @@ # Copyright (c) Microsoft Corporation. # Licensed under the MIT License. +from datetime import timedelta, timezone + import pytest from typeagent.knowpro.interfaces import DateRange, Datetime @@ -20,6 +22,7 @@ def to_ts_list(entries): @pytest.mark.asyncio async def test_lookup_range_half_open_and_point_query(): + # Stored timestamps are normalized to UTC with a "Z" suffix (naive == UTC). # Three sequential timestamps t0 = "2025-01-01T00:00:00" t1 = "2025-01-01T01:00:00" @@ -29,30 +32,95 @@ async def test_lookup_range_half_open_and_point_query(): # [t0, t1) includes t0, excludes t1 dr = DateRange(start=Datetime.fromisoformat(t0), end=Datetime.fromisoformat(t1)) results = await idx.lookup_range(dr) - assert to_ts_list(results) == [t0] + assert to_ts_list(results) == [t0 + ".000000Z"] # [t0, t2) includes t0 and t1, excludes t2 dr = DateRange(start=Datetime.fromisoformat(t0), end=Datetime.fromisoformat(t2)) results = await idx.lookup_range(dr) - assert to_ts_list(results) == [t0, t1] + assert to_ts_list(results) == [t0 + ".000000Z", t1 + ".000000Z"] # [t1, t2) includes only t1 dr = DateRange(start=Datetime.fromisoformat(t1), end=Datetime.fromisoformat(t2)) results = await idx.lookup_range(dr) - assert to_ts_list(results) == [t1] + assert to_ts_list(results) == [t1 + ".000000Z"] # Point query: end=None means [t1, t1+epsilon) -> exactly t1 dr = DateRange(start=Datetime.fromisoformat(t1), end=None) results = await idx.lookup_range(dr) - assert to_ts_list(results) == [t1] + assert to_ts_list(results) == [t1 + ".000000Z"] # Point query at t2 returns [t2] dr = DateRange(start=Datetime.fromisoformat(t2), end=None) results = await idx.lookup_range(dr) - assert to_ts_list(results) == [t2] + assert to_ts_list(results) == [t2 + ".000000Z"] # Point query at a time not present returns [] tmid = "2025-01-01T00:30:00" dr = DateRange(start=Datetime.fromisoformat(tmid), end=None) results = await idx.lookup_range(dr) assert to_ts_list(results) == [] + + +@pytest.mark.asyncio +async def test_lookup_range_mixed_utc_offsets(): + # 12:00+05:00 == 07:00 UTC; 10:00-08:00 == 18:00 UTC + idx = await make_index(["2024-01-01T12:00:00+05:00", "2024-01-01T10:00:00-08:00"]) + utc = timezone.utc + + dr = DateRange( + start=Datetime(2024, 1, 1, 6, tzinfo=utc), + end=Datetime(2024, 1, 1, 8, tzinfo=utc), + ) + assert [e.range.start.message_ordinal for e in await idx.lookup_range(dr)] == [0] + + dr = DateRange( + start=Datetime(2024, 1, 1, 6, tzinfo=utc), + end=Datetime(2024, 1, 1, 19, tzinfo=utc), + ) + assert [e.range.start.message_ordinal for e in await idx.lookup_range(dr)] == [0, 1] + + # Query bound with a non-UTC offset; point query on the exact instant. + plus5 = timezone(timedelta(hours=5)) + dr = DateRange(start=Datetime(2024, 1, 1, 12, tzinfo=plus5), end=None) + assert [e.range.start.message_ordinal for e in await idx.lookup_range(dr)] == [0] + + +@pytest.mark.asyncio +async def test_add_timestamp_incremental_keeps_chronological_order(): + idx = TimestampToTextRangeIndex() + await idx.add_timestamp(0, "2024-01-01T12:00:00+05:00") # 07:00Z + await idx.add_timestamp(1, "2024-01-01T08:00:00+00:00") # 08:00Z + await idx.add_timestamp(2, "2024-01-01T01:00:00-08:00") # 09:00Z + dr = DateRange( + start=Datetime(2024, 1, 1, tzinfo=timezone.utc), + end=Datetime(2024, 1, 2, tzinfo=timezone.utc), + ) + assert [e.range.start.message_ordinal for e in await idx.lookup_range(dr)] == [ + 0, + 1, + 2, + ] + + +@pytest.mark.asyncio +async def test_naive_timestamp_on_aware_range_start_is_included(): + idx = await make_index(["2024-01-01T00:00:00"]) + dr = DateRange( + start=Datetime(2024, 1, 1, tzinfo=timezone.utc), + end=Datetime(2024, 1, 1, 1, tzinfo=timezone.utc), + ) + assert len(await idx.lookup_range(dr)) == 1 + + +@pytest.mark.asyncio +async def test_fractional_seconds_sort_chronologically(): + idx = await make_index(["2024-01-01T12:00:00", "2024-01-01T12:00:00.5"]) + dr = DateRange( + start=Datetime(2024, 1, 1, 12, tzinfo=timezone.utc), + end=Datetime(2024, 1, 1, 12, 0, 1, tzinfo=timezone.utc), + ) + assert [e.range.start.message_ordinal for e in await idx.lookup_range(dr)] == [0, 1] + dr = DateRange( + start=Datetime(2024, 1, 1, 12, 0, 0, 250000, tzinfo=timezone.utc), end=None + ) + assert await idx.lookup_range(dr) == [] From 04606e7cb39560d081512a6aad04917245b6dcbf Mon Sep 17 00:00:00 2001 From: Bernhard Merkle Date: Tue, 29 Sep 2026 00:06:41 +0200 Subject: [PATCH 2/2] Compare SQLite timestamps as UTC instants, not raw strings Stored start_timestamp values can have any UTC offset and fractional precision (existing rows included), so raw string comparison misorders e.g. "07:00:00Z" vs "07:00:00.500000Z". Compare via strftime(), which converts to UTC and yields a fixed-width value, in lookup_range and get_timestamp_ranges. Add a matching expression index so range queries stay indexed; no data migration is needed. Co-Authored-By: Claude Sonnet 5.5 --- src/typeagent/storage/sqlite/schema.py | 10 ++++ .../storage/sqlite/timestampindex.py | 27 +++++++---- tests/test_sqlite_indexes.py | 47 +++++++++++++++++++ 3 files changed, 74 insertions(+), 10 deletions(-) diff --git a/src/typeagent/storage/sqlite/schema.py b/src/typeagent/storage/sqlite/schema.py index 99117c24..ca7927ca 100644 --- a/src/typeagent/storage/sqlite/schema.py +++ b/src/typeagent/storage/sqlite/schema.py @@ -33,10 +33,19 @@ ); """ +# Normalizes an ISO timestamp (any UTC offset, any fractional precision) to a +# fixed-width UTC string that sorts chronologically. Millisecond precision. +NORMALIZED_TIMESTAMP_SQL = "strftime('%Y-%m-%dT%H:%M:%f', {value})" + TIMESTAMP_INDEX_SCHEMA = """ CREATE INDEX IF NOT EXISTS idx_messages_start_timestamp ON Messages(start_timestamp); """ +NORMALIZED_TIMESTAMP_INDEX_SCHEMA = f""" +CREATE INDEX IF NOT EXISTS idx_messages_start_timestamp_utc + ON Messages({NORMALIZED_TIMESTAMP_SQL.format(value="start_timestamp")}); +""" + # Conversation metadata table (key-value pairs) CONVERSATION_METADATA_SCHEMA = """ CREATE TABLE IF NOT EXISTS ConversationMetadata ( @@ -292,6 +301,7 @@ def init_db_schema(db: sqlite3.Connection) -> None: cursor.execute(RELATED_TERMS_ALIASES_SCHEMA) cursor.execute(RELATED_TERMS_FUZZY_SCHEMA) cursor.execute(TIMESTAMP_INDEX_SCHEMA) + cursor.execute(NORMALIZED_TIMESTAMP_INDEX_SCHEMA) cursor.execute(INGESTED_SOURCES_SCHEMA) cursor.execute(CHUNK_FAILURES_SCHEMA) diff --git a/src/typeagent/storage/sqlite/timestampindex.py b/src/typeagent/storage/sqlite/timestampindex.py index a5394c59..65d8bf25 100644 --- a/src/typeagent/storage/sqlite/timestampindex.py +++ b/src/typeagent/storage/sqlite/timestampindex.py @@ -14,6 +14,7 @@ TimestampedTextRange, ) from ...knowpro.universal_message import format_timestamp_utc +from .schema import NORMALIZED_TIMESTAMP_SQL class SqliteTimestampToTextRangeIndex(ITimestampToTextRangeIndex): @@ -52,24 +53,26 @@ async def get_timestamp_ranges( """Get timestamp ranges from Messages table.""" cursor = self.db.cursor() + norm_col = NORMALIZED_TIMESTAMP_SQL.format(value="start_timestamp") + norm_arg = NORMALIZED_TIMESTAMP_SQL.format(value="?") if end_timestamp is None: # Single timestamp query cursor.execute( - """ + f""" SELECT msg_id, start_timestamp FROM Messages - WHERE start_timestamp = ? + WHERE {norm_col} = {norm_arg} ORDER BY msg_id """, (start_timestamp,), ) else: - # Range query + # Range query (inclusive) cursor.execute( - """ + f""" SELECT msg_id, start_timestamp FROM Messages - WHERE start_timestamp >= ? AND start_timestamp <= ? + WHERE {norm_col} >= {norm_arg} AND {norm_col} <= {norm_arg} ORDER BY msg_id """, (start_timestamp, end_timestamp), @@ -103,17 +106,21 @@ async def lookup_range(self, date_range: DateRange) -> list[TimestampedTextRange """Lookup messages in a date range.""" cursor = self.db.cursor() - # Convert datetime objects to ISO format strings with Z suffix + # Stored timestamps may have any UTC offset and any fractional precision, + # so raw strings don't sort chronologically. Compare via strftime(), which + # converts to UTC and renders a fixed-width value (also for existing rows). start_timestamp = format_timestamp_utc(date_range.start) end_timestamp = format_timestamp_utc(date_range.end) if date_range.end else None + norm_col = NORMALIZED_TIMESTAMP_SQL.format(value="start_timestamp") + norm_arg = NORMALIZED_TIMESTAMP_SQL.format(value="?") if date_range.end is None: # Point query cursor.execute( - """ + f""" SELECT msg_id, start_timestamp, chunks FROM Messages - WHERE start_timestamp = ? + WHERE {norm_col} = {norm_arg} ORDER BY msg_id """, (start_timestamp,), @@ -121,10 +128,10 @@ async def lookup_range(self, date_range: DateRange) -> list[TimestampedTextRange else: # Range query cursor.execute( - """ + f""" SELECT msg_id, start_timestamp, chunks FROM Messages - WHERE start_timestamp >= ? AND start_timestamp < ? + WHERE {norm_col} >= {norm_arg} AND {norm_col} < {norm_arg} ORDER BY msg_id """, (start_timestamp, end_timestamp), diff --git a/tests/test_sqlite_indexes.py b/tests/test_sqlite_indexes.py index 825f57d8..b547c6c9 100644 --- a/tests/test_sqlite_indexes.py +++ b/tests/test_sqlite_indexes.py @@ -3,6 +3,7 @@ """Tests for SQLite index implementations with real embeddings.""" +from datetime import timedelta, timezone import os import sqlite3 import tempfile @@ -15,6 +16,8 @@ from typeagent.knowpro import interfaces from typeagent.knowpro.convsettings import MessageTextIndexSettings from typeagent.knowpro.interfaces import ( + DateRange, + Datetime, SemanticRef, Term, TextLocation, @@ -188,6 +191,50 @@ async def test_timestamp_operations(self, sqlite_db: sqlite3.Connection): ) assert len(results) == 2 + @pytest.mark.asyncio + async def test_lookup_range_mixed_offsets_and_precision( + self, sqlite_db: sqlite3.Connection + ): + """Range lookup compares instants, not raw strings.""" + index = SqliteTimestampToTextRangeIndex(sqlite_db) + stored = [ + "2024-01-01T12:00:00+05:00", # 07:00:00 UTC (legacy email format) + "2024-01-01T07:00:00Z", # 07:00:00 UTC + "2024-01-01T07:00:00.250000Z", + "2024-01-01T07:00:01Z", + ] + cursor = sqlite_db.cursor() + for i, ts in enumerate(stored): + cursor.execute( + "INSERT INTO Messages (msg_id, chunks, start_timestamp) VALUES (?, '[\"m\"]', ?)", + (i, ts), + ) + sqlite_db.commit() + + def ordinals(results): + return [r.range.start.message_ordinal for r in results] + + utc = timezone.utc + dr = DateRange( + start=Datetime(2024, 1, 1, 7, tzinfo=utc), + end=Datetime(2024, 1, 1, 7, 0, 1, tzinfo=utc), + ) + assert ordinals(await index.lookup_range(dr)) == [0, 1, 2] + + # A fractional bound must exclude the whole-second value before it. + dr = DateRange( + start=Datetime(2024, 1, 1, 7, 0, 0, 500000, tzinfo=utc), + end=Datetime(2024, 1, 1, 8, tzinfo=utc), + ) + assert ordinals(await index.lookup_range(dr)) == [3] + + # Point query with a non-UTC offset matches both representations. + dr = DateRange( + start=Datetime(2024, 1, 1, 12, tzinfo=timezone(timedelta(hours=5))), + end=None, + ) + assert ordinals(await index.lookup_range(dr)) == [0, 1] + class TestSqliteRelatedTermsAliases: """Test SqliteRelatedTermsAliases functionality."""