Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
* Anthropic-backed providers (`ChatAnthropic()`, `ChatPosit()`, `ChatBedrock()`, etc.) no longer fail with `Invalid signature in thinking block` when the conversation history contains reasoning from a non-Claude model (e.g., after switching `chat.model` across families); such thinking is now replayed as plain text so the model can still see it.
* `content_image_file()` no longer fails with `ValueError: unknown file extension` when `resize` uses the `!` (ignore aspect ratio) flag on an image larger than the requested box, e.g. `resize="200x200!"`. (#433)
* `params(top_k=)` is no longer sent as `top_logprobs` for OpenAI-based providers (the two are unrelated; OpenAI has no `top_k` sampling parameter). `top_k` is now dropped with the standard unsupported-parameter warning. (#412)
* `Chat.export()` and `interpolate_file()` now always read and write files as UTF-8. Previously they relied on the platform's default locale encoding, so on a non-UTF-8 locale `export()` could raise `UnicodeEncodeError` and `interpolate_file()` could silently return a corrupted (mojibake) prompt instead of the file's contents. (#437)


## [0.23.0] - 2026-09-04
Expand Down
2 changes: 1 addition & 1 deletion chatlas/_chat.py
Original file line number Diff line number Diff line change
Expand Up @@ -2584,7 +2584,7 @@ def export(
if is_html:
contents = self._html_template(contents)

with open(filename, "w") as f:
with open(filename, "w", encoding="utf-8") as f:
f.write(contents)

return filename
Expand Down
2 changes: 1 addition & 1 deletion chatlas/_interpolate.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,7 +109,7 @@ def interpolate_file(
variables = _infer_variables(frame)
del frame

with open(path, "r") as file:
with open(path, "r", encoding="utf-8") as file:
return interpolate(
file.read(),
variables=variables,
Expand Down
33 changes: 33 additions & 0 deletions tests/test_chat.py
Original file line number Diff line number Diff line change
Expand Up @@ -161,6 +161,39 @@ def test_basic_export(snapshot):
assert snapshot == f.read()


def test_export_writes_utf8(monkeypatch):
# `Chat.export()` must write the transcript as UTF-8 regardless of the
# platform's default locale encoding, otherwise non-ASCII turn content
# raises UnicodeEncodeError instead of exporting successfully.
text = "Wie heißt die Hauptstadt? 中文测试 café"
chat = ChatOpenAI(
system_prompt="You're a helpful assistant that returns very minimal output",
)
chat.set_turns(
[
UserTurn(text),
AssistantTurn(text, tokens=(15, 5, 0)),
]
)

real_open = open

def open_using_locale_default(file, mode="r", *args, encoding=None, **kwargs):
# Simulate a non-UTF-8 preferred locale encoding by falling back to
# latin-1 whenever the caller doesn't pass an explicit encoding.
if "b" not in mode and encoding is None:
encoding = "latin-1"
return real_open(file, mode, *args, encoding=encoding, **kwargs)

monkeypatch.setattr("chatlas._chat.open", open_using_locale_default, raising=False)

with tempfile.TemporaryDirectory() as tmpdir:
tmpfile = tmpdir + "/chat.md"
chat.export(tmpfile)
with open(tmpfile, "r", encoding="utf-8") as f:
assert text in f.read()


@pytest.mark.vcr
def test_chat_structured():
chat = ChatOpenAI(model="gpt-5.4")
Expand Down
25 changes: 25 additions & 0 deletions tests/test_interpolate.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,3 +18,28 @@ def test_interpolate_file(tmp_path):

x = 1 # noqa
assert interpolate_file(path) == "1"


def test_interpolate_file_reads_as_utf8(tmp_path, monkeypatch):
# `interpolate_file()` must read the prompt file as UTF-8 regardless of
# the platform's default locale encoding, otherwise non-ASCII prompt
# content is silently corrupted (mojibake) instead of raising.
text = "Wie heißt die Hauptstadt? 中文测试 café {{ name }}"
path = tmp_path / "prompt.txt"
path.write_text(text, encoding="utf-8")

real_open = open

def open_using_locale_default(file, mode="r", *args, encoding=None, **kwargs):
# Simulate a non-UTF-8 preferred locale encoding by falling back to
# latin-1 whenever the caller doesn't pass an explicit encoding.
if "b" not in mode and encoding is None:
encoding = "latin-1"
return real_open(file, mode, *args, encoding=encoding, **kwargs)

monkeypatch.setattr(
"chatlas._interpolate.open", open_using_locale_default, raising=False
)

result = interpolate_file(path, variables={"name": "x"})
assert "Wie heißt die Hauptstadt? 中文测试 café" in result