Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
39 changes: 39 additions & 0 deletions backend/open_webui/retrieval/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -1369,6 +1369,41 @@ def filter_source_metadata(metadata: dict) -> dict:
return {key: metadata[key] for key in RAG_SOURCE_METADATA_KEYS if metadata.get(key) is not None}


def _is_indexed_filename(value) -> bool:
return isinstance(value, str) and bool(value) and not value.startswith(('http://', 'https://'))


async def apply_current_file_names(metadata_groups: list) -> None:
"""Point citation metadata at the file's current name after a rename.

Chunk metadata copies ``name`` and ``source`` when a file is indexed.
Renaming updates the file row only, so later retrievals would keep citing
the previous filename. URLs, external documents, and deleted files are left
as stored.
"""
names: dict[str, str | None] = {}

for group in metadata_groups:
if not isinstance(group, list):
continue
for meta in group:
if not isinstance(meta, dict) or meta.get('external'):
continue
file_id = meta.get('file_id')
if not isinstance(file_id, str) or not file_id or file_id.startswith('external-'):
continue
if file_id not in names:
file = await Files.get_file_by_id(file_id)
names[file_id] = file.filename if file and file.filename else None
filename = names[file_id]
if not filename:
continue
if _is_indexed_filename(meta.get('name')):
meta['name'] = filename
if _is_indexed_filename(meta.get('source')):
meta['source'] = filename


async def get_sources_from_items(
request,
items,
Expand Down Expand Up @@ -1719,6 +1754,10 @@ async def get_sources_from_items(
sources.append(source)
except Exception as e:
log.exception(e)

# Indexed chunks snapshot the filename. Read the current name so a
# knowledge-base rename shows up on the next citation.
await apply_current_file_names([source.get('metadata') for source in sources])
return sources


Expand Down
3 changes: 2 additions & 1 deletion backend/open_webui/tools/builtin.py
Original file line number Diff line number Diff line change
Expand Up @@ -3197,7 +3197,7 @@ async def query_knowledge_files(
from open_webui.models.knowledge import Knowledges
from open_webui.models.notes import Notes
from open_webui.retrieval.external import retrieve_external_knowledge
from open_webui.retrieval.utils import query_collection
from open_webui.retrieval.utils import apply_current_file_names, query_collection

user_id = __user__.get('id')
user_role = __user__.get('role', 'user')
Expand Down Expand Up @@ -3322,6 +3322,7 @@ async def query_knowledge_files(
documents = query_results.get('documents', [[]])[0]
metadatas = query_results.get('metadatas', [[]])[0]
distances = query_results.get('distances', [[]])[0]
await apply_current_file_names([metadatas])

for idx, doc in enumerate(documents):
chunk_info = {
Expand Down
Loading