From b037377be6e683a3089994448e13b48d6333cbfa Mon Sep 17 00:00:00 2001 From: macodev00 Date: Mon, 28 Sep 2026 06:34:00 +0000 Subject: [PATCH] fix: cite a renamed knowledge file by its current name Indexed chunks keep the filename from when the file was processed, so sources and the citation panel stayed on the old name after a rename. Read the current filename when those citations are built. --- backend/open_webui/retrieval/utils.py | 39 +++++++++++++++++++++++++++ backend/open_webui/tools/builtin.py | 3 ++- 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/backend/open_webui/retrieval/utils.py b/backend/open_webui/retrieval/utils.py index 58a4da1f1b82..097cf89d944d 100644 --- a/backend/open_webui/retrieval/utils.py +++ b/backend/open_webui/retrieval/utils.py @@ -1369,6 +1369,41 @@ def filter_source_metadata(metadata: dict) -> dict: return {key: metadata[key] for key in RAG_SOURCE_METADATA_KEYS if metadata.get(key) is not None} +def _is_indexed_filename(value) -> bool: + return isinstance(value, str) and bool(value) and not value.startswith(('http://', 'https://')) + + +async def apply_current_file_names(metadata_groups: list) -> None: + """Point citation metadata at the file's current name after a rename. + + Chunk metadata copies ``name`` and ``source`` when a file is indexed. + Renaming updates the file row only, so later retrievals would keep citing + the previous filename. URLs, external documents, and deleted files are left + as stored. + """ + names: dict[str, str | None] = {} + + for group in metadata_groups: + if not isinstance(group, list): + continue + for meta in group: + if not isinstance(meta, dict) or meta.get('external'): + continue + file_id = meta.get('file_id') + if not isinstance(file_id, str) or not file_id or file_id.startswith('external-'): + continue + if file_id not in names: + file = await Files.get_file_by_id(file_id) + names[file_id] = file.filename if file and file.filename else None + filename = names[file_id] + if not filename: + continue + if _is_indexed_filename(meta.get('name')): + meta['name'] = filename + if _is_indexed_filename(meta.get('source')): + meta['source'] = filename + + async def get_sources_from_items( request, items, @@ -1719,6 +1754,10 @@ async def get_sources_from_items( sources.append(source) except Exception as e: log.exception(e) + + # Indexed chunks snapshot the filename. Read the current name so a + # knowledge-base rename shows up on the next citation. + await apply_current_file_names([source.get('metadata') for source in sources]) return sources diff --git a/backend/open_webui/tools/builtin.py b/backend/open_webui/tools/builtin.py index c1d1b164d626..09d29b8dd86d 100644 --- a/backend/open_webui/tools/builtin.py +++ b/backend/open_webui/tools/builtin.py @@ -3197,7 +3197,7 @@ async def query_knowledge_files( from open_webui.models.knowledge import Knowledges from open_webui.models.notes import Notes from open_webui.retrieval.external import retrieve_external_knowledge - from open_webui.retrieval.utils import query_collection + from open_webui.retrieval.utils import apply_current_file_names, query_collection user_id = __user__.get('id') user_role = __user__.get('role', 'user') @@ -3322,6 +3322,7 @@ async def query_knowledge_files( documents = query_results.get('documents', [[]])[0] metadatas = query_results.get('metadatas', [[]])[0] distances = query_results.get('distances', [[]])[0] + await apply_current_file_names([metadatas]) for idx, doc in enumerate(documents): chunk_info = {