diff --git a/server/mcp_server_knowledgebase/README.md b/server/mcp_server_knowledgebase/README.md index f011786b..b0cca888 100644 --- a/server/mcp_server_knowledgebase/README.md +++ b/server/mcp_server_knowledgebase/README.md @@ -157,7 +157,8 @@ Example result: ```json { "collection_name": "product_docs", - "doc_id": "product_guide_2026" + "doc_id": "product_guide_2026", + "resource_id": "kb-example" } ``` @@ -168,13 +169,16 @@ Get a document's metadata and processing status. ```python get_doc( collection_name="product_docs", + resource_id="kb-example", doc_id="product_guide_2026", ) ``` Parameters: -- `collection_name` (required): Collection containing the document. +- `collection_name` (optional): Collection containing the document. +- `resource_id` (optional): Collection ID. Provide this or `collection_name`; + `resource_id` takes precedence when both are provided. - `doc_id` (required): Document ID. Example result: @@ -185,7 +189,6 @@ Example result: "doc_id": "product_guide_2026", "doc_name": "Product Guide", "doc_type": "pdf", - "url": "https://example.com/product-guide.pdf", "add_type": "url", "create_time": 1788220800, "update_time": 1788220860, @@ -193,13 +196,54 @@ Example result: "status": { "process_status": 0, "failed_code": null + }, + "title": "Product Guide", + "doc_summary": "Product setup and account management instructions.", + "brief_summary": "An introduction to the product.", + "meta": { + "category": "product", + "version": "2026" + }, + "video_outline": { + "title": "Product walkthrough", + "summary": "A walkthrough of the product.", + "chapters": [ + { + "title": "Account settings", + "content": "Open Account Settings.", + "start_time": "00:00:10", + "end_time": "00:00:30", + "element_content": { + "text": "Account Settings" + } + } + ] + }, + "audio_outline": { + "title": "Product introduction", + "summary": "An introduction to account management.", + "chapters": [ + { + "title": "Resetting your password", + "content": "Select Reset Password.", + "start_time": 10.0, + "end_time": 30.0, + "element_content": { + "text": "Reset Password" + } + } + ] } } ``` `process_status` values: `0` completed, `1` failed, `2` or `3` queued, `5` -deleting, and `6` processing. Fields not returned by Viking are `null`; -additional upstream document fields are preserved. +deleting, and `6` processing. Fields not returned by Viking are `null`. + +Additional result fields include `title` (document title), `doc_summary` (document +summary), `brief_summary` (short summary), `meta` (metadata), `video_outline`, and +`audio_outline`. Outlines contain `title`, `summary`, and `chapters`; chapters +contain `title`, `content`, `start_time`, `end_time`, and `element_content`. ### `list_docs` @@ -208,6 +252,7 @@ List documents in a collection using cursor pagination. ```python list_docs( collection_name="product_docs", + resource_id="kb-example", limit=2, next_token=None, ) @@ -215,9 +260,11 @@ list_docs( Parameters: -- `collection_name` (required): Collection whose documents will be listed. +- `collection_name` (optional): Collection whose documents will be listed. +- `resource_id` (optional): Collection ID. Provide this or `collection_name`; + `resource_id` takes precedence when both are provided. - `limit` (optional): Number of documents to return, from 1 to 100. Defaults - to `100`. + to `50`. - `next_token` (optional): Opaque cursor returned by the previous call. Omit it for the first page. An empty cursor in the result means all documents have been returned. @@ -228,23 +275,21 @@ Example result: { "collection_name": "product_docs", "total_num": 3, - "count": 2, + "count": 1, "doc_list": [ { - "collection_name": "product_docs", "doc_id": "product_guide_2026", "doc_name": "Product Guide", "doc_type": "pdf", - "url": "https://example.com/product-guide.pdf", - "add_type": "url", "create_time": 1788220800, "update_time": 1788220860, "point_num": 53, "status": { - "process_status": 0 + "process_status": 0, + "failed_code": null }, "brief_summary": "An introduction to the product.", - "total_tokens": 345 + "title": "Product Guide" } ], "has_more": true, @@ -252,20 +297,24 @@ Example result: } ``` -`total_num` is `null` when Viking does not provide it. Document entries -preserve additional upstream fields such as summaries and token counts. +`total_num` is `null` when Viking does not provide it. ### `get_collection` Get information and build status for a collection. ```python -get_collection(collection_name="product_docs") +get_collection( + collection_name="product_docs", + resource_id="kb-example", +) ``` Parameters: -- `collection_name` (required): Collection name. +- `collection_name` (optional): Collection name. +- `resource_id` (optional): Collection ID. Provide this or `collection_name`; + `resource_id` takes precedence when both are provided. Example result: @@ -273,7 +322,11 @@ Example result: { "collection_name": "product_docs", "description": "Product manuals and release notes", - "status": 1 + "status": 1, + "resource_id": "kb-example", + "doc_num": 3, + "create_time": 1788220800, + "update_time": 1788220860 } ``` @@ -297,13 +350,20 @@ Example result: "collection_list": [ { "collection_name": "product_docs", - "description": "Product manuals and release notes" + "description": "Product manuals and release notes", + "resource_id": "kb-example-1", + "create_time": 1788220800, + "update_time": 1788220860 }, { "collection_name": "support_faq", - "description": "Frequently asked support questions" + "description": "Frequently asked support questions", + "resource_id": "kb-example-2", + "create_time": 1788220800, + "update_time": 1788220860 } - ] + ], + "total_num": 2 } ``` @@ -316,7 +376,8 @@ include or exclude matching document field values. search_knowledge( query="How do I reset my password?", collection_name="support_faq", - limit=3, + resource_id="kb-example", + limit=10, doc_filter={ "op": "must", "field": "doc_id", @@ -327,10 +388,12 @@ search_knowledge( Parameters: -- `query` (required): Search query. -- `collection_name` (required): Collection to search. +- `query` (required): Search query, from 1 to 8000 characters. +- `collection_name` (optional): Collection to search. +- `resource_id` (optional): Collection ID. Provide this or `collection_name`; + `resource_id` takes precedence when both are provided. - `limit` (optional): Maximum number of chunks to return, from 1 to 100. - Defaults to `3`. + Defaults to `10`. - `doc_filter` (optional): Object with the following fields: - `op`: `"must"` to include matches or `"must_not"` to exclude them. - `field`: Document field to filter, such as `"doc_id"`. @@ -345,7 +408,23 @@ Example result: "id": "chunk_001", "content": "Open Account Settings and select Reset Password.", "doc_id": "account_guide", - "doc_name": "Account Guide" + "doc_name": "Account Guide", + "title": "Account Guide", + "doc_type": "pdf", + "score": 0.85, + "rerank_score": 0.92, + "chunk_title": "Resetting your password", + "audio_start_time": null, + "audio_end_time": null, + "video_start_time": null, + "video_end_time": null, + "chunk_attachment": [ + { + "uuid": "image_1", + "caption": "Account Settings", + "type": "image" + } + ] } ] } @@ -354,6 +433,27 @@ Example result: `doc_id` and `doc_name` are `null` when Viking does not provide document metadata. A non-null `doc_id` can be passed directly to `get_doc`. +Chunks also include `title` (document title), `doc_type` (document type), `score` +(search score), `rerank_score`, `chunk_title`, `audio_start_time`, `audio_end_time`, +`video_start_time`, `video_end_time`, and `chunk_attachment` (attachments). Each +attachment contains `uuid`, `caption`, and `type`. + +Image links for `image`, `doc-image`, and `table` attachments are returned as MCP +`ResourceLink` blocks with `uri`, `name`, `description`, and `mimeType` (`image/*`), +and URL is not included in `chunk_attachment`. + +Example image link content block: + +```json +{ + "type": "resource_link", + "uri": "https://example.com/account-settings.png", + "name": "image_1", + "description": "Account Settings", + "mimeType": "image/*" +} +``` + ## MCP client configuration Example stdio configuration using `uvx` and a Viking API key: @@ -365,7 +465,7 @@ Example stdio configuration using `uvx` and a Viking API key: "command": "uvx", "args": [ "--from", - "mcp-server-knowledgebase>=0.2.1", + "mcp-server-knowledgebase>=0.2.2", "mcp-server-knowledgebase" ], "env": { diff --git a/server/mcp_server_knowledgebase/README_zh.md b/server/mcp_server_knowledgebase/README_zh.md index 9cdf5e70..59faaaa4 100644 --- a/server/mcp_server_knowledgebase/README_zh.md +++ b/server/mcp_server_knowledgebase/README_zh.md @@ -153,7 +153,8 @@ add_doc( ```json { "collection_name": "product_docs", - "doc_id": "product_guide_2026" + "doc_id": "product_guide_2026", + "resource_id": "kb-example" } ``` @@ -164,13 +165,15 @@ add_doc( ```python get_doc( collection_name="product_docs", + resource_id="kb-example", doc_id="product_guide_2026", ) ``` 参数: -- `collection_name`(必填):文档所属的知识库。 +- `collection_name`(可选):文档所属的知识库。 +- `resource_id`(可选):知识库 ID。与 `collection_name` 至少提供一个;同时提供时优先使用 `resource_id`。 - `doc_id`(必填):文档 ID。 返回示例: @@ -181,7 +184,6 @@ get_doc( "doc_id": "product_guide_2026", "doc_name": "Product Guide", "doc_type": "pdf", - "url": "https://example.com/product-guide.pdf", "add_type": "url", "create_time": 1788220800, "update_time": 1788220860, @@ -189,13 +191,54 @@ get_doc( "status": { "process_status": 0, "failed_code": null + }, + "title": "Product Guide", + "doc_summary": "Product setup and account management instructions.", + "brief_summary": "An introduction to the product.", + "meta": { + "category": "product", + "version": "2026" + }, + "video_outline": { + "title": "Product walkthrough", + "summary": "A walkthrough of the product.", + "chapters": [ + { + "title": "Account settings", + "content": "Open Account Settings.", + "start_time": "00:00:10", + "end_time": "00:00:30", + "element_content": { + "text": "Account Settings" + } + } + ] + }, + "audio_outline": { + "title": "Product introduction", + "summary": "An introduction to account management.", + "chapters": [ + { + "title": "Resetting your password", + "content": "Select Reset Password.", + "start_time": 10.0, + "end_time": 30.0, + "element_content": { + "text": "Reset Password" + } + } + ] } } ``` `process_status` 状态值:`0` 表示处理完成,`1` 表示处理失败,`2` 或 `3` -表示排队中,`5` 表示删除中,`6` 表示处理中。Viking 未返回的字段为 `null`; -上游返回的其他文档字段也会保留。 +表示排队中,`5` 表示删除中,`6` 表示处理中。Viking 未返回的字段为 `null`。 + +返回字段还包括 `title`(文档标题)、`doc_summary`(文档摘要)、 +`brief_summary`(简短摘要)、`meta`(元数据)、`video_outline`(视频大纲)和 +`audio_outline`(音频大纲)。大纲包含 `title`、`summary` 和 `chapters`;章节包含 +`title`、`content`、`start_time`、`end_time` 和 `element_content`。 ### `list_docs` @@ -204,6 +247,7 @@ get_doc( ```python list_docs( collection_name="product_docs", + resource_id="kb-example", limit=2, next_token=None, ) @@ -211,8 +255,9 @@ list_docs( 参数: -- `collection_name`(必填):要获取文档列表的知识库。 -- `limit`(可选):单次返回的文档数量,范围为 1–100,默认值为 `100`。 +- `collection_name`(可选):要获取文档列表的知识库。 +- `resource_id`(可选):知识库 ID。与 `collection_name` 至少提供一个;同时提供时优先使用 `resource_id`。 +- `limit`(可选):单次返回的文档数量,范围为 1–100,默认值为 `50`。 - `next_token`(可选):上一次调用返回的不透明游标。首次请求不传;返回值中的 游标为空表示文档已全部返回。 @@ -222,23 +267,21 @@ list_docs( { "collection_name": "product_docs", "total_num": 3, - "count": 2, + "count": 1, "doc_list": [ { - "collection_name": "product_docs", "doc_id": "product_guide_2026", "doc_name": "Product Guide", "doc_type": "pdf", - "url": "https://example.com/product-guide.pdf", - "add_type": "url", "create_time": 1788220800, "update_time": 1788220860, "point_num": 53, "status": { - "process_status": 0 + "process_status": 0, + "failed_code": null }, "brief_summary": "An introduction to the product.", - "total_tokens": 345 + "title": "Product Guide" } ], "has_more": true, @@ -246,20 +289,23 @@ list_docs( } ``` -Viking 未提供 `total_num` 时,该字段为 `null`。文档条目会保留摘要、token 数 -等上游扩展字段。 +Viking 未提供 `total_num` 时,该字段为 `null`。 ### `get_collection` 获取知识库信息和构建状态。 ```python -get_collection(collection_name="product_docs") +get_collection( + collection_name="product_docs", + resource_id="kb-example", +) ``` 参数: -- `collection_name`(必填):知识库名称。 +- `collection_name`(可选):知识库名称。 +- `resource_id`(可选):知识库 ID。与 `collection_name` 至少提供一个;同时提供时优先使用 `resource_id`。 返回示例: @@ -267,7 +313,11 @@ get_collection(collection_name="product_docs") { "collection_name": "product_docs", "description": "Product manuals and release notes", - "status": 1 + "status": 1, + "resource_id": "kb-example", + "doc_num": 3, + "create_time": 1788220800, + "update_time": 1788220860 } ``` @@ -291,13 +341,20 @@ list_collections() "collection_list": [ { "collection_name": "product_docs", - "description": "Product manuals and release notes" + "description": "Product manuals and release notes", + "resource_id": "kb-example-1", + "create_time": 1788220800, + "update_time": 1788220860 }, { "collection_name": "support_faq", - "description": "Frequently asked support questions" + "description": "Frequently asked support questions", + "resource_id": "kb-example-2", + "create_time": 1788220800, + "update_time": 1788220860 } - ] + ], + "total_num": 2 } ``` @@ -309,7 +366,8 @@ list_collections() search_knowledge( query="How do I reset my password?", collection_name="support_faq", - limit=3, + resource_id="kb-example", + limit=10, doc_filter={ "op": "must", "field": "doc_id", @@ -320,9 +378,10 @@ search_knowledge( 参数: -- `query`(必填):检索问题。 -- `collection_name`(必填):要检索的知识库。 -- `limit`(可选):返回的最大切片数量,范围为 1–100,默认值为 `3`。 +- `query`(必填):检索问题,长度为 1–8000 个字符。 +- `collection_name`(可选):要检索的知识库。 +- `resource_id`(可选):知识库 ID。与 `collection_name` 至少提供一个;同时提供时优先使用 `resource_id`。 +- `limit`(可选):返回的最大切片数量,范围为 1–100,默认值为 `10`。 - `doc_filter`(可选):包含以下字段的对象: - `op`:`"must"` 表示包含匹配结果,`"must_not"` 表示排除匹配结果。 - `field`:要过滤的文档字段,例如 `"doc_id"`。 @@ -337,7 +396,23 @@ search_knowledge( "id": "chunk_001", "content": "Open Account Settings and select Reset Password.", "doc_id": "account_guide", - "doc_name": "Account Guide" + "doc_name": "Account Guide", + "title": "Account Guide", + "doc_type": "pdf", + "score": 0.85, + "rerank_score": 0.92, + "chunk_title": "Resetting your password", + "audio_start_time": null, + "audio_end_time": null, + "video_start_time": null, + "video_end_time": null, + "chunk_attachment": [ + { + "uuid": "image_1", + "caption": "Account Settings", + "type": "image" + } + ] } ] } @@ -346,6 +421,27 @@ search_knowledge( Viking 未提供文档元数据时,`doc_id` 和 `doc_name` 为 `null`。非空的 `doc_id` 可以直接传给 `get_doc`。 +切片还包括 `title`(文档标题)、`doc_type`(文档类型)、`score`(检索得分)、 +`rerank_score`(重排得分)、`chunk_title`(切片标题)、`audio_start_time`、 +`audio_end_time`、`video_start_time`、`video_end_time`(音视频起止时间)和 +`chunk_attachment`(附件列表)。附件包含 `uuid`、`caption` 和 `type`。 + +`image`、`doc-image` 和 `table` 类型附件的图片链接通过 MCP `ResourceLink` +返回,包含 `uri`、`name`、`description` 和 `mimeType`(`image/*`), +URL 不包含在 `chunk_attachment` 中。 + +图片链接内容块示例: + +```json +{ + "type": "resource_link", + "uri": "https://example.com/account-settings.png", + "name": "image_1", + "description": "Account Settings", + "mimeType": "image/*" +} +``` + ## MCP 客户端配置 下面是使用 `uvx` 和 Viking API Key 的 stdio 配置示例: @@ -357,7 +453,7 @@ Viking 未提供文档元数据时,`doc_id` 和 `doc_name` 为 `null`。非空 "command": "uvx", "args": [ "--from", - "mcp-server-knowledgebase>=0.2.1", + "mcp-server-knowledgebase>=0.2.2", "mcp-server-knowledgebase" ], "env": { diff --git a/server/mcp_server_knowledgebase/pyproject.toml b/server/mcp_server_knowledgebase/pyproject.toml index 6eb5309d..0174dc5c 100644 --- a/server/mcp_server_knowledgebase/pyproject.toml +++ b/server/mcp_server_knowledgebase/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "mcp-server-knowledgebase" -version = "0.2.1" +version = "0.2.2" description = "MCP server for Viking Knowledge Base Service" readme = "README.md" requires-python = ">=3.10" diff --git a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/config.py b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/config.py index 04002c26..d9376049 100644 --- a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/config.py +++ b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/config.py @@ -49,7 +49,7 @@ def load_config() -> KnowledgeBaseConfig: ak=ak, sk=sk, api_key=api_key, - project=os.environ.get("KNOWLEDGE_BASE_PROJECT", "default"), + project=os.getenv("KNOWLEDGE_BASE_PROJECT", "").strip() or "default", region=os.environ.get("KNOWLEDGE_BASE_REGION", "cn-north-1"), ) diff --git a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/models.py b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/models.py index 0a291162..c4c7ef0f 100644 --- a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/models.py +++ b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/models.py @@ -1,75 +1,127 @@ -from typing import Any, Literal, Optional, Union +"""MCP response fields and small upstream value conversions.""" +import json +from typing import Annotated, Literal, Optional, Union -from pydantic import BaseModel, ConfigDict, Field +from pydantic import BaseModel, Field, JsonValue, StringConstraints, field_validator +from pydantic.json_schema import SkipJsonSchema +NonBlank = Annotated[str, StringConstraints(strip_whitespace=True, min_length=1)] +Timestamp = Union[int, float, str] -class DocFilter(BaseModel): - """Filter applied to Viking Knowledge Base search results.""" - op: Literal["must", "must_not"] - field: str = Field(min_length=1) - conds: list[Any] = Field(min_length=1) +class DocFilter(BaseModel): + op: Literal["must", "must_not"] = Field(description='Use must to include matches or must_not to exclude them.') + field: NonBlank = Field(description='Document field to filter, such as doc_id.') + conds: list[JsonValue] = Field(min_length=1, description='Non-empty list of JSON values to match.') class AddDocumentResult(BaseModel): - collection_name: str - doc_id: str + collection_name: str = Field(description='Knowledge base collection name.') + doc_id: str = Field(description='Document ID; can be used with get_doc.') + resource_id: Optional[str] = Field(default=None, description='Knowledge base resource ID.') class DocumentStatus(BaseModel): - model_config = ConfigDict(extra="allow") - - process_status: Optional[int] = None - failed_code: Optional[Union[int, str]] = None - - -class DocumentInfo(BaseModel): - """Known document fields; extra upstream fields are preserved.""" - - model_config = ConfigDict(extra="allow") - - collection_name: Optional[str] = None - doc_id: Optional[str] = None - doc_name: Optional[str] = None - doc_type: Optional[str] = None - url: Optional[str] = None - add_type: Optional[str] = None - create_time: Optional[int] = None - update_time: Optional[int] = None - point_num: Optional[int] = None - status: Optional[DocumentStatus] = None + process_status: Optional[int] = Field(default=None, description='Processing status: 0 completed, 1 failed, 2 or 3 queued, 5 deleting, 6 processing.') + failed_code: Optional[Union[int, str]] = Field(default=None, description='Failure code returned by Viking, when available.') + + +class MediaChapter(BaseModel): + title: Optional[str] = Field(default=None, description='Chapter title.') + content: Optional[str] = Field(default=None, description='Content text.') + start_time: Optional[Timestamp] = Field(default=None, description='Chapter start time in the format and units returned by Viking.') + end_time: Optional[Timestamp] = Field(default=None, description='Chapter end time in the format and units returned by Viking.') + element_content: JsonValue = Field(default=None, description='Structured chapter content returned by Viking.') + + +class MediaOutline(BaseModel): + title: Optional[str] = Field(default=None, description='Media outline title.') + summary: Optional[str] = Field(default=None, description='Media outline summary.') + chapters: Optional[list[MediaChapter]] = Field(default=None, description='Media chapters.') + + +class DocumentSummary(BaseModel): + doc_id: Optional[str] = Field(default=None, description='Document ID; can be used with get_doc.') + doc_name: Optional[str] = Field(default=None, description='Document name.') + title: Optional[str] = Field(default=None, description='Document title.') + doc_type: Optional[str] = Field(default=None, description='Document format, such as pdf or markdown.') + status: Optional[DocumentStatus] = Field(default=None, description='Document processing status.') + point_num: Optional[int] = Field(default=None, description='Number of document chunks.') + brief_summary: Optional[str] = Field(default=None, description='Brief document summary.') + create_time: Optional[int] = Field(default=None, description='Creation timestamp returned by Viking.') + update_time: Optional[int] = Field(default=None, description='Last update timestamp returned by Viking.') + + +class DocumentInfo(DocumentSummary): + collection_name: Optional[str] = Field(default=None, description='Knowledge base collection name.') + add_type: Optional[str] = Field(default=None, description='Document import method, such as url.') + doc_summary: Optional[str] = Field(default=None, description='Document summary.') + meta: JsonValue = Field(default=None, description='Document metadata; valid JSON strings are parsed into JSON values.') + video_outline: Optional[MediaOutline] = Field(default=None, description='Video outline and chapters, when available.') + audio_outline: Optional[MediaOutline] = Field(default=None, description='Audio outline and chapters, when available.') + + @field_validator("meta", mode="before") + @classmethod + def parse_meta(cls, value): + if isinstance(value, str): + try: + return json.loads(value) + except (ValueError, RecursionError): + pass + return value class ListDocumentsResult(BaseModel): - collection_name: str - total_num: Optional[int] = None - count: int - doc_list: list[DocumentInfo] - has_more: bool - next_token: Optional[str] = None + collection_name: str = Field(description='Knowledge base collection name.') + total_num: Optional[int] = Field(default=None, description='Total number of documents reported by Viking; null if unavailable.') + count: int = Field(description='Number of documents in this page.') + doc_list: list[DocumentSummary] = Field(description='Document summaries in this page.') + has_more: bool = Field(description='Whether more documents are available.') + next_token: Optional[str] = Field(default=None, description='Cursor for the next page; pass to list_docs. Empty or null at the end.') -class CollectionInfoResult(BaseModel): - collection_name: str - description: str - status: int +class CollectionSummary(BaseModel): + collection_name: str = Field(description='Knowledge base collection name.') + description: str = Field(description='Collection description.') + resource_id: Optional[str] = Field(default=None, description='Knowledge base resource ID.') + create_time: Optional[int] = Field(default=None, description='Creation timestamp returned by Viking.') + update_time: Optional[int] = Field(default=None, description='Last update timestamp returned by Viking.') -class CollectionSummary(BaseModel): - collection_name: str - description: str +class CollectionInfoResult(CollectionSummary): + status: int = Field(description='First index build status: -1 pending, 0 building, 1 completed, 2 failed, 3 changing.') + doc_num: Optional[int] = Field(default=None, description='Number of documents in the collection.') class ListCollectionsResult(BaseModel): - collection_list: list[CollectionSummary] + collection_list: list[CollectionSummary] = Field(description='Collections in the configured project.') + total_num: Optional[int] = Field(default=None, description='Total number of collections reported by Viking; null if unavailable.') + + +class ChunkAttachment(BaseModel): + uuid: Optional[str] = Field(default=None, description='Attachment identifier.') + caption: Optional[str] = Field(default=None, description='Attachment caption.') + type: Optional[str] = Field(default=None, description='Attachment type; image, doc-image and table represent images.') + # Internal input for ResourceLink blocks only; never part of JSON output/schema. + link: SkipJsonSchema[Optional[str]] = Field(default=None, exclude=True) class SearchChunk(BaseModel): - id: str - content: str - doc_id: Optional[str] = None - doc_name: Optional[str] = None + id: str = Field(description='Chunk identifier.') + content: str = Field(description='Content text.') + doc_id: Optional[str] = Field(description='Document ID; can be used with get_doc.') + doc_name: Optional[str] = Field(description='Document name.') + title: Optional[str] = Field(default=None, description='Document title.') + doc_type: Optional[str] = Field(default=None, description='Document format, such as pdf or markdown.') + score: Optional[float] = Field(default=None, description='Retrieval score returned by Viking.') + rerank_score: Optional[float] = Field(default=None, description='Reranking score, when available.') + chunk_title: Optional[str] = Field(default=None, description='Chunk title.') + audio_start_time: Optional[Timestamp] = Field(default=None, description='Audio start time in the format and units returned by Viking.') + audio_end_time: Optional[Timestamp] = Field(default=None, description='Audio end time in the format and units returned by Viking.') + video_start_time: Optional[Timestamp] = Field(default=None, description='Video start time in the format and units returned by Viking.') + video_end_time: Optional[Timestamp] = Field(default=None, description='Video end time in the format and units returned by Viking.') + chunk_attachment: Optional[list[ChunkAttachment]] = Field(default=None, description='Attachment metadata (uuid, caption, type). Image URLs are returned separately as MCP ResourceLink blocks, not as link fields here.') class SearchKnowledgeResult(BaseModel): - result_list: list[SearchChunk] + result_list: list[SearchChunk] = Field(description='Matching knowledge chunks.') diff --git a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/server.py b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/server.py index 43820f08..42a251a0 100644 --- a/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/server.py +++ b/server/mcp_server_knowledgebase/src/mcp_server_knowledgebase/server.py @@ -7,20 +7,19 @@ from mcp.server import MCPServer from mcp.server.caching import CacheHint from mcp.server.mcpserver.exceptions import ToolError -from mcp.types import ToolAnnotations -from pydantic import Field +from mcp.types import CallToolResult, ResourceLink, TextContent, ToolAnnotations +from pydantic import BaseModel, Field, StringConstraints, ValidationError from mcp_server_knowledgebase.common.auth import prepare_request from mcp_server_knowledgebase.config import config from mcp_server_knowledgebase.models import ( AddDocumentResult, CollectionInfoResult, - CollectionSummary, DocFilter, DocumentInfo, ListCollectionsResult, ListDocumentsResult, - SearchChunk, + NonBlank, SearchKnowledgeResult, ) @@ -48,6 +47,23 @@ ) +def result_model(model: type[BaseModel], data: Any) -> BaseModel: + """Use declared fields and default Pydantic validation for upstream results.""" + try: + return model.model_validate(data) + except ValidationError: + # Validation errors contain upstream values; don't expose them verbatim. + raise ToolError("Invalid upstream response structure for " + model.__name__) from None + + +def _safe_error(message: object) -> str: + text = str(message) + for secret in (config.ak, config.sk, config.api_key): + if secret and secret in text: + return "Knowledge Base request failed (sensitive details omitted)" + return text + + def _transport_options(transport: str) -> Dict[str, Any]: """Build transport-specific options accepted by MCP SDK v2.""" if transport == "stdio": @@ -90,19 +106,20 @@ async def _call_kb(path: str, params: Dict[str, Any], tool_name: str) -> Any: try: result = await _request_knowledgebase(path, params) if result["code"] != 0: - raise ToolError(result["message"]) + raise ToolError(_safe_error(result.get("message", "Knowledge Base request failed"))) data = result.get("data") if not data: - raise ValueError(f"{tool_name} returned no data") + raise ToolError(f"{tool_name} returned no data") return data except ToolError as e: - logger.error(f"Error in {tool_name}: {str(e)}") + logger.error("Error in %s: %s", tool_name, _safe_error(str(e))) raise except Exception as e: - logger.error(f"Error in {tool_name}: {str(e)}") - raise ToolError(str(e)) from e + logger.error("Error in %s: %s", tool_name, type(e).__name__) + detail = "upstream timed out" if isinstance(e, TimeoutError) else "upstream request failed" + raise ToolError(f"{tool_name}: {detail} ({type(e).__name__})") from None @mcp.tool( @@ -114,14 +131,14 @@ async def _call_kb(path: str, params: Dict[str, Any], tool_name: str) -> Any: ) ) async def add_doc( - collection_name: Annotated[str, Field(min_length=1)], - add_type: Literal["url"], + collection_name: Annotated[str, Field(min_length=1, description="Target collection name in the configured project.")], + add_type: Annotated[Literal["url"], Field(description="Document import method; use url.")], doc_id: Annotated[ str, - Field(min_length=1, max_length=128, pattern=r"^[A-Za-z][A-Za-z0-9_]*$"), + Field(min_length=1, max_length=128, pattern=r"^[A-Za-z][A-Za-z0-9_]*$", description="Unique document ID, 1–128 letters, digits or underscores; must start with a letter."), ], - doc_name: Annotated[str, Field(min_length=1, max_length=256)], - doc_type: Literal[ + doc_name: Annotated[str, Field(min_length=1, max_length=256, description="Document name, 1–256 characters.")], + doc_type: Annotated[Literal[ "xlsx", "csv", "jsonl", @@ -132,27 +149,13 @@ async def add_doc( "markdown", "faq.xlsx", "pptx", - ], - url: Annotated[str, Field(min_length=1)], + ], Field(description="Document format matching the source file.")], + url: Annotated[str, Field(min_length=1, description="Document URL accessible to Viking for import.")], ) -> AddDocumentResult: - """ - Add a document to a collection in your project. - This tool allows you to add a document to a collection in your project by collection_name. - Args: - collection_name: the name of the knowledge base collection to add document to. - add_type: the type of the document to add. so far only support "url" now. so you must assign this parameter to "url". - doc_id: you should generate a unique doc_id based on user's given url and timestamp, the doc_id can only use English letters, numbers, and underscores , and must start with an English letter. It cannot be empty. - Length requirement: [1, 128], you can use a format like "mcp_server_auto_gen_doc_id_xxxxxxx. - doc_name: the name of the document to add. you can1 generate a unique doc_name based on user given url and timestamp. the length of doc_name must between 1 and 256. you can use a - format like "mcp_server_auto_gen_doc_name_xxxxxxx. - doc_type: the type of the document to add. for structured document, we support xlsx, csv,jsonl, for unstructured document, wu support txt, doc, docx, pdf, markdown, faq.xlsx, pptx". - you should judge the doc_type based on user's given url and judge if we support this doc type. if supported, assign this parameter. - url: the url of the document to add. user should give a valid url, we will add the doc to the collection. - - Returns: - collection_name: the name of the knowledge base collection. - doc_id: the doc_id of document user added to collection. + """Add a document by URL to a collection in the configured project. + Returns the collection name, document ID and available resource ID from Viking. + Use get_doc to check the document's processing status after submission. """ request_params = { "collection_name": collection_name, @@ -164,215 +167,174 @@ async def add_doc( "url": url, } - await _call_kb(doc_add_path, request_params, "add_doc") - return AddDocumentResult(collection_name=collection_name, doc_id=doc_id) + data = await _call_kb(doc_add_path, request_params, "add_doc") + return result_model(AddDocumentResult, data) + + +def _locator( + collection_name: Optional[str], resource_id: Optional[str], name_key: str +) -> dict[str, str]: + resource_id = resource_id.strip() if resource_id else "" + collection_name = collection_name.strip() if collection_name else "" + if resource_id: + return {"resource_id": resource_id} + if collection_name: + return {name_key: collection_name, "project": config.project} + raise ToolError("Provide a non-empty collection_name or resource_id") @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True)) async def get_doc( - collection_name: str, - doc_id: str, + collection_name: Optional[str] = Field(default=None, description="Collection name in the configured project. Provide this or resource_id."), + doc_id: NonBlank = Field(..., description="Document ID to look up; must not be blank."), + resource_id: Optional[str] = Field(default=None, description="Knowledge base ID. Takes precedence over collection_name when both are provided."), ) -> DocumentInfo: - """ - Get information about a document from your collection. - This tool allows you to get information about a document from your project by collection_name and doc_id. - Args: - collection_name: the name of the knowledge base collection to get document from. - doc_id: the doc_id of document user want to get information. - Returns: - collection_name: the name of the knowledge base collection. - doc_id: the doc_id of document user added to collection. - doc_name: the name of the document. - doc_type: the type of the document. - url: the url of the document. - add_type: the type how to add document. - create_time: the time when document added to collection. - update_time: the time when document updated. - point_num: The number of points extracted from the document. - status: the status of the document. the status struct has two fields: - - process_status: The processing status of the document. - 0 means the processing is completed, - 1 means the processing failed, - 2 or 3 means it is in queue, - 5 means it is being deleted, - and 6 means it is processing. - - failed_code: the status message of the document. - """ + """Get document metadata, processing status, summaries and media outlines. - request_params = { - "collection_name": collection_name, - "project": config.project, - "doc_id": doc_id, - } + Provide collection_name or resource_id; the ID takes precedence. - doc_info_data = await _call_kb(doc_info_path, request_params, "get_doc") - return DocumentInfo.model_validate(doc_info_data) + doc_id is required and trimmed. Project comes from server configuration. + Unavailable optional fields are null. JSON metadata strings are parsed where possible. + """ + params = _locator(collection_name, resource_id, "collection_name") + params["doc_id"] = doc_id + return result_model(DocumentInfo, await _call_kb(doc_info_path, params, "get_doc")) @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True)) async def list_docs( - collection_name: Annotated[str, Field(min_length=1)], - limit: Annotated[int, Field(ge=1, le=100)] = 100, - next_token: Optional[str] = None, + collection_name: Optional[str] = Field(default=None, description="Collection name in the configured project. Provide this or resource_id."), + limit: Annotated[int, Field(ge=1, le=100, description="Maximum documents per page (1–100); defaults to 50.")] = 50, + next_token: Optional[str] = Field(default=None, description="Cursor returned by the preceding list_docs call. Omit for the first page."), + resource_id: Optional[str] = Field(default=None, description="Knowledge base ID. Takes precedence over collection_name when both are provided."), ) -> ListDocumentsResult: - """List documents in a knowledge base collection using cursor pagination. + """List document summaries using cursor pagination (default 50, range 1–100). - Args: - collection_name: The knowledge base collection to list documents from. - limit: The maximum number of documents to return, from 1 to 100. - next_token: The opaque cursor returned by the previous request. Omit it - to retrieve the first page. - - Returns: - The documents in the current page and the cursor for the next page. - An empty next_token indicates that all documents have been returned. + Use collection_name or resource_id; ID takes precedence. Pass next_token + unchanged from the preceding page. Returns collection_name, total_num, count, has_more, next_token and doc_list + summaries. total_num is null when unavailable. Name-based requests use the + configured project. """ - request_params = { - "collection_name": collection_name, - "project": config.project, - "limit": limit, - } + params = _locator(collection_name, resource_id, "collection_name") + params["limit"] = limit if next_token is not None: - request_params["next_token"] = next_token - - list_data = await _call_kb(list_docs_path, request_params, "list_docs") - return ListDocumentsResult.model_validate(list_data) + params["next_token"] = next_token + data = await _call_kb(list_docs_path, params, "list_docs") + return result_model(ListDocumentsResult, data) @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True)) async def get_collection( - collection_name: str, + collection_name: Optional[str] = Field(default=None, description="Collection name in the configured project. Provide this or resource_id."), + resource_id: Optional[str] = Field(default=None, description="Knowledge base ID. Takes precedence over collection_name when both are provided."), ) -> CollectionInfoResult: - """ - Get information about a collection from your project. - This tool allows you to get information about a collection from your project by collection_name. - Args: - collection_name: the name of the knowledge base collection to get info for. - - Returns: - collection_name: the name of the knowledge base collection. - description: the description of the knowledge base collection. - status: the status of the knowledge base collection. - status: - -1: To be built - 0: Building - 1: Build completed - 2: Build failed - 3: Changing + """Get collection metadata and build status. - """ + Provide collection_name or resource_id; the ID takes precedence. Name-based + requests use the configured project. - request_params = { - "name": collection_name, - "project": config.project, - } - - collection_info = await _call_kb( - get_collections_path, request_params, "get_collection" - ) - return CollectionInfoResult( - collection_name=collection_info["collection_name"], - description=collection_info["description"], - status=collection_info["pipeline_list"][0]["index_list"][0]["status"], - ) + status is the first pipeline's first index status: -1 pending, 0 building, + 1 ready, 2 failed, 3 changing. Unknown integer states are preserved. + Returns name, description, status and available ID, counts and timestamps. + """ + params = _locator(collection_name, resource_id, "name") + data = await _call_kb(get_collections_path, params, "get_collection") + try: + status = data["pipeline_list"][0]["index_list"][0]["status"] + if status is None: + raise ValueError("missing status") + except (KeyError, IndexError, TypeError, ValueError) as exc: + raise ToolError("Invalid upstream response structure: missing first pipeline/index status") from exc + projected = dict(data, status=status) + if data.get("doc_num") is None: + for pipeline in data["pipeline_list"]: + stat = pipeline.get("pipeline_stat") if isinstance(pipeline, dict) else None + if isinstance(stat, dict) and stat.get("doc_num") is not None: + projected["doc_num"] = stat["doc_num"] + break + return result_model(CollectionInfoResult, projected) @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True)) async def list_collections() -> ListCollectionsResult: - """ - List all collections of the globally configured project from the Viking Knowledgebase service. - This tool allows you to list all collections in the Viking Knowledgebase service. - - Returns: - A list of collections in the project. - collection_name: the name of the knowledge base collection. - description: the description of the knowledge base collection. + """List all collections in the configured project. No input parameters are required. + Returns collection_list with name, description and available resource ID, + timestamps, plus the upstream total_num when supplied. """ - - request_params = { - "project": config.project, - } - - list_data = await _call_kb( - list_collections_path, request_params, "list_collections" + data = await _call_kb( + list_collections_path, + {"project": config.project, "brief": True}, + "list_collections", ) - collections = list_data["collection_list"] - - collection_list = [] - for collection in collections: - collection_list.append( - CollectionSummary( - collection_name=collection["collection_name"], - description=collection["description"], - ) - ) - - return ListCollectionsResult(collection_list=collection_list) + return result_model(ListCollectionsResult, data) @mcp.tool(annotations=ToolAnnotations(readOnlyHint=True, openWorldHint=True)) async def search_knowledge( - query: str, - collection_name: str, - limit: Annotated[int, Field(ge=1, le=100)] = 3, - doc_filter: Optional[DocFilter] = None, -) -> SearchKnowledgeResult: - """Search knowledge from the Viking Knowledgebase service And return Top limit related chunks of your query. - This tool allows you to search knowledge in provided collection based on the given query. - - Args: - query: the search query string. - limit: the maximum number of results to return (default: 3). - collection_name: the name of the knowledge base collection to search for. - doc_filter: the filter is used to filter search results(default: None), which is structured as a JSON object with - the following key components: - - 'op': (string, required) specifies the query operator that defines the filtering logic. Valid values are - 'must' and 'must_not', 'must' means results must satisfy the condition (inclusion filter),'must_not' means - results must not satisfy the condition (exclusion filter). - - 'field': (string, required) indicates the specific document field to apply the filter on (e.g., "doc_id"). - - 'conds': (array, required) contains the concrete values used for filtering. The data type - of elements in the array depends on the field. - - Returns: - A list of search results. - id: the id of the knowledge base chunk. - content: the content of the knowledge base chunk. - doc_id: the id of the document containing the chunk, when available. - doc_name: the name of the document containing the chunk, when available. + query: Annotated[ + str, + StringConstraints(strip_whitespace=True, min_length=1, max_length=8000), + Field(description="Search query, 1–8000 characters after trimming whitespace."), + ], + collection_name: Optional[str] = Field(default=None, description="Collection name in the configured project. Provide this or resource_id."), + limit: Annotated[int, Field(ge=1, le=100, description="Maximum chunks to return (1–100); defaults to 10.")] = 10, + doc_filter: Optional[DocFilter] = Field(default=None, description="Optional document filter to include or exclude matching field values."), + resource_id: Optional[str] = Field(default=None, description="Knowledge base ID. Takes precedence over collection_name when both are provided."), +) -> CallToolResult: + """Search for relevant chunks in a knowledge base by collection name or resource ID. + + Provide collection_name or resource_id; the ID takes precedence. Name-based + searches use the configured project. doc_filter includes or excludes documents. + Returns result_list with chunk text, document metadata, scores, media times + and chunk_attachment metadata (uuid, caption, type). Unavailable optional + fields are null. Image URLs for image, doc-image and table attachments are + returned separately as MCP ResourceLink content blocks with uri, name, + description and mimeType (image/*); chunk_attachment contains no link field. """ - - request_params = { - "query": query, - "limit": limit, - "name": collection_name, - "project": config.project, - } - - if doc_filter: - request_params['query_param'] = { - "doc_filter": doc_filter.model_dump(mode="json"), - } - - search_data = await _call_kb( - search_knowledge_path, request_params, "search_knowledge" - ) - chunks = search_data.get('result_list', []) - - search_result = [] - for chunk in chunks: - raw_doc_info = chunk.get("doc_info") - doc_info = raw_doc_info if isinstance(raw_doc_info, dict) else {} - search_result.append( - SearchChunk( - id=chunk["id"], - content=chunk["content"], - doc_id=doc_info.get("doc_id"), - doc_name=doc_info.get("doc_name"), - ) - ) - - return SearchKnowledgeResult(result_list=search_result) + params = _locator(collection_name, resource_id, "name") + params.update(query=query, limit=limit, post_processing={"get_attachment_link": True}) + if doc_filter is not None: + params["query_param"] = {"doc_filter": doc_filter.model_dump(mode="json")} + data = await _call_kb(search_knowledge_path, params, "search_knowledge") + if not isinstance(data, dict) or not isinstance(data.get("result_list"), list): + raise ToolError("Invalid upstream response structure: expected result_list") + chunks = [] + for chunk in data["result_list"]: + if not isinstance(chunk, dict): + raise ToolError("Invalid upstream response structure: expected a search chunk") + info = chunk.get("doc_info") + info = info if isinstance(info, dict) else {} + item = dict(chunk, doc_id=info.get("doc_id"), doc_name=info.get("doc_name")) + # Provenance always comes from doc_info, not similarly named chunk fields. + for name in ("title", "doc_type"): + item.pop(name, None) + if name in info: + item[name] = info[name] + chunks.append(item) + result = result_model(SearchKnowledgeResult, {"result_list": chunks}) + content: list[TextContent | ResourceLink] = [ + TextContent(type="text", text=result.model_dump_json()) + ] + # Process attachments as ResourceLink blocks. + for chunk in result.result_list: + for index, attachment in enumerate(chunk.chunk_attachment or []): + if attachment.type not in {"image", "doc-image", "table"}: + continue + if not attachment.link or not attachment.link.strip(): + continue + # Viking supplies no MIME type; do not guess the image format. + try: + content.append(ResourceLink( + type="resource_link", + uri=attachment.link, + name=attachment.uuid or f"{chunk.id}-image-{index + 1}", + description=attachment.caption or chunk.chunk_title, + mime_type="image/*", + )) + except ValidationError: + raise ToolError("Invalid upstream response structure: invalid image resource link") from None + return CallToolResult(content=content) def main(): diff --git a/server/mcp_server_knowledgebase/tests/test_tools.py b/server/mcp_server_knowledgebase/tests/test_tools.py index 26026e3b..149535f1 100644 --- a/server/mcp_server_knowledgebase/tests/test_tools.py +++ b/server/mcp_server_knowledgebase/tests/test_tools.py @@ -1,4 +1,5 @@ import asyncio +import json import logging import os from typing import Any @@ -6,7 +7,7 @@ import pytest from mcp.server.mcpserver.exceptions import ToolError -os.environ.setdefault("VIKING_API_KEY", "test-api-key") +os.environ["VIKING_API_KEY"] = "test-api-key" os.environ["KNOWLEDGE_BASE_PROJECT"] = "default" from mcp_server_knowledgebase import server @@ -51,7 +52,7 @@ async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: { "collection_name": "product-docs", "project": "default", - "limit": 100, + "limit": 50, }, ) ] @@ -98,29 +99,6 @@ async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: assert received_limit == limit -@pytest.mark.parametrize("limit", [0, 101]) -def test_list_docs_rejects_out_of_range_limits_before_request( - monkeypatch: pytest.MonkeyPatch, limit: int -) -> None: - request_was_sent = False - - async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: - nonlocal request_was_sent - request_was_sent = True - return empty_list_response() - - monkeypatch.setattr(server, "_request_knowledgebase", fake_request) - - with pytest.raises(ToolError, match="limit"): - asyncio.run( - server.mcp.call_tool( - "list_docs", {"collection_name": "product-docs", "limit": limit} - ) - ) - - assert request_was_sent is False - - @pytest.mark.parametrize( ("response", "expected_message"), [ @@ -159,7 +137,7 @@ async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: ) -def test_list_docs_preserves_optional_and_extension_fields( +def test_list_docs_preserves_whitelisted_optional_fields( monkeypatch: pytest.MonkeyPatch, ) -> None: async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: @@ -192,10 +170,12 @@ async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: assert result["total_num"] is None assert document["status"]["failed_code"] == 7001 assert document["brief_summary"] == "A short guide." - assert document["total_tokens"] == 345 + assert "total_tokens" not in document + assert "url" not in document + assert "add_type" not in document -def test_get_doc_accepts_numeric_failed_code_and_preserves_extensions( +def test_get_doc_accepts_numeric_failed_code_and_preserves_summary( monkeypatch: pytest.MonkeyPatch, ) -> None: requests: list[tuple[str, dict[str, Any]]] = [] @@ -252,3 +232,66 @@ async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: assert "Error in get_collection: permission denied" in caplog.text assert "Error in search_knowledge" not in caplog.text + + +def test_search_returns_image_resource_links(monkeypatch: pytest.MonkeyPatch) -> None: + attachments = [ + {"uuid": kind, "type": kind, "link": f"https://example.com/{kind}.png?signature=original"} + for kind in ("image", "doc-image", "table") + ] + attachments.extend([ + {"type": "image"}, + {"type": "table", "link": " "}, + {"type": "video", "link": "https://example.com/video.mp4"}, + ]) + requests = [] + + async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: + requests.append(params) + return {"code": 0, "data": {"result_list": [{ + "id": "chunk-1", "content": "Chart description", + "doc_info": {"doc_id": "doc-1", "doc_name": "Guide"}, + "chunk_attachment": attachments, + }]}} + + monkeypatch.setattr(server, "_request_knowledgebase", fake_request) + doc_filter = {"op": "must", "field": "doc_id", "conds": ["doc-1"]} + result = asyncio.run(server.mcp.call_tool("search_knowledge", { + "resource_id": "kb-1", "collection_name": "ignored", "query": "chart", + "doc_filter": doc_filter, + })) + + assert requests == [{ + "resource_id": "kb-1", "query": "chart", "limit": 10, + "query_param": {"doc_filter": doc_filter}, + "post_processing": {"get_attachment_link": True}, + }] + assert not result.is_error + assert [block.type for block in result.content] == ["text"] + ["resource_link"] * 3 + assert [str(block.uri) for block in result.content[1:]] == [ + attachment["link"] for attachment in attachments[:3] + ] + assert all(block.mime_type == "image/*" for block in result.content[1:]) + assert result.structured_content is None + chunk = json.loads(result.content[0].text)["result_list"][0] + assert chunk["doc_id"] == "doc-1" and chunk["doc_name"] == "Guide" + assert "doc_info" not in chunk + assert all("link" not in attachment for attachment in chunk["chunk_attachment"]) + + +def test_add_doc_returns_actual_response_without_project(monkeypatch: pytest.MonkeyPatch) -> None: + async def fake_request(path: str, params: dict[str, Any]) -> dict[str, Any]: + return {"code": 0, "data": { + "collection_name": "actual-name", "doc_id": "actual-doc", + "resource_id": "kb-1", "project": "default", + }} + + monkeypatch.setattr(server, "_request_knowledgebase", fake_request) + result = call_tool("add_doc", { + "collection_name": "requested-name", "doc_id": "requested_doc", + "doc_name": "Guide", "doc_type": "pdf", "add_type": "url", + "url": "https://example.com/guide.pdf", + }) + assert result == { + "collection_name": "actual-name", "doc_id": "actual-doc", "resource_id": "kb-1", + } diff --git a/server/mcp_server_knowledgebase/uv.lock b/server/mcp_server_knowledgebase/uv.lock index 34fb06b6..fd7b065f 100644 --- a/server/mcp_server_knowledgebase/uv.lock +++ b/server/mcp_server_knowledgebase/uv.lock @@ -756,7 +756,7 @@ cli = [ [[package]] name = "mcp-server-knowledgebase" -version = "0.2.1" +version = "0.2.2" source = { editable = "." } dependencies = [ { name = "aiohttp" },