From fb44eb98f2962a9cbbfe455eb46da57e499ebb35 Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 18:48:36 +0800 Subject: [PATCH 1/6] feat(agent): let read_file return images per image processing mode - read_file now accepts raster images (.jpg/.jpeg/.png/.gif/.bmp/.webp): native multimodal mode attaches the compressed image data as a kira_image_ref media part on the tool message; VLM description mode returns the transcribed description via the default VLM, reusing the shared md5-keyed description cache - both paths honor bot_config.image_compression via the new compress_image_file() path-based helper in image_compression - ToolResult gains media_refs and build_content() so tool messages can carry multimodal image content (resolved per LLM request, kept as compact paths in persisted history) - SVG and other binary/media extensions remain blocked; write_file and edit_file are unchanged --- core/agent/func_tool_manager.py | 1 + core/agent/tool.py | 15 ++ core/plugin/builtin_plugins/agent/main.py | 103 +++++++++++- core/utils/image_compression.py | 31 ++++ tests/test_agent_plugin.py | 194 ++++++++++++++++++++++ tests/test_image_compression.py | 35 ++++ 6 files changed, 373 insertions(+), 6 deletions(-) diff --git a/core/agent/func_tool_manager.py b/core/agent/func_tool_manager.py index bf0c7f34..59c23988 100644 --- a/core/agent/func_tool_manager.py +++ b/core/agent/func_tool_manager.py @@ -137,6 +137,7 @@ async def execute_tool(self, event: KiraMessageBatchEvent, resp: LLMResponse, to # Save tool results content = await tool_result_obj.assemble_result() + content = tool_result_obj.build_content(content) tool_logger.info(f"tool_result: {content}") resp.tool_results.append({ "role": "tool", diff --git a/core/agent/tool.py b/core/agent/tool.py index 9e986d81..dcea4381 100644 --- a/core/agent/tool.py +++ b/core/agent/tool.py @@ -60,6 +60,11 @@ class ToolResult: attachments: list[Union[Image, Record, File]] = field(default_factory=list) + # Provider-independent image references (``kira_image_ref`` parts) embedded + # into the tool message content, resolved to image parts per LLM request and + # kept as compact paths in persisted history. + media_refs: list[dict] = field(default_factory=list) + result_str: str = field(default="", init=False, repr=False) async def assemble_result(self): @@ -101,3 +106,13 @@ async def assemble_result(self): ) self.result_str = "".join(res_text) return self.result_str + + def build_content(self, text: str) -> Union[str, list]: + """Return the tool message content, embedding media refs when present. + + Without media refs the content stays a plain string; with them it becomes + a multimodal content part list the provider layer understands. + """ + if not self.media_refs: + return text + return [{"type": "text", "text": text}, *self.media_refs] diff --git a/core/plugin/builtin_plugins/agent/main.py b/core/plugin/builtin_plugins/agent/main.py index 2d0bed3e..73537467 100644 --- a/core/plugin/builtin_plugins/agent/main.py +++ b/core/plugin/builtin_plugins/agent/main.py @@ -14,9 +14,13 @@ from core.plugin import BasePlugin, logger, on, Priority, register from core.chat import KiraMessageBatchEvent, MessageChain -from core.chat.message_elements import Text +from core.chat.message_elements import Image, Text from core.provider import LLMRequest +from core.agent.tool import ToolResult +from core.utils.common_utils import desc_img +from core.utils.image_compression import compress_image_file +from core.utils.media_refs import store_session_media from core.utils.path_utils import get_config_path, get_data_path, get_root_path PLUGIN_ID = "agent" @@ -51,6 +55,11 @@ '.dll', '.so', '.dylib', '.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx', '.iso', '.img', '.dmg'} +# Raster image formats read_file can serve to the LLM (compression and VLM +# both handle these). SVG stays blocked: it is text-based and neither the +# image compressor nor vision models treat it as a raster image. +readable_image_extensions = {'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.webp'} + ALL_TOOL_NAMES = [ "read_file", "write_file", "edit_file", "list_files", "grep", "search_files", "exec", "manage_background_exec", @@ -603,18 +612,18 @@ def _find_files(path: Path, pattern: str) -> list[Path]: @register.tool( "read_file", - "Read a plain text file (txt, html, py, etc..) in allowed read paths", + "Read a plain text file (txt, html, py, etc..) or an image file (jpg, png, gif, etc..) in allowed read paths. Images follow the configured image processing mode: returned as raw image data attached to this result in native multimodal mode, or as a text description in VLM description mode.", { "type": "object", "properties": { "path": {"type": "string", "description": "File path, must start with an allowed path prefix"}, - "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1"}, - "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200"}, + "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1. Ignored for image files."}, + "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200. Ignored for image files."}, }, "required": ["path"] } ) - async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = 1, limit: int = 200) -> str: + async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = 1, limit: int = 200) -> str | ToolResult: if not self._is_file_session_allowed(event.sid): return "Permission denied: current session not allowed to access local files" @@ -631,7 +640,9 @@ async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = ext = Path(path).suffix.lower() if ext in blocked_extensions: - return "Multimedia and binary files are not allowed" + if ext not in readable_image_extensions: + return "Multimedia and binary files are not allowed" + return await self._read_image_file(event, path) try: abs_path = self._resolve_path(path) @@ -653,6 +664,86 @@ async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = except Exception as e: return f"[Failed to read file: {e}]" + async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str | ToolResult: + """Read an image file according to the image processing mode. + + Native multimodal mode attaches the (possibly compressed) image data as + a media reference so the main LLM sees it directly; VLM description mode + returns the transcribed description instead. Both paths honor the global + image compression settings, mirroring how incoming chat images are + bounded before reaching a model. + """ + abs_path = self._resolve_path(path) + if not abs_path.is_file(): + return f"[Failed to read file: file not found: {path}]" + + capabilities = self.ctx.get_session_capabilities(event.sid) if self.ctx else {} + image_recognition = capabilities.get("image_recognition") if isinstance(capabilities, dict) else None + if not isinstance(image_recognition, dict): + image_recognition = {} + mode = image_recognition.get("mode", "vlm_description") + + compression_config = ( + self.ctx.config.get_config("bot_config.image_compression", {}) if self.ctx else {} + ) + try: + image_path, mime = await compress_image_file(abs_path, compression_config) + except Exception as e: + return f"[Failed to read file: {e}]" + + if mode == "native": + message_id = event.messages[-1].message_id if event.messages else "read_file" + media_ref = await store_session_media( + Image(image=str(image_path), mime=mime), event.sid, message_id + ) + return ToolResult( + text=f"Image file: {path} ({mime}). The raw image data is attached as an image content part.", + media_refs=[media_ref], + ) + + desc = await self._describe_image_file(image_path, mime, image_recognition) + if not desc: + return f"[Image description unavailable, file_path: {path}]" + return f"[Image {desc}, file_path: {path}]" + + async def _describe_image_file(self, image_path: Path, mime: str, image_recognition: dict) -> str: + """Transcribe an image file with the default VLM, reusing the shared description cache.""" + if not image_recognition.get("enabled", True): + return "" + + desc_cache = None + md5 = None + try: + from core.message_manager import ImageDescCache + + desc_cache = ImageDescCache(self.ctx.db) + md5 = await Image(image=str(image_path), mime=mime).hash_image() + cached_desc = await desc_cache.get(md5) + if cached_desc: + return cached_desc + except Exception as e: + logger.warning(f"Failed to read image desc cache for read_file: {e}") + desc_cache = None + + desc = "" + try: + vlm_client = self.ctx.provider_mgr.get_default_vlm() + desc = await desc_img( + client=vlm_client, + image=Image(image=str(image_path), mime=mime), + prompt=str(image_recognition.get("desc_prompt", "") or "").strip() or None, + lang=self.ctx.get_lang(), + ) + except Exception as e: + logger.error(f"Failed to describe image file for read_file: {e}") + + if desc and desc_cache and md5: + try: + await desc_cache.set(md5, desc) + except Exception as e: + logger.warning(f"Failed to cache read_file image desc: {e}") + return desc + @register.tool( "write_file", "Write content to a plain text file in allowed write paths. Creates the file if it doesn't exist, overwrites if it does.", diff --git a/core/utils/image_compression.py b/core/utils/image_compression.py index 294e7da5..11c77832 100644 --- a/core/utils/image_compression.py +++ b/core/utils/image_compression.py @@ -3,6 +3,7 @@ from __future__ import annotations import asyncio +import mimetypes import uuid from pathlib import Path @@ -97,6 +98,36 @@ def _compress_image_sync( return None +async def compress_image_file( + source_path: Path, + config: dict | None, +) -> tuple[Path, str]: + """Compress an image file on disk according to the compression settings. + + Returns ``(path, mime)`` where the path is a bounded temporary copy when the + settings require one, or the original path when compression is disabled or + unnecessary. Works on plain paths so tools that read images directly from + disk can share the settings with the chat message flow. + """ + def guess_mime() -> str: + return mimetypes.guess_type(source_path.name)[0] or "image/jpeg" + + enabled, max_size, quality, min_file_size_bytes = _compression_options(config) + if not enabled: + return source_path, guess_mime() + + compressed = await asyncio.to_thread( + _compress_image_sync, + source_path, + max_size, + quality, + min_file_size_bytes, + ) + if compressed is None: + return source_path, guess_mime() + return compressed + + async def compress_image_element( media: Image | Sticker, config: dict | None, diff --git a/tests/test_agent_plugin.py b/tests/test_agent_plugin.py index 1ab93d50..fe509836 100644 --- a/tests/test_agent_plugin.py +++ b/tests/test_agent_plugin.py @@ -3,9 +3,12 @@ from unittest.mock import AsyncMock, Mock, patch import pytest +from PIL import Image as PILImage +from core.agent.tool import ToolResult from core.plugin.builtin_plugins.agent import main as agent_main from core.plugin.builtin_plugins.agent.main import BackgroundExecTask, AgentPlugin +from core.utils import image_compression, media_refs @pytest.fixture(autouse=True) @@ -101,6 +104,33 @@ def agent_plugin(): return plugin +def make_image_ctx(mode="vlm_description", enabled=True, compression=None, db=None): + """Build a minimal plugin ctx for image read tests. + + Compression defaults to disabled so tests opt in explicitly; the db mock + backs the shared VLM description cache used by read_file. + """ + config_values = { + "bot_config.image_compression": compression or {"enabled": False}, + } + return SimpleNamespace( + get_session_capabilities=Mock( + return_value={"image_recognition": {"mode": mode, "enabled": enabled}} + ), + config=SimpleNamespace( + get_config=lambda key, default=None: config_values.get(key, default) + ), + provider_mgr=SimpleNamespace(get_default_vlm=Mock(return_value=object())), + get_lang=Mock(return_value="en"), + db=db + or SimpleNamespace( + get_image_desc_cache=AsyncMock(return_value=None), + add_image_desc_cache=AsyncMock(), + update_image_desc_cache=AsyncMock(), + ), + ) + + @pytest.mark.anyio async def test_initialize_loads_exec_timeouts(): plugin = AgentPlugin( @@ -552,3 +582,167 @@ async def run_taskkill(*_, **__): assert process.killed is False assert process.returncode == 0 assert process.wait_count == 1 + + +def _png_file(tmp_path, name="photo.png", size=(4, 4)): + target = tmp_path / name + PILImage.new("RGB", size, "red").save(target) + return target + + +@pytest.mark.anyio +async def test_read_file_returns_vlm_description_for_images(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + db = SimpleNamespace( + get_image_desc_cache=AsyncMock(return_value=None), + add_image_desc_cache=AsyncMock(), + update_image_desc_cache=AsyncMock(), + ) + agent_plugin.ctx = make_image_ctx(mode="vlm_description", db=db) + target = _png_file(tmp_path) + + with patch.object(agent_main, "desc_img", AsyncMock(return_value="a red square")) as desc_img: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Image a red square, file_path: {normalized}]" + desc_img.assert_awaited_once() + db.add_image_desc_cache.assert_awaited_once() + + +@pytest.mark.anyio +async def test_read_file_reuses_cached_image_description(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + db = SimpleNamespace( + get_image_desc_cache=AsyncMock( + return_value={"description": "cached desc", "count": 1} + ), + add_image_desc_cache=AsyncMock(), + update_image_desc_cache=AsyncMock(), + ) + agent_plugin.ctx = make_image_ctx(mode="vlm_description", db=db) + target = _png_file(tmp_path) + + with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + assert "cached desc" in result + desc_img.assert_not_awaited() + db.add_image_desc_cache.assert_not_awaited() + + +@pytest.mark.anyio +async def test_read_file_reports_unavailable_description_when_recognition_disabled( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_image_ctx(mode="vlm_description", enabled=False) + target = _png_file(tmp_path) + + with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Image description unavailable, file_path: {normalized}]" + desc_img.assert_not_awaited() + + +@pytest.mark.anyio +async def test_read_file_attaches_raw_image_in_native_mode(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_image_ctx(mode="native") + target = _png_file(tmp_path) + event = SimpleNamespace( + sid="test:dm:1", messages=[SimpleNamespace(message_id="msg-1")] + ) + + with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: + result = await agent_plugin.read_file(event, str(target)) + + assert isinstance(result, ToolResult) + assert "Image file:" in result.text + assert len(result.media_refs) == 1 + ref = result.media_refs[0] + assert ref["type"] == media_refs.MEDIA_REF_TYPE + assert ref["mime_type"] == "image/png" + stored = tmp_path / ref["path"] + assert stored.read_bytes() == target.read_bytes() + desc_img.assert_not_awaited() + + +@pytest.mark.anyio +async def test_read_file_compresses_image_in_native_mode_per_settings( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + monkeypatch.setattr(image_compression, "get_data_path", lambda: tmp_path) + monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_image_ctx( + mode="native", + compression={"enabled": True, "max_size": 100, "quality": 80, "min_file_size_mb": 0}, + ) + target = _png_file(tmp_path, name="big.png", size=(2000, 1000)) + event = SimpleNamespace( + sid="test:dm:1", messages=[SimpleNamespace(message_id="msg-1")] + ) + + result = await agent_plugin.read_file(event, str(target)) + + assert isinstance(result, ToolResult) + stored = tmp_path / result.media_refs[0]["path"] + with PILImage.open(stored) as compressed: + assert max(compressed.size) == 100 + + +@pytest.mark.anyio +async def test_read_file_reports_missing_image_file(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_image_ctx() + missing = tmp_path / "missing.png" + + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(missing)) + + normalized = str(missing).replace("\\", "/") + assert result == f"[Failed to read file: file not found: {normalized}]" + + +@pytest.mark.anyio +async def test_read_file_still_blocks_svg_and_non_image_media(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_image_ctx() + + svg = tmp_path / "icon.svg" + svg.write_text("", encoding="utf-8") + audio = tmp_path / "audio.mp3" + audio.write_bytes(b"\x00" * 8) + + svg_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(svg)) + audio_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(audio)) + + assert svg_result == "Multimedia and binary files are not allowed" + assert audio_result == "Multimedia and binary files are not allowed" + + +@pytest.mark.anyio +async def test_tool_result_build_content_embeds_media_refs(): + ref = { + "type": "kira_image_ref", + "path": "session_media/a/b.png", + "mime_type": "image/png", + } + plain = ToolResult(text="plain result") + media = ToolResult(text="image result", media_refs=[ref]) + + assert plain.build_content(await plain.assemble_result()) == "plain result" + assert media.build_content(await media.assemble_result()) == [ + {"type": "text", "text": "image result"}, + ref, + ] diff --git a/tests/test_image_compression.py b/tests/test_image_compression.py index 43938a6e..e8dea6fd 100644 --- a/tests/test_image_compression.py +++ b/tests/test_image_compression.py @@ -74,3 +74,38 @@ async def test_compression_skips_images_over_pixel_limit(tmp_path, monkeypatch): assert changed is False assert image.file == str(source_path) + + +@pytest.mark.asyncio +async def test_compress_image_file_scales_oversized_image(tmp_path, monkeypatch): + monkeypatch.setattr(image_compression, "get_data_path", lambda: tmp_path) + source_path = tmp_path / "big.jpg" + PILImage.new("RGB", (2000, 1000), "red").save(source_path) + + path, mime = await image_compression.compress_image_file( + source_path, + {"enabled": True, "max_size": 500, "quality": 80, "min_file_size_mb": 0}, + ) + + assert mime == "image/jpeg" + assert path != source_path + with PILImage.open(path) as compressed: + assert max(compressed.size) == 500 + + +@pytest.mark.asyncio +async def test_compress_image_file_keeps_original_when_disabled_or_within_limits(tmp_path): + source_path = tmp_path / "small.png" + PILImage.new("RGB", (10, 10), "red").save(source_path) + + disabled_path, disabled_mime = await image_compression.compress_image_file( + source_path, + {"enabled": False}, + ) + assert (disabled_path, disabled_mime) == (source_path, "image/png") + + within_limits_path, within_limits_mime = await image_compression.compress_image_file( + source_path, + {"enabled": True, "max_size": 1280, "quality": 95, "min_file_size_mb": 1}, + ) + assert (within_limits_path, within_limits_mime) == (source_path, "image/png") From 3efc37d50576613c6425c2ce53e275c81f83a59a Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 19:13:08 +0800 Subject: [PATCH 2/6] feat(agent): let read_file transcribe audio files via ASR - read_file now routes audio extensions (.mp3/.wav/.ogg/.flac/.aac/.m4a/ .amr/.silk/.slk/.slac, mirroring the adapters' voice-recognition set) to a new _read_audio_file handler - transcription reuses the incoming-voice pipeline: default STT model via Record + speech_to_text, honoring the session's stt.enabled toggle - failures degrade to a path-only note so the turn can still forward the file; .m4a is routed even though it is not in blocked_extensions (routing now checks readable media sets before the block list) - offset/limit are documented as ignored for image and audio files --- core/plugin/builtin_plugins/agent/main.py | 55 ++++++++-- tests/test_agent_plugin.py | 118 +++++++++++++++++++--- 2 files changed, 149 insertions(+), 24 deletions(-) diff --git a/core/plugin/builtin_plugins/agent/main.py b/core/plugin/builtin_plugins/agent/main.py index 73537467..00897979 100644 --- a/core/plugin/builtin_plugins/agent/main.py +++ b/core/plugin/builtin_plugins/agent/main.py @@ -14,11 +14,11 @@ from core.plugin import BasePlugin, logger, on, Priority, register from core.chat import KiraMessageBatchEvent, MessageChain -from core.chat.message_elements import Image, Text +from core.chat.message_elements import Image, Record, Text from core.provider import LLMRequest from core.agent.tool import ToolResult -from core.utils.common_utils import desc_img +from core.utils.common_utils import desc_img, speech_to_text from core.utils.image_compression import compress_image_file from core.utils.media_refs import store_session_media from core.utils.path_utils import get_config_path, get_data_path, get_root_path @@ -60,6 +60,12 @@ # image compressor nor vision models treat it as a raster image. readable_image_extensions = {'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.webp'} +# Audio formats read_file transcribes via the default STT model, mirroring the +# voice-recognition set of the chat adapters (which also accepts .m4a, so it is +# routed here even though it is not in blocked_extensions). +readable_audio_extensions = {'.mp3', '.wav', '.ogg', '.flac', '.aac', '.m4a', + '.amr', '.silk', '.slk', '.slac'} + ALL_TOOL_NAMES = [ "read_file", "write_file", "edit_file", "list_files", "grep", "search_files", "exec", "manage_background_exec", @@ -612,13 +618,13 @@ def _find_files(path: Path, pattern: str) -> list[Path]: @register.tool( "read_file", - "Read a plain text file (txt, html, py, etc..) or an image file (jpg, png, gif, etc..) in allowed read paths. Images follow the configured image processing mode: returned as raw image data attached to this result in native multimodal mode, or as a text description in VLM description mode.", + "Read a plain text file (txt, html, py, etc..), an image file (jpg, png, gif, etc..) or an audio file (mp3, wav, etc..) in allowed read paths. Images follow the configured image processing mode: returned as raw image data attached to this result in native multimodal mode, or as a text description in VLM description mode. Audio files are transcribed to text via speech recognition.", { "type": "object", "properties": { "path": {"type": "string", "description": "File path, must start with an allowed path prefix"}, - "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1. Ignored for image files."}, - "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200. Ignored for image files."}, + "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1. Ignored for image and audio files."}, + "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200. Ignored for image and audio files."}, }, "required": ["path"] } @@ -639,10 +645,12 @@ async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = return f"Permission denied: Path must start with one of: {', '.join(self.allowed_read_paths)}" ext = Path(path).suffix.lower() - if ext in blocked_extensions: - if ext not in readable_image_extensions: - return "Multimedia and binary files are not allowed" + if ext in readable_image_extensions: return await self._read_image_file(event, path) + if ext in readable_audio_extensions: + return await self._read_audio_file(event, path) + if ext in blocked_extensions: + return "Multimedia and binary files are not allowed" try: abs_path = self._resolve_path(path) @@ -744,6 +752,37 @@ async def _describe_image_file(self, image_path: Path, mime: str, image_recognit logger.warning(f"Failed to cache read_file image desc: {e}") return desc + async def _read_audio_file(self, event: KiraMessageBatchEvent, path: str) -> str: + """Transcribe an audio file with the default STT model. + + Mirrors how incoming voice records are recognized: the session's + ``stt`` capability toggle is honored, and an unavailable STT model or a + failed transcription degrades to returning the file path only, so the + turn can still forward the file to the user. + """ + abs_path = self._resolve_path(path) + if not abs_path.is_file(): + return f"[Failed to read file: file not found: {path}]" + + capabilities = self.ctx.get_session_capabilities(event.sid) if self.ctx else {} + stt_caps = capabilities.get("stt") if isinstance(capabilities, dict) else None + if not isinstance(stt_caps, dict): + stt_caps = {} + if not stt_caps.get("enabled", True): + return f"[Record (speech recognition disabled), file_path: {path}]" + + try: + stt_client = self.ctx.provider_mgr.get_default_stt() + transcript = await speech_to_text( + client=stt_client, record=Record(record=str(abs_path)) + ) + except Exception as e: + logger.error(f"Failed to transcribe audio file for read_file: {e}") + transcript = "" + if not transcript: + return f"[Record (speech recognition unavailable), file_path: {path}]" + return f"[Record {transcript}, file_path: {path}]" + @register.tool( "write_file", "Write content to a plain text file in allowed write paths. Creates the file if it doesn't exist, overwrites if it does.", diff --git a/tests/test_agent_plugin.py b/tests/test_agent_plugin.py index fe509836..2da93c7c 100644 --- a/tests/test_agent_plugin.py +++ b/tests/test_agent_plugin.py @@ -104,8 +104,8 @@ def agent_plugin(): return plugin -def make_image_ctx(mode="vlm_description", enabled=True, compression=None, db=None): - """Build a minimal plugin ctx for image read tests. +def make_media_ctx(mode="vlm_description", enabled=True, stt_enabled=True, compression=None, db=None): + """Build a minimal plugin ctx for image/audio read tests. Compression defaults to disabled so tests opt in explicitly; the db mock backs the shared VLM description cache used by read_file. @@ -115,12 +115,18 @@ def make_image_ctx(mode="vlm_description", enabled=True, compression=None, db=No } return SimpleNamespace( get_session_capabilities=Mock( - return_value={"image_recognition": {"mode": mode, "enabled": enabled}} + return_value={ + "image_recognition": {"mode": mode, "enabled": enabled}, + "stt": {"enabled": stt_enabled}, + } ), config=SimpleNamespace( get_config=lambda key, default=None: config_values.get(key, default) ), - provider_mgr=SimpleNamespace(get_default_vlm=Mock(return_value=object())), + provider_mgr=SimpleNamespace( + get_default_vlm=Mock(return_value=object()), + get_default_stt=Mock(return_value=object()), + ), get_lang=Mock(return_value="en"), db=db or SimpleNamespace( @@ -599,7 +605,7 @@ async def test_read_file_returns_vlm_description_for_images(agent_plugin, tmp_pa add_image_desc_cache=AsyncMock(), update_image_desc_cache=AsyncMock(), ) - agent_plugin.ctx = make_image_ctx(mode="vlm_description", db=db) + agent_plugin.ctx = make_media_ctx(mode="vlm_description", db=db) target = _png_file(tmp_path) with patch.object(agent_main, "desc_img", AsyncMock(return_value="a red square")) as desc_img: @@ -622,7 +628,7 @@ async def test_read_file_reuses_cached_image_description(agent_plugin, tmp_path, add_image_desc_cache=AsyncMock(), update_image_desc_cache=AsyncMock(), ) - agent_plugin.ctx = make_image_ctx(mode="vlm_description", db=db) + agent_plugin.ctx = make_media_ctx(mode="vlm_description", db=db) target = _png_file(tmp_path) with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: @@ -639,7 +645,7 @@ async def test_read_file_reports_unavailable_description_when_recognition_disabl ): monkeypatch.setattr(agent_main, "restricted_paths", []) agent_plugin.allowed_read_paths = (str(tmp_path),) - agent_plugin.ctx = make_image_ctx(mode="vlm_description", enabled=False) + agent_plugin.ctx = make_media_ctx(mode="vlm_description", enabled=False) target = _png_file(tmp_path) with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: @@ -655,7 +661,7 @@ async def test_read_file_attaches_raw_image_in_native_mode(agent_plugin, tmp_pat monkeypatch.setattr(agent_main, "restricted_paths", []) monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) agent_plugin.allowed_read_paths = (str(tmp_path),) - agent_plugin.ctx = make_image_ctx(mode="native") + agent_plugin.ctx = make_media_ctx(mode="native") target = _png_file(tmp_path) event = SimpleNamespace( sid="test:dm:1", messages=[SimpleNamespace(message_id="msg-1")] @@ -683,7 +689,7 @@ async def test_read_file_compresses_image_in_native_mode_per_settings( monkeypatch.setattr(image_compression, "get_data_path", lambda: tmp_path) monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) agent_plugin.allowed_read_paths = (str(tmp_path),) - agent_plugin.ctx = make_image_ctx( + agent_plugin.ctx = make_media_ctx( mode="native", compression={"enabled": True, "max_size": 100, "quality": 80, "min_file_size_mb": 0}, ) @@ -704,7 +710,7 @@ async def test_read_file_compresses_image_in_native_mode_per_settings( async def test_read_file_reports_missing_image_file(agent_plugin, tmp_path, monkeypatch): monkeypatch.setattr(agent_main, "restricted_paths", []) agent_plugin.allowed_read_paths = (str(tmp_path),) - agent_plugin.ctx = make_image_ctx() + agent_plugin.ctx = make_media_ctx() missing = tmp_path / "missing.png" result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(missing)) @@ -714,21 +720,101 @@ async def test_read_file_reports_missing_image_file(agent_plugin, tmp_path, monk @pytest.mark.anyio -async def test_read_file_still_blocks_svg_and_non_image_media(agent_plugin, tmp_path, monkeypatch): +async def test_read_file_still_blocks_svg_and_non_media_files(agent_plugin, tmp_path, monkeypatch): monkeypatch.setattr(agent_main, "restricted_paths", []) agent_plugin.allowed_read_paths = (str(tmp_path),) - agent_plugin.ctx = make_image_ctx() + agent_plugin.ctx = make_media_ctx() svg = tmp_path / "icon.svg" svg.write_text("", encoding="utf-8") - audio = tmp_path / "audio.mp3" - audio.write_bytes(b"\x00" * 8) + video = tmp_path / "video.mp4" + video.write_bytes(b"\x00" * 8) + archive = tmp_path / "bundle.zip" + archive.write_bytes(b"\x00" * 8) svg_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(svg)) - audio_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(audio)) + video_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(video)) + archive_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(archive)) assert svg_result == "Multimedia and binary files are not allowed" - assert audio_result == "Multimedia and binary files are not allowed" + assert video_result == "Multimedia and binary files are not allowed" + assert archive_result == "Multimedia and binary files are not allowed" + + +@pytest.mark.anyio +async def test_read_file_transcribes_audio_files(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx() + target = tmp_path / "voice.mp3" + target.write_bytes(b"ID3 fake mp3 payload") + + with patch.object( + agent_main, "speech_to_text", AsyncMock(return_value="hello world") + ) as speech_to_text_mock: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Record hello world, file_path: {normalized}]" + speech_to_text_mock.assert_awaited_once() + record = speech_to_text_mock.await_args.kwargs["record"] + assert record.file_type == "path" + + +@pytest.mark.anyio +async def test_read_file_transcribes_m4a_files_outside_blocked_list( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx() + target = tmp_path / "voice.m4a" + target.write_bytes(b"\x00" * 8) + + with patch.object( + agent_main, "speech_to_text", AsyncMock(return_value="m4a transcript") + ): + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Record m4a transcript, file_path: {normalized}]" + + +@pytest.mark.anyio +async def test_read_file_reports_disabled_speech_recognition(agent_plugin, tmp_path, monkeypatch): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx(stt_enabled=False) + target = tmp_path / "voice.mp3" + target.write_bytes(b"ID3 fake mp3 payload") + + with patch.object(agent_main, "speech_to_text", AsyncMock()) as speech_to_text_mock: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Record (speech recognition disabled), file_path: {normalized}]" + speech_to_text_mock.assert_not_awaited() + + +@pytest.mark.anyio +async def test_read_file_reports_unavailable_transcription_on_stt_failure( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx() + agent_plugin.ctx.provider_mgr.get_default_stt = Mock( + side_effect=RuntimeError("no stt model") + ) + target = tmp_path / "voice.mp3" + target.write_bytes(b"ID3 fake mp3 payload") + + with patch.object(agent_main, "speech_to_text", AsyncMock()) as speech_to_text_mock: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Record (speech recognition unavailable), file_path: {normalized}]" + speech_to_text_mock.assert_not_awaited() @pytest.mark.anyio From 28b1a46477380d487826e0de61ce0ea5f60fb470 Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 19:34:44 +0800 Subject: [PATCH 3/6] refactor(agent): route tool media into a user message for provider compatibility Some providers reject image parts inside tool-role messages, so tool results carrying kira_image_ref parts are no longer sent as-is: the agent executor now collapses the tool message back to its plain text and relocates the media refs into a single user message appended right after the tool results, mirroring how incoming images reach the model. Both the request and the persisted history receive the same shape, so replays stay consistent with what the model saw; media-ref plumbing (ToolResult.media_refs, build_content, store_session_media) is unchanged and only the mounting point moved. --- core/agent/agent_executor.py | 59 ++++++++++- core/agent/tool.py | 13 ++- core/plugin/builtin_plugins/agent/main.py | 2 +- tests/test_agent_executor.py | 115 ++++++++++++++++++++++ 4 files changed, 182 insertions(+), 7 deletions(-) create mode 100644 tests/test_agent_executor.py diff --git a/core/agent/agent_executor.py b/core/agent/agent_executor.py index 282a26b3..b9e1db9b 100644 --- a/core/agent/agent_executor.py +++ b/core/agent/agent_executor.py @@ -14,6 +14,7 @@ from core.chat.message_utils import KiraExceptionEvent from core.plugin.plugin_handlers import event_handler_reg, EventType from core.prompt_manager import Prompt +from core.utils.media_refs import MEDIA_REF_TYPE if TYPE_CHECKING: from core.chat.message_utils import KiraMessageBatchEvent @@ -48,6 +49,62 @@ def __init__(self, tool_manager: FuncToolManager, tool_set: Optional[ToolSet] = self.tool_manager = tool_manager self.tool_set = tool_set + @staticmethod + def _build_tool_messages(tool_results: list[dict]) -> list[OpenAIMessage]: + """Build the tool messages of one agent step from raw tool results. + + Media references embedded by tools (``kira_image_ref`` parts) are + relocated into a single user message appended after the tool results: + some providers reject image parts inside tool-role messages, while a + user message carrying images is supported by every vision-capable + provider. Both the request and the persisted history receive the same + relocated shape, so replays stay consistent with what the model saw. + """ + messages: list[OpenAIMessage] = [] + media_parts: list[dict] = [] + media_tool_names: list[str] = [] + for result in tool_results: + content = result.get("content") + if isinstance(content, list): + text_parts = [ + part for part in content + if not (isinstance(part, dict) and part.get("type") == MEDIA_REF_TYPE) + ] + refs = [ + part for part in content + if isinstance(part, dict) and part.get("type") == MEDIA_REF_TYPE + ] + if refs: + media_parts.extend(refs) + tool_name = result.get("name") or "unknown_tool" + if tool_name not in media_tool_names: + media_tool_names.append(tool_name) + # Refs were the only non-text parts, so joining the rest + # always restores the original plain-text tool output. + tool_content = "".join( + part.get("text", "") if isinstance(part, dict) else str(part) + for part in text_parts + ) + messages.append(OpenAIMessage( + role="tool", + tool_call_id=result.get("tool_call_id"), + name=result.get("name"), + content=tool_content, + )) + continue + messages.append(OpenAIMessage(**result)) + + if media_parts: + joined_names = ", ".join(media_tool_names) + messages.append(OpenAIMessage( + role="user", + content=[ + {"type": "text", "text": f"Media returned by tool call(s): {joined_names}"}, + *media_parts, + ], + )) + return messages + async def run( self, ctx: AgentExecutionContext, @@ -225,7 +282,7 @@ async def run( ) request.messages.append(msg) ctx.new_messages.append(msg) - tool_msgs = [OpenAIMessage(**r) for r in llm_resp.tool_results] + tool_msgs = self._build_tool_messages(llm_resp.tool_results) request.messages.extend(tool_msgs) ctx.new_messages.extend(tool_msgs) diff --git a/core/agent/tool.py b/core/agent/tool.py index dcea4381..884b8a21 100644 --- a/core/agent/tool.py +++ b/core/agent/tool.py @@ -60,9 +60,11 @@ class ToolResult: attachments: list[Union[Image, Record, File]] = field(default_factory=list) - # Provider-independent image references (``kira_image_ref`` parts) embedded - # into the tool message content, resolved to image parts per LLM request and - # kept as compact paths in persisted history. + # Provider-independent image references (``kira_image_ref`` parts) attached + # to the tool result. The agent executor relocates them into a user message + # following the tool results (some providers reject image parts in + # tool-role messages); refs stay compact paths in persisted history and are + # resolved to image parts per LLM request. media_refs: list[dict] = field(default_factory=list) result_str: str = field(default="", init=False, repr=False) @@ -108,10 +110,11 @@ async def assemble_result(self): return self.result_str def build_content(self, text: str) -> Union[str, list]: - """Return the tool message content, embedding media refs when present. + """Return the tool result content, embedding media refs when present. Without media refs the content stays a plain string; with them it becomes - a multimodal content part list the provider layer understands. + a content part list the agent executor splits into the tool message text + and a following user message carrying the media. """ if not self.media_refs: return text diff --git a/core/plugin/builtin_plugins/agent/main.py b/core/plugin/builtin_plugins/agent/main.py index 00897979..94eb8887 100644 --- a/core/plugin/builtin_plugins/agent/main.py +++ b/core/plugin/builtin_plugins/agent/main.py @@ -705,7 +705,7 @@ async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str Image(image=str(image_path), mime=mime), event.sid, message_id ) return ToolResult( - text=f"Image file: {path} ({mime}). The raw image data is attached as an image content part.", + text=f"Image file: {path} ({mime}). The raw image data is attached in the message below.", media_refs=[media_ref], ) diff --git a/tests/test_agent_executor.py b/tests/test_agent_executor.py new file mode 100644 index 00000000..4980f0ad --- /dev/null +++ b/tests/test_agent_executor.py @@ -0,0 +1,115 @@ +import pytest + +from core.agent.agent_executor import AgentExecutor +from core.utils.media_refs import MEDIA_REF_TYPE + + +def _image_ref(path: str = "session_media/a/b.png") -> dict: + return {"type": MEDIA_REF_TYPE, "path": path, "mime_type": "image/png", "detail": "high"} + + +def test_build_tool_messages_keeps_plain_results_unchanged(): + results = [ + {"role": "tool", "tool_call_id": "call-1", "name": "grep", "content": "no matches"}, + {"role": "tool", "tool_call_id": "call-2", "name": "read_file", "content": "hello"}, + ] + + messages = AgentExecutor._build_tool_messages(results) + + assert [m.role for m in messages] == ["tool", "tool"] + assert messages[0].content == "no matches" + assert messages[1].content == "hello" + + +def test_build_tool_messages_relocates_media_refs_into_user_message(): + ref = _image_ref() + results = [ + { + "role": "tool", + "tool_call_id": "call-1", + "name": "read_file", + "content": [ + {"type": "text", "text": "Image file: data/files/photo.png"}, + ref, + ], + }, + ] + + messages = AgentExecutor._build_tool_messages(results) + + assert [m.role for m in messages] == ["tool", "user"] + tool_message, user_message = messages + # The tool message collapses back to its plain-text output. + assert tool_message.tool_call_id == "call-1" + assert tool_message.content == "Image file: data/files/photo.png" + # The media rides in a user message right after the tool results. + assert user_message.content == [ + {"type": "text", "text": "Media returned by tool call(s): read_file"}, + ref, + ] + + +def test_build_tool_messages_groups_media_from_multiple_tools_into_one_user_message(): + results = [ + { + "role": "tool", + "tool_call_id": "call-1", + "name": "read_file", + "content": [{"type": "text", "text": "first"}, _image_ref("session_media/a/1.png")], + }, + {"role": "tool", "tool_call_id": "call-2", "name": "grep", "content": "plain"}, + { + "role": "tool", + "tool_call_id": "call-3", + "name": "read_file", + "content": [{"type": "text", "text": "second"}, _image_ref("session_media/a/2.png")], + }, + ] + + messages = AgentExecutor._build_tool_messages(results) + + assert [m.role for m in messages] == ["tool", "tool", "tool", "user"] + assert messages[0].content == "first" + assert messages[2].content == "second" + user_message = messages[3] + assert user_message.content[0] == { + "type": "text", + "text": "Media returned by tool call(s): read_file", + } + assert [part["path"] for part in user_message.content[1:]] == [ + "session_media/a/1.png", + "session_media/a/2.png", + ] + + +@pytest.mark.anyio +async def test_tool_media_user_message_is_resolved_for_the_provider(tmp_path, monkeypatch): + from core.utils import media_refs + + monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) + stored = tmp_path / "session_media" / "a" / "b.png" + stored.parent.mkdir(parents=True, exist_ok=True) + stored.write_bytes(b"\x89PNG\r\n\x1a\nfake-png-bytes") + results = [ + { + "role": "tool", + "tool_call_id": "call-1", + "name": "read_file", + "content": [ + {"type": "text", "text": "Image file: data/files/photo.png"}, + _image_ref(), + ], + }, + ] + + messages = AgentExecutor._build_tool_messages(results) + request_messages = [m.to_dict() for m in messages] + resolved = await media_refs.resolve_media_references(request_messages) + + tool_content = resolved[0]["content"] + assert isinstance(tool_content, str) + assert tool_content == "Image file: data/files/photo.png" + user_parts = resolved[1]["content"] + assert user_parts[0] == {"type": "text", "text": "Media returned by tool call(s): read_file"} + assert user_parts[1]["type"] == "image_url" + assert user_parts[1]["image_url"]["url"].startswith("data:image/png;base64,") From 0624fe9af74eddc9f6f9d3e815718fbf6112453b Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 19:48:24 +0800 Subject: [PATCH 4/6] fix(prompts): tell the agent image descriptions are already complete The message annotation for [Image ... file_path: ...] only discouraged using list_files to find the path, so models still called read_file on already-described images (the VLM-description cache makes such a call return the same text). Both the CN and EN templates now state that the description is the complete recognition result and need not be re-viewed with read_file or similar tools. --- core/prompts/agent_tmpl.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/core/prompts/agent_tmpl.py b/core/prompts/agent_tmpl.py index bd6999f2..3cf491fb 100644 --- a/core/prompts/agent_tmpl.py +++ b/core/prompts/agent_tmpl.py @@ -122,7 +122,7 @@ [At all] # at全体成员消息 [Reply message_id/message_content] # message_id 为用户回复的消息的ID 或者 message_content 为引用的消息内容 [Poke 用户xxx戳了戳/捏了捏你的xxx] # 不要认为这是冒犯或真实的对话,这是社交平台的戳一戳互动提示,表示对方在轻轻提醒或调侃你。请理解为轻松、友好的互动 -[Image image_description file_path: xxx] # 用户发送的图片消息,通常情况下你无需使用工具如list_files等来获取图片路径的相关信息 +[Image image_description file_path: xxx] # 用户发送的图片消息,描述即完整的识图结果:通常情况下你无需使用list_files等工具获取图片路径的相关信息,也无需通过read_file等工具查看该路径 """ @@ -257,7 +257,7 @@ [At all] # message mentioning all members [Reply message_id/message_content] # ID or quoted message content [Poke User xxx poked/nudged you] # a friendly social-platform interaction, not an insult or literal dialogue -[Image image_description file_path: xxx] # an image message; normally no tool is needed to obtain the image path +[Image image_description file_path: xxx] # an image message; the description is the complete recognition result: normally no tool such as list_files is needed for image path information, and there is no need to view the path with read_file or similar tools """, "accounts": """\ From 6debd8d59af150910c2b2d928133c167805ad890 Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 20:30:25 +0800 Subject: [PATCH 5/6] fix(agent): harden path gate against symlinks and clean compressed temps - _is_path_allowed now resolves both the candidate and each allowed root to their real location before the containment check, so a symlink planted inside an allowed directory can no longer alias a read or write out of the configured roots; the gate covers every file tool (read_file text/media/audio, list_files, grep, search_files, write_file, edit_file) - _read_image_file removes the compressed copy from data/temp once its last consumer finishes (session-media copy in native mode, description in VLM mode); the original file and the cached description are unaffected --- core/plugin/builtin_plugins/agent/main.py | 60 ++++++++++++++++------- tests/test_agent_plugin.py | 59 ++++++++++++++++++++++ 2 files changed, 101 insertions(+), 18 deletions(-) diff --git a/core/plugin/builtin_plugins/agent/main.py b/core/plugin/builtin_plugins/agent/main.py index 94eb8887..56f4dc16 100644 --- a/core/plugin/builtin_plugins/agent/main.py +++ b/core/plugin/builtin_plugins/agent/main.py @@ -566,13 +566,27 @@ def _on_background_exec_done(self, task_id: str, session: str, task: asyncio.Tas notice_task.add_done_callback(self._background_notice_tasks.discard) def _is_path_allowed(self, path: str, allowed_prefixes: tuple) -> bool: - """Check if path starts with an allowed prefix directory.""" + """Check a normalized path against the allowed roots with symlinks resolved. + + Both the candidate and each allowed root are resolved to their real + location before the containment check, so a symlink planted inside an + allowed directory cannot alias a read or write out of the configured + roots. Not-yet-existing write targets resolve as far as their existing + ancestors, which is sufficient here. + """ + try: + candidate = self._resolve_path(path).resolve() + except (OSError, RuntimeError, ValueError): + return False for prefix in allowed_prefixes: - prefix = self._normalize_path(prefix) - if prefix is None: + normalized = self._normalize_path(prefix) + if normalized is None: + continue + try: + root = self._resolve_path(normalized).resolve() + except (OSError, RuntimeError, ValueError): continue - prefix = prefix.rstrip('/') - if path == prefix or path.startswith(prefix + '/'): + if candidate == root or root in candidate.parents: return True return False @@ -699,20 +713,30 @@ async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str except Exception as e: return f"[Failed to read file: {e}]" - if mode == "native": - message_id = event.messages[-1].message_id if event.messages else "read_file" - media_ref = await store_session_media( - Image(image=str(image_path), mime=mime), event.sid, message_id - ) - return ToolResult( - text=f"Image file: {path} ({mime}). The raw image data is attached in the message below.", - media_refs=[media_ref], - ) + try: + if mode == "native": + message_id = event.messages[-1].message_id if event.messages else "read_file" + media_ref = await store_session_media( + Image(image=str(image_path), mime=mime), event.sid, message_id + ) + return ToolResult( + text=f"Image file: {path} ({mime}). The raw image data is attached in the message below.", + media_refs=[media_ref], + ) - desc = await self._describe_image_file(image_path, mime, image_recognition) - if not desc: - return f"[Image description unavailable, file_path: {path}]" - return f"[Image {desc}, file_path: {path}]" + desc = await self._describe_image_file(image_path, mime, image_recognition) + if not desc: + return f"[Image description unavailable, file_path: {path}]" + return f"[Image {desc}, file_path: {path}]" + finally: + if image_path != abs_path: + # The compressed copy in data/temp is only ever consumed inside + # this call (the native branch copies it into session media, the + # VLM branch turns it into a description), so it can go now. + try: + await asyncio.to_thread(image_path.unlink, missing_ok=True) + except OSError as e: + logger.warning(f"Failed to remove compressed temp image {image_path.name}: {e}") async def _describe_image_file(self, image_path: Path, mime: str, image_recognition: dict) -> str: """Transcribe an image file with the default VLM, reusing the shared description cache.""" diff --git a/tests/test_agent_plugin.py b/tests/test_agent_plugin.py index 2da93c7c..a07508cc 100644 --- a/tests/test_agent_plugin.py +++ b/tests/test_agent_plugin.py @@ -704,6 +704,38 @@ async def test_read_file_compresses_image_in_native_mode_per_settings( stored = tmp_path / result.media_refs[0]["path"] with PILImage.open(stored) as compressed: assert max(compressed.size) == 100 + _assert_no_temp_leftovers(tmp_path) + + +@pytest.mark.anyio +async def test_read_file_cleans_compressed_temp_in_vlm_mode( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + monkeypatch.setattr(image_compression, "get_data_path", lambda: tmp_path) + monkeypatch.setattr(media_refs, "get_data_path", lambda: tmp_path) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx( + mode="vlm_description", + db=SimpleNamespace( + get_image_desc_cache=AsyncMock(return_value=None), + add_image_desc_cache=AsyncMock(), + update_image_desc_cache=AsyncMock(), + ), + compression={"enabled": True, "max_size": 100, "quality": 80, "min_file_size_mb": 0}, + ) + target = _png_file(tmp_path, name="big.png", size=(2000, 1000)) + + with patch.object(agent_main, "desc_img", AsyncMock(return_value="a big red image")): + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + assert result == f"[Image a big red image, file_path: {str(target).replace(chr(92), '/')}]" + _assert_no_temp_leftovers(tmp_path) + + +def _assert_no_temp_leftovers(tmp_path): + temp_dir = tmp_path / "temp" + assert not temp_dir.exists() or not [p for p in temp_dir.iterdir() if p.is_file()] @pytest.mark.anyio @@ -719,6 +751,33 @@ async def test_read_file_reports_missing_image_file(agent_plugin, tmp_path, monk assert result == f"[Failed to read file: file not found: {normalized}]" +@pytest.mark.anyio +async def test_read_file_rejects_symlink_escaping_allowed_roots( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + allowed = tmp_path / "files" + allowed.mkdir() + secret_dir = tmp_path / "secret" + secret_dir.mkdir() + secret = secret_dir / "note.txt" + secret.write_text("top secret", encoding="utf-8") + link = allowed / "link.txt" + try: + link.symlink_to(secret) + except (OSError, NotImplementedError): + pytest.skip("symlink creation requires privileges on this platform") + agent_plugin.allowed_read_paths = (str(allowed),) + + inside = allowed / "real.txt" + inside.write_text("hello", encoding="utf-8") + denied = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(link)) + allowed_result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(inside)) + + assert denied == f"Permission denied: Path must start with one of: {', '.join(agent_plugin.allowed_read_paths)}" + assert allowed_result == "hello" + + @pytest.mark.anyio async def test_read_file_still_blocks_svg_and_non_media_files(agent_plugin, tmp_path, monkeypatch): monkeypatch.setattr(agent_main, "restricted_paths", []) From b12a28a882c3fc0153ddd1626413f6f651ccfb8b Mon Sep 17 00:00:00 2001 From: Caleb Date: Sun, 20 Sep 2026 22:15:21 +0800 Subject: [PATCH 6/6] fix(agent): distinguish recognition-disabled from VLM-failure in read_file Image reads in VLM description mode previously collapsed both cases into one generic "description unavailable" note. The enabled check moves from _describe_image_file to its caller so read_file can return distinct notes: "(recognition disabled)" when image recognition is off, and "(description unavailable)" when the VLM is missing or fails, mirroring the audio branch's two-level wording. The shared description cache is untouched. --- core/plugin/builtin_plugins/agent/main.py | 12 ++++++++---- tests/test_agent_plugin.py | 24 +++++++++++++++++++++-- 2 files changed, 30 insertions(+), 6 deletions(-) diff --git a/core/plugin/builtin_plugins/agent/main.py b/core/plugin/builtin_plugins/agent/main.py index 56f4dc16..8ee749c0 100644 --- a/core/plugin/builtin_plugins/agent/main.py +++ b/core/plugin/builtin_plugins/agent/main.py @@ -724,9 +724,12 @@ async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str media_refs=[media_ref], ) + if not image_recognition.get("enabled", True): + return f"[Image (recognition disabled), file_path: {path}]" + desc = await self._describe_image_file(image_path, mime, image_recognition) if not desc: - return f"[Image description unavailable, file_path: {path}]" + return f"[Image (description unavailable), file_path: {path}]" return f"[Image {desc}, file_path: {path}]" finally: if image_path != abs_path: @@ -739,10 +742,11 @@ async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str logger.warning(f"Failed to remove compressed temp image {image_path.name}: {e}") async def _describe_image_file(self, image_path: Path, mime: str, image_recognition: dict) -> str: - """Transcribe an image file with the default VLM, reusing the shared description cache.""" - if not image_recognition.get("enabled", True): - return "" + """Transcribe an image file with the default VLM, reusing the shared description cache. + An empty return means the VLM produced no description; the caller tells + recognition-disabled and VLM-failure apart on its own. + """ desc_cache = None md5 = None try: diff --git a/tests/test_agent_plugin.py b/tests/test_agent_plugin.py index a07508cc..64376d9d 100644 --- a/tests/test_agent_plugin.py +++ b/tests/test_agent_plugin.py @@ -640,7 +640,7 @@ async def test_read_file_reuses_cached_image_description(agent_plugin, tmp_path, @pytest.mark.anyio -async def test_read_file_reports_unavailable_description_when_recognition_disabled( +async def test_read_file_reports_recognition_disabled_for_images( agent_plugin, tmp_path, monkeypatch ): monkeypatch.setattr(agent_main, "restricted_paths", []) @@ -652,7 +652,27 @@ async def test_read_file_reports_unavailable_description_when_recognition_disabl result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) normalized = str(target).replace("\\", "/") - assert result == f"[Image description unavailable, file_path: {normalized}]" + assert result == f"[Image (recognition disabled), file_path: {normalized}]" + desc_img.assert_not_awaited() + + +@pytest.mark.anyio +async def test_read_file_reports_unavailable_description_when_vlm_fails( + agent_plugin, tmp_path, monkeypatch +): + monkeypatch.setattr(agent_main, "restricted_paths", []) + agent_plugin.allowed_read_paths = (str(tmp_path),) + agent_plugin.ctx = make_media_ctx(mode="vlm_description") + agent_plugin.ctx.provider_mgr.get_default_vlm = Mock( + side_effect=RuntimeError("no vlm model") + ) + target = _png_file(tmp_path) + + with patch.object(agent_main, "desc_img", AsyncMock()) as desc_img: + result = await agent_plugin.read_file(SimpleNamespace(sid="test:dm:1"), str(target)) + + normalized = str(target).replace("\\", "/") + assert result == f"[Image (description unavailable), file_path: {normalized}]" desc_img.assert_not_awaited()