Repository navigation
feat(agent): let read_file return images and transcribe audio #324
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
fb44eb9
3efc37d
28b1a46
0624fe9
6debd8d
b12a28a
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -14,9 +14,13 @@ | |
|
|
||
| from core.plugin import BasePlugin, logger, on, Priority, register | ||
| from core.chat import KiraMessageBatchEvent, MessageChain | ||
| from core.chat.message_elements import Text | ||
| from core.chat.message_elements import Image, Record, Text | ||
| from core.provider import LLMRequest | ||
| from core.agent.tool import ToolResult | ||
|
|
||
| from core.utils.common_utils import desc_img, speech_to_text | ||
| from core.utils.image_compression import compress_image_file | ||
| from core.utils.media_refs import store_session_media | ||
| from core.utils.path_utils import get_config_path, get_data_path, get_root_path | ||
|
|
||
| PLUGIN_ID = "agent" | ||
|
|
@@ -51,6 +55,17 @@ | |
| '.dll', '.so', '.dylib', '.pdf', '.doc', '.docx', '.xls', '.xlsx', | ||
| '.ppt', '.pptx', '.iso', '.img', '.dmg'} | ||
|
|
||
| # Raster image formats read_file can serve to the LLM (compression and VLM | ||
| # both handle these). SVG stays blocked: it is text-based and neither the | ||
| # image compressor nor vision models treat it as a raster image. | ||
| readable_image_extensions = {'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.webp'} | ||
|
|
||
| # Audio formats read_file transcribes via the default STT model, mirroring the | ||
| # voice-recognition set of the chat adapters (which also accepts .m4a, so it is | ||
| # routed here even though it is not in blocked_extensions). | ||
| readable_audio_extensions = {'.mp3', '.wav', '.ogg', '.flac', '.aac', '.m4a', | ||
| '.amr', '.silk', '.slk', '.slac'} | ||
|
|
||
| ALL_TOOL_NAMES = [ | ||
| "read_file", "write_file", "edit_file", "list_files", "grep", "search_files", | ||
| "exec", "manage_background_exec", | ||
|
|
@@ -551,13 +566,27 @@ def _on_background_exec_done(self, task_id: str, session: str, task: asyncio.Tas | |
| notice_task.add_done_callback(self._background_notice_tasks.discard) | ||
|
|
||
| def _is_path_allowed(self, path: str, allowed_prefixes: tuple) -> bool: | ||
| """Check if path starts with an allowed prefix directory.""" | ||
| """Check a normalized path against the allowed roots with symlinks resolved. | ||
|
|
||
| Both the candidate and each allowed root are resolved to their real | ||
| location before the containment check, so a symlink planted inside an | ||
| allowed directory cannot alias a read or write out of the configured | ||
| roots. Not-yet-existing write targets resolve as far as their existing | ||
| ancestors, which is sufficient here. | ||
| """ | ||
| try: | ||
| candidate = self._resolve_path(path).resolve() | ||
| except (OSError, RuntimeError, ValueError): | ||
| return False | ||
| for prefix in allowed_prefixes: | ||
| prefix = self._normalize_path(prefix) | ||
| if prefix is None: | ||
| normalized = self._normalize_path(prefix) | ||
| if normalized is None: | ||
| continue | ||
| prefix = prefix.rstrip('/') | ||
| if path == prefix or path.startswith(prefix + '/'): | ||
| try: | ||
| root = self._resolve_path(normalized).resolve() | ||
| except (OSError, RuntimeError, ValueError): | ||
| continue | ||
| if candidate == root or root in candidate.parents: | ||
| return True | ||
| return False | ||
|
|
||
|
|
@@ -603,18 +632,18 @@ def _find_files(path: Path, pattern: str) -> list[Path]: | |
|
|
||
| @register.tool( | ||
| "read_file", | ||
| "Read a plain text file (txt, html, py, etc..) in allowed read paths", | ||
| "Read a plain text file (txt, html, py, etc..), an image file (jpg, png, gif, etc..) or an audio file (mp3, wav, etc..) in allowed read paths. Images follow the configured image processing mode: returned as raw image data attached to this result in native multimodal mode, or as a text description in VLM description mode. Audio files are transcribed to text via speech recognition.", | ||
| { | ||
| "type": "object", | ||
| "properties": { | ||
| "path": {"type": "string", "description": "File path, must start with an allowed path prefix"}, | ||
| "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1"}, | ||
| "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200"}, | ||
| "offset": {"type": "integer", "description": "Which line to start reading, defaults to 1. Ignored for image and audio files."}, | ||
| "limit": {"type": "integer", "description": "Maximum lines to read, defaults to 200. Ignored for image and audio files."}, | ||
| }, | ||
| "required": ["path"] | ||
| } | ||
| ) | ||
| async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = 1, limit: int = 200) -> str: | ||
| async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = 1, limit: int = 200) -> str | ToolResult: | ||
| if not self._is_file_session_allowed(event.sid): | ||
| return "Permission denied: current session not allowed to access local files" | ||
|
|
||
|
|
@@ -630,6 +659,10 @@ async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = | |
| return f"Permission denied: Path must start with one of: {', '.join(self.allowed_read_paths)}" | ||
|
|
||
| ext = Path(path).suffix.lower() | ||
| if ext in readable_image_extensions: | ||
| return await self._read_image_file(event, path) | ||
| if ext in readable_audio_extensions: | ||
| return await self._read_audio_file(event, path) | ||
|
coderabbitai[bot] marked this conversation as resolved.
|
||
| if ext in blocked_extensions: | ||
| return "Multimedia and binary files are not allowed" | ||
|
|
||
|
|
@@ -653,6 +686,131 @@ async def read_file(self, event: KiraMessageBatchEvent, path: str, offset: int = | |
| except Exception as e: | ||
| return f"[Failed to read file: {e}]" | ||
|
|
||
| async def _read_image_file(self, event: KiraMessageBatchEvent, path: str) -> str | ToolResult: | ||
| """Read an image file according to the image processing mode. | ||
|
|
||
| Native multimodal mode attaches the (possibly compressed) image data as | ||
| a media reference so the main LLM sees it directly; VLM description mode | ||
| returns the transcribed description instead. Both paths honor the global | ||
| image compression settings, mirroring how incoming chat images are | ||
| bounded before reaching a model. | ||
| """ | ||
| abs_path = self._resolve_path(path) | ||
| if not abs_path.is_file(): | ||
| return f"[Failed to read file: file not found: {path}]" | ||
|
|
||
| capabilities = self.ctx.get_session_capabilities(event.sid) if self.ctx else {} | ||
| image_recognition = capabilities.get("image_recognition") if isinstance(capabilities, dict) else None | ||
| if not isinstance(image_recognition, dict): | ||
| image_recognition = {} | ||
| mode = image_recognition.get("mode", "vlm_description") | ||
|
|
||
| compression_config = ( | ||
| self.ctx.config.get_config("bot_config.image_compression", {}) if self.ctx else {} | ||
| ) | ||
| try: | ||
| image_path, mime = await compress_image_file(abs_path, compression_config) | ||
|
coderabbitai[bot] marked this conversation as resolved.
|
||
| except Exception as e: | ||
| return f"[Failed to read file: {e}]" | ||
|
|
||
| try: | ||
| if mode == "native": | ||
| message_id = event.messages[-1].message_id if event.messages else "read_file" | ||
| media_ref = await store_session_media( | ||
| Image(image=str(image_path), mime=mime), event.sid, message_id | ||
| ) | ||
| return ToolResult( | ||
| text=f"Image file: {path} ({mime}). The raw image data is attached in the message below.", | ||
| media_refs=[media_ref], | ||
| ) | ||
|
|
||
| if not image_recognition.get("enabled", True): | ||
| return f"[Image (recognition disabled), file_path: {path}]" | ||
|
|
||
| desc = await self._describe_image_file(image_path, mime, image_recognition) | ||
| if not desc: | ||
| return f"[Image (description unavailable), file_path: {path}]" | ||
| return f"[Image {desc}, file_path: {path}]" | ||
| finally: | ||
| if image_path != abs_path: | ||
| # The compressed copy in data/temp is only ever consumed inside | ||
| # this call (the native branch copies it into session media, the | ||
| # VLM branch turns it into a description), so it can go now. | ||
| try: | ||
| await asyncio.to_thread(image_path.unlink, missing_ok=True) | ||
| except OSError as e: | ||
| logger.warning(f"Failed to remove compressed temp image {image_path.name}: {e}") | ||
|
|
||
| async def _describe_image_file(self, image_path: Path, mime: str, image_recognition: dict) -> str: | ||
| """Transcribe an image file with the default VLM, reusing the shared description cache. | ||
|
|
||
| An empty return means the VLM produced no description; the caller tells | ||
| recognition-disabled and VLM-failure apart on its own. | ||
| """ | ||
| desc_cache = None | ||
| md5 = None | ||
| try: | ||
| from core.message_manager import ImageDescCache | ||
|
|
||
| desc_cache = ImageDescCache(self.ctx.db) | ||
| md5 = await Image(image=str(image_path), mime=mime).hash_image() | ||
| cached_desc = await desc_cache.get(md5) | ||
| if cached_desc: | ||
| return cached_desc | ||
|
Comment on lines
+756
to
+759
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win 🔎 Supported by static analysis🏁 Script executed: sed -n '710,755p' core/plugin/builtin_plugins/agent/main.py
sed -n '90,155p' core/message_manager.py
rg -n 'ImageDescCache|desc_prompt|image_recognition' core tests/test_agent_plugin.pyRepository: xxynet/KiraAI Length of output: 8371 🏁 Script executed: #!/bin/bash
set -e
printf '%s\n' '--- cache model and database methods ---'
sed -n '1,190p' core/db/models.py
sed -n '70,175p' core/db/service.py
printf '%s\n' '--- cache callers and related definitions ---'
rg -n -C 5 'ImageDescCache\(|_describe_image_file|desc_img\(|get_default_vlm|get_lang\(' core --glob '*.py'
printf '%s\n' '--- configuration and provider definitions ---'
sed -n '1,70p' core/config/default.py
rg -n -C 8 'def get_default_vlm|class Provider|default_vlm|vlm' core --glob '*.py'
printf '%s\n' '--- agent construction/context and image capability flow ---'
sed -n '650,750p' core/plugin/builtin_plugins/agent/main.py
sed -n '220,370p' core/message_manager.pyRepository: xxynet/KiraAI Length of output: 50369 🏁 Script executed: #!/bin/bash
set -e
printf '%s\n' '--- description request and image identity ---'
sed -n '48,96p' core/utils/common_utils.py
rg -n -C 4 'def hash_image|class Image|to_data_url' core/chat core --glob '*.py' | head -180
printf '%s\n' '--- exact provider model identity and default selection ---'
sed -n '140,180p' core/provider/provider_manager.py
sed -n '230,280p' core/provider/provider_manager.py
sed -n '25,65p' core/provider/provider.py
printf '%s\n' '--- session capability binding ---'
rg -n -C 8 'def get_session_capabilities|class PluginContext|session_capabilities|effective_capabilities' core --glob '*.py'
printf '%s\n' '--- all cache call sites only ---'
rg -n -C 3 'image_desc_cache|ImageDescCache' core --glob '*.py'Repository: xxynet/KiraAI Length of output: 47222 🏁 Script executed: #!/bin/bash
set -e
printf '%s\n' '--- image hashing and data URL contract ---'
sed -n '300,420p' core/chat/message_elements.py
sed -n '450,520p' core/chat/message_elements.py
printf '%s\n' '--- compression result contract ---'
rg -n -C 8 'async def compress_image_file|def compress_image_file|return .*mime|mime' core/plugin/builtin_plugins/agent core --glob '*.py' | head -160Repository: xxynet/KiraAI Length of output: 20443 Scope the description cache by the effective recognition request.
Use a composite identity for the image hash, effective prompt/language, MIME type, and resolved VLM identity. Apply the same change to all 🤖 Prompt for AI Agents |
||
| except Exception as e: | ||
| logger.warning(f"Failed to read image desc cache for read_file: {e}") | ||
| desc_cache = None | ||
|
|
||
| desc = "" | ||
| try: | ||
| vlm_client = self.ctx.provider_mgr.get_default_vlm() | ||
| desc = await desc_img( | ||
| client=vlm_client, | ||
| image=Image(image=str(image_path), mime=mime), | ||
| prompt=str(image_recognition.get("desc_prompt", "") or "").strip() or None, | ||
| lang=self.ctx.get_lang(), | ||
| ) | ||
| except Exception as e: | ||
| logger.error(f"Failed to describe image file for read_file: {e}") | ||
|
|
||
| if desc and desc_cache and md5: | ||
| try: | ||
| await desc_cache.set(md5, desc) | ||
| except Exception as e: | ||
| logger.warning(f"Failed to cache read_file image desc: {e}") | ||
| return desc | ||
|
|
||
| async def _read_audio_file(self, event: KiraMessageBatchEvent, path: str) -> str: | ||
| """Transcribe an audio file with the default STT model. | ||
|
|
||
| Mirrors how incoming voice records are recognized: the session's | ||
| ``stt`` capability toggle is honored, and an unavailable STT model or a | ||
| failed transcription degrades to returning the file path only, so the | ||
| turn can still forward the file to the user. | ||
| """ | ||
| abs_path = self._resolve_path(path) | ||
| if not abs_path.is_file(): | ||
| return f"[Failed to read file: file not found: {path}]" | ||
|
|
||
| capabilities = self.ctx.get_session_capabilities(event.sid) if self.ctx else {} | ||
| stt_caps = capabilities.get("stt") if isinstance(capabilities, dict) else None | ||
| if not isinstance(stt_caps, dict): | ||
| stt_caps = {} | ||
| if not stt_caps.get("enabled", True): | ||
| return f"[Record (speech recognition disabled), file_path: {path}]" | ||
|
|
||
| try: | ||
| stt_client = self.ctx.provider_mgr.get_default_stt() | ||
| transcript = await speech_to_text( | ||
| client=stt_client, record=Record(record=str(abs_path)) | ||
| ) | ||
| except Exception as e: | ||
| logger.error(f"Failed to transcribe audio file for read_file: {e}") | ||
| transcript = "" | ||
| if not transcript: | ||
| return f"[Record (speech recognition unavailable), file_path: {path}]" | ||
| return f"[Record {transcript}, file_path: {path}]" | ||
|
|
||
| @register.tool( | ||
| "write_file", | ||
| "Write content to a plain text file in allowed write paths. Creates the file if it doesn't exist, overwrites if it does.", | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🔒 Security & Privacy | 🛡️ Analyzed with Security Review | 🟡 Minor | ⚡ Quick win
🧩 Analysis chain
🏁 Script executed:
Repository: xxynet/KiraAI
Length of output: 13025
🏁 Script executed:
Repository: xxynet/KiraAI
Length of output: 50436
Path Traversal
Reachability: External
Exploitability: Difficult
CWE: CWE-367 — Time-of-check Time-of-use (TOCTOU) Race Condition
Use the resolved candidate for read-only operations.
_is_path_allowedresolvescandidatebut returns only a boolean. Read-only tools then reopen the original path, so a concurrent directory-entry replacement can redirect a checked read. File-only sessions cannot create symlinks through these tools. Sessions grantedexeccan already run shell filesystem commands, so this remains a narrow, difficult race rather than a major sandbox escape. Return the resolvedPathand pass it to all read-only sinks. Write targets still require directory-handle oropenat-style hardening.🤖 Prompt for AI Agents