From ed8f7ae977cb392b24b9038a55c5c82454cb4e5f Mon Sep 17 00:00:00 2001 From: billowliu2 Date: Sun, 2 Aug 2026 01:24:28 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=20kimi-eyes?= =?UTF-8?q?=EF=BC=9A=E7=BB=99=20KimiCode=20=E9=9D=9E=E5=A4=9A=E6=A8=A1?= =?UTF-8?q?=E6=80=81=E6=A8=A1=E5=9E=8B=E8=A1=A5=E4=B8=8A=E8=A7=86=E8=A7=89?= =?UTF-8?q?=E8=83=BD=E5=8A=9B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .gitignore | 13 ++ LICENSE | 21 +++ README.en.md | 268 +++++++++++++++++++++++++++++++ README.md | 198 +++++++++++++++++++++++ SYSTEM.md | 27 ++++ kimi.plugin.json | 21 +++ mcp/models-db.json | 1 + mcp/server.mjs | 241 ++++++++++++++++++++++++++++ mcp/vision.mjs | 344 ++++++++++++++++++++++++++++++++++++++++ package.json | 40 +++++ scripts/sync-models.mjs | 95 +++++++++++ setup.mjs | 241 ++++++++++++++++++++++++++++ 12 files changed, 1510 insertions(+) create mode 100644 .gitignore create mode 100644 LICENSE create mode 100644 README.en.md create mode 100644 README.md create mode 100644 SYSTEM.md create mode 100644 kimi.plugin.json create mode 100644 mcp/models-db.json create mode 100644 mcp/server.mjs create mode 100644 mcp/vision.mjs create mode 100644 package.json create mode 100644 scripts/sync-models.mjs create mode 100644 setup.mjs diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..5d8b796 --- /dev/null +++ b/.gitignore @@ -0,0 +1,13 @@ +# dependencies +node_modules/ + +# npm publish artifacts +*.tgz + +# local secrets — never commit (credentials live in user-level ~/.npmrc) +.env +.env.* + +# local verification / scratch +.verify/ +*.log diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..1cf8e2b --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 billowliu2 + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.en.md b/README.en.md new file mode 100644 index 0000000..2c5f76c --- /dev/null +++ b/README.en.md @@ -0,0 +1,268 @@ +# Kimi Eyes 👀 + +> [中文](README.md) | English + +Give **non-multimodal models** in Kimi Code the ability to analyze images and +screenshots (inspired by [opencode-vision](https://github.com/JochenYang/opencode-vision), +but thinner: no hooks, no message transforms, no leftover state). + +**How it works**: the plugin declares an MCP stdio server exposing two tools — +`read_image` (read a local image) and `read_clipboard_image` (read a clipboard +screenshot) — and a `SYSTEM.md` guide tells the model when to call them. The server +sends the image to your own vision API (OpenAI-compatible or Anthropic protocol) and +returns the text description. + +``` +Multimodal models: paste with Alt+V, see natively — the plugin stays idle. +Non-multimodal models: @image-path / "analyze this screenshot" + → model calls mcp__kimi-eyes__read_image / read_clipboard_image + → your VLM returns a description +``` + +## Prerequisites + +- Kimi Code CLI (with `/plugins` and MCP support) +- Node.js ≥ 18 (`node --version`; native `fetch` requires 18+) +- A multimodal vision API of your own (OpenAI-compatible `chat/completions`, or + Anthropic `messages`) — you provide the key + +## Compatibility + +**The plugin puts no restriction on your main model.** No matter which model your +Kimi Code session is running, or which provider it comes from, as long as it lacks +native image input (`image_in`), this plugin adds vision to it: + +- **Common non-multimodal models**: DeepSeek (`deepseek-chat`), Qwen text-only + (`qwen-plus` / `qwen-turbo`), Llama text variants, and locally deployed text + models (Ollama / vLLM, etc.) +- **Any third-party model**: any OpenAI-compatible or custom model configured in + Kimi Code — if it cannot see images natively, the plugin guides it to call the + vision tools whenever an image is involved +- **Multimodal models**: skipped automatically (paste with Alt+V, see natively) + +Vision capability is provided by the **VLM you configure**, fully decoupled from +the main model: + +| Protocol | Example vision models | +| --- | --- | +| OpenAI-compatible | qwen-vl series, GLM-4V, GPT-4o / GPT-5 compatible endpoints, Gemini-compatible endpoints, etc. | +| Anthropic | Claude 3.5 / 3.7 / 4 series, etc. | + +In short: **the main model handles text, the external VLM handles images.** When +configuring, just pick the protocol that matches your vision API provider (step 1 +of the setup wizard); everything else is automatic. + +## Quick start + +### 1. Install the plugin + +In a Kimi Code session: + +``` +/plugins install D:\AIGC\Plugin\kimi-eyes +``` + +### 2. Configure the vision API (one-time) + +Run the setup wizard from the plugin directory: + +``` +cd D:\AIGC\Plugin\kimi-eyes +node setup.mjs +``` + +The wizard walks you through: **choose protocol (OpenAI-compatible / Anthropic) → +enter Base URL → enter API Key (masked input) → fetch model list and pick a +multimodal model → 1×1 image vision check → optionally enter your main model name +(for trigger decisions) → write config**. + +- The model list is fetched from `GET {BaseUrl}/models`; entries are tagged + **"✓vision / text"** using the bundled models.dev database, with a "★guess" fallback + for models the database does not know; if fetching fails it falls back to manual entry +- The vision check must pass before the config is saved, so you never end up with a + model that rejects images +- The last step asks for your **main model name** (optional): the plugin looks it up in + models.dev, and if that model supports native image input the tools refuse with a + hint — the code-level "multimodal → don't trigger" switch (see + [Triggering](#triggering-everyday-fully-automatic)) +- Config is written to `~/.kimi-code/kimi-eyes/config.json` (`chmod 600` on + non-Windows systems) + +### 3. Enable + +``` +/reload +``` + +The MCP server starts automatically with the session. + +### 4. Use it + +| Scenario | Action | +| --- | --- | +| Multimodal model | Paste with Alt+V directly — native vision, plugin not involved | +| Non-multimodal + image path | Type `@screenshot.png` or paste the path; the model calls `read_image` | +| Non-multimodal + just screenshotted/copied | Ask "analyze this screenshot"; the model calls `read_clipboard_image` to read the system clipboard | + +## Activation and triggering + +### Activation (one-time) + +``` +/plugins install D:\AIGC\Plugin\kimi-eyes # 1. Install +/reload # 2. Enable (or start a new session with /new) +``` + +Once enabled, the MCP server starts automatically with every session — there is no +separate "turn on the feature" step. To verify: + +- `/plugins list` → kimi-eyes should show as enabled +- `/mcp` → the kimi-eyes server should show as connected +- `/plugins info kimi-eyes` → should show no diagnostics errors + +### Triggering (everyday, fully automatic) + +The plugin is **passive**: `SYSTEM.md` plants the rules into the model, and the model +calls the tools automatically at the right moment — you do nothing: + +| Signal (anything in the message) | Model's automatic behavior | +| --- | --- | +| Image-format path or `@` reference (`.png/.jpg/.jpeg/.webp/.gif/.bmp`) | **Hard trigger**: unconditionally calls `read_image(path)`, regardless of wording | +| Media content you cannot interpret (e.g. a pasted image) | Ignores that media part, calls `read_clipboard_image()` (the pasted image is almost always still in the clipboard) | +| Wording implies image content: image / screenshot / photo / UI / chart / CAPTCHA / OCR, etc., but no path | Calls `read_clipboard_image()` | +| Model natively supports `image_in` (multimodal) | Skips all rules, sees images natively, never calls the tools | + +**Triggering does not depend on fixed wording** — the user does not need to say +"analyze the image". Image-format paths are an **unconditional hard trigger**; when +in doubt, the model is instructed to call a tool rather than guess. + +**On model-capability detection**: Kimi Code does not expose "is the current model +multimodal?" to plugins, so this plugin provides two layers: + +- **Code-level (recommended)**: declare `mainModel` in the config (last wizard step, + or the `VISION_MAIN_MODEL` environment variable). The plugin looks it up in the + bundled models.dev database — if it supports image input, the tools refuse with a + hint ("paste the image directly"); if it is text-only, the tools proceed normally. +- **Prompt-level**: without `mainModel`, the SYSTEM.md explicit skip rule handles it + (the model recognizes its own capability). + +Both layers are fail-safe: a wrong judgment costs at most one wasted external call +(a multimodal model calling a tool) or a missed trigger (a text model — covered by +the path/semantic signals). For a hard switch, simply `/plugins disable kimi-eyes` +when running multimodal models. + +**The first tool call prompts one approval** (MCP tool permission): choose *Approve +for this session* to skip prompts for the rest of the session; for permanent +approval, add to `~/.kimi-code/config.toml`: + +```toml +[[permission.rules]] +decision = "allow" +pattern = "mcp__kimi-eyes__*" +``` + +A successful trigger looks like: a tool call appears in the TUI before the answer, +and the model then answers based on the returned description. If the model does not +call the tool on its own (e.g. you pasted a path without asking a question), just +command it: "Call the read_image tool to analyze D:\xxx.png". + +## Environment variables (optional; override the config file) + +| Variable | Description | Priority | +| --- | --- | --- | +| `VISION_API_PROTOCOL` | `openai` or `anthropic`; force the protocol | Higher than config file | +| `VISION_API_KEY` | API key | Higher than config file | +| `VISION_API_URL` | Base URL | Higher than config file | +| `VISION_MODEL` | Model name | Higher than config file | +| `VISION_MAIN_MODEL` | Your main model name (optional), used for trigger decisions | Higher than config file | +| `VISION_MAX_TOKENS` | Max tokens for the vision response (default 1024) | — | +| `VISION_FETCH_TIMEOUT_MS` | Request timeout in ms (default 60000) | — | + +> When an environment variable conflicts with `config.json`, the variable wins. If +> neither is set, the tools return a clear error pointing at `setup.mjs`. +> Variable names are compatible with opencode-vision, so migrating is trivial. + +## Protocol details + +| | OpenAI-compatible | Anthropic | +| --- | --- | --- | +| Request endpoint | `{BaseUrl}/chat/completions` (`/v1` auto-appended) | `{BaseUrl}/v1/messages` | +| Auth | `Authorization: Bearer ` | `x-api-key: ` + `anthropic-version: 2023-06-01` | +| Image payload | `image_url` + base64 data URL | `source: {type:"base64"}` | +| Model list | `GET {BaseUrl}/models` | `GET {BaseUrl}/v1/models` (not an official Anthropic endpoint; falls back to manual entry) | + +## Model capability database (models.dev) + +The plugin ships a slim capability cache `mcp/models-db.json` synced from +[models.dev](https://models.dev) (currently 279+ models, tagged with whether each +supports image input, ~19 KB). It powers two things: + +- **Exact tagging** in the setup wizard's model picker ("✓vision / text"), replacing + pure keyword guessing +- The **code-level trigger switch**: once `mainModel` is declared, the plugin checks + whether your main model is multimodal + +**Sync at packaging time** (the data evolves — run before each release): + +``` +node scripts/sync-models.mjs +``` + +- Source: `https://models.dev/models.json` +- Behind a proxy: set `HTTPS_PROXY`, e.g. `HTTPS_PROXY=http://127.0.0.1:7897 node scripts/sync-models.mjs` +- Options: `--timeout ` (default 180), `--out ` (default + `mcp/models-db.json`), `--endpoint ` +- Matching strategy: exact id → `provider/model-name` suffix → case-insensitive name +- **Unknown models**: lookup returns unknown — the wizard falls back to keyword + guessing ("★guess") and triggering falls back to the SYSTEM.md rules; nothing breaks +- **Manual extras**: models.dev does not list every vision model (e.g. `k3-256k`, + `kimi-for-coding`). Keep them in `EXTRA_ENTRIES` inside + `scripts/sync-models.mjs` — every sync merges them in, so re-syncs never drop them + +## Tools + +| Tool | Arguments | Description | +| --- | --- | --- | +| `read_image` | `path` (required), `prompt` (optional) | Validates the file is an image (extension + magic bytes), then calls the VLM | +| `read_clipboard_image` | `prompt` (optional) | Captures the clipboard image to a temp file, calls the VLM, deletes the temp file | + +Clipboard capture depends on the platform: Windows uses PowerShell (built-in), +macOS needs `pngpaste` (`brew install pngpaste`), Linux needs `wl-paste` (Wayland) +or `xclip` (X11). + +## Troubleshooting + +- **Tool returns "Vision API is not configured"** → run `node setup.mjs`, or set + `VISION_API_KEY` / `VISION_API_URL` / `VISION_MODEL` +- **Model list fetch fails** (404/401) → the wizard falls back to manual entry; if + your provider has no `/models` endpoint, just type the model name +- **Vision check fails** → pick a model that really accepts image input (e.g. + `qwen-vl-max`, `glm-4v`, `gpt-4o`, `claude-3-5-sonnet` — check your provider's docs) +- **`read_clipboard_image` errors** → make sure the clipboard actually holds an image + (Ctrl+C an image or Win+Shift+S a screenshot first); on macOS/Linux check the + platform tool above is installed +- **Tools not showing up after install** → verify the plugin is enabled + (`/plugins list`) and run `/reload` or start a new session + +## Security notes + +- `config.json` stores your API key in plain text (`600` perms on non-Windows); + never commit it to a repository +- `read_clipboard_image` reads the system clipboard — it may contain sensitive + content you just copied. The tool call goes through the approval flow, so you + decide when it runs +- This project ships no credentials; vision requests go only to the Base URL you + configured + +## Limitations + +- No subagent delegation: Kimi Code's `model_preference` only supports + primary/secondary (it cannot name a specific vision model the way opencode can), + so the value is limited +- No `UserPromptSubmit` hook fallback: the `SYSTEM.md` guide covers the common + cases; if paste/reference behavior misbehaves in your TUI, a hook can be + re-evaluated then + +## License + +MIT diff --git a/README.md b/README.md new file mode 100644 index 0000000..1e8f622 --- /dev/null +++ b/README.md @@ -0,0 +1,198 @@ +# Kimi Eyes 👀 + +> [English](README.en.md) | 中文 + +让 KimiCode 的**非多模态模型**也能分析图片与截图(借鉴 [opencode-vision](https://github.com/JochenYang/opencode-vision) 的思路,但更薄:无 hook、无消息变换、无状态残留)。 + +**核心机制**:插件声明一个 MCP stdio 服务器,暴露两个工具——`read_image`(读本地图片)与 `read_clipboard_image`(读剪贴板截图),经 `SYSTEM.md` 引导模型调用;服务器按你配置的协议(OpenAI 兼容 / Anthropic)把图片发给你自己配置的多模态 API,返回文本描述。 + +``` +多模态模型:Alt+V 粘贴即看,插件自动闲置 +非多模态模型:@图片路径 / 截图后提问 → 模型调 mcp__kimi-eyes__read_image / read_clipboard_image → 你的 VLM 返回描述 +``` + +## 前置要求 + +- KimiCode CLI(本插件通过插件机制加载,需支持 `/plugins` 与 MCP) +- Node.js ≥ 18(`node --version` 检查;原生 fetch 需要 18+) +- 一个支持视觉输入的多模态 API(OpenAI 兼容 `chat/completions`,或 Anthropic `messages`),Key 由你自己提供 + +## 适用范围 + +**本插件对主模型没有任何限制**——无论你的 KimiCode 当前用哪个模型、来自哪家服务商,只要它不具备原生图片输入能力(`image_in`),本插件就能为它补上视觉: + +- **常见非多模态模型**:DeepSeek(deepseek-chat)、Qwen 纯文本版(qwen-plus / qwen-turbo)、Llama 文本系列、以及本地部署的文本模型(Ollama / vLLM 等) +- **任何第三方模型**:通过 KimiCode 配置的任意 OpenAI 兼容或自定义模型,只要它原生不看图,插件就会在消息含图片时引导模型调用视觉工具 +- **多模态模型**:自动跳过(Alt+V 粘贴即看,插件闲置) + +视觉能力由**你自己配置的 VLM** 提供,与主模型完全解耦: + +| 协议 | 常见视觉模型示例 | +| --- | --- | +| OpenAI 兼容 | qwen-vl 系列、GLM-4V、GPT-4o / GPT-5 兼容端点、Gemini 兼容端点等 | +| Anthropic | Claude 3.5 / 3.7 / 4 系列等 | + +一句话:**主模型只管理解文字,看图交给外部 VLM**。配置时只需在 setup 向导第一步选择与你视觉 API 服务商匹配的协议,其余全部自动。 + +## 快速开始 + +### 1. 安装插件 + +在 KimiCode 会话里执行: + +``` +/plugins install D:\AIGC\Plugin\kimi-eyes +``` + +### 2. 配置视觉 API(一次性) + +在插件目录运行配置向导: + +``` +cd D:\AIGC\Plugin\kimi-eyes +node setup.mjs +``` + +向导按顺序引导:**选择协议(OpenAI 兼容 / Anthropic)→ 填写 BaseUrl → 填写 API Key(掩码输入)→ 拉取模型列表选择多模态模型 → 1×1 图片视觉验证 → 可选填写当前主模型名(用于触发判断)→ 写入配置**。 + +- 模型列表自动从 `GET {BaseUrl}/models` 拉取,用内置的 models.dev 数据库**精确标注「✓视觉 / 文本」**,未收录的按关键词标「★疑似」;拉取失败时回退为手动输入 +- 视觉验证通过才落盘,防止配到一个不收图片的模型 +- 最后一步可填写**当前主模型名**(可选):插件用 models.dev 判定后,若该模型支持原生识图,工具会直接提示无需调用——即「多模态不触发」的代码级判断(详见[触发](#触发日常全自动)) +- 配置写入 `~/.kimi-code/kimi-eyes/config.json`(类 Unix 系统自动 `chmod 600`) + +### 3. 启用 + +``` +/reload +``` + +MCP 服务器会随会话自动启动。 + +### 4. 使用 + +| 场景 | 操作 | +| --- | --- | +| 多模态模型 | 直接 Alt+V 粘贴图片,原生看图,插件不参与 | +| 非多模态 + 有图片路径 | 输入 `@截图.png` 或直接给路径,模型自动调 `read_image` | +| 非多模态 + 刚截图/复制 | 截图后直接提问「分析这张截图」,模型自动调 `read_clipboard_image` 读系统剪贴板 | + +## 激活与触发 + +### 激活(一次性) + +``` +/plugins install D:\AIGC\Plugin\kimi-eyes # 1. 安装 +/reload # 2. 启用(或 /new 新开会话) +``` + +启用后 MCP 服务器随每个会话自动启动,没有单独的「开启功能」步骤。验证方式: + +- `/plugins list` → kimi-eyes 状态应为 enabled +- `/mcp` → kimi-eyes 服务器应显示 connected +- `/plugins info kimi-eyes` → 应无诊断错误 + +### 触发(日常,全自动) + +插件是**被动式**的:`SYSTEM.md` 为模型植入规则,模型在合适时机自动调用工具,无需手动操作: + +| 信号(消息里的任何一项) | 模型自动行为 | +| --- | --- | +| 图片格式路径或 `@` 引用(`.png/.jpg/.jpeg/.webp/.gif/.bmp`) | **硬触发**:无条件调 `read_image(path)`,与提问语料无关 | +| 消息里有你无法解读的媒体内容(如粘贴的图片) | 忽略该媒体 part,自动调 `read_clipboard_image()` 读剪贴板(粘贴后剪贴板仍保留原图) | +| 提问语义涉及图像内容:图 / 截图 / 照片 / 界面 / 图表 / 验证码 / OCR 等,但无路径 | 自动调 `read_clipboard_image()` | +| 模型原生支持 `image_in`(多模态) | 跳过全部规则,原生看图,不调工具 | + +**触发不依赖固定语料**——用户不需要说「分析这个图片」。图片格式路径是**无条件硬触发**;拿不准时模型会倾向于调工具而不是瞎猜(SYSTEM.md 里的明确规则)。 + +**关于模型能力判断**:KimiCode 不向插件暴露「当前模型是否多模态」的信号,本插件提供两层判断: + +- **代码级(推荐)**:在配置里声明 `mainModel`(setup 向导最后一步,或环境变量 `VISION_MAIN_MODEL`),插件用内置的 models.dev 数据库判断——命中且支持识图 → 工具直接提示「你是多模态模型,直接粘贴看图」;命中且纯文本 → 正常走工具 +- **引导级**:未声明 `mainModel` 时,靠 SYSTEM.md 的显式跳过规则(模型自我识别) + +两层都是 fail-safe:判断错误最坏只是多一次外部调用(多模态误调)或漏调(文本模型漏调可用路径/语义信号补上)。若想硬性关闭,多模态场景可直接 `/plugins disable kimi-eyes`。 + +**首次调用会弹一次审批**(MCP 工具权限):选 *Approve for this session* 本会话免问;想永久免审批,在 `~/.kimi-code/config.toml` 添加: + +```toml +[[permission.rules]] +decision = "allow" +pattern = "mcp__kimi-eyes__*" +``` + +触发成功的标志:回答前 TUI 出现工具调用记录,模型随后基于返回的描述作答。若模型未自动调用(例如只贴了路径未提问),可直接命令:「调用 read_image 工具分析 D:\xxx.png」。 + +## 环境变量(可选,覆盖配置文件) + +| 变量 | 说明 | 优先级 | +| --- | --- | --- | +| `VISION_API_PROTOCOL` | `openai` 或 `anthropic`,强制指定协议 | 高于配置文件 | +| `VISION_API_KEY` | API Key | 高于配置文件 | +| `VISION_API_URL` | BaseUrl | 高于配置文件 | +| `VISION_MODEL` | 模型名 | 高于配置文件 | +| `VISION_MAIN_MODEL` | 当前主模型名(可选),用于触发判断 | 高于配置文件 | +| `VISION_MAX_TOKENS` | 视觉 API 返回最大 token 数(默认 1024) | — | +| `VISION_FETCH_TIMEOUT_MS` | 请求超时毫秒(默认 60000) | — | + +> 环境变量与 `config.json` 同名配置冲突时,环境变量优先;两者都没配时工具会返回明确错误并提示先运行 `setup.mjs`。 +> 环境变量可沿用 opencode-vision 的迁移习惯——变量名完全兼容。 + +## 协议说明 + +| | OpenAI 兼容 | Anthropic | +| --- | --- | --- | +| 请求端点 | `{BaseUrl}/chat/completions`(自动补 `/v1`) | `{BaseUrl}/v1/messages` | +| 鉴权 | `Authorization: Bearer ` | `x-api-key: ` + `anthropic-version: 2023-06-01` | +| 图片传法 | `image_url` + base64 data URL | `source: {type:"base64"}` | +| 模型列表 | `GET {BaseUrl}/models` | `GET {BaseUrl}/v1/models`(官方无此端点,失败回退手动输入) | + +## 模型能力数据库(models.dev) + +插件内置一份从 [models.dev](https://models.dev) 同步的精简能力库 `mcp/models-db.json`(当前收录 279+ 模型,标注每个模型是否支持图片输入,约 19KB),用于两处: + +- 向导选择视觉模型时的**精确标注**(✓视觉 / 文本),替代纯关键词猜测 +- 声明 `mainModel` 后**判断当前主模型是否多模态**,实现代码级触发开关 + +**打包时同步**(数据会更新,建议每次发布前运行): + +``` +node scripts/sync-models.mjs +``` + +- 数据源:`https://models.dev/models.json` +- 走代理:设置 `HTTPS_PROXY`,例如 `HTTPS_PROXY=http://127.0.0.1:7897 node scripts/sync-models.mjs` +- 可选参数:`--timeout <秒>`(默认 180)、`--out <路径>`(默认 `mcp/models-db.json`)、`--endpoint ` +- 模型匹配策略:精确 id → `provider/模型名` 后缀 → 大小写不敏感的名称匹配 +- **未收录的模型**:查询返回未知——向导回退关键词猜测(★疑似),触发判断回退 SYSTEM.md 引导,不影响使用 +- **手动特例**:models.dev 未收录但确认支持识图的模型(如 `k3-256k`、`kimi-for-coding`),在 `scripts/sync-models.mjs` 的 `EXTRA_ENTRIES` 里维护,每次同步自动合入、不会被覆盖 + +## 工具 + +| 工具 | 参数 | 说明 | +| --- | --- | --- | +| `read_image` | `path`(必填)、`prompt`(可选) | 校验文件为图片(扩展名 + 魔数)后调 VLM | +| `read_clipboard_image` | `prompt`(可选) | 从系统剪贴板取图存临时文件后调 VLM,用完即删 | + +剪贴板取图依赖平台工具:Windows 用 PowerShell(内置)、macOS 需 `pngpaste`(`brew install pngpaste`)、Linux 需 `wl-paste`(Wayland)或 `xclip`(X11)。 + +## 故障排查 + +- **工具返回「Vision API is not configured」** → 运行 `node setup.mjs`,或设置 `VISION_API_KEY` / `VISION_API_URL` / `VISION_MODEL` +- **模型列表拉取失败**(404/401)→ 向导自动回退手动输入;若服务端不支持 `/models`,直接输入模型名即可 +- **视觉验证失败** → 换一个真正支持图片输入的模型(参考服务商文档,如 `qwen-vl-max`、`glm-4v`、`gpt-4o`、`claude-3-5-sonnet` 等) +- **`read_clipboard_image` 报错** → 确认剪贴板里确实有图片(先 Ctrl+C 复制或 Win+Shift+S 截图);macOS/Linux 检查上述平台工具是否安装 +- **安装后工具不生效** → 确认插件已启用(`/plugins list`),并运行过 `/reload` 或新开会话 + +## 安全说明 + +- `config.json` 以明文保存你的 API Key(类 Unix 已设 `600` 权限);请勿将该文件提交到版本库 +- `read_clipboard_image` 会读取系统剪贴板内容——剪贴板里可能刚复制过敏感信息。该工具调用会经过审批流,由你决定何时触发 +- 本项目不内置任何密钥,视觉请求只发往你配置的 BaseUrl + +## 局限 + +- 不做子代理委托方案:KimiCode 子代理的 `model_preference` 只能选择 primary/secondary,无法像 opencode 那样指定一个具体视觉模型,收益有限 +- 不做 `UserPromptSubmit` hook 兜底:当前 `SYSTEM.md` 引导已覆盖常规场景;若你的 TUI 环境下粘贴/引用行为异常,可再评估 hook + +## License + +MIT diff --git a/SYSTEM.md b/SYSTEM.md new file mode 100644 index 0000000..a39e9ac --- /dev/null +++ b/SYSTEM.md @@ -0,0 +1,27 @@ +# Kimi Eyes — Vision Assist Rules + +**If your model supports image input (`image_in`) natively, ignore every rule below — +you can see images directly, so never call these tools.** + +For models **without** native image input, the signals below are **hard triggers**: +call a vision tool whenever any of them appears, regardless of how the user phrased +their request. + +1. **Image-format path / `@` reference (hard trigger, unconditional)** — any path + ending in `.png`, `.jpg`, `.jpeg`, `.webp`, `.gif`, or `.bmp` (including an `@` + file reference to such a file) → call `mcp__kimi-eyes__read_image` with that + path. Pass `prompt` when the user has a specific question. +2. **Media content you cannot interpret** — a pasted image or any media part you + cannot read → ignore that part and call + `mcp__kimi-eyes__read_clipboard_image` (the image is almost always still in the + system clipboard). +3. **Wording implies an image, no path** — the user refers to "this", "the + screenshot", "the UI/interface", "the chart", "the photo", asks for OCR / CAPTCHA + reading, or otherwise implies image content, with no path and no visible media + part → call `mcp__kimi-eyes__read_clipboard_image`. + +When in doubt, call a tool rather than guessing blindly about the image. + +If a tool reports that the vision API is not configured, tell the user to run +`node setup.mjs` inside the kimi-eyes plugin directory (or set the +`VISION_API_KEY` / `VISION_API_URL` / `VISION_MODEL` environment variables). diff --git a/kimi.plugin.json b/kimi.plugin.json new file mode 100644 index 0000000..5e006e3 --- /dev/null +++ b/kimi.plugin.json @@ -0,0 +1,21 @@ +{ + "name": "kimi-eyes", + "version": "1.0.0", + "description": "Give non-multimodal models vision: analyze local images and clipboard screenshots through a user-configured VLM API (OpenAI-compatible or Anthropic protocol).", + "keywords": ["vision", "image", "screenshot", "multimodal", "vlm"], + "author": "kimi-eyes", + "license": "MIT", + "interface": { + "displayName": "Kimi Eyes", + "shortDescription": "视觉辅助:让非多模态模型也能分析图片/截图", + "longDescription": "读取本地图片或剪贴板截图,调用用户自配的多模态 API(OpenAI 兼容或 Anthropic 协议)返回描述。多模态模型不受影响,粘贴即原生看图。" + }, + "systemPromptPath": "./SYSTEM.md", + "mcpServers": { + "kimi-eyes": { + "command": "node", + "args": ["./mcp/server.mjs"], + "cwd": "./" + } + } +} diff --git a/mcp/models-db.json b/mcp/models-db.json new file mode 100644 index 0000000..a5adbe2 --- /dev/null +++ b/mcp/models-db.json @@ -0,0 +1 @@ +{"zhipuai/glm-4.6v":{"name":"GLM-4.6V","image":true},"zhipuai/glm-5":{"name":"GLM-5","image":false},"zhipuai/glm-4.5-air":{"name":"GLM-4.5-Air","image":false},"zhipuai/glm-5.1":{"name":"GLM-5.1","image":false},"zhipuai/glm-4.7-flash":{"name":"GLM-4.7-Flash","image":false},"zhipuai/glm-5.2":{"name":"GLM-5.2","image":false},"zhipuai/glm-4.7-flashx":{"name":"GLM-4.7-FlashX","image":false},"zhipuai/glm-4.6":{"name":"GLM-4.6","image":false},"zhipuai/glm-4.5":{"name":"GLM-4.5","image":false},"zhipuai/glm-4.5v":{"name":"GLM-4.5V","image":true},"zhipuai/glm-4.7":{"name":"GLM-4.7","image":false},"zhipuai/glm-5-turbo":{"name":"GLM-5-Turbo","image":false},"zhipuai/glm-5v-turbo":{"name":"GLM-5V-Turbo","image":true},"zhipuai/glm-4.5-flash":{"name":"GLM-4.5-Flash","image":false},"microsoft/mai-code-1-flash":{"name":"MAI-Code-1-Flash","image":false},"cohere/command-r7b-arabic-02-2025":{"name":"Command R7B Arabic","image":false},"cohere/command-r-08-2024":{"name":"Command R","image":false},"cohere/command-a-plus-05-2026":{"name":"Command A Plus","image":true},"cohere/command-a-translate-08-2025":{"name":"Command A Translate","image":false},"cohere/c4ai-aya-expanse-8b":{"name":"Aya Expanse 8B","image":false},"cohere/c4ai-aya-vision-32b":{"name":"Aya Vision 32B","image":true},"cohere/command-a-03-2025":{"name":"Command A","image":false},"cohere/command-r-plus-08-2024":{"name":"Command R+","image":false},"cohere/command-a-reasoning-08-2025":{"name":"Command A Reasoning","image":false},"cohere/command-a-vision-07-2025":{"name":"Command A Vision","image":true},"cohere/command-r7b-12-2024":{"name":"Command R7B","image":false},"cohere/c4ai-aya-vision-8b":{"name":"Aya Vision 8B","image":true},"cohere/north-mini-code-1-0":{"name":"North Mini Code","image":false},"cohere/c4ai-aya-expanse-32b":{"name":"Aya Expanse 32B","image":false},"nvidia/mistral-nemotron":{"name":"Mistral Nemotron","image":false},"nvidia/nemotron-nano-12b-v2-vl":{"name":"Nemotron Nano 12B v2 VL","image":true},"nvidia/nemotron-3-nano-30b-a3b":{"name":"Nemotron 3 Nano 30B A3B","image":false},"nvidia/llama-3.3-nemotron-super-49b-v1.5":{"name":"Llama 3.3 Nemotron Super 49B v1.5","image":false},"nvidia/llama-3.1-nemotron-safety-guard-8b-v3":{"name":"Llama 3.1 Nemotron Safety Guard 8B v3","image":false},"nvidia/llama-3.3-nemotron-super-49b-v1":{"name":"Llama 3.3 Nemotron Super 49B v1","image":false},"nvidia/llama-3.1-nemotron-70b-instruct":{"name":"Llama 3.1 Nemotron 70B Instruct","image":false},"nvidia/nemotron-3-super-120b-a12b":{"name":"Nemotron 3 Super 120B A12B","image":false},"nvidia/nemotron-3-content-safety":{"name":"Nemotron 3 Content Safety","image":false},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning":{"name":"Nemotron 3 Nano Omni 30B A3B Reasoning","image":true},"nvidia/llama-3.1-nemotron-ultra-253b":{"name":"Llama 3.1 Nemotron Ultra 253B","image":false},"nvidia/nemotron-voicechat":{"name":"Nemotron VoiceChat","image":false},"nvidia/llama-nemotron-embed-vl-1b-v2":{"name":"Llama Nemotron Embed VL 1B v2","image":true},"nvidia/nemotron-content-safety-reasoning-4b":{"name":"Nemotron Content Safety Reasoning 4B","image":false},"nvidia/nemotron-3-ultra-550b-a55b":{"name":"Nemotron 3 Ultra 550B A55B","image":false},"nvidia/llama-nemotron-rerank-vl-1b-v2":{"name":"Llama Nemotron Rerank VL 1B v2","image":true},"nvidia/nemotron-mini-4b-instruct":{"name":"Nemotron Mini 4B Instruct","image":false},"nvidia/nemotron-nano-9b-v2":{"name":"Nemotron Nano 9B v2","image":false},"nvidia/nemotron-cascade-2-30b-a3b":{"name":"Nemotron Cascade 2 30B A3B","image":false},"nvidia/nemotron-3.5-content-safety":{"name":"Nemotron 3.5 Content Safety","image":true},"google/gemini-2.5-computer-use-preview-10-2025":{"name":"Gemini 2.5 Computer Use Preview","image":true},"google/deep-research-preview-04-2026":{"name":"Gemini Deep Research Preview","image":true},"google/gemini-3.1-flash-tts-preview":{"name":"Gemini 3.1 Flash TTS Preview","image":false},"google/gemini-flash-latest":{"name":"Gemini Flash Latest","image":true},"google/gemini-embedding-2":{"name":"Gemini Embedding 2","image":true},"google/lyria-3-pro-preview":{"name":"Lyria 3 Pro Preview","image":true},"google/gemini-3.5-flash":{"name":"Gemini 3.5 Flash","image":true},"google/gemini-2.5-flash":{"name":"Gemini 2.5 Flash","image":true},"google/gemini-3.5-flash-lite":{"name":"Gemini 3.5 Flash Lite","image":true},"google/lyria-3-clip-preview":{"name":"Lyria 3 Clip Preview","image":true},"google/gemini-omni-flash-preview":{"name":"Gemini Omni Flash Preview","image":true},"google/veo-3.1-generate-preview":{"name":"Veo 3.1 Preview","image":true},"google/gemini-2.5-pro-tts":{"name":"Gemini 2.5 Pro TTS","image":false},"google/deep-research-max-preview-04-2026":{"name":"Deep Research Max Preview","image":true},"google/gemini-3-pro-image-preview":{"name":"Nano Banana Pro","image":true},"google/gemini-3.1-flash-lite-preview":{"name":"Gemini 3.1 Flash Lite Preview","image":true},"google/gemini-2.5-flash-tts":{"name":"Gemini 2.5 Flash TTS","image":false},"google/gemini-3.5-live-translate-preview":{"name":"Gemini 3.5 Live Translate Preview","image":false},"google/gemini-3-flash-preview":{"name":"Gemini 3 Flash Preview","image":true},"google/gemini-3.1-pro-preview-customtools":{"name":"Gemini 3.1 Pro Preview Custom Tools","image":true},"google/gemini-3.1-flash-lite-image":{"name":"Nano Banana 2 Lite","image":true},"google/gemini-3.1-flash-image-preview":{"name":"Nano Banana 2","image":true},"google/gemini-robotics-er-1.6-preview":{"name":"Gemini Robotics-ER 1.6 Preview","image":true},"google/gemma-4-26b-a4b-it":{"name":"Gemma 4 26B A4B IT","image":true},"google/gemini-embedding-001":{"name":"Gemini Embedding 001","image":false},"google/veo-3.1-lite-generate-preview":{"name":"Veo 3.1 Lite Preview","image":true},"google/gemini-2.0-flash":{"name":"Gemini 2.0 Flash","image":true},"google/gemini-3.6-flash":{"name":"Gemini 3.6 Flash","image":true},"google/veo-3.1-fast-generate-preview":{"name":"Veo 3.1 Fast Preview","image":true},"google/gemini-3.1-flash-lite":{"name":"Gemini 3.1 Flash Lite","image":true},"google/gemma-4-E2B-it":{"name":"Gemma 4 E2B IT","image":true},"google/gemma-4-31b-it":{"name":"Gemma 4 31B IT","image":true},"google/gemini-3.1-flash-image":{"name":"Nano Banana 2","image":true},"google/gemini-2.5-flash-image":{"name":"Nano Banana","image":true},"google/gemini-3-pro-image":{"name":"Nano Banana Pro","image":true},"google/gemini-3.1-pro-preview":{"name":"Gemini 3.1 Pro Preview","image":true},"google/gemini-flash-lite-latest":{"name":"Gemini Flash-Lite Latest","image":true},"google/gemini-2.5-pro":{"name":"Gemini 2.5 Pro","image":true},"google/gemma-4-E4B-it":{"name":"Gemma 4 E4B IT","image":true},"google/gemini-3-pro-preview":{"name":"Gemini 3 Pro Preview","image":true},"google/gemini-2.5-flash-lite":{"name":"Gemini 2.5 Flash-Lite","image":true},"google/gemini-2.0-flash-lite":{"name":"Gemini 2.0 Flash-Lite","image":true},"google/gemini-3.1-flash-live-preview":{"name":"Gemini 3.1 Flash Live Preview","image":true},"thinkingmachines/inkling":{"name":"Inkling","image":true},"sakana/fugu-ultra":{"name":"Fugu Ultra","image":true},"sakana/fugu":{"name":"Fugu","image":true},"tencent/hy3":{"name":"Hy3","image":false},"tencent/hy3-preview":{"name":"Hy3 preview","image":false},"meta/muse-spark-1.1":{"name":"Muse Spark 1.1","image":true},"meta/llama-4-scout-17b-instruct":{"name":"Llama 4 Scout 17B Instruct","image":true},"meta/llama-3.3-70b-instruct":{"name":"Llama-3.3-70B-Instruct","image":false},"meta/llama-4-maverick-17b-instruct":{"name":"Llama 4 Maverick 17B Instruct","image":true},"meituan/longcat-2.0":{"name":"LongCat-2.0","image":false},"xai/grok-4.3":{"name":"Grok 4.3","image":true},"xai/grok-4.20-0309-non-reasoning":{"name":"Grok 4.20 (Non-Reasoning)","image":true},"xai/grok-imagine-video-1.5":{"name":"Grok Imagine Video 1.5","image":true},"xai/grok-4.5":{"name":"Grok 4.5","image":true},"xai/grok-build-0.1":{"name":"Grok Build 0.1","image":true},"xai/grok-4.20-0309-reasoning":{"name":"Grok 4.20 (Reasoning)","image":true},"mistral/mistral-small-2506":{"name":"Mistral Small 3.2","image":true},"mistral/pixtral-large-latest":{"name":"Pixtral Large (latest)","image":true},"mistral/pixtral-12b":{"name":"Pixtral 12B","image":true},"mistral/magistral-medium-latest":{"name":"Magistral Medium (latest)","image":false},"mistral/mistral-large-2411":{"name":"Mistral Large 2.1","image":false},"mistral/mistral-medium-latest":{"name":"Mistral Medium (latest)","image":true},"mistral/devstral-small-2507":{"name":"Devstral Small","image":false},"mistral/mistral-medium-2505":{"name":"Mistral Medium 3","image":true},"mistral/mistral-small-latest":{"name":"Mistral Small (latest)","image":true},"mistral/mistral-nemo":{"name":"Mistral Nemo","image":false},"mistral/mistral-small-2603":{"name":"Mistral Small 4","image":true},"mistral/mistral-medium-2604":{"name":"Mistral Medium 3.5","image":true},"mistral/codestral-latest":{"name":"Codestral (latest)","image":false},"mistral/devstral-medium-latest":{"name":"Devstral 2 (latest)","image":false},"mistral/devstral-2512":{"name":"Devstral 2","image":false},"mistral/mistral-large-latest":{"name":"Mistral Large (latest)","image":true},"mistral/devstral-medium-2507":{"name":"Devstral Medium","image":false},"mistral/mistral-large-2512":{"name":"Mistral Large 3","image":true},"poolside/laguna-xs.2":{"name":"Laguna XS.2","image":false},"poolside/laguna-xs-2.1":{"name":"Laguna XS 2.1","image":false},"poolside/laguna-s-2.1":{"name":"Laguna S 2.1","image":false},"poolside/laguna-m.1":{"name":"Laguna M.1","image":false},"minimax/MiniMax-M2":{"name":"MiniMax-M2","image":false},"minimax/MiniMax-M2.7":{"name":"MiniMax-M2.7","image":false},"minimax/MiniMax-M2.1":{"name":"MiniMax-M2.1","image":false},"minimax/MiniMax-M2.5":{"name":"MiniMax-M2.5","image":false},"minimax/MiniMax-M2.5-highspeed":{"name":"MiniMax-M2.5-highspeed","image":false},"minimax/MiniMax-M2.7-highspeed":{"name":"MiniMax-M2.7-highspeed","image":false},"minimax/MiniMax-M3":{"name":"MiniMax-M3","image":true},"deepseek/deepseek-v4-flash":{"name":"DeepSeek V4 Flash","image":false},"deepseek/deepseek-v4-pro":{"name":"DeepSeek V4 Pro","image":false},"deepseek/deepseek-r1":{"name":"DeepSeek-R1","image":false},"deepseek/deepseek-chat":{"name":"DeepSeek Chat","image":false},"deepseek/deepseek-reasoner":{"name":"DeepSeek Reasoner","image":false},"deepreinforce/ornith-1.0-9b":{"name":"Ornith 1.0 9B","image":true},"deepreinforce/ornith-1.0-397b":{"name":"Ornith 1.0 397B","image":true},"deepreinforce/ornith-1.0-31b":{"name":"Ornith 1.0 31B","image":true},"deepreinforce/ornith-1.0-35b":{"name":"Ornith 1.0 35B","image":true},"alibaba/qwen3-coder-480b-a35b-instruct":{"name":"Qwen3-Coder 480B-A35B Instruct","image":false},"alibaba/qwen3.7-plus":{"name":"Qwen3.7 Plus","image":true},"alibaba/qwen3-vl-plus":{"name":"Qwen3-VL Plus","image":true},"alibaba/qwen3-32b":{"name":"Qwen3 32B","image":false},"alibaba/qwen3.8-max-preview":{"name":"Qwen3.8 Max Preview","image":true},"alibaba/qwen3.6-35b-a3b":{"name":"Qwen3.6 35B-A3B","image":true},"alibaba/qwen2-5-vl-72b-instruct":{"name":"Qwen2.5-VL 72B Instruct","image":true},"alibaba/qwen-max":{"name":"Qwen Max","image":false},"alibaba/qwen3.5-plus":{"name":"Qwen3.5 Plus","image":true},"alibaba/qwen-omni-turbo":{"name":"Qwen-Omni Turbo","image":true},"alibaba/qwen-vl-max":{"name":"Qwen-VL Max","image":true},"alibaba/qwen3-coder-30b-a3b-instruct":{"name":"Qwen3-Coder 30B-A3B Instruct","image":false},"alibaba/qwen3.5-27b":{"name":"Qwen3.5 27B","image":true},"alibaba/qwen3.7-max":{"name":"Qwen3.7 Max","image":false},"alibaba/qwen3-next-80b-a3b-thinking":{"name":"Qwen3-Next 80B-A3B (Thinking)","image":false},"alibaba/qwen3.5-9b":{"name":"Qwen3.5 9B","image":true},"alibaba/qwen3.6-27b":{"name":"Qwen3.6 27B","image":true},"alibaba/qwen3.5-35b-a3b":{"name":"Qwen3.5 35B-A3B","image":true},"alibaba/qwen3-235b-a22b":{"name":"Qwen3 235B-A22B","image":false},"alibaba/qwen3-max":{"name":"Qwen3 Max","image":false},"alibaba/qwen3-coder-plus":{"name":"Qwen3 Coder Plus","image":false},"alibaba/qwen3-coder-flash":{"name":"Qwen3 Coder Flash","image":false},"alibaba/qwen-flash":{"name":"Qwen Flash","image":false},"alibaba/qwen-plus":{"name":"Qwen Plus","image":false},"alibaba/qwen-turbo":{"name":"Qwen Turbo","image":false},"alibaba/qwq-plus":{"name":"QwQ Plus","image":false},"alibaba/qwen3.6-flash":{"name":"Qwen3.6 Flash","image":true},"alibaba/qwen3.5-397b-a17b":{"name":"Qwen3.5 397B-A17B","image":true},"alibaba/qwen3.6-max-preview":{"name":"Qwen3.6 Max Preview","image":false},"alibaba/qwen3.6-plus":{"name":"Qwen3.6 Plus","image":true},"alibaba/qwen3.5-122b-a10b":{"name":"Qwen3.5 122B-A10B","image":true},"alibaba/qwen3-next-80b-a3b-instruct":{"name":"Qwen3-Next 80B-A3B Instruct","image":false},"alibaba/qwen-vl-plus":{"name":"Qwen-VL Plus","image":true},"xiaomi/mimo-v2-omni":{"name":"MiMo-V2-Omni","image":true},"xiaomi/mimo-v2-flash":{"name":"MiMo-V2-Flash","image":false},"xiaomi/mimo-v2.5-pro-ultraspeed":{"name":"MiMo-V2.5-Pro-UltraSpeed","image":false},"xiaomi/mimo-v2-pro":{"name":"MiMo-V2-Pro","image":false},"xiaomi/mimo-v2.5":{"name":"MiMo-V2.5","image":true},"xiaomi/mimo-v2.5-pro":{"name":"MiMo-V2.5-Pro","image":false},"anthropic/claude-sonnet-4-6":{"name":"Claude Sonnet 4.6","image":true},"anthropic/claude-opus-4-0":{"name":"Claude Opus 4 (latest)","image":true},"anthropic/claude-haiku-4-5":{"name":"Claude Haiku 4.5 (latest)","image":true},"anthropic/claude-3-7-sonnet-20250219":{"name":"Claude Sonnet 3.7","image":true},"anthropic/claude-opus-4-6":{"name":"Claude Opus 4.6","image":true},"anthropic/claude-3-5-sonnet-20241022":{"name":"Claude Sonnet 3.5 v2","image":true},"anthropic/claude-fable-5":{"name":"Claude Fable 5","image":true},"anthropic/claude-opus-4-8":{"name":"Claude Opus 4.8","image":true},"anthropic/claude-sonnet-4-0":{"name":"Claude Sonnet 4 (latest)","image":true},"anthropic/claude-opus-4-1":{"name":"Claude Opus 4.1 (latest)","image":true},"anthropic/claude-sonnet-4-5":{"name":"Claude Sonnet 4.5 (latest)","image":true},"anthropic/claude-sonnet-4-5-20250929":{"name":"Claude Sonnet 4.5","image":true},"anthropic/claude-opus-4-5-20251101":{"name":"Claude Opus 4.5","image":true},"anthropic/claude-3-5-haiku-20241022":{"name":"Claude Haiku 3.5","image":true},"anthropic/claude-haiku-4-5-20251001":{"name":"Claude Haiku 4.5","image":true},"anthropic/claude-opus-4-7":{"name":"Claude Opus 4.7","image":true},"anthropic/claude-sonnet-4-20250514":{"name":"Claude Sonnet 4","image":true},"anthropic/claude-opus-4-5":{"name":"Claude Opus 4.5 (latest)","image":true},"anthropic/claude-sonnet-5":{"name":"Claude Sonnet 5","image":true},"anthropic/claude-3-haiku-20240307":{"name":"Claude Haiku 3","image":true},"anthropic/claude-opus-4-20250514":{"name":"Claude Opus 4","image":true},"anthropic/claude-opus-5":{"name":"Claude Opus 5","image":true},"anthropic/claude-opus-4-1-20250805":{"name":"Claude Opus 4.1","image":true},"perplexity/sonar":{"name":"Sonar","image":false},"perplexity/sonar-pro":{"name":"Sonar Pro","image":true},"perplexity/sonar-reasoning-pro":{"name":"Sonar Reasoning Pro","image":true},"moonshotai/kimi-k2.5":{"name":"Kimi K2.5","image":true},"moonshotai/kimi-k2-thinking-turbo":{"name":"Kimi K2 Thinking Turbo","image":false},"moonshotai/kimi-k2.7-code-highspeed":{"name":"Kimi K2.7 Code Highspeed","image":true},"moonshotai/kimi-k2.6":{"name":"Kimi K2.6","image":true},"moonshotai/kimi-k2.7-code":{"name":"Kimi K2.7 Code","image":true},"moonshotai/kimi-k2-thinking":{"name":"Kimi K2 Thinking","image":false},"moonshotai/kimi-k3":{"name":"Kimi K3","image":true},"openai/gpt-5.1-codex-mini":{"name":"GPT-5.1 Codex mini","image":true},"openai/gpt-image-2":{"name":"GPT-Image-2","image":true},"openai/gpt-5.2-pro":{"name":"GPT-5.2 Pro","image":true},"openai/gpt-5.5-pro":{"name":"GPT-5.5 Pro","image":true},"openai/gpt-4.1-mini":{"name":"GPT-4.1 mini","image":true},"openai/gpt-4o":{"name":"GPT-4o","image":true},"openai/gpt-5.4-pro":{"name":"GPT-5.4 Pro","image":true},"openai/gpt-5.6-sol":{"name":"GPT-5.6 Sol","image":true},"openai/gpt-oss-safeguard-120b":{"name":"GPT OSS Safeguard 120B","image":false},"openai/gpt-4.1-nano":{"name":"GPT-4.1 nano","image":true},"openai/gpt-oss-20b":{"name":"GPT OSS 20B","image":false},"openai/gpt-5.5-instant":{"name":"GPT-5.5 Instant","image":true},"openai/gpt-5.2-chat-latest":{"name":"GPT-5.2 Chat","image":true},"openai/o3-mini":{"name":"o3-mini","image":false},"openai/gpt-5.5":{"name":"GPT-5.5","image":true},"openai/o3-pro":{"name":"o3-pro","image":true},"openai/gpt-4o-mini":{"name":"GPT-4o mini","image":true},"openai/gpt-5":{"name":"GPT-5","image":true},"openai/o4-mini-deep-research":{"name":"o4-mini-deep-research","image":true},"openai/gpt-realtime-2.1":{"name":"GPT-Realtime-2.1","image":true},"openai/o3-deep-research":{"name":"o3-deep-research","image":true},"openai/gpt-5-codex":{"name":"GPT-5-Codex","image":true},"openai/gpt-3.5-turbo":{"name":"GPT-3.5-turbo","image":false},"openai/whisper-large-v3":{"name":"Whisper 3 Large","image":false},"openai/gpt-4o-2024-05-13":{"name":"GPT-4o (2024-05-13)","image":true},"openai/gpt-5-chat-latest":{"name":"GPT-5 Chat (latest)","image":true},"openai/gpt-4o-2024-11-20":{"name":"GPT-4o (2024-11-20)","image":true},"openai/gpt-5.4":{"name":"GPT-5.4","image":true},"openai/gpt-5.3-chat-latest":{"name":"GPT-5.3 Chat (latest)","image":true},"openai/gpt-realtime-whisper":{"name":"GPT Realtime Whisper","image":false},"openai/gpt-image-1.5":{"name":"GPT-Image-1.5","image":true},"openai/gpt-5.2-codex":{"name":"GPT-5.2 Codex","image":true},"openai/gpt-5.4-nano":{"name":"GPT-5.4 nano","image":true},"openai/gpt-5-pro":{"name":"GPT-5 Pro","image":true},"openai/gpt-5.4-mini":{"name":"GPT-5.4 mini","image":true},"openai/whisper-large-v3-turbo":{"name":"Whisper Large v3 Turbo","image":false},"openai/gpt-oss-120b":{"name":"GPT OSS 120B","image":false},"openai/o1":{"name":"o1","image":true},"openai/gpt-5.6-luna":{"name":"GPT-5.6 Luna","image":true},"openai/gpt-5.2":{"name":"GPT-5.2","image":true},"openai/gpt-5.3-codex":{"name":"GPT-5.3 Codex","image":true},"openai/gpt-5-mini":{"name":"GPT-5 Mini","image":true},"openai/o1-pro":{"name":"o1-pro","image":true},"openai/gpt-5.1":{"name":"GPT-5.1","image":true},"openai/gpt-4-turbo":{"name":"GPT-4 Turbo","image":true},"openai/gpt-4o-2024-08-06":{"name":"GPT-4o (2024-08-06)","image":true},"openai/gpt-5.1-codex":{"name":"GPT-5.1 Codex","image":true},"openai/gpt-5-nano":{"name":"GPT-5 Nano","image":true},"openai/o3":{"name":"o3","image":true},"openai/gpt-5.6-terra":{"name":"GPT-5.6 Terra","image":true},"openai/gpt-image-1":{"name":"GPT-Image-1","image":true},"openai/gpt-4.1":{"name":"GPT-4.1","image":true},"openai/o4-mini":{"name":"o4-mini","image":true},"openai/gpt-5.1-codex-max":{"name":"GPT-5.1 Codex Max","image":true},"openai/gpt-4":{"name":"GPT-4","image":false},"openai/gpt-5.1-chat-latest":{"name":"GPT-5.1 Chat","image":true},"sarvam/sarvam-30b":{"name":"Sarvam 30B","image":false},"sarvam/sarvam-105b":{"name":"Sarvam 105B","image":false},"stepfun/step-3.5-flash":{"name":"Step 3.5 Flash","image":false},"stepfun/step-3.7-flash":{"name":"Step 3.7 Flash","image":true},"stepfun/step-3.5-flash-2603":{"name":"Step 3.5 Flash 2603","image":false},"k3-256k":{"name":"K3-256k","image":true},"kimi-k3-256k":{"name":"K3-256k","image":true},"moonshot/k3-256k":{"name":"K3-256k","image":true},"kimi-for-coding":{"name":"Kimi For Coding","image":true},"moonshotai/kimi-for-coding":{"name":"Kimi For Coding","image":true},"kimi-for-coding-highspeed":{"name":"Kimi For Coding Highspeed","image":true},"moonshotai/kimi-for-coding-highspeed":{"name":"Kimi For Coding Highspeed","image":true}} diff --git a/mcp/server.mjs b/mcp/server.mjs new file mode 100644 index 0000000..48406c4 --- /dev/null +++ b/mcp/server.mjs @@ -0,0 +1,241 @@ +// kimi-eyes MCP stdio server — zero-dependency. +// Speaks JSON-RPC 2.0 over LSP-style framing (Content-Length headers) on stdio. + +import fs from 'node:fs'; +import path from 'node:path'; +import { + VERSION, + getConfig, + requireConfigured, + callVLM, + sniffImage, + clipboardImageToTemp, + lookupModelCapability, +} from './vision.mjs'; + +const TOOLS = [ + { + name: 'read_image', + description: + 'Analyze a local image file using the configured vision API and return a text description. ' + + 'Use when the user provides an image path or an @ file reference and your model cannot see images directly.', + inputSchema: { + type: 'object', + properties: { + path: { + type: 'string', + description: 'Absolute path to a local image file (png, jpg, jpeg, webp, gif, bmp).', + }, + prompt: { + type: 'string', + description: 'Optional instruction describing what to analyze (default: describe the image).', + }, + }, + required: ['path'], + }, + }, + { + name: 'read_clipboard_image', + description: + 'Capture the image currently in the system clipboard (e.g. a screenshot) and analyze it with the ' + + 'configured vision API. Use when the user says they just took a screenshot or copied an image and no file path is given.', + inputSchema: { + type: 'object', + properties: { + prompt: { + type: 'string', + description: 'Optional instruction describing what to analyze (default: describe the image).', + }, + }, + }, + }, +]; + +// --------------------------------------------------------------------------- +// Tools +// --------------------------------------------------------------------------- + +/** + * When the user declares their main model (config.mainModel) and the models.dev + * database says it supports image input, the tools are unnecessary — the model can + * see images natively. Refuse with a hint instead of wasting a VLM call. + */ +function assertNeedsExternalVision(cfg) { + if (!cfg.mainModel) return; + const cap = lookupModelCapability(cfg.mainModel); + if (cap.known && cap.image) { + throw new Error( + `Model "${cfg.mainModel}" supports native image input (per the models.dev database), ` + + `so the kimi-eyes tools are not needed. Paste the image directly and the model will see it. ` + + `If you still want kimi-eyes active, remove "mainModel" from the kimi-eyes config.` + ); + } +} + +async function readImageTool(args) { + const p = args?.path; + if (typeof p !== 'string' || !p.trim()) { + throw new Error('Missing required parameter: path'); + } + const cfg = getConfig(); + assertNeedsExternalVision(cfg); + const abs = path.resolve(p.trim()); + let stat; + try { + stat = fs.statSync(abs); + } catch { + throw new Error(`File not found: ${abs}`); + } + if (!stat.isFile()) throw new Error(`Not a file: ${abs}`); + const img = sniffImage(abs); + if (!img) { + throw new Error( + `Unsupported or invalid image file: ${abs} (supported: png, jpg, jpeg, webp, gif, bmp)` + ); + } + requireConfigured(cfg); + return callVLM({ ...img, prompt: args?.prompt }, cfg); +} + +async function readClipboardImageTool(args) { + const cfg = getConfig(); + assertNeedsExternalVision(cfg); + requireConfigured(cfg); + let tmp = null; + try { + tmp = clipboardImageToTemp(); + const img = sniffImage(tmp); + if (!img) throw new Error('Clipboard does not contain a valid image.'); + return callVLM({ ...img, prompt: args?.prompt }, cfg); + } finally { + if (tmp) { + try { + fs.unlinkSync(tmp); + } catch { + // best-effort cleanup + } + } + } +} + +async function callTool(name, args) { + if (name === 'read_image') return readImageTool(args); + if (name === 'read_clipboard_image') return readClipboardImageTool(args); + throw new Error(`Unknown tool: ${name}`); +} + +// --------------------------------------------------------------------------- +// JSON-RPC over stdio (MCP framing) +// --------------------------------------------------------------------------- + +function send(obj) { + const body = Buffer.from(JSON.stringify(obj), 'utf8'); + process.stdout.write(`Content-Length: ${body.length}\r\n\r\n`); + process.stdout.write(body); +} + +function sendError(id, code, message) { + send({ jsonrpc: '2.0', id, error: { code, message } }); +} + +async function handleRequest(msg) { + const { id, method, params } = msg; + try { + switch (method) { + case 'initialize': + send({ + jsonrpc: '2.0', + id, + result: { + protocolVersion: params?.protocolVersion || '2025-06-18', + capabilities: { tools: { listChanged: false } }, + serverInfo: { name: 'kimi-eyes', version: VERSION }, + }, + }); + return; + case 'ping': + send({ jsonrpc: '2.0', id, result: {} }); + return; + case 'tools/list': + send({ jsonrpc: '2.0', id, result: { tools: TOOLS } }); + return; + case 'tools/call': { + const name = params?.name; + const args = params?.arguments || {}; + try { + const text = await callTool(name, args); + send({ + jsonrpc: '2.0', + id, + result: { content: [{ type: 'text', text }] }, + }); + } catch (err) { + send({ + jsonrpc: '2.0', + id, + result: { + content: [{ type: 'text', text: err.message || String(err) }], + isError: true, + }, + }); + } + return; + } + default: + sendError(id, -32601, `Method not found: ${method}`); + } + } catch (err) { + sendError(id, -32603, err.message || String(err)); + } +} + +function handleMessage(msg) { + if (!msg || typeof msg !== 'object' || typeof msg.method !== 'string') return; + const isRequest = msg.id !== undefined && msg.id !== null; + if (isRequest) { + handleRequest(msg).catch((err) => sendError(msg.id, -32603, err.message || String(err))); + } + // Notifications (e.g. notifications/initialized) are intentionally ignored. +} + +// --- framing parser --------------------------------------------------------- + +let buf = Buffer.alloc(0); + +function tryParseFrame() { + const headerEnd = buf.indexOf('\r\n\r\n'); + if (headerEnd === -1) return null; + const header = buf.subarray(0, headerEnd).toString('ascii'); + const m = /Content-Length:\s*(\d+)/i.exec(header); + if (!m) return null; + const len = Number(m[1]); + const bodyStart = headerEnd + 4; + if (buf.length < bodyStart + len) return null; + const body = buf.subarray(bodyStart, bodyStart + len).toString('utf8'); + buf = buf.subarray(bodyStart + len); + return body; +} + +function pump() { + for (;;) { + const body = tryParseFrame(); + if (body === null) break; + let msg; + try { + msg = JSON.parse(body); + } catch (err) { + console.error(`[kimi-eyes] invalid JSON frame: ${err.message}`); + continue; + } + handleMessage(msg); + } +} + +process.stdin.on('data', (chunk) => { + buf = Buffer.concat([buf, chunk]); + pump(); +}); + +process.stdin.on('end', () => process.exit(0)); + +console.error(`[kimi-eyes] MCP server started (v${VERSION}, node ${process.version})`); diff --git a/mcp/vision.mjs b/mcp/vision.mjs new file mode 100644 index 0000000..c79197b --- /dev/null +++ b/mcp/vision.mjs @@ -0,0 +1,344 @@ +// kimi-eyes vision core — shared by the MCP server and the setup wizard. +// Zero-dependency. Node 18+ (native fetch, AbortSignal.timeout). + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { execFileSync } from 'node:child_process'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +export const MODELS_DB_PATH = path.join(__dirname, 'models-db.json'); + +export const VERSION = '1.0.0'; + +export const EXT_MEDIA = { + png: 'image/png', + jpg: 'image/jpeg', + jpeg: 'image/jpeg', + webp: 'image/webp', + gif: 'image/gif', + bmp: 'image/bmp', +}; + +export const DEFAULT_PROMPT = 'Describe this image in detail.'; + +// A minimal 1x1 PNG used for setup verification. +export const ONE_PX_PNG_BASE64 = + 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=='; + +// --------------------------------------------------------------------------- +// Configuration +// --------------------------------------------------------------------------- + +// Resolution order: environment variables (VISION_*) > config.json. +// protocol resolves to 'openai' by default when nothing is set. +export function getConfig(env = process.env) { + const home = env.KIMI_CODE_HOME || path.join(os.homedir(), '.kimi-code'); + const cfgPath = path.join(home, 'kimi-eyes', 'config.json'); + let file = {}; + try { + file = JSON.parse(fs.readFileSync(cfgPath, 'utf8')); + } catch { + // no config file yet — fine + } + const envProtocol = (env.VISION_API_PROTOCOL || '').toLowerCase(); + const protocol = + envProtocol === 'openai' || envProtocol === 'anthropic' + ? envProtocol + : file.protocol || 'openai'; + return { + configPath: cfgPath, + protocol, + apiKey: env.VISION_API_KEY || file.apiKey || '', + baseUrl: env.VISION_API_URL || file.baseUrl || '', + model: env.VISION_MODEL || file.model || '', + mainModel: env.VISION_MAIN_MODEL || file.mainModel || '', + }; +} + +export function checkConfig(cfg) { + const missing = []; + if (!cfg.apiKey) missing.push('VISION_API_KEY'); + if (!cfg.baseUrl) missing.push('VISION_API_URL'); + if (!cfg.model) missing.push('VISION_MODEL'); + return missing; +} + +export function requireConfigured(cfg) { + const missing = checkConfig(cfg); + if (missing.length > 0) { + throw new Error( + `Vision API is not configured. Missing: ${missing.join(', ')}. ` + + `Run "node setup.mjs" in the kimi-eyes plugin directory (or set the environment variables).` + ); + } +} + +// --------------------------------------------------------------------------- +// Models.dev capability database (see scripts/sync-models.mjs) +// --------------------------------------------------------------------------- + +let modelsDbCache = null; // null = not loaded yet, undefined = unavailable + +/** Load the slimmed models.dev database; returns null if unavailable. */ +export function loadModelsDb() { + if (modelsDbCache !== null) return modelsDbCache; + try { + modelsDbCache = JSON.parse(fs.readFileSync(MODELS_DB_PATH, 'utf8')); + } catch { + modelsDbCache = undefined; + } + return modelsDbCache; +} + +/** + * Look up whether a model name supports image input, using the models.dev cache. + * Matching: exact id → provider/id suffix → case-insensitive display name. + * @returns {{known: boolean, image: boolean}} + */ +export function lookupModelCapability(modelName) { + const db = loadModelsDb(); + if (!db) return { known: false, image: false }; + const n = String(modelName || '').trim(); + if (!n) return { known: false, image: false }; + if (db[n]) return { known: true, image: !!db[n].image }; + const lower = n.toLowerCase(); + for (const [id, m] of Object.entries(db)) { + if (id.toLowerCase().endsWith(`/${lower}`)) { + return { known: true, image: !!m.image }; + } + if (typeof m.name === 'string' && m.name.toLowerCase() === lower) { + return { known: true, image: !!m.image }; + } + } + return { known: false, image: false }; +} + +/** Text tag used by the setup wizard: exact capability when known, keyword hint otherwise. */ +export function modelCapabilityTag(modelName, visionHintRe) { + const cap = lookupModelCapability(modelName); + if (cap.known) return cap.image ? '✓视觉' : '文本'; + return visionHintRe && visionHintRe.test(modelName) ? '★疑似' : ''; +} + +// --------------------------------------------------------------------------- +// Endpoints +// --------------------------------------------------------------------------- + +export function openaiEndpoint(baseUrl) { + let b = String(baseUrl || '').trim().replace(/\/+$/, ''); + if (!b) return ''; + if (/\/chat\/completions$/i.test(b)) return b; + if (/\/v\d+$/i.test(b)) return `${b}/chat/completions`; + return `${b}/v1/chat/completions`; +} + +export function anthropicEndpoint(baseUrl) { + let b = String(baseUrl || '').trim().replace(/\/+$/, ''); + if (!b) return ''; + if (/\/v1\/messages$/i.test(b)) return b; + if (/\/v\d+$/i.test(b)) return `${b}/messages`; + return `${b}/v1/messages`; +} + +export function modelsEndpoint(baseUrl, protocol) { + let b = String(baseUrl || '').trim().replace(/\/+$/, ''); + if (!b) return ''; + if (/\/v\d+$/i.test(b)) return `${b}/models`; + return `${b}/v1/models`; +} + +// --------------------------------------------------------------------------- +// Vision API calls +// --------------------------------------------------------------------------- + +const FETCH_TIMEOUT_MS = Number(process.env.VISION_FETCH_TIMEOUT_MS) || 60000; +const MAX_TOKENS = Number(process.env.VISION_MAX_TOKENS) || 1024; + +/** + * Call the user-configured VLM with a base64 image. + * @param {{imageBase64:string, mediaType:string, prompt?:string}} input + * @param {{protocol:string, apiKey:string, baseUrl:string, model:string}} cfg + * @returns {Promise} text description + */ +export async function callVLM({ imageBase64, mediaType, prompt }, cfg) { + requireConfigured(cfg); + if (cfg.protocol === 'anthropic') { + return callAnthropic(imageBase64, mediaType, prompt, cfg); + } + return callOpenAI(imageBase64, mediaType, prompt, cfg); +} + +async function callOpenAI(imageBase64, mediaType, prompt, cfg) { + const endpoint = openaiEndpoint(cfg.baseUrl); + const body = { + model: cfg.model, + max_tokens: MAX_TOKENS, + messages: [ + { + role: 'user', + content: [ + { type: 'text', text: prompt || DEFAULT_PROMPT }, + { type: 'image_url', image_url: { url: `data:${mediaType};base64,${imageBase64}` } }, + ], + }, + ], + }; + const res = await fetch(endpoint, { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + Authorization: `Bearer ${cfg.apiKey}`, + }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + }); + if (!res.ok) { + throw new Error(`Vision API HTTP ${res.status}: ${(await safeText(res)).slice(0, 300)}`); + } + const data = await res.json(); + const content = data?.choices?.[0]?.message?.content; + if (typeof content === 'string' && content.trim()) return content.trim(); + if (Array.isArray(content)) { + const text = content + .filter((p) => p && p.type === 'text' && typeof p.text === 'string') + .map((p) => p.text) + .join('\n') + .trim(); + if (text) return text; + } + throw new Error(`Unexpected response from vision API: ${JSON.stringify(data).slice(0, 300)}`); +} + +async function callAnthropic(imageBase64, mediaType, prompt, cfg) { + const endpoint = anthropicEndpoint(cfg.baseUrl); + const body = { + model: cfg.model, + max_tokens: MAX_TOKENS, + messages: [ + { + role: 'user', + content: [ + { type: 'image', source: { type: 'base64', media_type: mediaType, data: imageBase64 } }, + { type: 'text', text: prompt || DEFAULT_PROMPT }, + ], + }, + ], + }; + const res = await fetch(endpoint, { + method: 'POST', + headers: { + 'Content-Type': 'application/json', + 'x-api-key': cfg.apiKey, + 'anthropic-version': '2023-06-01', + }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + }); + if (!res.ok) { + throw new Error(`Vision API HTTP ${res.status}: ${(await safeText(res)).slice(0, 300)}`); + } + const data = await res.json(); + const texts = (Array.isArray(data?.content) ? data.content : []) + .filter((p) => p && p.type === 'text' && typeof p.text === 'string') + .map((p) => p.text) + .join('\n') + .trim(); + if (texts) return texts; + throw new Error(`Unexpected response from vision API: ${JSON.stringify(data).slice(0, 300)}`); +} + +async function safeText(res) { + try { + return await res.text(); + } catch { + return '(could not read response body)'; + } +} + +// --------------------------------------------------------------------------- +// Image sniffing +// --------------------------------------------------------------------------- + +/** + * Validate that a file is a supported image and return base64 + media type. + * @returns {{imageBase64:string, mediaType:string}|null} + */ +export function sniffImage(absPath) { + const ext = path.extname(absPath).toLowerCase().replace('.', ''); + if (!EXT_MEDIA[ext]) return null; + let buf; + try { + buf = fs.readFileSync(absPath); + } catch { + return null; + } + if (buf.length === 0) return null; + const ascii4 = buf.toString('ascii', 0, 4); + let ok = false; + if (ext === 'png') ok = buf.subarray(0, 8).toString('hex').startsWith('89504e47'); + else if (ext === 'jpg' || ext === 'jpeg') ok = buf.subarray(0, 3).toString('hex') === 'ffd8ff'; + else if (ext === 'gif') ok = ascii4 === 'GIF8'; + else if (ext === 'bmp') ok = ascii4 === 'BM'; + else if (ext === 'webp') { + ok = buf.length > 12 && ascii4 === 'RIFF' && buf.toString('ascii', 8, 12) === 'WEBP'; + } + if (!ok) return null; + return { imageBase64: buf.toString('base64'), mediaType: EXT_MEDIA[ext] }; +} + +// --------------------------------------------------------------------------- +// Clipboard capture +// --------------------------------------------------------------------------- + +/** + * Grab the image currently in the system clipboard and save it to a temp PNG. + * @returns {string} path to the temp PNG file + */ +export function clipboardImageToTemp() { + const out = path.join( + os.tmpdir(), + `kimi-eyes-clip-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.png` + ); + if (process.platform === 'win32') { + const script = [ + 'Add-Type -AssemblyName System.Windows.Forms;', + '$img = [System.Windows.Forms.Clipboard]::GetImage();', + "if ($img -eq $null) { Write-Error 'No image in clipboard'; exit 1 };", + `$img.Save('${out.replace(/'/g, "''")}', [System.Drawing.Imaging.ImageFormat]::Png)`, + ].join(' '); + execFileSync('powershell', ['-NoProfile', '-NonInteractive', '-Command', script], { + stdio: 'pipe', + timeout: 20000, + }); + } else if (process.platform === 'darwin') { + try { + execFileSync('pngpaste', [out], { stdio: 'pipe', timeout: 20000 }); + } catch (e) { + throw new Error('Clipboard capture failed. Install "pngpaste" (brew install pngpaste). ' + e.message); + } + } else { + try { + const png = execFileSync('wl-paste', ['-t', 'image/png', '--no-newline'], { + stdio: 'pipe', + timeout: 15000, + }); + fs.writeFileSync(out, png); + } catch (e1) { + try { + const png = execFileSync('xclip', ['-selection', 'clipboard', '-t', 'image/png', '-o'], { + stdio: 'pipe', + timeout: 15000, + }); + fs.writeFileSync(out, png); + } catch (e2) { + throw new Error( + 'Clipboard capture failed. Need "wl-paste" (Wayland) or "xclip" (X11). ' + + `${e1.message}; ${e2.message}` + ); + } + } + } + return out; +} diff --git a/package.json b/package.json new file mode 100644 index 0000000..8d6cf21 --- /dev/null +++ b/package.json @@ -0,0 +1,40 @@ +{ + "name": "kimi-eyes", + "version": "1.0.0", + "description": "Give non-multimodal models in Kimi Code the ability to analyze images and screenshots via a user-configured VLM (OpenAI-compatible or Anthropic protocol).", + "keywords": [ + "kimi", + "kimi-code", + "vision", + "image", + "screenshot", + "multimodal", + "vlm", + "mcp", + "plugin" + ], + "license": "MIT", + "author": "billowliu2", + "type": "module", + "bin": { + "kimi-eyes": "setup.mjs" + }, + "files": [ + "mcp/", + "scripts/", + "SYSTEM.md", + "kimi.plugin.json", + "setup.mjs", + "README.md", + "README.en.md", + "LICENSE" + ], + "engines": { + "node": ">=18" + }, + "scripts": { + "setup": "node setup.mjs", + "sync:models": "node scripts/sync-models.mjs", + "prepublishOnly": "node scripts/sync-models.mjs" + } +} diff --git a/scripts/sync-models.mjs b/scripts/sync-models.mjs new file mode 100644 index 0000000..579ce7a --- /dev/null +++ b/scripts/sync-models.mjs @@ -0,0 +1,95 @@ +#!/usr/bin/env node +// kimi-eyes — sync the models.dev capability database into a slim local cache. +// Run this when packaging/releasing the plugin: +// +// node scripts/sync-models.mjs +// +// Options: +// --timeout curl timeout (default 180) +// --out output file (default ./mcp/models-db.json) +// --endpoint override the source URL (default https://models.dev/models.json) +// +// Proxy: the script uses the HTTPS_PROXY / https_proxy environment variable if set +// (for users behind Clash/V2Ray etc.), falling back to a direct connection. + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { execFileSync } from 'node:child_process'; + +const DEFAULT_ENDPOINT = 'https://models.dev/models.json'; +const DEFAULT_OUT = path.join('mcp', 'models-db.json'); + +// Models known to support image input but missing from models.dev. +// Maintained manually — merged into the db on every sync so re-syncs never drop them. +const EXTRA_ENTRIES = { + 'k3-256k': { name: 'K3-256k', image: true }, + 'kimi-k3-256k': { name: 'K3-256k', image: true }, + 'moonshot/k3-256k': { name: 'K3-256k', image: true }, + 'kimi-for-coding': { name: 'Kimi For Coding', image: true }, + 'moonshotai/kimi-for-coding': { name: 'Kimi For Coding', image: true }, + 'kimi-for-coding-highspeed': { name: 'Kimi For Coding Highspeed', image: true }, + 'moonshotai/kimi-for-coding-highspeed': { name: 'Kimi For Coding Highspeed', image: true }, +}; + +function parseArgs(argv) { + const args = { timeout: 180, out: DEFAULT_OUT, endpoint: DEFAULT_ENDPOINT }; + for (let i = 0; i < argv.length; i++) { + if (argv[i] === '--timeout') args.timeout = Number(argv[++i]) || 180; + else if (argv[i] === '--out') args.out = argv[++i]; + else if (argv[i] === '--endpoint') args.endpoint = argv[++i]; + } + return args; +} + +function download(endpoint, outFile, timeout) { + const proxy = process.env.HTTPS_PROXY || process.env.https_proxy; + const curl = ['-sSL', '--max-time', String(timeout), '-o', outFile, endpoint]; + const args = proxy ? ['-x', proxy, ...curl] : curl; + try { + execFileSync('curl', args, { stdio: 'inherit', timeout: (timeout + 30) * 1000 }); + } catch (err) { + throw new Error( + `Failed to download ${endpoint}${proxy ? ` via proxy ${proxy}` : ''}. ` + + `If you are behind a proxy, set HTTPS_PROXY (e.g. HTTPS_PROXY=http://127.0.0.1:7897). ` + + `(${err.message.split('\n')[0]})` + ); + } +} + +const args = parseArgs(process.argv.slice(2)); +const tmpFile = path.join(os.tmpdir(), `models-dev-${Date.now()}.json`); + +console.log(`[sync-models] downloading ${args.endpoint} ...`); +download(args.endpoint, tmpFile, args.timeout); + +const raw = JSON.parse(fs.readFileSync(tmpFile, 'utf8')); +if (typeof raw !== 'object' || raw === null || Array.isArray(raw)) { + fs.rmSync(tmpFile, { force: true }); + throw new Error(`Unexpected response shape from ${args.endpoint} (expected a flat model map).`); +} + +// Slim down: keep only what the plugin needs. +const slim = {}; +let imageCount = 0; +for (const [id, m] of Object.entries(raw)) { + if (!m || typeof m !== 'object') continue; + const input = Array.isArray(m.modalities?.input) ? m.modalities.input : []; + const image = input.includes('image'); + if (image) imageCount++; + slim[id] = { name: typeof m.name === 'string' ? m.name : id, image }; +} +// Merge manual extras (they override the downloaded data on id collisions). +for (const [id, m] of Object.entries(EXTRA_ENTRIES)) { + slim[id] = m; + if (m.image) imageCount++; +} + +fs.rmSync(tmpFile, { force: true }); +fs.mkdirSync(path.dirname(args.out), { recursive: true }); +fs.writeFileSync(args.out, `${JSON.stringify(slim)}\n`, 'utf8'); + +const kb = (fs.statSync(args.out).size / 1024).toFixed(0); +console.log( + `[sync-models] wrote ${Object.keys(slim).length} models (${imageCount} support image input) to ${args.out} (${kb} KB)` +); diff --git a/setup.mjs b/setup.mjs new file mode 100644 index 0000000..79c7a04 --- /dev/null +++ b/setup.mjs @@ -0,0 +1,241 @@ +#!/usr/bin/env node +// kimi-eyes setup wizard — interactive one-time configuration. +// Zero-dependency: raw-mode stdin input + native fetch. +// +// Flow: choose protocol → base URL → API key → pick multimodal model → verify → write config. + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { callVLM, modelsEndpoint, ONE_PX_PNG_BASE64, modelCapabilityTag } from './mcp/vision.mjs'; + +// --------------------------------------------------------------------------- +// Raw-mode line input (supports masked echo) +// --------------------------------------------------------------------------- + +let pipeLines = null; +function ensurePipeLines() { + if (!pipeLines) { + pipeLines = new Promise((resolve, reject) => { + let s = ''; + process.stdin.setEncoding('utf8'); + process.stdin.on('data', (c) => (s += c)); + process.stdin.on('end', () => resolve(s.split(/\r?\n/))); + process.stdin.on('error', reject); + process.stdin.resume(); + }); + } + return pipeLines; +} + +function ask(query, { hidden = false } = {}) { + const stdin = process.stdin; + if (!stdin.isTTY) { + // Non-TTY fallback (piped input in tests/CI): plain line read, no masking. + process.stdout.write(query); + return ensurePipeLines().then((lines) => lines.shift() ?? ''); + } + process.stdout.write(query); + stdin.setRawMode(true); + stdin.resume(); + return new Promise((resolve) => { + let val = ''; + let esc = false; + const cleanup = () => { + stdin.setRawMode(false); + stdin.removeListener('data', onData); + process.stdout.write('\n'); + }; + const onData = (chunk) => { + for (const ch of chunk.toString('utf8')) { + if (esc) { + if ((ch >= 'A' && ch <= 'D') || ch === '~') esc = false; + continue; + } + if (ch === '\u001b') { + esc = true; + continue; + } + if (ch === '\r' || ch === '\n') { + cleanup(); + resolve(val); + return; + } + if (ch === '\u0003') process.exit(130); + if (ch === '\u007f' || ch === '\b') { + if (val.length > 0) { + val = val.slice(0, -1); + process.stdout.write('\b \b'); + } + continue; + } + val += ch; + process.stdout.write(hidden ? '*' : ch); + } + }; + stdin.on('data', onData); + }); +} + +async function askRequired(query, { hidden = false, validate = null } = {}) { + for (;;) { + const v = (await ask(query, { hidden })).trim(); + if (!v) { + console.log(' 输入不能为空,请重试。'); + continue; + } + if (validate) { + const err = validate(v); + if (err) { + console.log(` ${err}`); + continue; + } + } + return v; + } +} + +// --------------------------------------------------------------------------- +// Helpers +// --------------------------------------------------------------------------- + +const VISION_HINTS = + /(vision|visual|vlm|multimodal|4o|omni|glm-4v|qwen-vl|qwen2.*?vl|internvl|minicpm|gpt-4v|gpt-5|claude-3|claude-sonnet|claude-opus|gemini)/i; + +async function fetchModels(protocol, baseUrl, apiKey) { + const url = modelsEndpoint(baseUrl, protocol); + const headers = + protocol === 'anthropic' + ? { 'x-api-key': apiKey, 'anthropic-version': '2023-06-01' } + : { Authorization: `Bearer ${apiKey}` }; + const res = await fetch(url, { headers, signal: AbortSignal.timeout(15000) }); + if (!res.ok) throw new Error(`HTTP ${res.status}`); + const data = await res.json(); + const list = Array.isArray(data?.data) + ? data.data.map((m) => m.id || m.name).filter((x) => typeof x === 'string' && x) + : []; + if (list.length === 0) throw new Error('接口未返回任何模型'); + return [...new Set(list)]; +} + +async function chooseProtocol() { + console.log(' 选择视觉 API 协议:'); + console.log(' [1] OpenAI 兼容 (Chat Completions)'); + console.log(' [2] Anthropic (Messages API)'); + for (;;) { + const v = (await ask(' 输入序号 [1/2]: ')).trim(); + if (v === '1') return 'openai'; + if (v === '2') return 'anthropic'; + console.log(' 无效输入,请输入 1 或 2。'); + } +} + +async function chooseModel(protocol, baseUrl, apiKey) { + console.log(' 拉取模型列表 ...'); + let models = null; + try { + models = await fetchModels(protocol, baseUrl, apiKey); + } catch (err) { + console.log(` 拉取模型列表失败 (${err.message}),将改为手动输入模型名。`); + } + if (models) { + const shown = models.slice(0, 50); + console.log( + ` 可用模型 (✓视觉=支持识图 / 文本=纯文本 / ★疑似=未收录按关键词猜测): ` + + `${models.length > 50 ? `显示前 50 / 共 ${models.length}` : ''}` + ); + shown.forEach((m, i) => { + const tag = modelCapabilityTag(m, VISION_HINTS); + console.log(` [${String(i + 1).padStart(3)}] ${m}${tag ? ` ${tag}` : ''}`); + }); + console.log(' 输入序号选择模型,或直接输入自定义模型名。'); + for (;;) { + const v = (await ask(' 模型: ')).trim(); + if (v === '') { + console.log(' 输入不能为空。'); + continue; + } + const idx = Number.parseInt(v, 10); + if (!Number.isNaN(idx) && idx >= 1 && idx <= shown.length) return shown[idx - 1]; + return v; // treat as custom model name + } + } + return askRequired(' 模型名: '); +} + +async function verifyModel(cfg) { + console.log(' 视觉验证:发送 1×1 测试图到模型 ...'); + try { + const text = await callVLM( + { imageBase64: ONE_PX_PNG_BASE64, mediaType: 'image/png', prompt: 'Reply with the single word OK.' }, + cfg + ); + console.log(` ✓ 通过,模型可接收图片 (返回: ${text.slice(0, 60)})`); + return true; + } catch (err) { + console.log(` ✗ 失败: ${err.message.slice(0, 200)}`); + return false; + } +} + +// --------------------------------------------------------------------------- +// Main +// --------------------------------------------------------------------------- + +async function main() { + const home = process.env.KIMI_CODE_HOME || path.join(os.homedir(), '.kimi-code'); + const configPath = path.join(home, 'kimi-eyes', 'config.json'); + + console.log('=== Kimi Eyes 配置向导 ==='); + console.log(`配置将写入: ${configPath}\n`); + + const protocol = await chooseProtocol(); + const baseUrl = await askRequired(' 视觉 API BaseUrl (如 https://api.openai.com/v1): ', { + validate: (v) => (/^https?:\/\//i.test(v) ? null : 'BaseUrl 必须以 http:// 或 https:// 开头。'), + }); + const apiKey = await askRequired(' API Key: ', { hidden: true }); + const model = await chooseModel(protocol, baseUrl, apiKey); + + const cfg = { protocol, baseUrl, apiKey, model }; + + console.log(''); + for (let attempt = 0; attempt < 3; attempt++) { + if (await verifyModel(cfg)) break; + if (attempt === 2) { + console.log(' 连续验证失败,跳过验证(配置仍将写入)。'); + break; + } + const act = (await ask(' [r] 重试 [c] 换模型 [s] 跳过验证: ')).trim().toLowerCase(); + if (act === 'c') { + cfg.model = await chooseModel(protocol, baseUrl, apiKey); + continue; + } + if (act === 's') break; + // 'r' or anything else → retry the loop + } + + const mainModel = (await ask(' 当前主模型名(可选,用于触发判断;留空跳过): ')).trim(); + if (mainModel) cfg.mainModel = mainModel; + + const dir = path.dirname(configPath); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(configPath, `${JSON.stringify(cfg, null, 2)}\n`, 'utf8'); + if (process.platform !== 'win32') fs.chmodSync(configPath, 0o600); + + console.log(''); + console.log(`✓ 配置已写入: ${configPath}`); + console.log(` protocol: ${cfg.protocol}`); + console.log(` baseUrl : ${cfg.baseUrl}`); + console.log(` model : ${cfg.model}`); + if (cfg.mainModel) console.log(` mainModel: ${cfg.mainModel} (触发判断用)`); + console.log(''); + console.log('下一步:'); + console.log(` 1. 安装插件: /plugins install ${process.cwd()}`); + console.log(' 2. 启用插件: /reload'); + console.log(' 3. 开始使用: 非多模态模型下 @图片路径 或 截图后提问'); +} + +main().catch((err) => { + console.error(`\n[setup] 出错: ${err.message}`); + process.exit(1); +});