From f151dd1a98db7a1c436e2a71e6253b4584ba80ee Mon Sep 17 00:00:00 2001 From: Naser Derakhshan <9349840+nasser1941@users.noreply.github.com> Date: Thu, 1 Oct 2026 13:05:11 +0200 Subject: [PATCH 1/2] Let lcode look at images, and generate them through ComfyUI Images in: attach a screenshot, mockup or diagram with @path, or the model calls the new view_image tool. A model that can see describes the image (all text transcribed, layout and colors, focused on the question) and the description goes into the conversation, so any model can use it: - vision_model = "auto": the session's model if it can see; otherwise the original Ollama model behind lcode's text-only variant (same weights plus the vision projector, already downloaded by lcode setup), such as qwen3.6:35b-a3b-coding; or a configured model (e.g. qwen3-vl:8b). - A model that can see keeps the session's options, so Ollama doesn't reload it. - Images returned by MCP tools (Playwright screenshots) are described too. - read_file on an image points to view_image; the sandbox's path limits apply; lcode doctor shows the vision model. Images out: `lcode mcp add comfyui` adds ComfyUI's official MCP server (comfy-mcp via uvx, with comfy-cli), with only its 16 local tools enabled (no Comfy Cloud partner tools), to generate and edit images with local models such as FLUX.2 Klein, FLUX.1 Dev/Kontext, SDXL, SD 1.5 and Qwen-Image. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 12 ++++ README.md | 5 +- docs/configuration.md | 1 + docs/how-it-works.md | 1 + docs/mcp.md | 33 +++++++++++ docs/models.md | 6 +- docs/usage.md | 28 ++++++++++ src/lcode/agent.py | 57 ++++++++++++++++++- src/lcode/cli.py | 12 ++++ src/lcode/config.py | 1 + src/lcode/mcp/catalog.toml | 26 +++++++++ src/lcode/mcp/manager.py | 6 +- src/lcode/mcp/protocol.py | 7 ++- src/lcode/ollama.py | 4 ++ src/lcode/tools.py | 22 +++++++- src/lcode/vision.py | 109 +++++++++++++++++++++++++++++++++++++ tests/conftest.py | 10 ++++ tests/test_vision.py | 104 +++++++++++++++++++++++++++++++++++ 18 files changed, 432 insertions(+), 12 deletions(-) create mode 100644 src/lcode/vision.py create mode 100644 tests/test_vision.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 05d44a4..b96574e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,18 @@ All notable changes to lcode are documented here. The format follows ## [Unreleased] +### Added + +- lcode can look at images: attach a screenshot, mockup or diagram with `@path`, and the model can + open images itself with the new `view_image` tool. A model that can see describes the image in + detail (all text transcribed, with your question in mind): your model itself if it can see, or + the original model behind lcode's text-only variant, which `lcode setup` already downloaded + (`vision_model` setting; `lcode doctor` shows which). Screenshots returned by MCP tools, such as + Playwright's, are described too. +- `lcode mcp add comfyui`: generate and edit images with models you run locally in ComfyUI (FLUX, + SDXL, Stable Diffusion 1.5, Qwen-Image), through ComfyUI's official MCP server, with only its local + tools enabled. + ## [0.5.0] - 2026-10-01 ### Added diff --git a/README.md b/README.md index 169ea11..f26f5c8 100644 --- a/README.md +++ b/README.md @@ -52,7 +52,10 @@ through [Ollama](https://ollama.com), and when it needs current information it c with tool calling. - **MCP servers, ready to go.** Connect Jira and Confluence, GitHub, AWS, Google Drive, Grafana, Google Cloud, Sentry, Linear, Notion, Postgres, Kubernetes and more with one command - (`lcode mcp add atlassian`), or any other MCP server. Browser sign-in (OAuth) is built in. + (`lcode mcp add atlassian`), image generation with ComfyUI, or any other MCP server. Browser sign-in (OAuth) is built in. +- **Sees images.** Attach a screenshot or mockup with `@path` and lcode looks at it, using the + vision part of your model. With ComfyUI (`lcode mcp add comfyui`) it can also generate and edit + images with local models such as FLUX, SDXL and Qwen-Image. - **Web search when needed.** Looks up the latest versions, docs and error messages with Ollama web search, Brave, Tavily or your own SearXNG, and reads pages as clean text. - **Private.** The model runs on your machine, lcode never uploads your files and has no telemetry. diff --git a/docs/configuration.md b/docs/configuration.md index d6dc185..097656b 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -28,6 +28,7 @@ lcode config path # print the file location | `sandbox` | `off` | Run shell commands in a container: `docker` or `podman` ([Sandbox](sandbox.md)) | | `sandbox_image` | lcode's image | Container image for the sandbox; any image with bash and setsid | | `sandbox_network` | `false` | Let commands in the sandbox use the network | +| `vision_model` | `auto` | The model that [looks at images](usage.md#images): `auto`, `off` or an Ollama model that can see | | `mcp_tools` | `auto` | How [MCP](mcp.md#context) tool definitions reach the model: `auto`, `direct` or `search` (on demand) | | `checkpoints` | `true` | Save a checkpoint before the model changes files, so [`/undo`](usage.md#undo-and-checkpoints) can restore them | diff --git a/docs/how-it-works.md b/docs/how-it-works.md index 56643ee..44907ac 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -77,6 +77,7 @@ the Ollama server you configure, it only contacts the web when the model searche | `permissions.py` | Approval prompts and the read-only allowlist | | `mcp/` | MCP client: stdio and HTTP transports, OAuth sign-in, server settings, the catalog, `lcode mcp` | | `bench.py` | `lcode bench`: the benchmark tasks, their checks and the reports | +| `vision.py` | Looking at images with a model that can see | | `sandbox.py` | The optional container for shell commands | | `checkpoints.py` | Snapshots before the model changes files, for `/undo` and `/rewind` | | `catalog.py`, `models.toml` | Model catalog and memory estimates | diff --git a/docs/mcp.md b/docs/mcp.md index b909a50..06a71aa 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -28,6 +28,7 @@ and find the code for the top one."* | `gcp` | `gcloud` commands on your Google Cloud projects | `npx` and the `gcloud` CLI, signed in | | `github` | Repositories, issues, pull requests, Actions | a GitHub token (or the GitHub CLI) | | `playwright` | A real (headless) browser: open pages, click, fill forms | `npx` | +| `comfyui` | Generate and edit images with local models (FLUX, SDXL, SD 1.5, Qwen-Image) | `uvx` and [ComfyUI](#images-with-comfyui) running locally | | `context7` | Up-to-date docs and examples for thousands of libraries | nothing (an API key is optional) | | `sentry` | Errors, issues, traces and releases | a Sentry account (browser sign-in) | | `postgres` | Schemas, read-only queries, query performance | `uvx`, a connection URL | @@ -82,6 +83,38 @@ Google Cloud project: See [Google's guide](https://developers.google.com/workspace/guides/configure-mcp-servers) for details. +### Images with ComfyUI + +[ComfyUI](https://github.com/comfyanonymous/ComfyUI) runs image generation and editing models +locally; its official MCP server lets lcode use it, for app icons, illustrations, mockups or +placeholder art. + +```bash +uv tool install comfy-cli && comfy install && comfy launch # ComfyUI on http://127.0.0.1:8188 +lcode mcp add comfyui +``` + +Then add models in ComfyUI (its model manager, or ask lcode: *"download FLUX.2 Klein 4B in +ComfyUI"*), and ask for images: *"make a 512×512 app icon of a paper plane and save it in +assets/"*. To edit an image, ask lcode to upload it and run an editing workflow (Kontext or +Qwen-Image-Edit). + +| Model | Good for | On a 12 GB GPU | +|---|---|---| +| FLUX.2 Klein 4B | Generation and editing, fast | Fits (about 8 GB) | +| SDXL and its community models | Generation, inpainting | Fits | +| Stable Diffusion 1.5 and its community models | Light generation, inpainting | Fits easily | +| FLUX.1 Dev and finetunes | High-quality generation | Needs an fp8 or GGUF version; slower | +| FLUX.1 Kontext Dev | Editing an image from instructions | Needs an fp8 or GGUF version; slower | +| Qwen-Image / Qwen-Image-Edit | Generation and editing, good text in images | Heavy: a GGUF version and RAM offloading | + +Only ComfyUI's local tools are turned on; Comfy Cloud's partner tools are left out, so prompts and +images stay on your machine. One GPU can't hold a large coding model and an image model at the +same time: while lcode's model is loaded, ComfyUI runs slowly or runs out of memory. Use a smaller +image model, generate between requests (the `free_memory` tool frees ComfyUI's memory; Ollama frees +lcode's after `keep_alive`), or run ComfyUI on another machine and give its address when adding. +lcode can also look at the results ([Images](usage.md#images)). + ## Adding your own servers Any MCP server works. For a remote server, give its URL; for a local one, the command that starts it: diff --git a/docs/models.md b/docs/models.md index 382c947..b7b7cb5 100644 --- a/docs/models.md +++ b/docs/models.md @@ -225,7 +225,9 @@ GPU's share of unified memory. Results from other Macs are very welcome in the ## Text-only variants -Some models (the Qwen3.6 family, for example) ship with a vision encoder that a coding agent doesn't -use. `lcode setup` creates a text-only variant named `lcode-` that reuses the downloaded +Some models (the Qwen3.6 family, for example) ship with a vision encoder that a coding agent rarely +needs. `lcode setup` creates a text-only variant named `lcode-` that reuses the downloaded weights, so it takes no extra disk space, and frees about 1 GB of GPU memory for the context cache (and for a larger prompt batch, where it fits; see [tuning](configuration.md#tuning-for-speed-and-memory)). +When you show lcode an image, it borrows the original model, vision encoder included, to look at it +(see [Images](usage.md#images)). diff --git a/docs/usage.md b/docs/usage.md index 2cb54e8..e82e291 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -86,6 +86,34 @@ reasoning is on. lcode refuses to edit a file the model hasn't read in the session, or one that changed on disk since it was read, so the model always edits the current version. +## Images + +lcode can look at screenshots, mockups, diagrams and photos: + +```text +❯ the layout breaks on mobile, see @screenshots/mobile.png +❯ make the settings page match @design/settings.png +``` + +- Attach an image with `@path`, like a file. lcode describes it in detail, with all visible text + transcribed and with your question in mind, and the model works from that description. +- The model can open images itself with the `view_image` tool, for example a screenshot a test wrote. +- Screenshots that [MCP](mcp.md) tools return, such as the Playwright browser's, are described too. + +**Which model looks.** lcode's text-only model variants leave out the vision part to save GPU +memory, so lcode borrows the original model, which `lcode setup` already downloaded (for example +`qwen3.6:35b-a3b-coding` for `qwen3.6-35b`). If your model can see, it looks itself. `lcode doctor` +shows which one is used. With another model, install one that can see and point lcode to it: + +```bash +ollama pull qwen3-vl:8b +lcode config set vision_model qwen3-vl:8b +``` + +Loading a second model takes a moment: on a 12 GB GPU, describing an image with the original +`qwen3.6:35b-a3b-coding` takes about a minute, and the next request reloads the main model. Turn +image support off with `lcode config set vision_model off`. + ## Permissions | Mode | File edits | Shell commands | diff --git a/src/lcode/agent.py b/src/lcode/agent.py index 360f072..48dcb1b 100644 --- a/src/lcode/agent.py +++ b/src/lcode/agent.py @@ -2,6 +2,7 @@ from __future__ import annotations +import base64 import datetime as dt import json import platform @@ -19,7 +20,7 @@ from rich.panel import Panel from rich.text import Text -from lcode import catalog, limits, sessions, web +from lcode import catalog, limits, sessions, vision, web from lcode.checkpoints import Checkpoints from lcode.config import format_tokens from lcode.mcp import McpManager @@ -29,9 +30,11 @@ from lcode.sandbox import Sandbox, SandboxError, project_root from lcode.tools import ( SCHEMAS, + VIEW_IMAGE_SCHEMA, WEB_FETCH_SCHEMA, WEB_SEARCH_SCHEMA, Toolbox, + ToolError, is_binary, parse_text_tool_calls, tree, @@ -144,6 +147,7 @@ class Settings: sandbox: str = "off" # off, docker or podman: where the model's shell commands run sandbox_image: str | None = None sandbox_network: bool = False + vision_model: str = "auto" # auto, off or an Ollama model that can see images class Agent: @@ -163,6 +167,7 @@ def __init__(self, ollama: Ollama, settings: Settings, cwd: Path, console: Conso else None ) self._mcp_prompt = "" # the MCP part at the end of the system prompt + self._vision: str | bool | None = False # the model that looks at images; False = not decided yet self.session_name = "" self.session_title = "" self.ctx_used = 0 @@ -204,10 +209,48 @@ def tool_schemas(self) -> list[dict]: schemas = list(SCHEMAS) if self.settings.web != "off": schemas += [*([WEB_SEARCH_SCHEMA] if self.search_backend() else []), WEB_FETCH_SCHEMA] + if self.vision_model(): + schemas.append(VIEW_IMAGE_SCHEMA) if self.mcp: schemas += self.mcp.schemas(self.settings.context) return schemas + def vision_model(self) -> str | None: + """The model that looks at images for this session, if any (see lcode.vision).""" + if self._vision is False: + self._vision = vision.pick_model(self.ollama, self.settings.model, self.settings.vision_model) + return self._vision or None + + def look(self, path: Path, question: str = "") -> str: + """Describe an image with the vision model. Raises ToolError if it can't.""" + model = self.vision_model() + if not model: + raise ToolError(f"can't look at {path.name}: {vision.INSTALL_HINT}") + try: + image = vision.read_image(path) + # The session's own model keeps its settings, so Ollama doesn't reload it. + options = self.options() if model == self.settings.model else None + loading = "" if model == self.settings.model else " (loads it; the next request reloads the main model)" + with self.console.status(f"Looking at {path.name} with {model}{loading}…", spinner="dots"): + return vision.describe(self.ollama, model, image, question, options, self.settings.keep_alive) + except vision.VisionError as e: + raise ToolError(str(e)) from e + + def describe_image_data(self, data: str, mime: str) -> str: + """For images that tools return (e.g. an MCP browser's screenshots): a description, or a note.""" + model = self.vision_model() + if not model: + return f"[{mime} image not shown: {vision.INSTALL_HINT}]" + options = self.options() if model == self.settings.model else None + try: + with self.console.status(f"Looking at the {mime} image with {model}…", spinner="dots"): + text = vision.describe( + self.ollama, model, base64.b64decode(data), "", options, self.settings.keep_alive + ) + except (vision.VisionError, ValueError) as e: + return f"[{mime} image not shown: {e}]" + return f"[{mime} image, as described by {model}]\n{text}" + def prepare_mcp(self) -> None: """Before a request: wait for MCP servers still starting and describe them in the system prompt.""" if not self.mcp: @@ -559,7 +602,17 @@ def expand_mentions(self, text: str) -> str: attached = [] for ref in re.findall(r"(?\n{description}\n') + elif p.is_file() and not is_binary(p) and p.stat().st_size < 200_000: attached.append(f'\n{p.read_text(errors="replace")}\n') self.tools.read_mtimes[str(p)] = p.stat().st_mtime self.console.print(Text(f" ⎿ attached {self.tools.rel(p)}", style="dim")) diff --git a/src/lcode/cli.py b/src/lcode/cli.py index bee7211..98d0f13 100644 --- a/src/lcode/cli.py +++ b/src/lcode/cli.py @@ -273,6 +273,17 @@ def line(label: str, value: str, good: bool | None = True) -> None: if spec: detail += f" · ~{spec.memory_gib(ctx):.0f} GB needed, ~{hw.budget_gib:.0f} GB available" line("Context", detail + (f" ({note})" if note else ""), None if note else True) + from lcode import vision + + seer = vision.pick_model(ollama, model, cfg["vision_model"]) + if seer: + line( + "Vision", + f"{seer} looks at screenshots and images" + ("" if seer == model else " (loaded when needed)"), + True, + ) + else: + line("Vision", "off" if cfg["vision_model"] == "off" else vision.INSTALL_HINT, None) if limits.get(model): line( "Limit", @@ -497,6 +508,7 @@ def cmd_chat(args) -> None: sandbox=(cfg["sandbox"] if cfg["sandbox"] != "off" else "docker") if args.sandbox else cfg["sandbox"], sandbox_image=cfg["sandbox_image"], sandbox_network=cfg["sandbox_network"], + vision_model=cfg["vision_model"], ) agent = Agent(ollama, settings, cwd, console=console) if agent.sandbox: diff --git a/src/lcode/config.py b/src/lcode/config.py index f6d2b55..f21147c 100644 --- a/src/lcode/config.py +++ b/src/lcode/config.py @@ -43,6 +43,7 @@ "sandbox": ("off", str, "run the model's shell commands in a container: off | docker | podman"), "sandbox_image": (None, str, "container image for the sandbox (default: lcode's, built on first use)"), "sandbox_network": (False, bool, "let commands in the sandbox use the network"), + "vision_model": ("auto", str, "model that looks at images: auto | off | an Ollama model with vision"), "mcp_tools": ("auto", str, "how MCP tools reach the model: auto | direct | search (on demand, saves context)"), } ENV_OVERRIDES = { diff --git a/src/lcode/mcp/catalog.toml b/src/lcode/mcp/catalog.toml index 4022cc3..ec2bedc 100644 --- a/src/lcode/mcp/catalog.toml +++ b/src/lcode/mcp/catalog.toml @@ -130,6 +130,32 @@ requires = ["npx"] setup = "Runs a headless browser; the first use may download it." server = { command = "npx", args = ["-y", "@playwright/mcp@latest", "--headless"] } +[comfyui] +name = "ComfyUI" +description = "Generate and edit images with models you run locally in ComfyUI: FLUX, SDXL, SD 1.5, Qwen-Image" +homepage = "https://github.com/Comfy-Org/comfy-mcp" +requires = ["uvx"] +setup = """Needs ComfyUI on this machine (http://127.0.0.1:8188) with at least one image model: + uv tool install comfy-cli && comfy install && comfy launch +Download models with ComfyUI's model manager or ask lcode to (search_models, download_model). Only the +local tools are on: generating, running workflows and templates, models, uploading an image to edit, +jobs and memory. Comfy Cloud's partner tools are left out, so nothing leaves this machine.""" + +[comfyui.server] +command = "uvx" +args = ["--python", "3.12", "--from", "comfy-mcp", "--with", "comfy-cli>=1.14.0", "comfy-mcp"] +env = { COMFYUI_URL = "${COMFYUI_URL}" } +tools = [ + "server_info", "generate_image", "run_workflow", "run_template", "search_templates", "get_template", + "list_workflow_slots", "set_workflow_slot", "job", "fetch_outputs", "upload_file", "search_models", + "download_model", "system_stats", "free_memory", "launch_comfyui", +] + +[[comfyui.inputs]] +var = "COMFYUI_URL" +prompt = "ComfyUI address (Enter for this machine, 127.0.0.1:8188)" +optional = true + [context7] name = "Context7" description = "Current documentation and code examples for thousands of libraries" diff --git a/src/lcode/mcp/manager.py b/src/lcode/mcp/manager.py index 3ddc477..33d7d79 100644 --- a/src/lcode/mcp/manager.py +++ b/src/lcode/mcp/manager.py @@ -321,11 +321,11 @@ def allowed(self, state: ServerState, tool: dict) -> bool: """Tools the user allowed in mcp.json run without asking.""" return "*" in state.cfg.allow or tool["name"] in state.cfg.allow - def call(self, state: ServerState, tool: dict, arguments: dict) -> str: + def call(self, state: ServerState, tool: dict, arguments: dict, image_text=None) -> str: try: if state.conn is None: raise TransportError("the server isn't connected") - return result_text(state.conn.call_tool(tool, arguments, state.cfg.timeout)) + return result_text(state.conn.call_tool(tool, arguments, state.cfg.timeout), image_text) except AuthRequired: state.status = "login" state.error = f"sign-in expired: run /mcp login {state.name}" @@ -336,7 +336,7 @@ def call(self, state: ServerState, tool: dict, arguments: dict) -> str: self.restart(state.name) if state.status == "ready" and state.conn: try: - return result_text(state.conn.call_tool(tool, arguments, state.cfg.timeout)) + return result_text(state.conn.call_tool(tool, arguments, state.cfg.timeout), image_text) except McpError as again: return f"Error: {again}" return f"Error: {state.name}: {e}" diff --git a/src/lcode/mcp/protocol.py b/src/lcode/mcp/protocol.py index ab3b79c..ccf0df2 100644 --- a/src/lcode/mcp/protocol.py +++ b/src/lcode/mcp/protocol.py @@ -10,6 +10,7 @@ import base64 import json import re +from collections.abc import Callable from lcode import __version__ @@ -99,8 +100,8 @@ def walk(node: dict, values) -> bool: return headers if walk(schema, arguments) else None -def result_text(result: dict) -> str: - """Turn a tools/call result into text for the model.""" +def result_text(result: dict, image_text: Callable[[str, str], str] | None = None) -> str: + """Turn a tools/call result into text for the model. `image_text(data, mime)` describes images.""" parts = [] for item in result.get("content") or []: if not isinstance(item, dict): @@ -108,6 +109,8 @@ def result_text(result: dict) -> str: kind = item.get("type") if kind == "text": parts.append(str(item.get("text", ""))) + elif kind == "image" and image_text and item.get("data"): + parts.append(image_text(item["data"], item.get("mimeType", "image/png"))) elif kind in ("image", "audio"): parts.append(f"[{kind} ({item.get('mimeType', 'unknown type')}) not shown]") elif kind == "resource_link": diff --git a/src/lcode/ollama.py b/src/lcode/ollama.py index 81113cf..022957f 100644 --- a/src/lcode/ollama.py +++ b/src/lcode/ollama.py @@ -68,6 +68,10 @@ def installed_names(self) -> set[str]: names.add(m["name"][: -len(":latest")]) return names + def chat(self, payload: dict, timeout: float = 900) -> dict: + """One chat request without streaming.""" + return self._post("/api/chat", {**payload, "stream": False}, timeout=timeout) + def show(self, model: str) -> dict: return self._post("/api/show", {"model": model}) diff --git a/src/lcode/tools.py b/src/lcode/tools.py index f4fa5af..1fe4f4d 100644 --- a/src/lcode/tools.py +++ b/src/lcode/tools.py @@ -25,6 +25,7 @@ from lcode import web from lcode.permissions import bash_key, is_read_only from lcode.sandbox import SandboxError +from lcode.vision import is_image if TYPE_CHECKING: from lcode.agent import Agent @@ -162,7 +163,17 @@ def _fn(name: str, description: str, properties: dict, required: list[str]) -> d }, ["url"], ) -TOOL_NAMES = {s["function"]["name"] for s in [*SCHEMAS, WEB_SEARCH_SCHEMA, WEB_FETCH_SCHEMA]} +VIEW_IMAGE_SCHEMA = _fn( + "view_image", + "Look at an image file (screenshot, diagram, photo, mockup): returns a detailed description with all visible " + "text transcribed. Ask a specific question to focus it.", + { + "path": {"type": "string", "description": "Image file: png, jpg, webp or gif"}, + "question": {"type": "string", "description": "What you need to know from the image (optional)"}, + }, + ["path"], +) +TOOL_NAMES = {s["function"]["name"] for s in [*SCHEMAS, WEB_SEARCH_SCHEMA, WEB_FETCH_SCHEMA, VIEW_IMAGE_SCHEMA]} # ----------------------------------------------------------------------------- helpers @@ -329,6 +340,11 @@ def _argument_problem(fn, args: dict) -> str: valid = ", ".join(p if params[p].default is inspect.Parameter.empty else f"{p} (optional)" for p in params) return f"{'; '.join(parts)}. Valid arguments: {valid}." + # -- images + def t_view_image(self, path: str, question: str = "") -> str: + p = self.path(path) + return f"[{self.rel(p)}, as described by {self.agent.vision_model()}]\n" + self.agent.look(p, question) + # -- MCP servers def _mcp(self, name: str, args: dict) -> str: mcp = self.agent.mcp @@ -348,7 +364,7 @@ def _mcp(self, name: str, args: dict) -> str: ok, feedback = self.agent.perms.request(f"mcp:{state.name}:{tool['name']}", "mcp", title, body) if not ok: return feedback - result = mcp.call(state, tool, arguments) + result = mcp.call(state, tool, arguments, image_text=self.agent.describe_image_data) if result.startswith("Error:"): return result self.console.print(Text(f" ⎿ {result.count(chr(10)) + 1} line(s) from {state.name}", style="dim")) @@ -361,6 +377,8 @@ def t_read_file(self, path: str, offset: int = 1, limit: int = 2000) -> str: raise ToolError(f"{path} does not exist") if p.is_dir(): raise ToolError(f"{path} is a directory; use list_dir") + if is_image(p): + raise ToolError(f"{path} is an image; look at it with view_image") if is_binary(p): raise ToolError(f"{path} is a binary file ({p.stat().st_size} bytes)") lines = p.read_text(errors="replace").splitlines() diff --git a/src/lcode/vision.py b/src/lcode/vision.py new file mode 100644 index 0000000..1d79db5 --- /dev/null +++ b/src/lcode/vision.py @@ -0,0 +1,109 @@ +"""Looking at images: screenshots, diagrams and photos the user attaches or the model opens. + +The model that runs the conversation often can't see: lcode's text-only variants leave out the +vision projector to save GPU memory. So lcode asks a model that can see to describe the image in +detail (all text transcribed, layout, colors), with the user's question in mind, and gives the +description to the conversation. Which model looks (`vision_model = "auto"`): + +1. the session's model, if it has vision; +2. otherwise the original Ollama model behind a text-only catalog variant (the same weights plus + the vision projector; `lcode setup` already downloaded it), e.g. qwen3.6:35b-a3b-coding; +3. otherwise none, unless `vision_model` names one (for example qwen3-vl:8b). +""" + +from __future__ import annotations + +import base64 +from pathlib import Path + +from lcode import catalog +from lcode.ollama import Ollama, OllamaError + +IMAGE_TYPES = { + ".png": "image/png", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".webp": "image/webp", + ".gif": "image/gif", +} +MAX_IMAGE_BYTES = 20 * 1024 * 1024 +DESCRIBE_CONTEXT = 16384 # enough for one image and a long description + +PROMPT = """You are the eyes of a coding assistant that cannot see images. Describe this image so the \ +assistant can act on it without seeing it. +- Transcribe all visible text exactly: code, error messages, terminal output, labels, menu items, numbers. +- Describe the layout and every important element: what it is, where it is, its size, color and state \ +(enabled, selected, highlighted, misaligned, overlapping, cut off). +- For diagrams, charts and architecture drawings, describe the parts and how they connect. +- Say what looks wrong or unusual, if anything. +Don't guess at what isn't visible.{question}""" + +INSTALL_HINT = ( + "no model that can see images is available: run `ollama pull qwen3-vl:8b` and " + "`lcode config set vision_model qwen3-vl:8b`, or use a catalog model (lcode setup)" +) + + +class VisionError(Exception): + pass + + +def is_image(path: Path) -> bool: + return path.suffix.lower() in IMAGE_TYPES + + +def can_see(ollama: Ollama, model: str) -> bool: + try: + return "vision" in (ollama.show(model).get("capabilities") or []) + except (OllamaError, AttributeError, TypeError): + return False + + +def pick_model(ollama: Ollama, model: str, setting: str) -> str | None: + """The model that looks at images for a session running `model` (see the module docstring).""" + if setting == "off": + return None + if setting != "auto": + return setting + if can_see(ollama, model): + return model + spec = catalog.find(model) + if spec and spec.tag != model and can_see(ollama, spec.tag): + return spec.tag + return None + + +def read_image(path: Path) -> bytes: + if not path.is_file(): + raise VisionError(f"{path} doesn't exist") + if not is_image(path): + raise VisionError(f"{path.name} isn't an image lcode can read ({', '.join(sorted(IMAGE_TYPES))})") + size = path.stat().st_size + if size > MAX_IMAGE_BYTES: + raise VisionError(f"{path.name} is {size // (1024 * 1024)} MB; images up to 20 MB are supported") + return path.read_bytes() + + +def describe(ollama: Ollama, model: str, image: bytes, question: str = "", options: dict | None = None, + keep_alive: str = "30m") -> str: # fmt: skip + """A detailed description of the image, focused on the question if there is one.""" + focus = ( + f"\n\nThe user asks: {question.strip()}\nPay special attention to what that needs." if question.strip() else "" + ) + payload = { + "model": model, + "messages": [ + {"role": "user", "content": PROMPT.format(question=focus), "images": [base64.b64encode(image).decode()]} + ], + "think": False, + "keep_alive": keep_alive, + "options": options or {"num_ctx": DESCRIBE_CONTEXT}, + } + try: + reply = ollama.chat(payload) + except OllamaError as e: + raise VisionError(f"{model} couldn't look at the image: {e}") from e + text = ((reply.get("message") or {}).get("content") or "").strip() + if not text: + raise VisionError(f"{model} returned no description") + return text diff --git a/tests/conftest.py b/tests/conftest.py index a23a009..b50b3d7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -17,6 +17,16 @@ def __init__(self, scripts: list[list[dict]] | None = None, max_ctx: int = 26214 self.payloads: list[dict] = [] self.max_ctx = max_ctx self.host = "http://fake" + self.capabilities: dict[str, list[str]] = {} # model -> Ollama capabilities, e.g. ["vision"] + self.chats: list[dict] = [] # non-streaming requests (image descriptions) + self.description = "A login form with a red 'Sign in' button." + + def show(self, model: str) -> dict: + return {"capabilities": self.capabilities.get(model, ["completion", "tools"])} + + def chat(self, payload: dict) -> dict: + self.chats.append(payload) + return {"message": {"role": "assistant", "content": self.description}} def chat_stream(self, payload: dict): self.payloads.append(payload) diff --git a/tests/test_vision.py b/tests/test_vision.py new file mode 100644 index 0000000..407d249 --- /dev/null +++ b/tests/test_vision.py @@ -0,0 +1,104 @@ +import base64 + +import pytest + +from conftest import FakeOllama, call, output, reply +from lcode import vision +from lcode.mcp.protocol import result_text + +PNG = base64.b64decode( # a 1x1 transparent PNG + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==" +) + + +def seeing(agent, *models): + for model in models: + agent.ollama.capabilities[model] = ["completion", "vision", "tools"] + agent._vision = False # decide again + return agent + + +def test_which_model_looks(): + ollama = FakeOllama() + assert vision.pick_model(ollama, "lcode-qwen3.6-35b", "auto") is None + ollama.capabilities["qwen3.6:35b-a3b-coding"] = ["vision"] + assert vision.pick_model(ollama, "lcode-qwen3.6-35b", "auto") == "qwen3.6:35b-a3b-coding" # the base model + ollama.capabilities["qwen3.5:9b"] = ["vision"] + assert vision.pick_model(ollama, "qwen3.5:9b", "auto") == "qwen3.5:9b" # the model itself + assert vision.pick_model(ollama, "qwen3.5:9b", "off") is None + assert vision.pick_model(ollama, "custom:7b", "qwen3-vl:8b") == "qwen3-vl:8b" + + +def test_images_are_checked(tmp_path, monkeypatch): + (tmp_path / "a.png").write_bytes(PNG) + (tmp_path / "notes.txt").write_text("x") + assert vision.read_image(tmp_path / "a.png") == PNG + with pytest.raises(vision.VisionError, match="isn't an image"): + vision.read_image(tmp_path / "notes.txt") + with pytest.raises(vision.VisionError, match="doesn't exist"): + vision.read_image(tmp_path / "missing.png") + monkeypatch.setattr(vision, "MAX_IMAGE_BYTES", 10) + with pytest.raises(vision.VisionError, match="up to 20 MB"): + vision.read_image(tmp_path / "a.png") + + +def test_the_model_can_look_at_screenshots(make_agent, repo): + (repo / "screenshot.png").write_bytes(PNG) + agent = make_agent( + [reply(tool_calls=[call("view_image", path="screenshot.png", question="what does the button say?")]), + reply("The button says 'Sign in'.")] + ) # fmt: skip + assert "view_image" not in {s["function"]["name"] for s in agent.tool_schemas()} # nothing can see yet + seeing(agent, "qwen3.6:35b-a3b-coding") + assert "view_image" in {s["function"]["name"] for s in agent.tool_schemas()} + agent.run_turn("check the login screenshot") + result = next(m["content"] for m in agent.messages if m.get("tool_name") == "view_image") + assert result.startswith("[screenshot.png, as described by qwen3.6:35b-a3b-coding]") and "red 'Sign in'" in result + request = agent.ollama.chats[0] + assert request["model"] == "qwen3.6:35b-a3b-coding" and request["think"] is False + assert request["messages"][0]["images"] == [base64.b64encode(PNG).decode()] + assert "what does the button say?" in request["messages"][0]["content"] + assert request["options"] == {"num_ctx": vision.DESCRIBE_CONTEXT} + + +def test_a_model_that_sees_keeps_its_settings(make_agent, repo): + (repo / "a.png").write_bytes(PNG) + agent = seeing(make_agent(model="qwen3.5:9b", context=65536, num_batch=512), "qwen3.5:9b") + agent.look(repo / "a.png") + assert agent.ollama.chats[0]["options"] == {"num_ctx": 65536, "num_batch": 512} # no reload + + +def test_attached_images_are_described_with_the_question(make_agent, repo): + (repo / "bug.png").write_bytes(PNG) + agent = seeing(make_agent([reply("Fixed.")]), "qwen3.6:35b-a3b-coding") + agent.run_turn("why is the layout broken in @bug.png ?") + sent = agent.messages[1]["content"] + assert '' in sent and "red 'Sign in'" in sent + assert "why is the layout broken in ?" in agent.ollama.chats[0]["messages"][0]["content"] + assert "looked at bug.png" in output(agent) + + +def test_without_a_vision_model_lcode_says_how_to_get_one(make_agent, repo): + (repo / "bug.png").write_bytes(PNG) + agent = make_agent([reply("ok")]) + agent.run_turn("look at @bug.png") + assert "ollama pull qwen3-vl:8b" in agent.messages[1]["content"] + assert agent.tools.run("read_file", {"path": "bug.png"}) == "Error: bug.png is an image; look at it with view_image" + + +def test_images_from_mcp_tools_are_described(make_agent): + agent = seeing(make_agent(), "qwen3.6:35b-a3b-coding") + image = {"type": "image", "mimeType": "image/png", "data": base64.b64encode(PNG).decode()} + shot = {"content": [{"type": "text", "text": "Took a screenshot"}, image]} + text = result_text(shot, agent.describe_image_data) + assert text.startswith("Took a screenshot\n[image/png image, as described by qwen3.6:35b-a3b-coding]") + assert "red 'Sign in'" in text + blind = make_agent() + assert "not shown" in result_text(shot, blind.describe_image_data) + + +def test_view_image_respects_the_sandbox(make_agent, tmp_path_factory): + outside = tmp_path_factory.mktemp("elsewhere") / "x.png" + outside.write_bytes(PNG) + agent = seeing(make_agent(sandbox="docker"), "qwen3.6:35b-a3b-coding") + assert agent.tools.run("view_image", {"path": str(outside)}).startswith("Error:") From c3273310b84ed1e5f58df8dbccf824038de300e5 Mon Sep 17 00:00:00 2001 From: Naser Derakhshan <9349840+nasser1941@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:35:13 +0200 Subject: [PATCH 2/2] Free the GPU for image tools, measured on a 12 GB laptop A 12 GB GPU can't hold lcode's model and an image model at once: with qwen3.6-35b loaded, ComfyUI's FLUX.2 Klein run fails with "VRAM grow failed". New per-server setting free_gpu (true, or a list of tools): before such a tool runs, lcode unloads its own Ollama models (the session's and the vision model, nobody else's); the next request reloads them. The comfyui preset sets it for generate_image, run_workflow and run_template, and its setup now starts ComfyUI with --disable-smart-memory so ComfyUI hands the GPU back after each image. fetch_template joins the preset's tools (needed to adjust a template). Measured (RTX 4080 Laptop 12 GB, Klein Base 4B fp8, 512x512): unload 0.2 s, about 12 s per image, about 20 s to reload qwen3.6-35b. A full lcode request (find the template, adjust it, generate, save to assets/, look at the result) worked end to end; documented with these numbers. Co-Authored-By: Claude Opus 5.5 --- docs/mcp.md | 32 ++++++++++++++++++++++++++------ src/lcode/agent.py | 17 +++++++++++++++++ src/lcode/mcp/catalog.toml | 12 +++++++----- src/lcode/mcp/config.py | 6 +++++- src/lcode/mcp/manager.py | 5 +++++ src/lcode/tools.py | 6 ++++++ tests/conftest.py | 6 ++++-- tests/test_mcp.py | 18 ++++++++++++++++++ 8 files changed, 88 insertions(+), 14 deletions(-) diff --git a/docs/mcp.md b/docs/mcp.md index 06a71aa..54f55f1 100644 --- a/docs/mcp.md +++ b/docs/mcp.md @@ -90,7 +90,8 @@ locally; its official MCP server lets lcode use it, for app icons, illustrations placeholder art. ```bash -uv tool install comfy-cli && comfy install && comfy launch # ComfyUI on http://127.0.0.1:8188 +uv tool install comfy-cli && comfy install # ComfyUI, once +comfy launch --background -- --disable-smart-memory # start it (http://127.0.0.1:8188) lcode mcp add comfyui ``` @@ -109,11 +110,29 @@ Qwen-Image-Edit). | Qwen-Image / Qwen-Image-Edit | Generation and editing, good text in images | Heavy: a GGUF version and RAM offloading | Only ComfyUI's local tools are turned on; Comfy Cloud's partner tools are left out, so prompts and -images stay on your machine. One GPU can't hold a large coding model and an image model at the -same time: while lcode's model is loaded, ComfyUI runs slowly or runs out of memory. Use a smaller -image model, generate between requests (the `free_memory` tool frees ComfyUI's memory; Ollama frees -lcode's after `keep_alive`), or run ComfyUI on another machine and give its address when adding. -lcode can also look at the results ([Images](usage.md#images)). +images stay on your machine. + +**Sharing one GPU.** A GPU rarely holds a coding model and an image model at the same time, so the +two take turns: before ComfyUI generates, lcode unloads its own model (the `free_gpu` setting), and +ComfyUI started with `--disable-smart-memory` gives the GPU back after each image. lcode's model +then reloads for the next step, which adds a few seconds to half a minute per image request. With +enough GPU memory for both (or on a Mac with plenty of memory), remove `free_gpu` from the server's +settings in `mcp.json`. lcode can also look at the results ([Images](usage.md#images)). + +Measured on an RTX 4080 Laptop GPU (12 GB) with qwen3.6-35b at 64K context and FLUX.2 Klein Base 4B +(fp8, 512×512, 20 steps): + +| | | +|---|---| +| Generating while lcode's model is loaded | Fails: ComfyUI runs out of GPU memory | +| Freeing the GPU (lcode unloads its model) | 0.2 s | +| Generating one image | about 12 s | +| Reloading lcode's model afterwards | about 20 s | + +The first time, the model has to find the right template and fill in its settings (model file names, +size, prompt), which can take several minutes of trial and error. Once an image comes out right, ask +lcode to save that workflow in your project (for example `assets/icon.workflow.json`) and reuse it: +later images are a single `run_workflow` call. ## Adding your own servers @@ -155,6 +174,7 @@ document, so you can also paste a server's example config into that file: | `allow` | Tools that run without asking; `["*"]` for all of the server's tools | | `timeout` | Seconds a tool call may take (default 300) | | `disabled` | `true` to turn the server off (`lcode mcp disable `) | +| `free_gpu` | Tools that need the GPU to themselves (`true` for all): lcode unloads its model first and reloads it afterwards | | `oauth` | For servers that need a pre-registered OAuth client: `client_id`, `client_secret`, `scopes`, `authorize_params` | Values can use environment variables: `${NAME}`, or `${NAME:-default}`. Keep tokens in environment diff --git a/src/lcode/agent.py b/src/lcode/agent.py index 48dcb1b..07247e8 100644 --- a/src/lcode/agent.py +++ b/src/lcode/agent.py @@ -221,6 +221,23 @@ def vision_model(self) -> str | None: self._vision = vision.pick_model(self.ollama, self.settings.model, self.settings.vision_model) return self._vision or None + def free_gpu(self) -> list[str]: + """Unload lcode's models from Ollama so another program can use the GPU; the next request reloads. + + Only lcode's own models: the session's and the one that looks at images. + """ + mine = {self.settings.model, *([self._vision] if isinstance(self._vision, str) else [])} + freed = [] + for entry in self.ollama.running(): + name = entry.get("name") or entry.get("model") or "" + if name in mine or name.removesuffix(":latest") in mine: + try: + self.ollama.unload(name) + freed.append(name.removesuffix(":latest")) + except OllamaError: + pass + return freed + def look(self, path: Path, question: str = "") -> str: """Describe an image with the vision model. Raises ToolError if it can't.""" model = self.vision_model() diff --git a/src/lcode/mcp/catalog.toml b/src/lcode/mcp/catalog.toml index ec2bedc..ba6be00 100644 --- a/src/lcode/mcp/catalog.toml +++ b/src/lcode/mcp/catalog.toml @@ -136,20 +136,22 @@ description = "Generate and edit images with models you run locally in ComfyUI: homepage = "https://github.com/Comfy-Org/comfy-mcp" requires = ["uvx"] setup = """Needs ComfyUI on this machine (http://127.0.0.1:8188) with at least one image model: - uv tool install comfy-cli && comfy install && comfy launch -Download models with ComfyUI's model manager or ask lcode to (search_models, download_model). Only the -local tools are on: generating, running workflows and templates, models, uploading an image to edit, -jobs and memory. Comfy Cloud's partner tools are left out, so nothing leaves this machine.""" + uv tool install comfy-cli && comfy install + comfy launch --background -- --disable-smart-memory +--disable-smart-memory makes ComfyUI give the GPU back after each image. Before generating, lcode +unloads its own model so ComfyUI has the GPU; it reloads afterwards. Only the local tools are on; +Comfy Cloud's partner tools are left out, so nothing leaves this machine.""" [comfyui.server] command = "uvx" args = ["--python", "3.12", "--from", "comfy-mcp", "--with", "comfy-cli>=1.14.0", "comfy-mcp"] env = { COMFYUI_URL = "${COMFYUI_URL}" } tools = [ - "server_info", "generate_image", "run_workflow", "run_template", "search_templates", "get_template", + "server_info", "generate_image", "run_workflow", "run_template", "search_templates", "get_template", "fetch_template", "list_workflow_slots", "set_workflow_slot", "job", "fetch_outputs", "upload_file", "search_models", "download_model", "system_stats", "free_memory", "launch_comfyui", ] +free_gpu = ["generate_image", "run_workflow", "run_template"] [[comfyui.inputs]] var = "COMFYUI_URL" diff --git a/src/lcode/mcp/config.py b/src/lcode/mcp/config.py index 4bab921..e93d3e0 100644 --- a/src/lcode/mcp/config.py +++ b/src/lcode/mcp/config.py @@ -2,7 +2,8 @@ Both use the `mcpServers` format that most MCP servers document, so a snippet from a server's README can be pasted in as is. lcode adds a few optional keys per server: `disabled`, `tools` (only offer -these tools to the model), `allow` (tools that don't need approval), `timeout` and `oauth`. +these tools to the model), `allow` (tools that don't need approval), `timeout`, `oauth` and +`free_gpu` (tools that need the GPU to themselves, so lcode unloads its model first). Values can refer to environment variables as ${NAME} or ${NAME:-default}. """ @@ -69,6 +70,7 @@ class ServerConfig: timeout: float = DEFAULT_TIMEOUT oauth: dict = field(default_factory=dict) disabled: bool = False + free_gpu: list[str] | bool = False # tools (or all, if True) that need lcode's model off the GPU error: str = "" # a problem with the settings; the server can't start @property @@ -121,6 +123,8 @@ def parse(name: str, raw: dict, source: str = "user") -> ServerConfig: cfg.error = f"the environment variable {e} isn't set" tools = raw.get("tools") cfg.tools = [str(t) for t in tools] if isinstance(tools, list) else None + free_gpu = raw.get("free_gpu", False) + cfg.free_gpu = [str(t) for t in free_gpu] if isinstance(free_gpu, list) else bool(free_gpu) cfg.allow = [str(t) for t in raw.get("allow") or []] if isinstance(raw.get("allow"), list) else [] try: cfg.timeout = float(raw.get("timeout", DEFAULT_TIMEOUT)) diff --git a/src/lcode/mcp/manager.py b/src/lcode/mcp/manager.py index 33d7d79..0e8cf79 100644 --- a/src/lcode/mcp/manager.py +++ b/src/lcode/mcp/manager.py @@ -317,6 +317,11 @@ def resolve(self, name: str, args: dict) -> tuple[ServerState, dict, dict]: raise McpError("`arguments` must be a JSON object") return found[0], found[1], arguments + def needs_gpu(self, state: ServerState, tool: dict) -> bool: + """Whether lcode should free the GPU (unload its model) before this tool runs.""" + wanted = state.cfg.free_gpu + return wanted is True or (isinstance(wanted, list) and tool["name"] in wanted) + def allowed(self, state: ServerState, tool: dict) -> bool: """Tools the user allowed in mcp.json run without asking.""" return "*" in state.cfg.allow or tool["name"] in state.cfg.allow diff --git a/src/lcode/tools.py b/src/lcode/tools.py index 1fe4f4d..1a6684d 100644 --- a/src/lcode/tools.py +++ b/src/lcode/tools.py @@ -364,6 +364,12 @@ def _mcp(self, name: str, args: dict) -> str: ok, feedback = self.agent.perms.request(f"mcp:{state.name}:{tool['name']}", "mcp", title, body) if not ok: return feedback + if mcp.needs_gpu(state, tool): + freed = self.agent.free_gpu() + if freed: + self.console.print( + Text(f" ⎿ freed the GPU for {state.name} ({', '.join(freed)} reloads afterwards)", style="dim") + ) result = mcp.call(state, tool, arguments, image_text=self.agent.describe_image_data) if result.startswith("Error:"): return result diff --git a/tests/conftest.py b/tests/conftest.py index b50b3d7..6d6f095 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -19,6 +19,8 @@ def __init__(self, scripts: list[list[dict]] | None = None, max_ctx: int = 26214 self.host = "http://fake" self.capabilities: dict[str, list[str]] = {} # model -> Ollama capabilities, e.g. ["vision"] self.chats: list[dict] = [] # non-streaming requests (image descriptions) + self.loaded = [{"name": "lcode-qwen3.6-35b:latest", "size": 23_000_000_000, "size_vram": 11_500_000_000}] + self.unloaded: list[str] = [] self.description = "A login form with a red 'Sign in' button." def show(self, model: str) -> dict: @@ -42,10 +44,10 @@ def load(self, *args, **kwargs) -> None: pass def unload(self, model: str) -> None: - pass + self.unloaded.append(model) def running(self) -> list[dict]: - return [{"name": "lcode-qwen3.6-35b:latest", "size": 23_000_000_000, "size_vram": 11_500_000_000}] + return self.loaded def reply(content: str = "", tool_calls: list[dict] | None = None, thinking: str = "") -> list[dict]: diff --git a/tests/test_mcp.py b/tests/test_mcp.py index 267a6da..4196088 100644 --- a/tests/test_mcp.py +++ b/tests/test_mcp.py @@ -522,3 +522,21 @@ def test_options_and_the_server_command_are_kept_apart(): args = build_parser().parse_args(["add", "db", "--env", "URL=postgres://x", "--", "uvx", "pg", "--read-only"]) assert (args.name, args.env, args.command) == ("db", ["URL=postgres://x"], ["uvx", "pg", "--read-only"]) + + +def test_gpu_hungry_tools_get_the_gpu(make_agent, tmp_path, monkeypatch): + """Before a tool listed under free_gpu runs, lcode unloads its own models (and nobody else's).""" + agent = make_agent( + [ + reply(tool_calls=[call("mcp__img__add", a=1, b=2)]), # listed under free_gpu + reply(tool_calls=[call("mcp__img__echo", text="x")]), # not listed + reply("done"), + ] + ) + agent.ollama.loaded = [{"name": "lcode-qwen3.6-35b:latest"}, {"name": "someone-elses:7b"}] + agent.mcp = ready_manager(tmp_path, server_config("img", free_gpu=["add"])) + agent.run_turn("make an image") + assert agent.ollama.unloaded == ["lcode-qwen3.6-35b:latest"] + assert "freed the GPU for img (lcode-qwen3.6-35b reloads afterwards)" in output(agent) + assert mcp_config.parse("x", {"command": "y", "free_gpu": True}).free_gpu is True + agent.mcp.close()