From cab6bb94288f9354a22e64958b0ad5ad5189199d Mon Sep 17 00:00:00 2001 From: Lorchie Date: Sat, 3 Oct 2026 13:43:10 +0200 Subject: [PATCH 1/7] feat(llm): llama.cpp engine pool, model library and cloud providers Replace the Ollama-backed agent with a managed llama.cpp pool (one llama-server per loaded model, LRU + VRAM-budget eviction, idle reaper), a shared model library with resumable downloads, and external providers (OpenAI, Anthropic, Mistral, Groq, OpenRouter, Ollama, custom endpoint) with API keys encrypted through Electron safeStorage. Review fixes: - restart a crashed slot on a fresh port instead of killing the server that inherited its port - unload_all() spares slots answering a request or being started; the agent releases its slot before the post-workflow unload - "Free memory" also unloads the local LLMs - POSIX stale-server cleanup only kills llama-server processes - retry an external chat without images when a text-only model rejects them - only treat OSCrypt-tagged hex as ciphertext so legacy hex keys survive - external model picker no longer shows a model the draft does not hold - drop the CAD and Custom tabs from the model library --- api/main.py | 7 +- api/resources/llm_catalog.json | 169 +++ api/routers/agent.py | 260 +++-- api/routers/extensions.py | 5 + api/routers/llm.py | 418 +++++++ api/routers/model.py | 16 +- api/services/extension_process.py | 11 + api/services/generator_registry.py | 11 + api/services/llm_server.py | 1016 +++++++++++++++++ api/tests/test_agent_workflow_unload.py | 166 ++- api/tests/test_extension_process.py | 18 + api/tests/test_llm_downloads.py | 61 + api/tests/test_llm_pool_budget.py | 279 +++++ electron/main/hf-token.ts | 61 + electron/main/ipc-handlers.ts | 35 +- electron/main/model-downloader.ts | 38 +- electron/main/python-bridge.ts | 8 +- electron/main/secure-store.ts | 78 ++ electron/preload/electron-api.ts | 8 + scripts/react-test-env.mjs | 103 ++ src/areas/generate/components/ChatPanel.tsx | 59 +- .../settings/components/AgentSection.tsx | 301 +++-- src/shared/components/ui/LlmModelSelect.tsx | 62 + .../components/ui/ModelLibraryModal.tsx | 412 +++++++ src/shared/components/ui/agentGrade.test.mjs | 53 + src/shared/components/ui/agentGrade.ts | 44 + src/shared/components/ui/vramFit.test.mjs | 48 + src/shared/components/ui/vramFit.ts | 20 + src/shared/services/llmDownloads.ts | 104 ++ src/shared/stores/agentStore.ts | 152 ++- .../stores/llmModelsStore.react.test.mjs | 82 ++ src/shared/stores/llmModelsStore.test.mjs | 106 ++ src/shared/stores/llmModelsStore.ts | 112 ++ src/shared/types/electron.d.ts | 5 + 34 files changed, 4073 insertions(+), 255 deletions(-) create mode 100644 api/resources/llm_catalog.json create mode 100644 api/routers/llm.py create mode 100644 api/services/llm_server.py create mode 100644 api/tests/test_llm_downloads.py create mode 100644 api/tests/test_llm_pool_budget.py create mode 100644 electron/main/hf-token.ts create mode 100644 electron/main/secure-store.ts create mode 100644 scripts/react-test-env.mjs create mode 100644 src/shared/components/ui/LlmModelSelect.tsx create mode 100644 src/shared/components/ui/ModelLibraryModal.tsx create mode 100644 src/shared/components/ui/agentGrade.test.mjs create mode 100644 src/shared/components/ui/agentGrade.ts create mode 100644 src/shared/components/ui/vramFit.test.mjs create mode 100644 src/shared/components/ui/vramFit.ts create mode 100644 src/shared/services/llmDownloads.ts create mode 100644 src/shared/stores/llmModelsStore.react.test.mjs create mode 100644 src/shared/stores/llmModelsStore.test.mjs create mode 100644 src/shared/stores/llmModelsStore.ts diff --git a/api/main.py b/api/main.py index 382ed67e..20159f1c 100644 --- a/api/main.py +++ b/api/main.py @@ -12,7 +12,7 @@ from services.stdio_utf8 import ensure_utf8_stdio ensure_utf8_stdio() # must run before any print/logging hits the pipe -from routers import generation, model, optimize, status, settings, extensions, export, workflow_runs, agent +from routers import generation, model, optimize, status, settings, extensions, export, workflow_runs, agent, llm @asynccontextmanager @@ -21,8 +21,10 @@ async def lifespan(app: FastAPI): from services.generator_registry import generator_registry generator_registry.initialize() yield - # Shutdown: unload all models + # Shutdown: unload all models and stop the local LLM server generator_registry.unload_all() + from services.llm_server import llama_pool + llama_pool.unload_all(force=True) class _StatusFilter(logging.Filter): @@ -57,6 +59,7 @@ def filter(self, record): app.include_router(export.router, prefix="/export") app.include_router(workflow_runs.router, prefix="/workflow-runs") app.include_router(agent.router) +app.include_router(llm.router, prefix="/llm") # Serve generated files from workspace — dynamic so path changes take effect immediately @app.get("/workspace/{full_path:path}") diff --git a/api/resources/llm_catalog.json b/api/resources/llm_catalog.json new file mode 100644 index 00000000..928f4a2a --- /dev/null +++ b/api/resources/llm_catalog.json @@ -0,0 +1,169 @@ +[ + { + "id": "qwen3.5-4b", + "name": "Qwen 3.5 4B", + "description": "Newest small Qwen. Tops independent tool-calling tests for its size and stays fast on modest GPUs.", + "hf_repo": "unsloth/Qwen3.5-4B-GGUF", + "hf_filename": "Qwen3.5-4B-Q4_K_M.gguf", + "size_bytes": 2740937888, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 4600, + "ngl_suggestion": 99, + "tags": [ + "fast", + "4b" + ], + "sampling": { + "temperature": 0.7, + "top_p": 0.8, + "top_k": 20, + "presence_penalty": 1.5 + } + }, + { + "id": "qwen3.5-9b", + "name": "Qwen 3.5 9B", + "description": "Best balance of reliability and speed on an 8 GB+ GPU. Noticeably steadier than 4B over long conversations.", + "hf_repo": "unsloth/Qwen3.5-9B-GGUF", + "hf_filename": "Qwen3.5-9B-Q4_K_M.gguf", + "size_bytes": 5680522464, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 7600, + "ngl_suggestion": 99, + "tags": [ + "balanced", + "9b" + ], + "sampling": { + "temperature": 0.7, + "top_p": 0.8, + "top_k": 20, + "presence_penalty": 1.5 + } + }, + { + "id": "qwen3-4b", + "name": "Qwen 3 4B Instruct", + "description": "Latest Qwen generation. Fast, excellent tool-calling for its size — the recommended starting point.", + "hf_repo": "unsloth/Qwen3-4B-Instruct-2507-GGUF", + "hf_filename": "Qwen3-4B-Instruct-2507-Q4_K_M.gguf", + "size_bytes": 2497281120, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 4200, + "ngl_suggestion": 99, + "tags": [ + "fast", + "4b", + "default" + ] + }, + { + "id": "qwen3-8b", + "name": "Qwen 3 8B", + "description": "Great quality/speed balance with hybrid thinking. Fits in 8 GB VRAM.", + "hf_repo": "unsloth/Qwen3-8B-GGUF", + "hf_filename": "Qwen3-8B-Q4_K_M.gguf", + "size_bytes": 5027784512, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 7000, + "ngl_suggestion": 99, + "tags": [ + "balanced", + "8b", + "thinking" + ] + }, + { + "id": "qwen3-14b", + "name": "Qwen 3 14B", + "description": "High quality with hybrid thinking (reasons before answering when useful). Needs ~11 GB VRAM.", + "hf_repo": "unsloth/Qwen3-14B-GGUF", + "hf_filename": "Qwen3-14B-Q4_K_M.gguf", + "size_bytes": 9001753984, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 11000, + "ngl_suggestion": 99, + "tags": [ + "quality", + "14b", + "thinking" + ] + }, + { + "id": "gpt-oss-20b", + "name": "GPT-OSS 20B (OpenAI)", + "description": "OpenAI's open-weight MoE (3.6B active params — fast for its size). Top-tier tool calling and reasoning. Best with 16 GB VRAM.", + "hf_repo": "ggml-org/gpt-oss-20b-GGUF", + "hf_filename": "gpt-oss-20b-MXFP4.gguf", + "size_bytes": 12109566624, + "quant": "MXFP4", + "ctx": 16384, + "vram_estimate_mb": 13500, + "ngl_suggestion": 99, + "tags": [ + "quality", + "20b", + "thinking", + "moe" + ] + }, + { + "id": "qwen3-vl-4b", + "name": "Qwen 3 VL 4B (vision, light)", + "description": "Small vision model — sees attached images while fitting in ~5 GB VRAM. Pick this over the 8B on smaller GPUs.", + "hf_repo": "unsloth/Qwen3-VL-4B-Instruct-GGUF", + "hf_filename": "Qwen3-VL-4B-Instruct-Q4_K_M.gguf", + "hf_mmproj_filename": "mmproj-F16.gguf", + "mmproj_size_bytes": 836180640, + "size_bytes": 2497282336, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 4600, + "ngl_suggestion": 99, + "tags": [ + "vision", + "4b", + "fast" + ] + }, + { + "id": "qwen3-vl-8b", + "name": "Qwen 3 VL 8B (vision)", + "description": "Sees images — the agent can look at attached pictures before picking a workflow. Needs ~8 GB VRAM.", + "hf_repo": "unsloth/Qwen3-VL-8B-Instruct-GGUF", + "hf_filename": "Qwen3-VL-8B-Instruct-Q4_K_M.gguf", + "hf_mmproj_filename": "mmproj-F16.gguf", + "mmproj_size_bytes": 1159030336, + "size_bytes": 5027785568, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 7800, + "ngl_suggestion": 99, + "tags": [ + "vision", + "8b" + ] + }, + { + "id": "cadquery-coder-7b", + "name": "CadQuery Coder 7B (CAD-specialized)", + "description": "Qwen2.5-Coder-7B fine-tuned on CadQuery generation — best pick for the Text to CAD node. Community model, CC-BY-NC-SA license.", + "hf_repo": "yuvit-batra/qwen2.5-coder-7b-cadquery-gguf", + "hf_filename": "qwen2.5-coder-7b-cadquery-Q4_K_M.gguf", + "size_bytes": 4680000000, + "quant": "Q4_K_M", + "ctx": 16384, + "vram_estimate_mb": 6200, + "ngl_suggestion": 99, + "tags": [ + "code", + "cad", + "7b" + ] + } +] \ No newline at end of file diff --git a/api/routers/agent.py b/api/routers/agent.py index bf640b5d..1fb389eb 100644 --- a/api/routers/agent.py +++ b/api/routers/agent.py @@ -1,11 +1,19 @@ """ -Agent chat endpoint — runs an Ollama-powered tool-use loop against Modly's API. +Agent chat endpoint — runs a tool-use loop against Modly's API, on the managed +local llama.cpp server or any OpenAI-compatible provider. """ +import asyncio +import json import re import uuid +from typing import Optional + import httpx from fastapi import APIRouter -from pydantic import BaseModel +from pydantic import BaseModel, Field + +from services import llm_server +from services.llm_server import llama_pool router = APIRouter(prefix="/agent", tags=["agent"]) @@ -374,13 +382,19 @@ async def execute_tool(name: str, arguments: dict, context: dict) -> tuple[str, class ChatMessage(BaseModel): role: str content: str - images: list[str] = [] + images: list[str] = [] # data URLs + + +class ProviderConfig(BaseModel): + type: str = "local" # "local" | "external" + base_url: Optional[str] = None # external only, e.g. https://api.openai.com/v1 + api_key: Optional[str] = None class AgentChatRequest(BaseModel): messages: list[ChatMessage] - ollama_url: str = "http://localhost:11434" - model: str = "qwen2.5:3b" + model: str = Field(default_factory=llm_server.default_model_id) # local: catalog/custom id — external: provider model name + provider: ProviderConfig = ProviderConfig() context: dict = {} thinking: str = "auto" # "auto" | "on" | "off" @@ -398,9 +412,10 @@ class AgentChatResponse(BaseModel): def _extract_thinking(msg: dict) -> tuple[str, str | None]: - """Return (clean_content, thinking_text). Handles both Ollama native field and tags.""" - content = msg.get("content", "") - thinking = msg.get("thinking") or None + """Return (clean_content, thinking_text). Handles llama-server's + reasoning_content field and inline tags.""" + content = msg.get("content") or "" + thinking = msg.get("reasoning_content") or None if not thinking: match = re.search(r"(.*?)", content, re.DOTALL) if match: @@ -409,37 +424,100 @@ def _extract_thinking(msg: dict) -> tuple[str, str | None]: return content, thinking -@router.get("/models") -async def list_ollama_models(ollama_url: str = "http://localhost:11434"): - async with httpx.AsyncClient(timeout=5.0) as client: +def _auth_headers(base_url: str, api_key: Optional[str]) -> dict: + headers: dict[str, str] = {} + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + if "anthropic" in base_url: + # Anthropic's OpenAI-compat layer also accepts the native headers. + headers["x-api-key"] = api_key + headers["anthropic-version"] = "2023-06-01" + return headers + + +class ExternalModelsRequest(BaseModel): + base_url: str + api_key: str = "" + + +@router.post("/external/models") +async def list_external_models(req: ExternalModelsRequest): + """Proxy the provider's /models listing (avoids CORS issues from the renderer). + + POST with the key in the body, never a GET query string: uvicorn's access log + records the full request line, and that log ends up in runtime.log. + """ + headers = _auth_headers(req.base_url, req.api_key) + async with httpx.AsyncClient(timeout=10.0) as client: try: - r = await client.get(f"{ollama_url}/api/tags") + r = await client.get(f"{req.base_url.rstrip('/')}/models", headers=headers) r.raise_for_status() - models = [m["name"] for m in r.json().get("models", [])] - return {"models": models} + data = r.json().get("data", []) + return {"models": sorted(m["id"] for m in data if isinstance(m, dict) and m.get("id"))} except Exception: return {"models": []} -async def _unload_llm_after_workflow( - client: httpx.AsyncClient, - request: AgentChatRequest, - actions_done: list[ActionDone], -) -> None: - """Free the Ollama model's VRAM once a workflow has been dispatched, so the +async def _unload_llm_after_workflow(request: AgentChatRequest, actions_done: list[ActionDone]) -> None: + """Free the local LLM's VRAM once a workflow has been dispatched, so the workflow gets the full GPU. Best-effort — never fail the chat over it.""" + if request.provider.type != "local": + return if not any(a.tool == "run_workflow" for a in actions_done): return try: - await client.post( - f"{request.ollama_url}/api/generate", - json={"model": request.model, "keep_alive": 0}, - timeout=5.0, - ) + await asyncio.to_thread(llama_pool.unload_all) except Exception: pass +def _error_detail(r: httpx.Response) -> str: + """The provider's own message (OpenAI-style `{"error": {"message": …}}`), else the raw body.""" + try: + err = r.json().get("error") + if isinstance(err, dict) and err.get("message"): + return str(err["message"])[:300] + if isinstance(err, str): + return err[:300] + except (ValueError, AttributeError): + pass + return r.text[:300] + + +def _tool_arguments(raw) -> dict: + """OpenAI-compatible servers send tool arguments as a JSON string.""" + if isinstance(raw, dict): + return raw + try: + parsed = json.loads(raw or "{}") + except (TypeError, ValueError): + return {} + return parsed if isinstance(parsed, dict) else {} + + +def _user_entry(m: ChatMessage, send_images: bool) -> dict: + if not (m.images and send_images): + return {"role": m.role, "content": m.content} + parts: list[dict] = [{"type": "text", "text": m.content}] + for data_url in m.images: + parts.append({"type": "image_url", "image_url": {"url": data_url}}) + return {"role": m.role, "content": parts} + + +def _drop_images(messages: list[dict]) -> bool: + """Replace image parts with a text note, in place. True if any were dropped.""" + dropped = False + for m in messages: + if not isinstance(m.get("content"), list): + continue + texts = [p["text"] for p in m["content"] if p.get("type") == "text"] + if any(p.get("type") == "image_url" for p in m["content"]): + dropped = True + texts.append("[An image was attached, but this model reads text only.]") + m["content"] = "\n".join(texts) + return dropped + + @router.post("/chat", response_model=AgentChatResponse) async def agent_chat(request: AgentChatRequest): messages: list[dict] = [{"role": "system", "content": SYSTEM_PROMPT}] @@ -471,59 +549,93 @@ async def agent_chat(request: AgentChatRequest): ), }) + # ONE system message, first. Qwen3.5's chat template raises "System message + # must be at the beginning", which llama-server returns as a flat HTTP 400; + # other templates silently drop the extra ones. + messages = [{"role": "system", "content": "\n\n".join(m["content"] for m in messages)}] + + slot = None + send_images = True + extra: dict = {} + if request.provider.type == "local": + try: + spec = llm_server.resolve_model(request.model) + # hold=True: the slot stays claimed for the whole loop, so neither the + # idle reaper nor a model loading elsewhere can evict it mid-answer. + slot = await asyncio.to_thread(llama_pool.ensure, request.model, spec, True) + except Exception as e: + return AgentChatResponse(message=f"Could not start the local LLM: {e}") + base_url = slot.base_url + headers: dict = {} + send_images = spec["vision"] + extra.update(llm_server.sampling_for(request.model)) + if request.thinking == "off": + extra["chat_template_kwargs"] = {"enable_thinking": False} + else: + if not request.provider.base_url: + return AgentChatResponse(message="No provider URL configured. Set one in Settings → Agent.") + base_url = request.provider.base_url.rstrip("/") + headers = _auth_headers(base_url, request.provider.api_key) + for m in request.messages: - entry: dict = {"role": m.role, "content": m.content} - if m.images: - entry["images"] = m.images - messages.append(entry) + messages.append(_user_entry(m, send_images)) actions_done: list[ActionDone] = [] all_thinking: list[str] = [] + final: AgentChatResponse | None = None - # Build Ollama think param - ollama_extra: dict = {} - if request.thinking == "on": - ollama_extra["think"] = True - elif request.thinking == "off": - ollama_extra["think"] = False - - async with httpx.AsyncClient(timeout=120.0) as client: - for _ in range(10): # max tool-call rounds - r = await client.post( - f"{request.ollama_url}/api/chat", - json={"model": request.model, "messages": messages, "tools": TOOLS, "stream": False, **ollama_extra}, - ) - - if r.status_code != 200: - return AgentChatResponse( - message=f"Ollama error ({r.status_code}). Is Ollama running at {request.ollama_url}?", - ) - - msg = r.json()["message"] - messages.append(msg) - - clean_content, thinking_text = _extract_thinking(msg) - if thinking_text: - all_thinking.append(thinking_text) - - tool_calls = msg.get("tool_calls") or [] - if not tool_calls: - await _unload_llm_after_workflow(client, request, actions_done) - combined_thinking = "\n\n---\n\n".join(all_thinking) if all_thinking else None - return AgentChatResponse( - message=clean_content, - actions=actions_done, - thinking=combined_thinking, + try: + async with httpx.AsyncClient(timeout=120.0) as client: + for _ in range(10): # max tool-call rounds + r = await client.post( + f"{base_url}/chat/completions", + headers=headers, + json={"model": request.model, "messages": messages, "tools": TOOLS, "stream": False, **extra}, ) + # An external model's vision support is unknown up front, and a + # text-only one rejects image parts with a 400 that failed the + # whole turn. Retry once with the images described as omitted. + if r.status_code == 400 and request.provider.type != "local" and _drop_images(messages): + r = await client.post( + f"{base_url}/chat/completions", + headers=headers, + json={"model": request.model, "messages": messages, "tools": TOOLS, "stream": False, **extra}, + ) - for tc in tool_calls: - fn = tc["function"] - result_text, payload = await execute_tool(fn["name"], fn.get("arguments") or {}, request.context) - actions_done.append(ActionDone(tool=fn["name"], result=result_text, payload=payload)) - messages.append({"role": "tool", "content": result_text}) - - # Loop exhausted without a final answer: still free VRAM if a workflow ran. - await _unload_llm_after_workflow(client, request, actions_done) - - combined_thinking = "\n\n---\n\n".join(all_thinking) if all_thinking else None - return AgentChatResponse(message="Reached maximum tool iterations.", actions=actions_done, thinking=combined_thinking) + if r.status_code != 200: + return AgentChatResponse(message=f"LLM error ({r.status_code}): {_error_detail(r)}") + + msg = r.json()["choices"][0]["message"] + tool_calls = msg.get("tool_calls") or [] + assistant: dict = {"role": "assistant", "content": msg.get("content") or ""} + if tool_calls: + assistant["tool_calls"] = tool_calls + messages.append(assistant) + + clean_content, thinking_text = _extract_thinking(msg) + if thinking_text: + all_thinking.append(thinking_text) + + if not tool_calls: + final = AgentChatResponse(message=clean_content, actions=actions_done) + break + + for tc in tool_calls: + fn = tc["function"] + result_text, payload = await execute_tool(fn["name"], _tool_arguments(fn.get("arguments")), request.context) + actions_done.append(ActionDone(tool=fn["name"], result=result_text, payload=payload)) + messages.append({"role": "tool", "tool_call_id": tc.get("id", ""), "content": result_text}) + except httpx.HTTPError as e: + return AgentChatResponse(message=f"Could not reach the LLM at {base_url}: {e}", actions=actions_done) + finally: + if slot is not None: + slot.release() + + # After the release: unload_all() spares a held slot, so unloading while + # still holding ours would keep the agent's own model on the GPU. + await _unload_llm_after_workflow(request, actions_done) + + if final is None: + final = AgentChatResponse(message="Reached maximum tool iterations.", actions=actions_done) + final.thinking = "\n\n---\n\n".join(all_thinking) if all_thinking else None + return final diff --git a/api/routers/extensions.py b/api/routers/extensions.py index 25ef3ba6..fdfbf0e0 100644 --- a/api/routers/extensions.py +++ b/api/routers/extensions.py @@ -87,6 +87,11 @@ async def setup_extension(ext_id: str): [sys.executable, str(setup_py), args], capture_output=True, text=True, + # pip's own output carries box-drawing characters; the locale codec + # (cp1252 on Windows) raises on some of them and would turn a + # successful setup into a 500 with no usable message. + encoding="utf-8", + errors="replace", ) ) diff --git a/api/routers/llm.py b/api/routers/llm.py new file mode 100644 index 00000000..33e1534c --- /dev/null +++ b/api/routers/llm.py @@ -0,0 +1,418 @@ +""" +Local LLM engine endpoints — model catalog, GGUF downloads, llama-server lifecycle. +Reuses the streamed/resumable downloader from routers.model. +""" +import asyncio +import json +import threading +from pathlib import Path +from typing import Optional +import httpx +from fastapi import APIRouter, HTTPException +from fastapi.responses import StreamingResponse +from pydantic import BaseModel + +from routers.model import DownloadCancelled, DownloadPaused, _download_file_streamed +from services import llm_server +from services.llm_server import llama_pool + +router = APIRouter(tags=["llm"]) + +_controls: dict[str, dict[str, threading.Event]] = {} + + +class _DownloadState: + """Tracks one model's in-flight download independent of any single SSE + connection, so closing/reopening the Model Library modal reattaches to the + same download instead of racing a second one against the same .part file.""" + + def __init__(self) -> None: + self.last_msg: dict = {} + self.subscribers: list[asyncio.Queue] = [] + self.task: asyncio.Task | None = None + + +_downloads: dict[str, _DownloadState] = {} + + +def _new_control(key: str) -> dict[str, threading.Event]: + control: dict[str, threading.Event] = {"pause": threading.Event(), "cancel": threading.Event()} + _controls[key] = control + return control + + +def _check_control(control: dict[str, threading.Event]) -> None: + if control["cancel"].is_set(): + raise DownloadCancelled() + if control["pause"].is_set(): + raise DownloadPaused() + + +def _fmt(data: dict) -> str: + return f"data: {json.dumps(data)}\n\n" + + +# ─── Catalog / status ───────────────────────────────────────────────────────── + +@router.get("/models") +async def list_models(tag: Optional[str] = None, downloaded: bool = False): + """Model library. `tag` filters catalog entries by category (e.g. tag=code + for coder models); custom user GGUFs are always included since their + capabilities are unknown. `downloaded=true` keeps only ready-to-use models.""" + models = llm_server.list_models() + if tag: + models = [m for m in models if m.get("source") == "custom" or tag in (m.get("tags") or [])] + if downloaded: + models = [m for m in models if m.get("downloaded")] + return {"models": models} + + +@router.get("/status") +async def status(): + return { + "binary_installed": llm_server.binary_installed(), + "has_nvidia_gpu": llm_server.has_nvidia_gpu(), + "vram_gb": llm_server.detect_vram_gb() or None, + "models_dir": str(llm_server.LLM_MODELS_DIR), + "server": await asyncio.to_thread(llama_pool.snapshot), + } + + +class LlmConfigRequest(BaseModel): + max_models: str | int # "auto" or 1..MAX_SLOT_PORTS + + +@router.get("/config") +async def get_config(): + return { + "max_models": llm_server.load_config().get("max_models", "auto"), + "resolved_max_models": llm_server.resolve_max_models(), + "vram_gb": llm_server.detect_vram_gb() or None, + "max_slot_ports": llm_server.MAX_SLOT_PORTS, + } + + +@router.post("/config") +async def set_config(request: LlmConfigRequest): + value: str | int = request.max_models + if isinstance(value, str) and value.lower() != "auto": + try: + value = int(value) + except ValueError: + raise HTTPException(status_code=422, detail="max_models must be 'auto' or an integer") + if isinstance(value, int) and not (1 <= value <= llm_server.MAX_SLOT_PORTS): + raise HTTPException(status_code=422, detail=f"max_models must be between 1 and {llm_server.MAX_SLOT_PORTS}") + cfg = llm_server.load_config() + cfg["max_models"] = value + llm_server.save_config(cfg) + # A lowered limit applies immediately — small-VRAM users count on it. + await asyncio.to_thread(llama_pool.enforce_limit) + return {"max_models": value, "resolved_max_models": llm_server.resolve_max_models()} + + +@router.post("/unload") +async def unload(): + await asyncio.to_thread(llama_pool.unload_all) + return {"unloaded": True} + + +@router.delete("/models/{model_id}") +async def delete_model(model_id: str): + try: + spec = llm_server.resolve_model(model_id) + except KeyError as e: + raise HTTPException(status_code=404, detail=str(e)) + if await asyncio.to_thread(llama_pool.is_loaded, model_id): + await asyncio.to_thread(llama_pool.unload, model_id) + spec["gguf_path"].unlink(missing_ok=True) + if spec.get("mmproj_path"): + spec["mmproj_path"].unlink(missing_ok=True) + return {"deleted": True} + + +# ─── Chat completion through the managed server ─────────────────────────────── + +class LlmChatRequest(BaseModel): + model: str + messages: list[dict] + temperature: Optional[float] = None + max_tokens: Optional[int] = None + stream: Optional[bool] = False + + +@router.post("/chat") +async def llm_chat(request: LlmChatRequest): + """Load `model` (hot-swapping if needed) and run one OpenAI-format chat completion. + + Used by workflow nodes (LLM, Text to CAD, …) so they share the managed + llama-server instead of loading their own copy of the model. When + `stream` is set, the llama-server SSE chunks are proxied straight through so + the caller can render tokens as they arrive. + """ + try: + spec = llm_server.resolve_model(request.model) + except KeyError as e: + raise HTTPException(status_code=404, detail=str(e)) + try: + # hold=True: the slot comes back claimed, so a model loading in another + # thread cannot evict it between here and the request below. + slot = await asyncio.to_thread(llama_pool.ensure, request.model, spec, True) + except Exception as e: + raise HTTPException(status_code=503, detail=f"Could not start the local LLM: {e}") + + payload: dict = {"model": request.model, "messages": request.messages, "stream": bool(request.stream)} + if request.temperature is not None: + payload["temperature"] = request.temperature + if request.max_tokens is not None: + payload["max_tokens"] = request.max_tokens + + # The claim spans the whole completion, not just its end: a generation longer + # than the idle TTL (a Text-to-CAD node can sit here for minutes) used to be + # reaped mid-answer, since only a *finished* call marked the slot as used. + if request.stream: + async def proxy(): + try: + async with httpx.AsyncClient(timeout=600.0) as client: + async with client.stream("POST", f"{slot.base_url}/chat/completions", json=payload) as r: + if r.status_code != 200: + body = (await r.aread()).decode("utf-8", "replace")[:300] + yield f"data: {json.dumps({'error': f'llama-server error ({r.status_code}): {body}'})}\n\n" + return + async for line in r.aiter_lines(): + if line.startswith("data: "): + yield f"{line}\n\n" + finally: + slot.release() + return StreamingResponse(proxy(), media_type="text/event-stream") + + try: + async with httpx.AsyncClient(timeout=600.0) as client: + r = await client.post(f"{slot.base_url}/chat/completions", json=payload) + if r.status_code != 200: + raise HTTPException(status_code=502, detail=f"llama-server error ({r.status_code}): {r.text[:300]}") + return r.json() + finally: + slot.release() + + +# ─── GGUF download (SSE) ────────────────────────────────────────────────────── + +@router.post("/download/pause") +async def pause_download(model_id: str): + control = _controls.get(model_id) + if control: + control["pause"].set() + return {"paused": True} + + +@router.post("/download/cancel") +async def cancel_download(model_id: str): + """Cancel an in-flight download, or clean up a paused one. + + A paused download has already left `_run_download` — its control is gone + from `_controls`, so setting the cancel event would be a no-op. Reporting + success there left the user with a multi-GB .part file they believed they + had deleted, so the partial files are removed here instead.""" + control = _controls.get(model_id) + if control: + control["cancel"].set() + return {"cancelled": True, "removed_partials": []} + + entry = next((e for e in llm_server.load_catalog() if e["id"] == model_id), None) + removed = _discard_incomplete(entry) if entry else [] + _downloads.pop(model_id, None) + return {"cancelled": bool(removed), "removed": removed} + + +def _model_files(entry: dict) -> list[Path]: + """Every file a model needs on disk: the weights, plus the vision projector.""" + paths = [llm_server.LLM_MODELS_DIR / entry["hf_filename"]] + if entry.get("hf_mmproj_filename"): + paths.append(llm_server.LLM_MODELS_DIR / llm_server.mmproj_local_name(entry)) + return paths + + +def _discard_incomplete(entry: dict) -> list[str]: + """Drop what a cancelled download left behind, and report what went. + + Not just the .part: a vision model downloads weights then projector, so + cancelling during the second one left 2.5 GB of finished weights on disk + under a model still reported as `downloaded: false` — no trash button is + offered for those, so the space could not be reclaimed from the UI at all. + A model whose files are all present is complete, not in flight, and is + never touched here (deleting it is what DELETE /llm/models/{id} is for).""" + paths = _model_files(entry) + if all(p.exists() for p in paths): + return [] + removed = [] + for path in paths: + for target in (path.with_suffix(path.suffix + ".part"), path): + if target.exists(): + target.unlink(missing_ok=True) + removed.append(target.name) + return removed + + +def _is_terminal(msg: dict) -> bool: + return msg.get("status") == "done" or "error" in msg or msg.get("cancelled") or msg.get("paused") + + +async def _run_download(model_id: str, entry: dict, control: dict[str, threading.Event], state: _DownloadState) -> None: + """Owns one model's download end to end, independent of any SSE connection. + Broadcasts progress to every currently-attached watcher (see `_broadcast`).""" + loop = asyncio.get_running_loop() + + def _broadcast(msg: dict) -> None: + # _progress (below) runs in a worker thread — hop back onto the loop. + def _do() -> None: + state.last_msg = msg + for q in list(state.subscribers): + q.put_nowait(msg) + loop.call_soon_threadsafe(_do) + + from huggingface_hub import hf_hub_url + + # Vision models ship a companion mmproj file — download it alongside the + # weights, stored under a per-model local name (HF names collide). + files: list[tuple[str, str, int]] = [(entry["hf_filename"], entry["hf_filename"], entry.get("size_bytes") or 0)] + if entry.get("hf_mmproj_filename"): + files.append(( + entry["hf_mmproj_filename"], + llm_server.mmproj_local_name(entry), + entry.get("mmproj_size_bytes") or 0, + )) + grand_total = sum(size for _, _, size in files) + + try: + _broadcast({"percent": 0, "status": "Starting download…"}) + completed_bytes = 0 + + for index, (hf_filename, local_filename, _size) in enumerate(files): + url = hf_hub_url(repo_id=entry["hf_repo"], filename=hf_filename) + + def _progress(msg: dict, _base: int = completed_bytes) -> None: + done = _base + msg.get("bytesDownloaded", 0) + if grand_total: + pct = min(99, round(done / grand_total * 100)) + # The streaming helper reports per-file figures. A vision + # model downloads two files, so the bar (combined percent) + # and the label baked into `status` (per-file percent) drew + # two different numbers side by side. Restate all of it + # against the combined total; `file`/`fileIndex` stay + # per-file, which is what they're for. + msg["percent"] = pct + msg["bytesDownloaded"] = done + msg["totalBytes"] = grand_total + status = msg.get("status") or "" + if status.endswith("%"): + msg["status"] = f"{status.rsplit(' ', 1)[0]} {pct}%" + _broadcast(msg) + + await loop.run_in_executor( + None, + lambda url=url, filename=local_filename, cb=_progress: _download_file_streamed( + url=url, + filename=filename, + dest_dir=str(llm_server.LLM_MODELS_DIR), + file_index=index + 1, + total_files=len(files), + base_percent=0, + progress_cb=cb, + control=control, + ), + ) + completed_bytes += _size + + _broadcast({"percent": 100, "status": "done"}) + + except DownloadPaused: + # Carry the last percent so the modal's bar stays where it stopped + # instead of snapping back to 0 behind the Resume button. + _broadcast({"paused": True, "status": "paused", "percent": state.last_msg.get("percent", 0)}) + except DownloadCancelled: + _discard_incomplete(entry) + _broadcast({"cancelled": True, "status": "cancelled", "percent": 0}) + except Exception as exc: + _broadcast({"error": str(exc)}) + finally: + if _controls.get(model_id) is control: + _controls.pop(model_id, None) + + +@router.get("/download") +async def download_model(model_id: str): + entry = next((e for e in llm_server.load_catalog() if e["id"] == model_id), None) + if entry is None: + raise HTTPException(status_code=404, detail=f"Unknown catalog model: {model_id}") + + # Reconnect-safe: if this model is already downloading, attach a new watcher + # instead of starting a second download racing the same .part file — this is + # what lets the Model Library modal be closed and reopened mid-download. + state = _downloads.get(model_id) + if state is None or state.task is None or state.task.done(): + control = _new_control(model_id) + state = _DownloadState() + _downloads[model_id] = state + state.task = asyncio.create_task(_run_download(model_id, entry, control, state)) + + queue: asyncio.Queue[dict] = asyncio.Queue() + state.subscribers.append(queue) + if state.last_msg: + queue.put_nowait(state.last_msg) # replay current progress immediately on (re)connect + + async def stream(): + try: + while True: + msg = await queue.get() + yield _fmt(msg) + if _is_terminal(msg): + break + finally: + if queue in state.subscribers: + state.subscribers.remove(queue) + + return StreamingResponse(stream(), media_type="text/event-stream") + + +# ─── Engine (llama-server binary) install (SSE) ─────────────────────────────── + +@router.get("/binary/install") +async def install_binary(): + if llm_server.binary_installed(): + async def already(): + yield _fmt({"percent": 100, "status": "done"}) + return StreamingResponse(already(), media_type="text/event-stream") + + control = _new_control("__binary__") + + async def stream(): + loop = asyncio.get_running_loop() + queue: asyncio.Queue[dict] = asyncio.Queue() + + def _progress(msg: dict) -> None: + loop.call_soon_threadsafe(queue.put_nowait, msg) + + try: + future = loop.run_in_executor( + None, + lambda: llm_server.install_binary(_progress, lambda: _check_control(control)), + ) + while not future.done(): + try: + msg = await asyncio.wait_for(queue.get(), timeout=2.0) + except asyncio.TimeoutError: + continue + else: + yield _fmt(msg) + await future + while not queue.empty(): + yield _fmt(queue.get_nowait()) + except (DownloadPaused, DownloadCancelled): + yield _fmt({"cancelled": True, "status": "cancelled"}) + except Exception as exc: + yield _fmt({"error": str(exc)}) + finally: + if _controls.get("__binary__") is control: + _controls.pop("__binary__", None) + + return StreamingResponse(stream(), media_type="text/event-stream") diff --git a/api/routers/model.py b/api/routers/model.py index 66e301bf..aab4c3f4 100644 --- a/api/routers/model.py +++ b/api/routers/model.py @@ -8,7 +8,7 @@ from typing import Optional from urllib.error import HTTPError, URLError from urllib.request import Request, urlopen -from fastapi import APIRouter, HTTPException, Request as FastAPIRequest +from fastapi import APIRouter, Header, HTTPException, Request as FastAPIRequest from fastapi.responses import StreamingResponse from services.generator_registry import generator_registry import services.generator_registry as registry_module @@ -94,6 +94,9 @@ async def unload_all_models(): """Unloads all models from memory to free VRAM/RAM.""" # Off the event loop: unloading waits for any in-progress model load. await asyncio.to_thread(generator_registry.unload_all) + # The local LLMs hold VRAM too; "Free memory" left them resident. + from services.llm_server import llama_pool + await asyncio.to_thread(llama_pool.unload_all) # Force Python to release memory back to the OS import gc gc.collect() @@ -301,7 +304,7 @@ async def hf_download( model_id: str, skip_prefixes: Optional[str] = None, include_prefixes: Optional[str] = None, - token: Optional[str] = None, + x_hf_token: Optional[str] = Header(default=None), ): """ Streams a HuggingFace Hub model download via SSE. @@ -310,11 +313,16 @@ async def hf_download( skip_prefixes: JSON-encoded list of path prefixes to exclude. include_prefixes: JSON-encoded list of path prefixes to include (whitelist). - token: HuggingFace access token for gated repos (from Electron settings). - All three fall back to the extension's manifest / environment when not supplied. + X-HF-Token: HuggingFace access token for gated repos (from Electron settings). + A HEADER, not a query param: uvicorn logs the full request + line to stdout, python-bridge.ts pipes that into runtime.log, + and `log:readAll` hands that file to the user for bug + reports — the token used to travel all the way there. + All fall back to the extension's manifest / environment when not supplied. SSE format: data: {"percent": 0-100, "file": "...", "status": "..."} """ + token = x_hf_token import json as _json import os dest_dir = str(registry_module.MODELS_DIR / model_id) diff --git a/api/services/extension_process.py b/api/services/extension_process.py index efb9f11e..5eaed4b8 100644 --- a/api/services/extension_process.py +++ b/api/services/extension_process.py @@ -74,9 +74,12 @@ def _build_env(self) -> dict: env["MODELS_DIR"] = str(MODELS_DIR) env["WORKSPACE_DIR"] = str(WORKSPACE_DIR) env["MODLY_API_DIR"] = str(Path(__file__).parent.parent) + # Lets extensions call back into the Modly API (e.g. /llm/chat for the shared LLM). + env.setdefault("MODLY_API_URL", "http://127.0.0.1:8765") # Force the worker's Python stdio to UTF-8 so it matches the UTF-8 # pipe readers below regardless of the OS locale (cp1252/cp932). env["PYTHONUTF8"] = "1" + env["PYTHONIOENCODING"] = "utf-8" if sys.platform == "darwin": env.setdefault("NUMBA_DISABLE_JIT", "1") # Must be set before the subprocess's first `import torch` — PyTorch @@ -126,6 +129,12 @@ def _start(self) -> None: stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + # Without these, text=True decodes with the locale codec — cp1252 + # on Windows. tqdm draws its partial blocks with U+258D/U+258F, + # whose UTF-8 bytes (0x8d/0x8f) are undefined there: the decode + # raises, _stderr_loop dies, nobody drains the pipe, and the + # child blocks forever on write once it fills. errors="replace" + # keeps a stray non-UTF-8 byte from resurrecting that failure. encoding="utf-8", errors="replace", bufsize=1, @@ -191,6 +200,8 @@ def _install_missing_package(self, python: Path, module_name: str, package_name: stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + encoding="utf-8", + errors="replace", ) except subprocess.CalledProcessError as exc: details = (exc.stderr or exc.stdout or "").strip() diff --git a/api/services/generator_registry.py b/api/services/generator_registry.py index d34b1603..57f21ff3 100644 --- a/api/services/generator_registry.py +++ b/api/services/generator_registry.py @@ -847,6 +847,17 @@ def switch_model(self, model_id: str) -> None: f"Unknown model ID: '{model_id}'. " f"Available: {list(self._generators.keys())}" ) + # 3D generation owns the GPU: evict the chat LLMs before anything is + # about to allocate on it. The trigger is "the target model is not + # resident", not "the target model changed" — the common case is the + # default generator, which is already `_active_id` at boot and still + # has to load its weights. Gating on the id alone let a full LLM pool + # (2 slots, ~11.6 GB of 12) sit through an entire generation. + target = self._generators[model_id] + if model_id != self._active_id or not target.is_loaded(): + from services.llm_server import llama_pool + llama_pool.unload_all() + if model_id != self._active_id: if self._active_id in self._generators: self._generators[self._active_id].unload() diff --git a/api/services/llm_server.py b/api/services/llm_server.py new file mode 100644 index 00000000..0d4a7b23 --- /dev/null +++ b/api/services/llm_server.py @@ -0,0 +1,1016 @@ +""" +Local LLM engine — manages a llama.cpp `llama-server` subprocess. + +Everything lives under the per-user directory ~/.modly/llm/: + bin/ llama-server binary + DLLs (auto-downloaded from GitHub releases) + models/ GGUF files (catalog downloads + any custom .gguf the user drops in) + +Nothing is hardcoded to a machine: the binary variant is picked per-platform +(CUDA if an NVIDIA driver is present, otherwise Vulkan, otherwise CPU) and +models are chosen by the user from the catalog. +""" +import contextlib +import os +import re +import shutil +import subprocess +import sys +import tempfile +import threading +import time +import zipfile +from pathlib import Path +from typing import Callable, Optional +from urllib.request import Request, urlopen +import json as _json + +LLM_DIR = Path(os.environ.get("MODLY_LLM_DIR") or Path.home() / ".modly" / "llm") +BIN_DIR = LLM_DIR / "bin" +LLM_MODELS_DIR = LLM_DIR / "models" +LOGS_DIR = LLM_DIR / "logs" +SERVER_PORT = int(os.environ.get("MODLY_LLM_PORT", "8791")) + +# Modly is first and foremost a 3D-generation app: the LLM must never sit on +# VRAM it isn't using. Idle models are evicted after this many seconds. +IDLE_TTL_SECONDS = int(os.environ.get("MODLY_LLM_IDLE_TTL", "300")) + +# Multi-model pool: each loaded model gets its own llama-server process on its +# own port (SERVER_PORT .. SERVER_PORT+MAX_SLOT_PORTS-1). How many may run at +# once is user-configurable ("auto" sizes it from the GPU's VRAM — small cards +# stay at 1, exactly like the old single-slot behavior). +MAX_SLOT_PORTS = 4 +CONFIG_PATH = LLM_DIR / "config.json" + +LLM_MODELS_DIR.mkdir(parents=True, exist_ok=True) + + +def load_config() -> dict: + try: + return _json.loads(CONFIG_PATH.read_text(encoding="utf-8")) + except Exception: + return {} + + +def save_config(cfg: dict) -> None: + CONFIG_PATH.write_text(_json.dumps(cfg, indent=2), encoding="utf-8") + + +_vram_gb_cache: Optional[float] = None + + +def detect_vram_gb() -> float: + """Total VRAM of the first NVIDIA GPU in GiB (0.0 if none/unknown).""" + global _vram_gb_cache + if _vram_gb_cache is not None: + return _vram_gb_cache + try: + out = subprocess.run( + ["nvidia-smi", "--query-gpu=memory.total", "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ).stdout.strip().splitlines() + _vram_gb_cache = round(float(out[0]) / 1024.0, 1) if out else 0.0 + except Exception: + _vram_gb_cache = 0.0 + return _vram_gb_cache + + +def resolve_max_models() -> int: + """How many llama-server processes may run at once. + + Priority: MODLY_LLM_MAX_MODELS env → user config → auto from VRAM. + Auto is deliberately conservative: 8 GB cards keep the single-slot + behavior, VRAM stays available for the 3D pipeline.""" + raw = os.environ.get("MODLY_LLM_MAX_MODELS") or load_config().get("max_models") or "auto" + if isinstance(raw, str) and raw.lower() == "auto": + # Thresholds sit just under the marketing size on purpose: a card sold + # as "12 GB" reports 12227 MiB = 11.9 GiB, so a `>= 12` test excluded + # every 12 GB card there is (5070, 4070, 3060 12G…) from the 2-slot + # tier it was written for. Same story at 20 for a 20/24 GB card. + vram = detect_vram_gb() + n = 3 if vram >= 19.5 else 2 if vram >= 11.5 else 1 + else: + try: + n = int(raw) + except (TypeError, ValueError): + n = 1 + return max(1, min(n, MAX_SLOT_PORTS)) + +_BINARY_NAME = "llama-server.exe" if sys.platform == "win32" else "llama-server" +_UA = {"User-Agent": "modly-llm"} + +# Pinned to a known-good tag by default so installs are reproducible and never +# silently pick up a broken/incompatible "latest" release. Set MODLY_LLM_RELEASE +# to another llama.cpp tag (e.g. "b10075") to override, or to "latest" to track +# the newest release. +_RELEASE_TAG = os.environ.get("MODLY_LLM_RELEASE", "b10075") +_RELEASES_API = ( + "https://api.github.com/repos/ggml-org/llama.cpp/releases/latest" + if _RELEASE_TAG.lower() == "latest" + else f"https://api.github.com/repos/ggml-org/llama.cpp/releases/tags/{_RELEASE_TAG}" +) + +CATALOG_PATH = Path(__file__).resolve().parent.parent / "resources" / "llm_catalog.json" + +DEFAULT_CTX = 16384 +DEFAULT_NGL = 99 + + +def load_catalog() -> list[dict]: + return _json.loads(CATALOG_PATH.read_text(encoding="utf-8")) + + +_FALLBACK_DEFAULT_MODEL = "qwen3-4b" + +# What the agent sends when a catalog entry declares nothing of its own. Low +# temperature is what keeps small models from emitting malformed tool-call JSON. +_DEFAULT_SAMPLING = {"temperature": 0.2} + + +def sampling_for(model_id: str) -> dict: + """Sampling parameters for one local model. + + Model families publish the settings they were tuned for, and ignoring them + is not neutral: served at a flat temperature with no presence penalty, a + Qwen 3.5 repeated the same lookup ten times in one turn and never acted. + A catalog entry can therefore carry its own `sampling` block.""" + entry = next((e for e in load_catalog() if e["id"] == model_id), None) + declared = (entry or {}).get("sampling") or {} + return {**_DEFAULT_SAMPLING, **{k: v for k, v in declared.items() if v is not None}} + + +def default_model_id() -> str: + """Catalog entry tagged "default", falling back to a known-good id if the + catalog has no such tag (keeps old behavior if llm_catalog.json changes).""" + for entry in load_catalog(): + if "default" in (entry.get("tags") or []): + return entry["id"] + return _FALLBACK_DEFAULT_MODEL + + +def mmproj_local_name(entry: dict) -> str: + """Local filename for a vision projector. HF repos all name it mmproj-F16.gguf, + which would collide across models in the shared models dir.""" + return f"mmproj-{entry['id']}.gguf" + + +def list_models() -> list[dict]: + """Catalog entries with a `downloaded` flag + custom GGUFs found on disk.""" + catalog = load_catalog() + known_files = set() + for entry in catalog: + path = LLM_MODELS_DIR / entry["hf_filename"] + downloaded = path.exists() + if entry.get("hf_mmproj_filename"): + downloaded = downloaded and (LLM_MODELS_DIR / mmproj_local_name(entry)).exists() + known_files.add(mmproj_local_name(entry)) + # Show the real download size (weights + vision projector) + entry["size_bytes"] = (entry.get("size_bytes") or 0) + (entry.get("mmproj_size_bytes") or 0) + entry["downloaded"] = downloaded + entry["source"] = "catalog" + known_files.add(entry["hf_filename"]) + + custom = [ + { + "id": f"custom:{f.name}", + "name": f.stem, + "description": "Custom model found in the models folder.", + "hf_filename": f.name, + "size_bytes": f.stat().st_size, + "downloaded": True, + "source": "custom", + "ctx": DEFAULT_CTX, + "ngl_suggestion": DEFAULT_NGL, + "tags": ["custom"], + } + for f in sorted(LLM_MODELS_DIR.glob("*.gguf")) + if f.name not in known_files and not f.name.lower().startswith("mmproj") + ] + return catalog + custom + + +def estimate_vram_mb(entry: dict) -> int: + """How much VRAM this model is expected to hold, in MiB. + + Catalog entries declare it. A user's own GGUF doesn't, so it is derived from + the file: weights land in VRAM about 1:1, plus the KV cache and compute + buffers for a 16k context. Measured against the catalog's own numbers, whose + declared/file ratio runs 1.22–1.32, so 1.25 + 500 MiB sits in the middle.""" + declared = entry.get("vram_estimate_mb") + if isinstance(declared, (int, float)) and declared > 0: + return int(declared) + size_mb = (entry.get("size_bytes") or 0) / (1024 * 1024) + return int(size_mb * 1.25 + 500) if size_mb else 0 + + +def vram_budget_mb() -> int: + """VRAM the LLM pool may fill, in MiB — total minus what the desktop needs. + 0 when no NVIDIA GPU is detected, which disables budgeting entirely.""" + total = detect_vram_gb() * 1024 + if total <= 0: + return 0 + reserve = _env_int("MODLY_LLM_VRAM_RESERVE_MB", 768) + return max(0, int(total) - reserve) + + +def _env_int(name: str, default: int) -> int: + try: + return int(os.environ.get(name) or default) + except ValueError: + return default + + +def resolve_model(model_id: str) -> dict: + """Return {gguf_path, mmproj_path, ngl, ctx, vision, vram_mb} for a catalog id + or a custom: id.""" + if model_id.startswith("custom:"): + # The name is user/agent-supplied and this path is both loaded and + # DELETEd (DELETE /llm/models/{model_id}). Confine it to the models dir: + # on Windows a backslash is a separator too, so "custom:..\..\x.gguf" + # would otherwise resolve — and unlink — outside it. + name = model_id[len("custom:"):] + path = (LLM_MODELS_DIR / name).resolve() + root = LLM_MODELS_DIR.resolve() + if path.parent != root or not path.exists() or path.suffix != ".gguf": + raise KeyError(f"Custom model not found: {model_id}") + return { + "gguf_path": path, "mmproj_path": None, "ngl": DEFAULT_NGL, + "ctx": DEFAULT_CTX, "vision": False, + "vram_mb": estimate_vram_mb({"size_bytes": path.stat().st_size}), + } + for entry in load_catalog(): + if entry["id"] == model_id: + has_mmproj = bool(entry.get("hf_mmproj_filename")) + return { + "gguf_path": LLM_MODELS_DIR / entry["hf_filename"], + "mmproj_path": LLM_MODELS_DIR / mmproj_local_name(entry) if has_mmproj else None, + "ngl": entry.get("ngl_suggestion", DEFAULT_NGL), + "ctx": entry.get("ctx", DEFAULT_CTX), + "vision": "vision" in (entry.get("tags") or []), + "vram_mb": estimate_vram_mb(entry), + } + raise KeyError(f"Unknown model id: {model_id}") + + +def binary_path() -> Path: + return BIN_DIR / _BINARY_NAME + + +def binary_installed() -> bool: + return binary_path().exists() + + +def has_nvidia_gpu() -> bool: + return shutil.which("nvidia-smi") is not None + + +# ─── Binary bootstrap ───────────────────────────────────────────────────────── + +def _fetch_release() -> dict: + with urlopen(Request(_RELEASES_API, headers=_UA), timeout=30) as r: + return _json.loads(r.read()) + + +def _cuda_ver(name: str) -> tuple[int, int]: + m = re.search(r"cuda-(?:cu)?(\d+)[._](\d+)", name.lower()) + return (int(m.group(1)), int(m.group(2))) if m else (0, 0) + + +def _driver_cuda_version() -> tuple[int, int]: + """Highest CUDA version the installed NVIDIA driver supports (0,0 if unknown).""" + try: + out = subprocess.run(["nvidia-smi"], capture_output=True, text=True, timeout=10).stdout + m = re.search(r"CUDA Version:\s*(\d+)\.(\d+)", out) + return (int(m.group(1)), int(m.group(2))) if m else (0, 0) + except Exception: + return (0, 0) + + +def _pick_assets(assets: list[dict]) -> list[dict]: + """Ordered list of assets to install for this machine. + + Windows + NVIDIA: the CUDA build (self-contained, includes CPU fallback) + plus its matching `cudart-…` runtime package. Otherwise Vulkan, then CPU. + macOS / Linux: the platform tarball (Vulkan variant preferred on Linux). + """ + import platform as _platform + + def matching(ext: tuple[str, ...], *needles: str) -> list[dict]: + return [ + a for a in assets + if a["name"].endswith(ext) and all(n in a["name"].lower() for n in needles) + ] + + if sys.platform == "win32": + if has_nvidia_gpu(): + builds = matching((".zip",), "win", "x64", "cuda") + builds = [b for b in builds if not b["name"].startswith("cudart")] + if builds: + driver = _driver_cuda_version() + # Newest build the driver can run; if the driver version is + # unknown, the oldest build has the widest compatibility. + supported = [b for b in builds if _cuda_ver(b["name"]) <= driver] + pool = supported or builds + pick = max(pool, key=lambda b: _cuda_ver(b["name"])) if supported else \ + min(pool, key=lambda b: _cuda_ver(b["name"])) + ver = _cuda_ver(pick["name"]) + cudart = [ + a for a in assets + if a["name"].startswith("cudart") and _cuda_ver(a["name"]) == ver + ] + return [pick, *cudart[:1]] + return (matching((".zip",), "win", "x64", "vulkan") + or matching((".zip",), "win", "x64", "cpu") + or matching((".zip",), "win", "x64", "avx2"))[:1] + + if sys.platform == "darwin": + arch = "arm64" if _platform.machine().lower() in ("arm64", "aarch64") else "x64" + exts = (".zip", ".tar.gz") + return (matching(exts, "macos", arch) or matching(exts, "macos"))[:1] + + exts = (".zip", ".tar.gz") + return (matching(exts, "ubuntu", "vulkan", "x64") + or matching(exts, "ubuntu", "x64") + or matching(exts, "linux", "x64"))[:1] + + +def _download_asset(asset: dict, progress_cb: Callable[[dict], None], control_check: Callable[[], None], label: str) -> Path: + suffix = ".tar.gz" if asset["name"].endswith(".tar.gz") else ".zip" + fd, tmp_name = tempfile.mkstemp(suffix=suffix) + os.close(fd) + tmp = Path(tmp_name) + with urlopen(Request(asset["browser_download_url"], headers=_UA), timeout=30) as resp: + total = int(resp.headers.get("Content-Length", 0)) or asset.get("size", 0) + done = 0 + last_emit = 0.0 + with open(tmp, "wb") as fh: + while chunk := resp.read(1 << 20): + control_check() + fh.write(chunk) + done += len(chunk) + now = time.monotonic() + if now - last_emit >= 0.5: + progress_cb({ + "status": f"Downloading {label} ({asset['name']})", + "bytesDownloaded": done, + "totalBytes": total, + "percent": round(done / total * 100) if total else 0, + }) + last_emit = now + return tmp + + +_LIB_SUFFIXES = (".so", ".dylib", ".metal") + + +def _extract_archive(tmp: Path, asset_name: str) -> int: + """Extract runtime files flat into BIN_DIR (exe/dll on Windows, bin/* + libs elsewhere).""" + BIN_DIR.mkdir(parents=True, exist_ok=True) + count = 0 + + def wanted(member_path: str, fname_low: str) -> bool: + if sys.platform == "win32": + return fname_low.endswith((".exe", ".dll")) + return "/bin/" in member_path.replace("\\", "/") or fname_low.endswith(_LIB_SUFFIXES) + + if asset_name.endswith(".tar.gz"): + import tarfile + with tarfile.open(tmp, "r:gz") as tf: + for m in tf.getmembers(): + if not m.isfile(): + continue + fname = Path(m.name).name + if not fname or not wanted(m.name, fname.lower()): + continue + src = tf.extractfile(m) + if src is None: + continue + dest = BIN_DIR / fname + with open(dest, "wb") as dst: + shutil.copyfileobj(src, dst) + if not fname.lower().endswith(_LIB_SUFFIXES): + dest.chmod(0o755) + count += 1 + return count + + with zipfile.ZipFile(tmp) as zf: + for item in zf.infolist(): + if item.is_dir(): + continue + fname = Path(item.filename).name + if not fname or not wanted(item.filename, fname.lower()): + continue + dest = BIN_DIR / fname + with zf.open(item) as src, open(dest, "wb") as dst: + shutil.copyfileobj(src, dst) + if sys.platform != "win32" and not fname.lower().endswith(_LIB_SUFFIXES): + dest.chmod(0o755) + count += 1 + return count + + +def install_binary(progress_cb: Callable[[dict], None], control_check: Callable[[], None]) -> None: + """Download and install the best llama-server build for this machine.""" + progress_cb({"status": "Fetching latest llama.cpp release…", "percent": 0}) + release = _fetch_release() + progress_cb({"status": f"Release {release['tag_name']} — selecting build for this machine…", "percent": 1}) + + assets = _pick_assets(release["assets"]) + if not assets: + raise RuntimeError("No compatible llama.cpp build found for this platform.") + + for i, asset in enumerate(assets): + label = "engine" if i == 0 else "CUDA runtime" + tmp = _download_asset(asset, progress_cb, control_check, label) + try: + count = _extract_archive(tmp, asset["name"]) + progress_cb({"status": f"Extracted {count} files from {asset['name']}"}) + finally: + tmp.unlink(missing_ok=True) + + if not binary_installed(): + raise RuntimeError("Install finished but llama-server binary is missing.") + progress_cb({"status": "done", "percent": 100}) + + +# ─── Orphan prevention ──────────────────────────────────────────────────────── +# If Modly is force-killed, a plain child process would keep its model in +# memory forever. On Windows we put llama-server in a Job object with +# KILL_ON_JOB_CLOSE (the OS kills it when Modly dies); on Linux we ask the +# kernel to deliver SIGKILL on parent death. + +_win_job_handle = None + + +def _windows_job() -> Optional[int]: + global _win_job_handle + if _win_job_handle is not None: + return _win_job_handle + try: + import ctypes + from ctypes import wintypes + + class JOBOBJECT_BASIC_LIMIT_INFORMATION(ctypes.Structure): + _fields_ = [ + ("PerProcessUserTimeLimit", ctypes.c_int64), + ("PerJobUserTimeLimit", ctypes.c_int64), + ("LimitFlags", wintypes.DWORD), + ("MinimumWorkingSetSize", ctypes.c_size_t), + ("MaximumWorkingSetSize", ctypes.c_size_t), + ("ActiveProcessLimit", wintypes.DWORD), + ("Affinity", ctypes.c_size_t), + ("PriorityClass", wintypes.DWORD), + ("SchedulingClass", wintypes.DWORD), + ] + + class IO_COUNTERS(ctypes.Structure): + _fields_ = [(n, ctypes.c_uint64) for n in ( + "ReadOperationCount", "WriteOperationCount", "OtherOperationCount", + "ReadTransferCount", "WriteTransferCount", "OtherTransferCount")] + + class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure): + _fields_ = [ + ("BasicLimitInformation", JOBOBJECT_BASIC_LIMIT_INFORMATION), + ("IoInfo", IO_COUNTERS), + ("ProcessMemoryLimit", ctypes.c_size_t), + ("JobMemoryLimit", ctypes.c_size_t), + ("PeakProcessMemoryUsed", ctypes.c_size_t), + ("PeakJobMemoryUsed", ctypes.c_size_t), + ] + + kernel32 = ctypes.windll.kernel32 + job = kernel32.CreateJobObjectW(None, None) + info = JOBOBJECT_EXTENDED_LIMIT_INFORMATION() + info.BasicLimitInformation.LimitFlags = 0x2000 # JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE + kernel32.SetInformationJobObject(job, 9, ctypes.byref(info), ctypes.sizeof(info)) # JobObjectExtendedLimitInformation + _win_job_handle = job + return job + except Exception: + return None + + +def _tie_to_parent(process: subprocess.Popen) -> None: + if sys.platform == "win32": + job = _windows_job() + if job is not None: + try: + import ctypes + ctypes.windll.kernel32.AssignProcessToJobObject(job, int(process._handle)) # type: ignore[attr-defined] + except Exception: + pass + + +def _linux_preexec(): + try: + import ctypes + import signal as _signal + libc = ctypes.CDLL("libc.so.6", use_errno=True) + libc.prctl(1, _signal.SIGKILL) # PR_SET_PDEATHSIG + except Exception: + pass + + +def _kill_stale_server(port: int = SERVER_PORT) -> None: + """Kill a leftover llama-server holding `port` (e.g. after a force-kill of Modly).""" + try: + with urlopen(Request(f"http://127.0.0.1:{port}/health", headers=_UA), timeout=1) as r: + if r.status != 200: + return + except Exception: + return # nothing listening — the normal case + try: + if sys.platform == "win32": + out = subprocess.run( + ["netstat", "-ano", "-p", "tcp"], capture_output=True, text=True, timeout=10, + ).stdout + for line in out.splitlines(): + if f":{port}" in line and "LISTENING" in line.upper(): + pid = line.split()[-1] + name = subprocess.run( + ["tasklist", "/FI", f"PID eq {pid}", "/FO", "CSV", "/NH"], + capture_output=True, text=True, timeout=10, + ).stdout.lower() + if "llama-server" in name: + subprocess.run(["taskkill", "/PID", pid, "/F"], capture_output=True, timeout=10) + break + else: + out = subprocess.run( + ["lsof", "-ti", f"tcp:{port}"], capture_output=True, text=True, timeout=10, + ).stdout + for pid in out.split(): + name = subprocess.run( + ["ps", "-p", pid, "-o", "comm="], capture_output=True, text=True, timeout=10, + ).stdout.lower() + if "llama-server" in name: + subprocess.run(["kill", "-9", pid], capture_output=True, timeout=10) + except Exception: + pass + + +# ─── Managed server ─────────────────────────────────────────────────────────── + +class LlamaServerManager: + """One llama-server slot — a single loaded GGUF on a dedicated port. + + Slots are owned by LlamaPool, which decides how many run at once, evicts + idle ones (IDLE_TTL_SECONDS), and unloads everything when the 3D pipeline + needs the VRAM. + """ + + def __init__(self, port: int = SERVER_PORT) -> None: + self.port = port + self._lock = threading.Lock() + self._process: Optional[subprocess.Popen] = None + self._current_id: Optional[str] = None + self._started_at: Optional[float] = None + self._last_used: float = 0.0 + self._log_file = None + # Requests currently being answered by this slot. Guarded by its own + # lock: _lock is held for the whole of a cold start (up to 180 s), and a + # request must never wait on that to declare itself in flight. + self._inflight: int = 0 + self._inflight_lock = threading.Lock() + # Expected VRAM of whatever this slot is serving, for the pool's budget. + self.vram_mb: int = 0 + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self.port}/v1" + + def _alive(self) -> bool: + return self._process is not None and self._process.poll() is None + + def ensure(self, model_id: str, spec: dict) -> None: + """Blocking: make sure `model_id` is loaded, swapping the server if needed. + `spec` comes from resolve_model().""" + with self._lock: + self._last_used = time.monotonic() + if self._current_id == model_id and self._alive(): + return + if not binary_installed(): + raise RuntimeError("llama-server is not installed. Install the engine in Settings → Agent.") + if not spec["gguf_path"].exists(): + raise RuntimeError(f"Model file not found: {spec['gguf_path'].name}. Download it in Settings → Agent.") + + self._terminate_locked() + _kill_stale_server(self.port) + self._spawn(spec) + self._current_id = model_id + self._started_at = time.monotonic() + self._last_used = time.monotonic() + + def touch(self) -> None: + """Mark the server as recently used (postpones idle eviction).""" + self._last_used = time.monotonic() + + @property + def busy_count(self) -> int: + with self._inflight_lock: + return self._inflight + + def hold(self) -> None: + """Claim the slot for a request that is about to start.""" + with self._inflight_lock: + self._inflight += 1 + self._last_used = time.monotonic() + + def release(self) -> None: + with self._inflight_lock: + self._inflight -= 1 + self._last_used = time.monotonic() + + @contextlib.contextmanager + def busy(self): + """Hold the slot for the duration of one request. + + _last_used only moved when a completion *finished*, so a generation + longer than IDLE_TTL_SECONDS — a Text-to-CAD node parked on /llm/chat, + or an agent round on a 14B running on CPU — looked idle to the reaper, + which terminated the server mid-answer: truncated stream, node failed + with no message. Marking the request in flight covers the whole call, + including the prompt-eval stall before the first token.""" + self.hold() + try: + yield self + finally: + self.release() + + def _spawn(self, spec: dict) -> None: + n_threads = max(1, int((os.cpu_count() or 4) * 0.8)) + cmd = [ + str(binary_path()), + "-m", str(spec["gguf_path"]), + "--port", str(self.port), + "--host", "127.0.0.1", + "-c", str(spec["ctx"]), + "-ngl", str(spec["ngl"]), + "--jinja", + "-np", "1", + "--threads", str(n_threads), + "--threads-batch", str(n_threads), + ] + if spec.get("mmproj_path"): + cmd += ["--mmproj", str(spec["mmproj_path"])] + if has_nvidia_gpu(): + # Halve KV-cache VRAM so the larger context still fits 12 GB cards. + # V-cache quantization requires flash attention (fine on CUDA). + cmd += ["-fa", "on", "-ctk", "q8_0", "-ctv", "q8_0"] + # llama-server resolves its plugin DLLs (ggml-cuda, ggml-cpu-*) relative + # to cwd, not the exe path — running from bin/ is required. + env = {**os.environ, "PATH": str(BIN_DIR) + os.pathsep + os.environ.get("PATH", "")} + kwargs: dict = {} + if sys.platform.startswith("linux"): + kwargs["preexec_fn"] = _linux_preexec + + # Overwritten on every spawn (one log per slot/port) so a crash at + # startup (OOM, corrupt GGUF) can be diagnosed instead of vanishing + # into DEVNULL. + LOGS_DIR.mkdir(parents=True, exist_ok=True) + self._log_file = open(LOGS_DIR / f"slot-{self.port}.log", "w", encoding="utf-8") + + self._process = subprocess.Popen( + cmd, + stdout=self._log_file, + stderr=subprocess.STDOUT, + cwd=str(BIN_DIR), + env=env, + **kwargs, + ) + _tie_to_parent(self._process) + self._wait_for_health() + + def _wait_for_health(self, timeout: float = 180.0, interval: float = 1.0) -> None: + url = f"http://127.0.0.1:{self.port}/health" + deadline = time.time() + timeout + while time.time() < deadline: + if not self._alive(): + raise RuntimeError("llama-server exited during startup (bad model file or out of memory?).") + try: + with urlopen(Request(url, headers=_UA), timeout=2) as r: + if r.status == 200: + return + except Exception: + pass + time.sleep(interval) + self._terminate_locked() + raise TimeoutError(f"llama-server did not become ready within {timeout:.0f}s") + + def unload(self) -> None: + with self._lock: + self._terminate_locked() + + def _terminate_locked(self) -> None: + if self._process is None: + return + try: + self._process.terminate() + try: + self._process.wait(timeout=10) + except subprocess.TimeoutExpired: + self._process.kill() + except Exception: + pass + finally: + self._process = None + self._current_id = None + self._started_at = None + if self._log_file is not None: + self._log_file.close() + self._log_file = None + + def snapshot(self) -> dict: + alive = self._alive() + return { + "alive": alive, + "model_id": self._current_id if alive else None, + "port": self.port, + "uptime_seconds": round(time.monotonic() - self._started_at, 1) if alive and self._started_at else None, + "vram_mb": self.vram_mb if alive else None, + } + + +class LlamaPool: + """Pool of llama-server slots — one process per loaded model, each on its + own port, capped by BOTH resolve_max_models() and vram_budget_mb() + (LRU eviction). + + The idle reaper and unload_all() keep the old guarantees: models are + evicted after IDLE_TTL_SECONDS unused, and the 3D pipeline can reclaim all + VRAM at once. + """ + + def __init__(self) -> None: + self._lock = threading.Lock() + self._slots: dict[str, LlamaServerManager] = {} + # Ports held by a load in flight. A reserved slot has no process yet, so + # _alive() cannot tell it apart from a dead one; tracking the port rather + # than the model_id keeps the reservation valid even if unload() drops + # the slot from _slots while it is still starting. + self._loading_ports: set[int] = set() + # Signalled whenever a load lands, so callers held back by a concurrent + # cold start can re-run the limit check instead of loading on top of it. + self._cond = threading.Condition(self._lock) + self._reaper_started = False + + def ensure(self, model_id: str, spec: dict, hold: bool = False) -> LlamaServerManager: + """Blocking: return a ready slot serving `model_id`, starting it (and + evicting the least-recently-used slots) if needed. + + With `hold`, the slot is returned already claimed for one request and the + caller MUST call slot.release() when done (see busy()). Without it there + is a window between "loaded" and "in flight" in which a competing load + can evict the slot, and the caller then posts to a dead server: with + max_models = 1, two nodes asking for different models had one of them + answer 500 Internal Server Error. + + The slot is reserved under the pool lock but LOADED outside it. A cold + start blocks in _wait_for_health for up to 180 s, and holding the pool + lock that long froze every other caller: /llm/status, polled by an open + Settings → Agent page, hung for the whole load, and switch_model's + unload_all() — which exists to reclaim VRAM before a 3D model — queued + behind the very LLM competing for it. + + Loading outside the lock means two callers can be starting different + models at the same time (an agent turn plus a workflow LLM node). A slot + being started holds no process, so it cannot be evicted to make room: + when only such loads keep the pool over its limit, this waits for one to + land — it then becomes an ordinary eviction candidate — instead of + putting a second model on a card sized for one. + """ + incoming_mb = spec.get("vram_mb") or 0 + # A last resort, not a policy: past this the thing we are waiting for is + # stuck, and blocking forever would be worse than loading on top of it. + deadline = time.monotonic() + 300.0 + with self._lock: + while True: + slot = self._slots.get(model_id) + if slot is not None and slot._alive(): + break + if slot is not None and slot.port not in self._loading_ports: + # A server that crashed (OOM, CUDA error) leaves its slot + # behind, and its port is free for the next model. Respawning + # in place would then find that model's server healthy on + # "our" port and kill it. Start over on a fresh port instead. + self._drop_dead_locked(model_id, slot) + slot = None + # Also for a dead slot respawned in place, so reviving one + # cannot push the pool past the limit. Only alive slots that are + # not answering a request are eviction candidates, so this never + # evicts our own — nor anyone's live stream. + self._enforce_limit_locked(reserve=1, incoming_mb=incoming_mb) + remaining = deadline - time.monotonic() + if (not self._over_capacity_locked(incoming_mb) + or not self._transient_blockers_locked() + or remaining <= 0): + break + # Re-checked about once a second: capacity also frees up when a + # request finishes, which is not a load event and so notifies + # nothing. + self._cond.wait(min(1.0, remaining)) + if slot is None: + used_ports = {s.port for s in self._slots.values() if s._alive()} | self._loading_ports + port = next( + (SERVER_PORT + i for i in range(MAX_SLOT_PORTS) + if SERVER_PORT + i not in used_ports), + None, + ) + if port is None: + # A bare next() raised StopIteration here, which surfaced to + # the caller as "Could not start the local LLM: " — an empty + # message. Say what actually happened. + raise RuntimeError( + f"All {MAX_SLOT_PORTS} local-LLM slots are in use or starting up. " + "Wait for a run to finish, or lower the model limit in Settings → Agent." + ) + slot = LlamaServerManager(port) + self._slots[model_id] = slot + slot.vram_mb = incoming_mb + self._loading_ports.add(slot.port) + if hold: + # Claimed BEFORE the load, not after: a caller woken by this load + # landing would otherwise find the slot idle and evict it, and + # two racing callers spent their time killing each other's + # freshly loaded model until one gave up. + slot.hold() + self._start_reaper() + + # Outside the pool lock. LlamaServerManager holds its own, so concurrent + # callers for the same model serialise here and the later ones no-op. + try: + slot.ensure(model_id, spec) + except Exception: + if hold: + slot.release() # nothing to answer with — do not pin the slot + with self._lock: + if self._slots.get(model_id) is slot and not slot._alive(): + self._slots.pop(model_id, None) + raise + finally: + with self._lock: + self._loading_ports.discard(slot.port) + self._cond.notify_all() # waiters can re-run the limit check + return slot + + def _evictable_locked(self, slot: LlamaServerManager) -> bool: + """May this slot be unloaded to make room? Shared by the limit check and + the idle reaper, which both unload while holding the pool lock — a rule + applied to only one of them is a rule that does not hold. + + A slot answering a request is spared because terminating it truncates + the caller's stream mid-generation. A slot another thread is loading is + spared for a harder reason: from the moment Popen returns it is _alive() + (the long wait is _wait_for_health, up to 180 s) while its own lock is + held by the loader, so unload() would block on that lock and freeze the + whole pool with it — /llm/status, every ensure(), and the unload_all() + the 3D pipeline needs to reclaim VRAM. Its cost is still counted, it + just cannot be the victim.""" + return slot.busy_count == 0 and slot.port not in self._loading_ports + + def _enforce_limit_locked(self, reserve: int = 0, incoming_mb: int = 0) -> None: + """Evict LRU slots until the pool fits BOTH limits: the configured model + count, and the VRAM budget. + + The count alone was not enough. `max_models: 2` let any two models in, + so a 9 GB custom model plus a vision model on a 12 GB card oversubscribed + the card; Windows spills to shared memory instead of failing, so nothing + errored — a cold start just went from 5 s to 24 s with no explanation. + Budgeting on declared estimates keeps the pairs that actually fit (a 4B + and a 7B come to 10.4 of 11.5 GiB) and refuses the ones that never did. + + The incoming model is never rejected: if it does not fit even alone, + everything else is evicted and it loads by itself.""" + alive = [(mid, s) for mid, s in self._slots.items() if s._alive()] + alive.sort(key=lambda kv: kv[1]._last_used) # oldest first + # Slots reserved by a concurrent ensure() hold no process yet, so + # _alive() cannot see them. Counting only live slots let two concurrent + # ensure() calls — an agent turn plus a workflow LLM node — each + # conclude the pool was empty and both load, which on an 8 GB card + # (max_models = 1) is exactly the oversubscription this rule prevents. + loading = [s for s in self._slots.values() + if s.port in self._loading_ports and not s._alive()] + candidates = [kv for kv in alive if self._evictable_locked(kv[1])] + + def _evict_oldest() -> None: + mid, slot = candidates.pop(0) + alive.remove((mid, slot)) + slot.unload() + self._slots.pop(mid, None) + + max_n = resolve_max_models() + while len(alive) + len(loading) + reserve > max_n and candidates: + _evict_oldest() + + budget = vram_budget_mb() + if not budget or not incoming_mb: + return # no GPU detected, or an unknown estimate — count rule only + + def _committed_mb() -> int: + return sum(s.vram_mb for _mid, s in alive) + sum(s.vram_mb for s in loading) + + while candidates and _committed_mb() + incoming_mb > budget: + _evict_oldest() + + def _transient_blockers_locked(self) -> bool: + """Is the pool full only of things that end on their own — a load in + flight, or a slot answering a request? Those are worth waiting for; a + merely idle slot is not (it gets evicted instead).""" + return bool(self._loading_ports) or any( + s.busy_count for s in self._slots.values() if s._alive() + ) + + def _over_capacity_locked(self, incoming_mb: int) -> bool: + """Would loading one more model break either limit, counting the slots + another thread is starting right now?""" + alive = [s for s in self._slots.values() if s._alive()] + loading = [s for s in self._slots.values() + if s.port in self._loading_ports and not s._alive()] + if len(alive) + len(loading) + 1 > resolve_max_models(): + return True + budget = vram_budget_mb() + if not budget or not incoming_mb: + return False # no GPU detected, or an unknown estimate + committed = sum(s.vram_mb for s in alive) + sum(s.vram_mb for s in loading) + return committed + incoming_mb > budget + + def enforce_limit(self) -> None: + """Apply the configured limit right away (used when the user lowers it).""" + with self._lock: + self._enforce_limit_locked() + + def is_loaded(self, model_id: str) -> bool: + with self._lock: + slot = self._slots.get(model_id) + return slot is not None and slot._alive() + + def touch(self, model_id: str) -> None: + with self._lock: + slot = self._slots.get(model_id) + if slot is not None: + slot.touch() + + def unload(self, model_id: str) -> None: + with self._lock: + slot = self._slots.pop(model_id, None) + if slot is not None: + slot.unload() + + def unload_all(self, force: bool = False) -> None: + """Free the pool's VRAM. Slots answering a request or being started are + spared (see _evictable_locked): killing one cut a Text-to-CAD stream off + mid-answer when a 3D model loaded, and a cold start in another thread + handed its caller a dead slot. `force` is for shutdown only.""" + if force: + with self._lock: + slots = list(self._slots.values()) + self._slots.clear() + for slot in slots: + slot.unload() + return + with self._lock: + for mid, slot in list(self._slots.items()): + if self._evictable_locked(slot): + slot.unload() + self._slots.pop(mid, None) + + def _drop_dead_locked(self, model_id: str, slot: LlamaServerManager) -> None: + self._slots.pop(model_id, None) + slot.unload() # reaps the exited process and closes its log + + def _start_reaper(self) -> None: + if self._reaper_started or IDLE_TTL_SECONDS <= 0: + return + self._reaper_started = True + threading.Thread(target=self._reap_idle, daemon=True, name="llm-idle-reaper").start() + + def _reap_idle(self) -> None: + while True: + time.sleep(15) + self._reap_once() + + def _reap_once(self, now: Optional[float] = None) -> None: + now = time.monotonic() if now is None else now + with self._lock: + for mid, slot in list(self._slots.items()): + if not self._evictable_locked(slot): + continue # answering right now, or being loaded — never idle + if not slot._alive(): + self._drop_dead_locked(mid, slot) + elif now - slot._last_used > IDLE_TTL_SECONDS: + slot.unload() + self._slots.pop(mid, None) + + def snapshot(self) -> dict: + with self._lock: + slots = list(self._slots.values()) + servers = [s.snapshot() for s in slots if s._alive()] + first = servers[0] if servers else {"alive": False, "model_id": None, "port": SERVER_PORT, "uptime_seconds": None} + return { + **first, # legacy single-server shape + "servers": servers, + "max_models": resolve_max_models(), + "vram_gb": detect_vram_gb() or None, + "vram_budget_mb": vram_budget_mb() or None, + "vram_used_mb": sum(s.get("vram_mb") or 0 for s in servers) or None, + } + + +llama_pool = LlamaPool() diff --git a/api/tests/test_agent_workflow_unload.py b/api/tests/test_agent_workflow_unload.py index d5d7724c..665ea90d 100644 --- a/api/tests/test_agent_workflow_unload.py +++ b/api/tests/test_agent_workflow_unload.py @@ -8,22 +8,19 @@ import routers.agent as agent -class _ScriptedOllama: - """MockTransport wiring: serves a scripted /api/chat sequence and records - every /api/generate (VRAM keep_alive) call agent_chat makes.""" +class _ScriptedServer: + """MockTransport wiring: serves a scripted /chat/completions sequence and + records every request body agent_chat sends.""" def __init__(self, chat_bodies) -> None: self._chat = list(chat_bodies) - self.generate_calls: list[dict] = [] + self.requests: list[dict] = [] self._real = httpx.AsyncClient def _handler(self, request: httpx.Request) -> httpx.Response: - path = request.url.path - if path == "/api/chat": + if request.url.path.endswith("/chat/completions"): + self.requests.append({"headers": dict(request.headers), "body": json.loads(request.content)}) return httpx.Response(200, json=self._chat.pop(0)) - if path == "/api/generate": - self.generate_calls.append(json.loads(request.content)) - return httpx.Response(200, json={}) return httpx.Response(404, json={}) def __call__(self, *args, **kwargs): @@ -31,46 +28,153 @@ def __call__(self, *args, **kwargs): return self._real(*args, **kwargs) +class _FakeSlot: + base_url = "http://127.0.0.1:8791/v1" + + def __init__(self) -> None: + self.released = 0 + + def release(self) -> None: + self.released += 1 + + def _assistant(content: str = "", tool_calls=None) -> dict: msg: dict = {"role": "assistant", "content": content} if tool_calls: msg["tool_calls"] = tool_calls - return {"message": msg} + return {"choices": [{"message": msg}]} + +_RUN_WF1 = {"id": "call_1", "type": "function", "function": {"name": "run_workflow", "arguments": '{"workflow_id": "wf1"}'}} +_SPEC = {"vision": False} -def _run(chat_bodies, context) -> _ScriptedOllama: - scripted = _ScriptedOllama(chat_bodies) + +def _run_local(chat_bodies, context): + scripted = _ScriptedServer(chat_bodies) + slot = _FakeSlot() request = agent.AgentChatRequest( messages=[agent.ChatMessage(role="user", content="make me a thing")], - ollama_url="http://ollama.test", - model="llama-test", + model="qwen3-4b", context=context, ) - with mock.patch.object(agent.httpx, "AsyncClient", scripted): - asyncio.run(agent.agent_chat(request)) - return scripted + with mock.patch.object(agent.httpx, "AsyncClient", scripted), \ + mock.patch.object(agent.llm_server, "resolve_model", return_value=_SPEC), \ + mock.patch.object(agent.llama_pool, "ensure", return_value=slot), \ + mock.patch.object(agent.llama_pool, "unload_all") as unload_all: + response = asyncio.run(agent.agent_chat(request)) + return scripted, slot, unload_all, response class WorkflowVramUnloadTests(unittest.TestCase): def test_llm_unloaded_on_the_normal_return_after_a_workflow(self) -> None: # Round 1 dispatches a workflow; round 2 is the final answer (no tools) and - # returns early. The VRAM unload must still fire — before the fix it lived - # after the loop and this common path skipped it entirely. - chat = [ - _assistant(tool_calls=[ - {"function": {"name": "run_workflow", "arguments": {"workflow_id": "wf1"}}}, - ]), - _assistant(content="Running your workflow now."), - ] - scripted = _run(chat, {"workflows": [{"id": "wf1", "name": "My Workflow"}]}) - self.assertTrue( - any(call.get("keep_alive") == 0 for call in scripted.generate_calls), - f"expected a keep_alive:0 unload, got {scripted.generate_calls}", + # returns early. The VRAM unload must still fire on that path. + chat = [_assistant(tool_calls=[_RUN_WF1]), _assistant(content="Running your workflow now.")] + _, slot, unload_all, response = _run_local(chat, {"workflows": [{"id": "wf1", "name": "My Workflow"}]}) + unload_all.assert_called_once() + self.assertEqual(slot.released, 1) + self.assertEqual([a.tool for a in response.actions], ["run_workflow"]) + + def test_slot_is_released_before_the_unload(self) -> None: + # unload_all() spares held slots, so unloading before the release kept + # the agent's own model on the GPU through the whole workflow. + chat = [_assistant(tool_calls=[_RUN_WF1]), _assistant(content="ok")] + released_at_unload: list[int] = [] + slot = _FakeSlot() + request = agent.AgentChatRequest( + messages=[agent.ChatMessage(role="user", content="go")], + model="qwen3-4b", + context={"workflows": [{"id": "wf1", "name": "My Workflow"}]}, ) + with mock.patch.object(agent.httpx, "AsyncClient", _ScriptedServer(chat)), \ + mock.patch.object(agent.llm_server, "resolve_model", return_value=_SPEC), \ + mock.patch.object(agent.llama_pool, "ensure", return_value=slot), \ + mock.patch.object(agent.llama_pool, "unload_all", lambda: released_at_unload.append(slot.released)): + asyncio.run(agent.agent_chat(request)) + self.assertEqual(released_at_unload, [1]) def test_no_unload_when_no_workflow_was_dispatched(self) -> None: - scripted = _run([_assistant(content="Here is some info.")], {}) - self.assertEqual(scripted.generate_calls, []) + _, slot, unload_all, _ = _run_local([_assistant(content="Here is some info.")], {}) + unload_all.assert_not_called() + self.assertEqual(slot.released, 1) + + def test_a_single_system_message_leads_the_conversation(self) -> None: + # Qwen3.5's template rejects a system message anywhere but first (HTTP 400). + context = {"currentMeshPath": "/tmp/a.glb", "extensions": [{"id": "remesh", "name": "Remesh"}]} + scripted, *_ = _run_local([_assistant(content="ok")], context) + roles = [m["role"] for m in scripted.requests[0]["body"]["messages"]] + self.assertEqual(roles, ["system", "user"]) + system = scripted.requests[0]["body"]["messages"][0]["content"] + self.assertIn("Current mesh path: /tmp/a.glb", system) + self.assertIn("remesh", system) + + def test_tool_result_answers_its_call_id(self) -> None: + chat = [_assistant(tool_calls=[_RUN_WF1]), _assistant(content="done")] + scripted, *_ = _run_local(chat, {"workflows": [{"id": "wf1", "name": "My Workflow"}]}) + tool_msgs = [m for m in scripted.requests[1]["body"]["messages"] if m["role"] == "tool"] + self.assertEqual(tool_msgs[0]["tool_call_id"], "call_1") + + +class ExternalProviderTests(unittest.TestCase): + def test_external_provider_sends_the_key_and_never_touches_the_local_pool(self) -> None: + scripted = _ScriptedServer([_assistant(tool_calls=[_RUN_WF1]), _assistant(content="ok")]) + request = agent.AgentChatRequest( + messages=[agent.ChatMessage(role="user", content="hi")], + model="gpt-test", + provider=agent.ProviderConfig(type="external", base_url="https://llm.test/v1/", api_key="sk-test"), + context={"workflows": [{"id": "wf1", "name": "My Workflow"}]}, + ) + with mock.patch.object(agent.httpx, "AsyncClient", scripted), \ + mock.patch.object(agent.llama_pool, "ensure") as ensure, \ + mock.patch.object(agent.llama_pool, "unload_all") as unload_all: + response = asyncio.run(agent.agent_chat(request)) + self.assertEqual(response.message, "ok") + self.assertEqual(scripted.requests[0]["headers"]["authorization"], "Bearer sk-test") + ensure.assert_not_called() + unload_all.assert_not_called() + + def test_text_only_provider_gets_a_retry_without_the_image(self) -> None: + bodies: list[dict] = [] + + def handler(request: httpx.Request) -> httpx.Response: + body = json.loads(request.content) + bodies.append(body) + if isinstance(body["messages"][1]["content"], list): + return httpx.Response(400, json={"error": {"message": "image input not supported"}}) + return httpx.Response(200, json=_assistant(content="ok")) + + real = httpx.AsyncClient + request = agent.AgentChatRequest( + messages=[agent.ChatMessage(role="user", content="what is this", images=["data:image/png;base64,AAAA"])], + model="text-only", + provider=agent.ProviderConfig(type="external", base_url="https://llm.test/v1", api_key="k"), + ) + with mock.patch.object(agent.httpx, "AsyncClient", lambda *a, **k: real(*a, transport=httpx.MockTransport(handler), **k)): + response = asyncio.run(agent.agent_chat(request)) + self.assertEqual(response.message, "ok") + self.assertEqual(len(bodies), 2) + self.assertIn("what is this", bodies[1]["messages"][1]["content"]) + + def test_provider_error_shows_its_message_not_raw_json(self) -> None: + def handler(_: httpx.Request) -> httpx.Response: + return httpx.Response(401, json={"error": {"message": "Incorrect API key provided.", "type": "invalid_request_error"}}) + + real = httpx.AsyncClient + request = agent.AgentChatRequest( + messages=[agent.ChatMessage(role="user", content="hi")], + provider=agent.ProviderConfig(type="external", base_url="https://llm.test/v1", api_key="bad"), + ) + with mock.patch.object(agent.httpx, "AsyncClient", lambda *a, **k: real(*a, transport=httpx.MockTransport(handler), **k)): + response = asyncio.run(agent.agent_chat(request)) + self.assertEqual(response.message, "LLM error (401): Incorrect API key provided.") + + def test_missing_provider_url_is_reported(self) -> None: + request = agent.AgentChatRequest( + messages=[agent.ChatMessage(role="user", content="hi")], + provider=agent.ProviderConfig(type="external"), + ) + response = asyncio.run(agent.agent_chat(request)) + self.assertIn("No provider URL", response.message) if __name__ == "__main__": diff --git a/api/tests/test_extension_process.py b/api/tests/test_extension_process.py index 204162a2..29602cd6 100644 --- a/api/tests/test_extension_process.py +++ b/api/tests/test_extension_process.py @@ -183,6 +183,24 @@ def test_returns_none_for_unknown_module(self) -> None: self.assertIsNone(proc._resolve_auto_repair_package("totally_unknown_pkg")) +class StderrDecodingTests(unittest.TestCase): + """tqdm draws partial blocks with U+258D/U+258F, whose UTF-8 bytes are + undefined in cp1252. Decoding the child's pipes with the locale codec killed + _stderr_loop mid-run; nothing then drained stderr and the child blocked + forever on write once the pipe filled (a 3D generation froze at 80%).""" + + def test_tqdm_partial_blocks_survive_the_loop(self) -> None: + proc = _make_proc() + bar = "Volume Decoding: 1%|█▍▏| 123/13827\r" + fake_proc = type("FakeProc", (), {"stderr": io.StringIO(bar)})() + + proc._stderr_loop(fake_proc) # must not raise + + def test_child_is_told_to_write_utf8(self) -> None: + proc = _make_proc() + self.assertEqual(proc._build_env().get("PYTHONIOENCODING"), "utf-8") + + class RecvTests(unittest.TestCase): def test_returns_message_from_queue(self) -> None: proc = _make_proc() diff --git a/api/tests/test_llm_downloads.py b/api/tests/test_llm_downloads.py new file mode 100644 index 00000000..6450fa24 --- /dev/null +++ b/api/tests/test_llm_downloads.py @@ -0,0 +1,61 @@ +import importlib +import tempfile +import unittest +from pathlib import Path + +llm_server = importlib.import_module("services.llm_server") +llm_router = importlib.import_module("routers.llm") + +_VISION = { + "id": "vl", + "hf_filename": "weights.gguf", + "hf_mmproj_filename": "mmproj-F16.gguf", +} + + +class DiscardIncompleteTests(unittest.TestCase): + """Cancelling a download has to leave nothing behind. A vision model fetches + weights then projector, so a cancel during the second one used to leave the + finished weights on disk under a model still reported `downloaded: false` — + the UI offers no trash button for those, so the space was unreclaimable.""" + + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + self._orig = llm_server.LLM_MODELS_DIR + llm_server.LLM_MODELS_DIR = Path(self._tmp.name) + + def tearDown(self) -> None: + llm_server.LLM_MODELS_DIR = self._orig + self._tmp.cleanup() + + def _touch(self, name: str) -> Path: + p = llm_server.LLM_MODELS_DIR / name + p.write_bytes(b"x") + return p + + def test_removes_a_part_file(self): + self._touch("weights.gguf.part") + removed = llm_router._discard_incomplete(_VISION) + self.assertEqual(removed, ["weights.gguf.part"]) + self.assertEqual(list(llm_server.LLM_MODELS_DIR.iterdir()), []) + + def test_removes_a_sibling_that_finished_before_the_cancel(self): + self._touch("weights.gguf") # done + self._touch("mmproj-vl.gguf.part") # in flight + removed = llm_router._discard_incomplete(_VISION) + self.assertIn("weights.gguf", removed) + self.assertIn("mmproj-vl.gguf.part", removed) + self.assertEqual(list(llm_server.LLM_MODELS_DIR.iterdir()), []) + + def test_leaves_a_complete_model_alone(self): + self._touch("weights.gguf") + self._touch("mmproj-vl.gguf") + self.assertEqual(llm_router._discard_incomplete(_VISION), []) + self.assertEqual(len(list(llm_server.LLM_MODELS_DIR.iterdir())), 2) + + def test_nothing_on_disk_is_not_an_error(self): + self.assertEqual(llm_router._discard_incomplete(_VISION), []) + + +if __name__ == "__main__": + unittest.main() diff --git a/api/tests/test_llm_pool_budget.py b/api/tests/test_llm_pool_budget.py new file mode 100644 index 00000000..9bc34a95 --- /dev/null +++ b/api/tests/test_llm_pool_budget.py @@ -0,0 +1,279 @@ +import importlib +import unittest + +llm_server = importlib.import_module("services.llm_server") + + +class _FakeSlot: + """Stands in for a loaded LlamaServerManager: alive, with a last-used stamp + and a VRAM estimate.""" + + def __init__(self, last_used: float, vram_mb: int, busy: int = 0, port: int = 0) -> None: + self._last_used = last_used + self.vram_mb = vram_mb + self.port = port + self.busy_count = busy + self.unloaded = False + + def _alive(self) -> bool: + return not self.unloaded + + def unload(self) -> None: + self.unloaded = True + + +class EstimateVramTests(unittest.TestCase): + def test_declared_estimate_wins(self): + self.assertEqual(llm_server.estimate_vram_mb({"vram_estimate_mb": 4200}), 4200) + + def test_custom_model_is_derived_from_file_size(self): + # 9 GB of weights: enough on its own to fill a 12 GB card. + mb = llm_server.estimate_vram_mb({"size_bytes": 9 * 1024**3}) + self.assertGreater(mb, 11000) + self.assertLess(mb, 12500) + + def test_unknown_size_yields_zero(self): + self.assertEqual(llm_server.estimate_vram_mb({}), 0) + + +class PoolBudgetTests(unittest.TestCase): + """Numbers are the ones measured on the reference machine: an RTX 5070 + reporting 11.9 GiB, so a 768 MiB reserve leaves a 11.4 GiB budget.""" + + def setUp(self) -> None: + self.pool = llm_server.LlamaPool() + self._orig_max = llm_server.resolve_max_models + self._orig_budget = llm_server.vram_budget_mb + llm_server.resolve_max_models = lambda: 2 + llm_server.vram_budget_mb = lambda: int(11.9 * 1024) - 768 # 11417 + + def tearDown(self) -> None: + llm_server.resolve_max_models = self._orig_max + llm_server.vram_budget_mb = self._orig_budget + + def _load(self, **slots) -> None: + for i, (mid, vram) in enumerate(slots.items()): + self.pool._slots[mid] = _FakeSlot(last_used=float(i), vram_mb=vram) + + def test_a_pair_that_fits_is_kept(self): + # qwen3-4b (4200) already loaded, cadquery-coder-7b (6200) incoming. + self._load(qwen4b=4200) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=6200) + self.assertEqual(list(self.pool._slots), ["qwen4b"]) + + def test_a_pair_that_does_not_fit_evicts_the_oldest(self): + # qwen3-4b (4200) + qwen3-vl-8b (7800) = 12000 > 11417. + self._load(qwen4b=4200) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=7800) + self.assertEqual(list(self.pool._slots), []) + + def test_an_oversized_model_gets_the_card_to_itself(self): + # The custom 14B alone exceeds the budget: it still loads, alone. + self._load(qwen4b=4200, coder7b=6200) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=11700) + self.assertEqual(list(self.pool._slots), []) + + def test_count_limit_still_applies_under_the_budget(self): + # Three tiny models fit the VRAM budget but not `max_models: 2`. + self._load(a=500, b=500) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=500) + self.assertEqual(list(self.pool._slots), ["b"]) + + def test_evicts_least_recently_used_first(self): + self._load(oldest=4200, newest=4200) + llm_server.resolve_max_models = lambda: 3 + self.pool._enforce_limit_locked(reserve=1, incoming_mb=7800) + self.assertNotIn("oldest", self.pool._slots) + + def test_no_gpu_falls_back_to_the_count_rule(self): + llm_server.vram_budget_mb = lambda: 0 + self._load(a=9000) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=9000) + self.assertEqual(list(self.pool._slots), ["a"]) + + def test_unknown_estimate_never_evicts(self): + # A model with no size on disk must not push everything out. + self._load(a=4200) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=0) + self.assertEqual(list(self.pool._slots), ["a"]) + + def test_a_concurrent_load_counts_against_the_budget(self): + # ensure() reserves under the lock but loads outside it, so a slot being + # started has no process yet. Counting only live slots let two callers + # (agent turn + workflow LLM node) each load 7.8 GB on an 11.4 GB card. + loading = _FakeSlot(last_used=0.0, vram_mb=7800, port=llm_server.SERVER_PORT) + loading.unloaded = True # reserved: not alive yet + self.pool._slots["incoming-a"] = loading + self.pool._loading_ports.add(llm_server.SERVER_PORT) + self._load(alive_one=4200) + + self.pool._enforce_limit_locked(reserve=1, incoming_mb=7800) + # 7800 (in flight) + 7800 (incoming) already blows the budget, so the + # live 4.2 GB model goes. + self.assertNotIn("alive_one", self.pool._slots) + + def test_a_concurrent_load_counts_against_the_model_limit(self): + loading = _FakeSlot(last_used=0.0, vram_mb=500, port=llm_server.SERVER_PORT) + loading.unloaded = True + self.pool._slots["incoming-a"] = loading + self.pool._loading_ports.add(llm_server.SERVER_PORT) + self._load(a=500) # 1 live + 1 loading + 1 incoming > max_models (2) + + self.pool._enforce_limit_locked(reserve=1, incoming_mb=500) + self.assertNotIn("a", self.pool._slots) + + def test_a_slot_answering_a_request_is_never_evicted(self): + # Evicting it would terminate llama-server mid-generation: the caller + # gets a truncated stream and the node fails with no message. + self.pool._slots["busy"] = _FakeSlot(last_used=0.0, vram_mb=7800, busy=1) + self.pool._enforce_limit_locked(reserve=1, incoming_mb=7800) + self.assertIn("busy", self.pool._slots) + + def test_a_slot_being_started_is_never_evicted(self): + # From the moment Popen returns, a slot is _alive() while the loading + # thread still holds its lock for the whole of _wait_for_health (up to + # 180 s). unload() runs UNDER the pool lock, so evicting it there would + # block every other pool operation — /llm/status, ensure(), and the + # unload_all() the 3D pipeline calls to reclaim VRAM — for that long. + starting = _FakeSlot(last_used=0.0, vram_mb=7800, port=llm_server.SERVER_PORT) + self.pool._slots["starting"] = starting + self.pool._loading_ports.add(llm_server.SERVER_PORT) + + self.pool._enforce_limit_locked(reserve=1, incoming_mb=7800) + self.assertIn("starting", self.pool._slots) + self.assertFalse(starting.unloaded) + + def test_over_capacity_counts_loads_in_flight(self): + loading = _FakeSlot(last_used=0.0, vram_mb=7800, port=llm_server.SERVER_PORT) + loading.unloaded = True # reserved, no process yet + self.pool._slots["incoming-a"] = loading + self.pool._loading_ports.add(llm_server.SERVER_PORT) + # Nothing is alive, yet the card is already spoken for. + self.assertTrue(self.pool._over_capacity_locked(7800)) + self.pool._loading_ports.clear() + self.assertFalse(self.pool._over_capacity_locked(7800)) + + def test_only_loads_and_live_requests_are_worth_waiting_for(self): + # ensure() waits for these to end instead of loading a second model on a + # one-model card; a merely idle slot is evicted rather than waited on. + idle = _FakeSlot(last_used=0.0, vram_mb=4200, port=llm_server.SERVER_PORT) + self.pool._slots["idle"] = idle + self.assertFalse(self.pool._transient_blockers_locked()) + + idle.busy_count = 1 + self.assertTrue(self.pool._transient_blockers_locked()) + + idle.busy_count = 0 + self.pool._loading_ports.add(llm_server.SERVER_PORT + 1) + self.assertTrue(self.pool._transient_blockers_locked()) + + def test_a_crashed_slot_is_restarted_on_a_fresh_port(self): + # Its port was handed to another model meanwhile; respawning in place + # killed that model's server through _kill_stale_server. + dead = _FakeSlot(last_used=0.0, vram_mb=500, port=llm_server.SERVER_PORT) + dead.unloaded = True + self.pool._slots["x"] = dead + self.pool._slots["y"] = _FakeSlot(last_used=1.0, vram_mb=500, port=llm_server.SERVER_PORT) + orig = llm_server.LlamaServerManager.ensure + llm_server.LlamaServerManager.ensure = lambda self, mid, spec: None + try: + slot = self.pool.ensure("x", {"vram_mb": 500}) + finally: + llm_server.LlamaServerManager.ensure = orig + self.assertIsNot(slot, dead) + self.assertNotEqual(slot.port, llm_server.SERVER_PORT) + self.assertFalse(self.pool._slots["y"].unloaded) + + def test_unload_all_spares_requests_and_cold_starts(self): + idle = _FakeSlot(last_used=0.0, vram_mb=4200, port=llm_server.SERVER_PORT) + busy = _FakeSlot(last_used=0.0, vram_mb=4200, busy=1, port=llm_server.SERVER_PORT + 1) + starting = _FakeSlot(last_used=0.0, vram_mb=4200, port=llm_server.SERVER_PORT + 2) + self.pool._slots.update(idle=idle, busy=busy, starting=starting) + self.pool._loading_ports.add(starting.port) + + self.pool.unload_all() + self.assertEqual(sorted(self.pool._slots), ["busy", "starting"]) + self.assertTrue(idle.unloaded) + + self.pool.unload_all(force=True) + self.assertEqual(list(self.pool._slots), []) + self.assertTrue(busy.unloaded and starting.unloaded) + + def test_no_free_port_raises_a_readable_error(self): + for i in range(llm_server.MAX_SLOT_PORTS): + self.pool._loading_ports.add(llm_server.SERVER_PORT + i) + with self.assertRaises(RuntimeError) as ctx: + self.pool.ensure("whatever", {"vram_mb": 500}) + self.assertIn("slots", str(ctx.exception)) + + +class ReaperTests(unittest.TestCase): + def setUp(self) -> None: + self.pool = llm_server.LlamaPool() + + def test_an_idle_slot_is_unloaded(self): + old = _FakeSlot(last_used=0.0, vram_mb=4200) + self.pool._slots["old"] = old + self.pool._reap_once(now=llm_server.IDLE_TTL_SECONDS + 1) + self.assertEqual(list(self.pool._slots), []) + self.assertTrue(old.unloaded) + + def test_a_slot_answering_a_long_request_survives(self): + # A generation longer than the TTL (Text-to-CAD, a 14B on CPU) used to + # be killed mid-answer: _last_used only moved once a call had finished. + busy = _FakeSlot(last_used=0.0, vram_mb=4200, busy=1) + self.pool._slots["busy"] = busy + self.pool._reap_once(now=llm_server.IDLE_TTL_SECONDS * 10) + self.assertEqual(list(self.pool._slots), ["busy"]) + self.assertFalse(busy.unloaded) + + def test_a_crashed_slot_is_dropped(self): + dead = _FakeSlot(last_used=0.0, vram_mb=4200) + dead.unloaded = True + self.pool._slots["dead"] = dead + self.pool._reap_once(now=1.0) + self.assertEqual(list(self.pool._slots), []) + + + def test_a_slot_being_started_is_not_idle(self): + # Same reason as the eviction case: the reaper also unloads under the + # pool lock, so a slot mid-cold-start must not be its victim. + starting = _FakeSlot(last_used=0.0, vram_mb=4200, port=llm_server.SERVER_PORT) + self.pool._slots["starting"] = starting + self.pool._loading_ports.add(llm_server.SERVER_PORT) + self.pool._reap_once(now=llm_server.IDLE_TTL_SECONDS * 10) + self.assertEqual(list(self.pool._slots), ["starting"]) + self.assertFalse(starting.unloaded) + + +class BusyContextTests(unittest.TestCase): + def test_busy_counts_nest_and_stamp_last_used(self): + slot = llm_server.LlamaServerManager(port=1) + self.assertEqual(slot.busy_count, 0) + with slot.busy(): + self.assertEqual(slot.busy_count, 1) + with slot.busy(): + self.assertEqual(slot.busy_count, 2) + self.assertEqual(slot.busy_count, 1) + self.assertEqual(slot.busy_count, 0) + self.assertGreater(slot._last_used, 0.0) + + def test_hold_and_release_pair_up(self): + # ensure(hold=True) claims the slot before returning it; the caller + # releases it once the request is over. + slot = llm_server.LlamaServerManager(port=1) + slot.hold() + self.assertEqual(slot.busy_count, 1) + slot.release() + self.assertEqual(slot.busy_count, 0) + + def test_busy_is_released_when_the_request_raises(self): + slot = llm_server.LlamaServerManager(port=1) + with self.assertRaises(ValueError): + with slot.busy(): + raise ValueError("client disconnected") + self.assertEqual(slot.busy_count, 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/electron/main/hf-token.ts b/electron/main/hf-token.ts new file mode 100644 index 00000000..2ea9061a --- /dev/null +++ b/electron/main/hf-token.ts @@ -0,0 +1,61 @@ +import { getSettings, setSettings } from './settings-store' +import { encryptSecretSync, decryptSecretSync } from './secure-store' +import { logger } from './logger' + +/** + * Hugging Face token, encrypted at rest in settings.json the same way the + * provider API keys are (see secure-store.ts). It used to sit there in plain + * text while every other secret in the app was encrypted. + * + * The plaintext lives only in this module's cache: `getHfToken()` is sync + * because the callers that need it are sync (child-process env building), while + * encrypt/decrypt are async. + */ +let cached = '' + +export function getHfToken(): string { + return cached +} + +/** + * Decrypt the stored token into the cache, re-encrypting a legacy plaintext one. + * Synchronous on purpose: the FastAPI bridge reads the token while building its + * child-process env at startup, so an async init could lose a race with it. + */ +export function initHfToken(userData: string): string { + const stored = getSettings(userData).hfToken ?? '' + if (!stored) { + cached = '' + return cached + } + + const decrypted = decryptSecretSync(stored) + if (decrypted === null) { + // One of our blobs, but not decryptable here (different OS user/machine). + // Leave settings.json alone — rewriting would turn an unreadable-but-intact + // blob into a permanently lost one — and expose no token rather than + // handing the ciphertext out as a credential. + cached = '' + logger.error('[hf-token] the stored Hugging Face token could not be decrypted — re-enter it in Settings → Integrations') + return cached + } + + cached = decrypted + // An unchanged value means it was never encrypted (saved before encryption + // existed) — upgrade it in place. + if (cached === stored) { + try { + setSettings(userData, { hfToken: encryptSecretSync(cached) }) + logger.info('[hf-token] migrated a plaintext token to encrypted storage') + } catch (err) { + logger.error(`[hf-token] could not migrate the stored token: ${err}`) + } + } + return cached +} + +/** Persist the token encrypted and refresh the cache. */ +export function setHfToken(userData: string, token: string): void { + cached = token + setSettings(userData, { hfToken: token ? encryptSecretSync(token) : '' }) +} diff --git a/electron/main/ipc-handlers.ts b/electron/main/ipc-handlers.ts index 7d206bde..abcf7869 100644 --- a/electron/main/ipc-handlers.ts +++ b/electron/main/ipc-handlers.ts @@ -84,6 +84,8 @@ import { registerWorkspaceAssetLibraryIpcHandlers } from './artifact-registry-se import { updatesSupported } from './updater' import { ModelWeightOperations } from './model-weight-operations' import { readLocalFileBase64 } from './bounded-file-reader' +import { encryptSecret, decryptSecret } from './secure-store' +import { getHfToken, initHfToken, setHfToken } from './hf-token' type WindowGetter = () => BrowserWindow | null const pExecFile = promisify(execFile) @@ -243,6 +245,11 @@ export function setupIpcHandlers(pythonBridge: PythonBridge, getWindow: WindowGe } }) + // Secure storage — OS-level encryption (Keychain/DPAPI/libsecret) for secrets + // the renderer would otherwise have to keep in plain-text localStorage (API keys). + ipcMain.handle('secure:encrypt', (_, plainText: string) => encryptSecret(plainText)) + ipcMain.handle('secure:decrypt', (_, stored: string) => decryptSecret(stored)) + // Window controls (frameless window) ipcMain.on('window:minimize', () => getWindow()?.minimize()) ipcMain.on('window:maximize', () => { @@ -837,36 +844,40 @@ export function setupIpcHandlers(pythonBridge: PythonBridge, getWindow: WindowGe arch: process.arch, })) - // Settings — seed HF token into main-process env at startup + // Settings — decrypt the HF token (migrating a legacy plaintext one) and seed + // it into the main-process env at startup. { - const initialToken = getSettings(app.getPath('userData')).hfToken ?? '' - if (initialToken) { - process.env['HUGGING_FACE_HUB_TOKEN'] = initialToken - process.env['HF_TOKEN'] = initialToken + const token = initHfToken(app.getPath('userData')) + if (token) { + process.env['HUGGING_FACE_HUB_TOKEN'] = token + process.env['HF_TOKEN'] = token } } ipcMain.handle('settings:get', () => { - return getSettings(app.getPath('userData')) + // hfToken is stored encrypted — hand the renderer the usable value. + return { ...getSettings(app.getPath('userData')), hfToken: getHfToken() } }) ipcMain.handle('settings:set', async (_event, patch: { modelsDir?: string; workspaceDir?: string; extensionsDir?: string; hfToken?: string }) => { if (patch.modelsDir !== undefined && weightOperations.busy) { throw new Error('Cannot change model storage while model weights are busy') } - const updated = setSettings(app.getPath('userData'), patch) + const { hfToken, ...dirs } = patch + setSettings(app.getPath('userData'), dirs) // Keep main-process env in sync so child processes spawned after token change inherit it - if (patch.hfToken !== undefined) { - process.env['HUGGING_FACE_HUB_TOKEN'] = patch.hfToken - process.env['HF_TOKEN'] = patch.hfToken + if (hfToken !== undefined) { + setHfToken(app.getPath('userData'), hfToken) + process.env['HUGGING_FACE_HUB_TOKEN'] = hfToken + process.env['HF_TOKEN'] = hfToken // Also push the token into the live FastAPI process env so extension // subprocesses spawned by ExtensionProcess._build_env() pick it up // without requiring a full app restart. try { - await axios.post(`${API_BASE_URL}/settings/hf-token`, { token: patch.hfToken }, { timeout: 3000 }) + await axios.post(`${API_BASE_URL}/settings/hf-token`, { token: hfToken }, { timeout: 3000 }) } catch { /* FastAPI may not be running yet — ignore */ } } - return updated + return { ...getSettings(app.getPath('userData')), hfToken: getHfToken() } }) // Directory picker diff --git a/electron/main/model-downloader.ts b/electron/main/model-downloader.ts index 174b474a..abad7b81 100644 --- a/electron/main/model-downloader.ts +++ b/electron/main/model-downloader.ts @@ -4,8 +4,7 @@ */ import { existsSync, readdirSync, statSync, readFileSync } from 'fs' import { join } from 'path' -import { getSettings } from './settings-store' -import { app } from 'electron' +import { getHfToken } from './hf-token' import type { ModelSource } from './model-sources' export interface DownloadProgress { @@ -129,12 +128,11 @@ export async function downloadModelFromHF( if (includePrefixes && includePrefixes.length > 0) { url += `&include_prefixes=${encodeURIComponent(JSON.stringify(includePrefixes))}` } - const hfToken = getSettings(app.getPath('userData')).hfToken - if (hfToken) { - url += `&token=${encodeURIComponent(hfToken)}` - } - - const res = await net.fetch(url) + // Header, never a query param: uvicorn logs the full request line to stdout, + // python-bridge pipes that into runtime.log, and `log:readAll` hands that file + // to the user for bug reports. The token used to ride all the way there. + const hfToken = getHfToken() // decrypted cache — settings.json holds the ciphertext + const res = await net.fetch(url, hfToken ? { headers: { 'X-HF-Token': hfToken } } : undefined) if (!res.ok) throw new Error(`HuggingFace download failed: HTTP ${res.status}`) await consumeDownloadStream(res, onProgress) } @@ -147,7 +145,7 @@ export async function downloadModelSourcesFromHF( ): Promise { const { net } = require('electron') const headers: Record = { 'Content-Type': 'application/json' } - const hfToken = getSettings(app.getPath('userData')).hfToken + const hfToken = getHfToken() if (hfToken) headers.Authorization = `Bearer ${hfToken}` const url = `${PYTHON_API_URL}/model/hf-download-sources?model_id=${encodeURIComponent(modelId)}` const res = await net.fetch(url, { @@ -171,12 +169,22 @@ async function consumeDownloadStream( const STALL_TIMEOUT_MS = 120_000 async function readWithTimeout() { - return await Promise.race([ - reader.read(), - new Promise((_, reject) => { - setTimeout(() => reject(new Error(`Model download stalled for ${Math.round(STALL_TIMEOUT_MS / 1000)}s`)), STALL_TIMEOUT_MS) - }), - ]) + // The timer has to be cleared: an SSE stream emitting ~10 events/s otherwise + // keeps every timer created in the last 120 s alive in the main process. + let timer: NodeJS.Timeout | undefined + try { + return await Promise.race([ + reader.read(), + new Promise((_, reject) => { + timer = setTimeout( + () => reject(new Error(`Model download stalled for ${Math.round(STALL_TIMEOUT_MS / 1000)}s`)), + STALL_TIMEOUT_MS, + ) + }), + ]) + } finally { + if (timer) clearTimeout(timer) + } } while (true) { diff --git a/electron/main/python-bridge.ts b/electron/main/python-bridge.ts index 0f8a7d82..82edb9ea 100644 --- a/electron/main/python-bridge.ts +++ b/electron/main/python-bridge.ts @@ -4,6 +4,7 @@ import { app, BrowserWindow } from 'electron' import { existsSync, mkdirSync } from 'fs' import axios from 'axios' import { getSettings } from './settings-store' +import { getHfToken } from './hf-token' import { logger } from './logger' import { cleanPythonEnv, getVenvPythonExe } from './python-setup' @@ -55,6 +56,11 @@ export class PythonBridge { // mesh_ops uses Electron in Node mode for the existing meshoptimizer // backend, so packaged builds do not depend on a system Node install. MODLY_NODE_EXECUTABLE: process.execPath, + // We decode both pipes as UTF-8 below (Buffer.toString() default), and + // the API forwards extension output — tqdm bars included — through its + // own stderr. Without this the API would encode them with the Windows + // locale codec and the HUD log pane would show █ escapes. + PYTHONIOENCODING: 'utf-8', // No PYTHONPATH needed - the venv's Python has its own isolated site-packages MODELS_DIR: this.resolveModelsDir(), WORKSPACE_DIR: this.resolveWorkspaceDir(), @@ -241,6 +247,6 @@ export class PythonBridge { } private resolveHfToken(): string { - return getSettings(app.getPath('userData')).hfToken ?? '' + return getHfToken() // decrypted cache — settings.json holds the ciphertext } } diff --git a/electron/main/secure-store.ts b/electron/main/secure-store.ts new file mode 100644 index 00000000..7c70cd27 --- /dev/null +++ b/electron/main/secure-store.ts @@ -0,0 +1,78 @@ +import { safeStorage } from 'electron' +import { logger } from './logger' + +// Secrets that safeStorage couldn't encrypt (no OS keychain available, e.g. some +// Linux setups without a keyring) are stored with this prefix so decryptSecret +// can tell them apart from a real ciphertext and hand them back unchanged. +const PLAINTEXT_PREFIX = 'plain:' + +let warnedOnce = false + +// Electron 33 (pinned in package.json) only ships the synchronous safeStorage +// API — encryptStringAsync/decryptStringAsync/isAsyncEncryptionAvailable were +// added in a later Electron release. The sync calls are cheap (no disk I/O, +// just OS keychain/DPAPI crypto), so wrapping them in an async function here +// is only to keep the IPC handler signature uniform, not for a real await. + +export function encryptSecretSync(plainText: string): string { + if (!plainText) return plainText + + if (!safeStorage.isEncryptionAvailable()) { + if (!warnedOnce) { + warnedOnce = true + logger.warn('[secure-store] OS-level encryption unavailable — storing secrets in plain text') + } + return PLAINTEXT_PREFIX + plainText + } + + return safeStorage.encryptString(plainText).toString('hex') +} + +/** + * True when `stored` has the shape encryptSecretSync produces: a hex blob + * starting with Chromium OSCrypt's version tag ("v10", "v11", …). The tag + * matters — hex alone also matched plaintext API keys that happen to be hex + * (common for self-hosted endpoints), which then "failed to decrypt" and were + * wiped during the migration instead of encrypted. Used to tell "this was never encrypted" apart from "this IS one of + * our blobs but decryption just failed" — the two must not be confused: DPAPI + * keys are per-OS-user, so a restored backup or a different Windows account + * makes decryption fail on a perfectly valid ciphertext. Treating that as + * plaintext would hand the ciphertext out as a credential and then re-encrypt + * it, destroying the secret for good. + */ +function looksEncrypted(stored: string): boolean { + return stored.length >= 32 && /^76(3[0-9]){2}([0-9a-f]{2})+$/i.test(stored) +} + +/** + * Returns the plaintext, or `null` when `stored` is one of our blobs that could + * not be decrypted. A value that was never encrypted comes back unchanged, so + * callers can migrate it. + */ +export function decryptSecretSync(stored: string): string | null { + if (!stored) return stored + if (stored.startsWith(PLAINTEXT_PREFIX)) return stored.slice(PLAINTEXT_PREFIX.length) + + if (!looksEncrypted(stored)) return stored // legacy plaintext, saved before encryption existed + + try { + return safeStorage.decryptString(Buffer.from(stored, 'hex')) + } catch (err) { + logger.error( + '[secure-store] a stored secret could not be decrypted — it was encrypted by a ' + + `different OS user or machine. It must be re-entered. (${err})`, + ) + return null + } +} + +// Async wrappers kept for the IPC handlers, whose signatures are all async. +export async function encryptSecret(plainText: string): Promise { + return encryptSecretSync(plainText) +} + +/** `null` on an undecryptable blob — never the ciphertext, which callers would + * otherwise use as a credential and re-encrypt. */ +export async function decryptSecret(stored: string): Promise { + return decryptSecretSync(stored) +} diff --git a/electron/preload/electron-api.ts b/electron/preload/electron-api.ts index 0c7f74a9..d03656f8 100644 --- a/electron/preload/electron-api.ts +++ b/electron/preload/electron-api.ts @@ -103,6 +103,14 @@ export function createElectronApi(ipcRenderer: IpcRendererLike, webFrame: WebFra ipcRenderer.invoke('fs:readScreenshotDataUrl', filename) as Promise, }, + // Secure storage — OS-level encryption for secrets (API keys, …) + secureStore: { + encrypt: (plainText: string): Promise => ipcRenderer.invoke('secure:encrypt', plainText) as Promise, + // null = one of our blobs that couldn't be decrypted here (different OS + // user/machine). Never the ciphertext — see secure-store.ts. + decrypt: (stored: string): Promise => ipcRenderer.invoke('secure:decrypt', stored) as Promise, + }, + // Settings settings: { get: (): Promise<{ modelsDir: string; workspaceDir: string; workflowsDir: string; extensionsDir: string; hfToken?: string }> => diff --git a/scripts/react-test-env.mjs b/scripts/react-test-env.mjs new file mode 100644 index 00000000..d2bd8d2e --- /dev/null +++ b/scripts/react-test-env.mjs @@ -0,0 +1,103 @@ +/** + * Minimal React test environment for `node --test`. + * + * The repo tests everything except the UI layer, which is exactly where the + * bugs the user actually sees have come from: a hook returning a fresh callback + * on every render made a modal re-run its effect forever, flickering and + * hammering the API. That class of bug is invisible to a type-checker and to + * store-level tests — it only exists once a component renders more than once. + * + * Deliberately built on jsdom + react-dom directly rather than a testing + * library: what we need is "render, re-render, count the effects", not queries + * and user events. + */ +import { JSDOM } from 'jsdom' +import { buildSync } from 'esbuild' +import { createRequire } from 'node:module' +import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { join, resolve } from 'node:path' + +/** Install a DOM into the globals React expects. Call once, at module scope. */ +export function setupDom() { + const dom = new JSDOM('', { url: 'http://localhost/' }) + const { window } = dom + + // Node 24 defines some of these (`navigator`) as getter-only on globalThis, + // so a plain assignment throws — go through defineProperty for all of them. + const define = (name, value) => + Object.defineProperty(globalThis, name, { value, writable: true, configurable: true }) + + define('window', window) + define('document', window.document) + define('navigator', window.navigator) + define('localStorage', window.localStorage) + define('requestAnimationFrame', (cb) => setTimeout(() => cb(Date.now()), 0)) + define('cancelAnimationFrame', (id) => clearTimeout(id)) + // React 18 refuses to run `act` without it, and warns on every update. + define('IS_REACT_ACT_ENVIRONMENT', true) + for (const name of ['HTMLElement', 'Element', 'Node', 'Event', 'MouseEvent', 'getComputedStyle']) { + define(name, window[name]) + } + return dom +} + +/** + * Bundle a source module and load it as CommonJS. + * + * `react` and `react-dom` stay external so the module under test and the test + * file share one React instance — two copies produce "invalid hook call", + * which reads as a bug in the code under test and is not one. + */ +export function loadModule(entryPath) { + // Inside the project, not the system temp dir: the bundle keeps `react` as a + // bare require, and that only resolves from a path under this node_modules. + const cacheRoot = resolve('node_modules/.cache/modly-react-tests') + mkdirSync(cacheRoot, { recursive: true }) + const outfile = join(mkdtempSync(join(cacheRoot, 'm-')), 'module.cjs') + const result = buildSync({ + entryPoints: [resolve(entryPath)], + bundle: true, + platform: 'node', + format: 'cjs', + jsx: 'automatic', + external: ['react', 'react-dom', 'react-dom/client', 'react/jsx-runtime'], + write: false, + }) + writeFileSync(outfile, result.outputFiles[0].text, 'utf8') + return createRequire(import.meta.url)(outfile) +} + +/** + * Render `element`, and hand back a way to re-render it with the same root — + * which is the whole point: a hook that misbehaves does so on the SECOND render. + */ +export async function mount(element) { + const require = createRequire(import.meta.url) + const { createRoot } = require('react-dom/client') + const { act } = require('react-dom/test-utils') + const { cloneElement } = require('react') + + // Through globalThis: these are globals this module installed itself in + // setupDom(), not ambient browser ones. + const container = globalThis.document.createElement('div') + globalThis.document.body.appendChild(container) + const root = createRoot(container) + + await act(async () => { root.render(element) }) + + return { + container, + /** Re-render. The element is cloned by default: handed the very same + * element object, React bails out and nothing re-renders — which silently + * turns a re-render test into a no-op. */ + rerender: async (next) => { + await act(async () => { root.render(next ?? cloneElement(element)) }) + }, + /** Let effects, promises and state updates settle. */ + flush: async () => { await act(async () => { await Promise.resolve() }) }, + unmount: async () => { + await act(async () => { root.unmount() }) + container.remove() + }, + } +} diff --git a/src/areas/generate/components/ChatPanel.tsx b/src/areas/generate/components/ChatPanel.tsx index 7618f75d..053e780a 100644 --- a/src/areas/generate/components/ChatPanel.tsx +++ b/src/areas/generate/components/ChatPanel.tsx @@ -1,6 +1,7 @@ import { useEffect, useMemo, useRef, useState } from 'react' import { useAppStore } from '@shared/stores/appStore' -import { useAgentStore } from '@shared/stores/agentStore' +import { useAgentStore, PROVIDERS, providerBaseUrl } from '@shared/stores/agentStore' +import { useLlmModels } from '@shared/stores/llmModelsStore' import { useWorkflowsStore } from '@shared/stores/workflowsStore' import { useExtensionsStore } from '@shared/stores/extensionsStore' import { useWorkflowRunStore } from '@areas/workflows/workflowRunStore' @@ -242,16 +243,22 @@ function WorkflowProgressCard({ name }: { name: string }): JSX.Element { // ─── Main component ────────────────────────────────────────────────────────── export default function ChatPanel(): JSX.Element { - const { ollamaUrl, defaultModel, defaultThinking } = useAgentStore() + const { provider, localModel, external, defaultThinking } = useAgentStore() + const { models: llmModels } = useLlmModels() + const isLocal = provider === 'local' + const externalConfig = external[provider] const [messages, setMessages] = useState([]) const [input, setInput] = useState('') const [isLoading, setIsLoading] = useState(false) const [error, setError] = useState(null) const [showAll, setShowAll] = useState(false) - const [model, setModel] = useState(defaultModel) + // null = follow the default from Settings, which hydrates asynchronously. + const [localPick, setLocalPick] = useState(null) + const model = isLocal ? (localPick ?? localModel) : (externalConfig?.model ?? '') + // code/cad models are node tools, not chat models. + const localModels = llmModels.filter((m) => m.downloaded && !(m.tags ?? []).some((t) => t === 'code' || t === 'cad')) const [showModelPicker, setShowModelPicker] = useState(false) - const [ollamaModels, setOllamaModels] = useState([]) const [pendingWorkflow, setPendingWorkflow] = useState<{ id: string; name: string } | null>(null) const [attachments, setAttachments] = useState([]) // data URLs const [isDragging, setIsDragging] = useState(false) @@ -349,7 +356,7 @@ export default function ChatPanel(): JSX.Element { content: m.content, } if (m.imageDataUrls?.length) { - entry.images = m.imageDataUrls.map((url) => url.split(',')[1]) + entry.images = m.imageDataUrls } return entry }) @@ -361,7 +368,15 @@ export default function ChatPanel(): JSX.Element { const res = await fetch(`${apiUrl}/agent/chat`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ messages: apiMessages, ollama_url: ollamaUrl, model, context, thinking: thinkingMode }), + body: JSON.stringify({ + messages: apiMessages, + model, + provider: isLocal + ? { type: 'local' } + : { type: 'external', base_url: providerBaseUrl(provider, externalConfig), api_key: externalConfig?.apiKey ?? '' }, + context, + thinking: thinkingMode, + }), }) if (!res.ok) throw new Error(`API error ${res.status}`) @@ -406,16 +421,6 @@ export default function ChatPanel(): JSX.Element { } } - async function fetchOllamaModels() { - try { - const res = await fetch(`${apiUrl}/agent/models?ollama_url=${encodeURIComponent(ollamaUrl)}`) - const data = await res.json() - setOllamaModels(data.models ?? []) - } catch { - setOllamaModels([]) - } - } - function handleFiles(files: File[]) { files.forEach((file) => { if (!file.type.startsWith('image/')) return @@ -646,10 +651,10 @@ export default function ChatPanel(): JSX.Element { {/* Model selector */}
- {testResult === 'ok' && ( -

Connection successful — {models.length} model{models.length > 1 ? 's' : ''} found

- )} - {testResult === 'error' && ( -

Could not reach Ollama at this address

- )} - +

+ Browse models by category (General, Vision), download only what you need, and pick the chat default. +

- - {models.length > 0 ? ( + - ) : ( - setModelDraft(e.target.value)} - className="bg-zinc-900 border border-zinc-700/60 rounded-lg px-3 py-2 text-[12.5px] text-zinc-200 focus:outline-none focus:border-zinc-500" - placeholder="gemma4:e4b" - /> + + + ) : ( + + {provider === 'custom' && ( + + { setBaseUrlDraft(e.target.value); setExtResult(null) }} + placeholder="http://192.168.1.20:8080/v1" + className={inputCls} + /> + )} - - - + +
+ { setKeyDraft(e.target.value); setExtResult(null) }} + placeholder={PROVIDERS[provider].noKey ? '' : 'sk-…'} + className={`${inputCls} flex-1`} + /> + +
+ {extResult === 'ok' && ( +

Connected — {extModels.length} model{extModels.length > 1 ? 's' : ''} available

+ )} + {extResult === 'error' && ( +

+ {provider === 'ollama' + ? 'Could not list models — is Ollama running?' + : `Could not list models — check the key${provider === 'custom' ? ' and URL' : ''}`} +

+ )} +
+ + + {extModels.length > 0 ? ( + + ) : ( + setExtModelDraft(e.target.value)} + placeholder={provider === 'anthropic' ? 'claude-sonnet-5' : provider === 'openai' ? 'gpt-5.2' : provider === 'ollama' ? 'qwen2.5:3b' : 'model name'} + className={inputCls} + /> + )} + + + + + )} {/* Thinking */} @@ -221,6 +364,10 @@ export function AgentSection(): JSX.Element { ))} + + {showLibrary && ( + { setShowLibrary(false); void refreshLocal() }} /> + )} ) } diff --git a/src/shared/components/ui/LlmModelSelect.tsx b/src/shared/components/ui/LlmModelSelect.tsx new file mode 100644 index 00000000..44bb1da1 --- /dev/null +++ b/src/shared/components/ui/LlmModelSelect.tsx @@ -0,0 +1,62 @@ +import { useLlmModels } from '@shared/stores/llmModelsStore' + +/** + * Dropdown of the shared local-LLM library, optionally filtered by category + * (`tag`). Catalog models that aren't downloaded yet stay visible (marked) so + * the user can pick ONE and download only that one in Settings → Agent. + * + * Shared by the LLM node and by extension params of type `llm-model`, so both + * show the same list, the same names and the same warning. + */ +/** A GGUF the user dropped in the models folder has no tags, so the API returns + * it for every category ("capabilities unknown" — api/routers/llm.py). Under a + * category filter that reads as an endorsement: a general-purpose model showed + * up as a CAD model in Text to CAD's picker with nothing to say otherwise. */ +function label(m: { source?: string }, tag?: string): string { + return tag && m.source === 'custom' ? ' — custom, not verified for this use' : '' +} + +export default function LlmModelSelect({ value, tag, disabled, className, onChange }: { + value: string + tag?: string + disabled?: boolean + className?: string + onChange: (v: string) => void +}) { + const { models } = useLlmModels(tag) + + const ready = models.filter((m) => m.downloaded) + const missing = models.filter((m) => !m.downloaded) + const selectedMissing = missing.some((m) => m.id === value) + + return ( +
+ + {selectedMissing && ( +

+ Download this model in Settings → Agent before running. +

+ )} +
+ ) +} diff --git a/src/shared/components/ui/ModelLibraryModal.tsx b/src/shared/components/ui/ModelLibraryModal.tsx new file mode 100644 index 00000000..81e6afe5 --- /dev/null +++ b/src/shared/components/ui/ModelLibraryModal.tsx @@ -0,0 +1,412 @@ +import { useCallback, useEffect, useRef, useState } from 'react' +import { createPortal } from 'react-dom' +import { useAppStore } from '@shared/stores/appStore' +import { useAgentStore } from '@shared/stores/agentStore' +import { useLlmModels, type LlmModel } from '@shared/stores/llmModelsStore' +import { consumeSse, useLlmDownloadsStore, type SseEvent } from '@shared/services/llmDownloads' +import { formatBytes as fmtBytes } from '@shared/utils/format' +import { vramFit } from './vramFit' +import { agentGrade } from './agentGrade' + +// ─── Types ──────────────────────────────────────────────────────────────────── + +// Single source of truth lives in the shared catalog store; re-exported here so +// existing importers (AgentSection) keep resolving `LlmModel` from this module. +export type { LlmModel } + +interface LlmStatus { + binary_installed: boolean + has_nvidia_gpu: boolean + vram_gb: number | null + models_dir: string + server: { alive: boolean; model_id: string | null } +} + +type CategoryId = 'all' | 'general' | 'vision' + +const CATEGORIES: { id: CategoryId; label: string }[] = [ + { id: 'all', label: 'All' }, + { id: 'general', label: 'General' }, + { id: 'vision', label: 'Vision' }, +] + +function inCategory(m: LlmModel, cat: CategoryId): boolean { + const tags = m.tags ?? [] + switch (cat) { + case 'all': return true + case 'vision': return tags.includes('vision') + case 'general': return !tags.includes('vision') + } +} + +// ─── Helpers ────────────────────────────────────────────────────────────────── + +export function formatBytes(n?: number): string { + return n ? fmtBytes(n) : '—' +} + +function ProgressBar({ event }: { event: SseEvent }): JSX.Element { + return ( +
+
+ {event.status} + + {event.totalBytes ? `${formatBytes(event.bytesDownloaded)} / ${formatBytes(event.totalBytes)}` : `${event.percent ?? 0}%`} + +
+
+
+
+
+ ) +} + +// ─── Component ──────────────────────────────────────────────────────────────── + +export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Element { + const apiUrl = useAppStore((s) => s.apiUrl) + const localModel = useAgentStore((s) => s.localModel) + const setLocalModel = useAgentStore((s) => s.setLocalModel) + + const [status, setStatus] = useState(null) + const [category, setCategory] = useState('all') + const [installing, setInstalling] = useState(null) + const [error, setError] = useState(null) + + // The model list comes from the shared catalog store, so a download or delete + // here immediately updates every other picker (chat, extension params, + // chat) and vice-versa — no independent per-surface fetch. + const { models, refresh: refreshModels } = useLlmModels() + + // Downloads live in a module-level store (src/shared/services/llmDownloads.ts) + // so they keep running — and stay visible on reopen — after this modal closes. + const downloads = useLlmDownloadsStore((s) => s.downloads) + const downloadError = useLlmDownloadsStore((s) => s.error) + const startDownload = useLlmDownloadsStore((s) => s.start) + const pauseDownload = useLlmDownloadsStore((s) => s.pause) + const cancelDownload = useLlmDownloadsStore((s) => s.cancel) + const dismissDownloadError = useLlmDownloadsStore((s) => s.dismissError) + + // Aborts every in-flight fetch (status/model list + any SSE stream) when the + // modal unmounts, so closing it mid-download doesn't leak a fetch or call + // setState on a component that's gone. + const aliveRef = useRef(true) + const abortControllersRef = useRef(new Set()) + const dialogRef = useRef(null) + + function withAbort(run: (signal: AbortSignal) => Promise): Promise { + const controller = new AbortController() + abortControllersRef.current.add(controller) + return run(controller.signal).finally(() => { abortControllersRef.current.delete(controller) }) + } + + useEffect(() => { + aliveRef.current = true + const controllers = abortControllersRef.current + return () => { + aliveRef.current = false + for (const controller of controllers) controller.abort() + controllers.clear() + } + }, []) + + // Escape closes the modal; Tab is trapped inside it while it's open. + useEffect(() => { + function onKeyDown(e: KeyboardEvent) { + if (e.key === 'Escape') { onClose(); return } + if (e.key !== 'Tab' || !dialogRef.current) return + const focusable = dialogRef.current.querySelectorAll( + 'button, [href], input, select, textarea, [tabindex]:not([tabindex="-1"])', + ) + if (focusable.length === 0) return + const first = focusable[0] + const last = focusable[focusable.length - 1] + if (e.shiftKey && document.activeElement === first) { e.preventDefault(); last.focus() } + else if (!e.shiftKey && document.activeElement === last) { e.preventDefault(); first.focus() } + } + document.addEventListener('keydown', onKeyDown) + dialogRef.current?.focus() + return () => document.removeEventListener('keydown', onKeyDown) + }, [onClose]) + + const refresh = useCallback(async () => { + void refreshModels() // shared model catalog (propagates to every picker) + try { + const s = await withAbort((signal) => + fetch(`${apiUrl}/llm/status`, { signal }).then((r) => r.json()), + ) + if (!aliveRef.current) return + setStatus(s) + } catch { + if (!aliveRef.current) return + setStatus(null) + } + }, [apiUrl, refreshModels]) + + // Once, on open (and if the API URL changes) — `refresh` is stable, see + // useLlmModels. It used to be rebuilt on every render, so this effect re-ran + // on every render and each pass forced another /llm/models + /llm/status. + useEffect(() => { void refresh() }, [refresh]) + + async function handleInstallEngine() { + setInstalling({ percent: 0, status: 'Starting…' }) + setError(null) + try { + await withAbort((signal) => consumeSse(`${apiUrl}/llm/binary/install`, (e) => { + if (!aliveRef.current) return + if (e.error) { setError(e.error); return } + setInstalling(e) + }, signal)) + } catch (e) { + if (aliveRef.current) setError(e instanceof Error ? e.message : String(e)) + } finally { + if (aliveRef.current) setInstalling(null) + void refresh() + } + } + + async function handleDelete(id: string) { + await fetch(`${apiUrl}/llm/models/${encodeURIComponent(id)}`, { method: 'DELETE' }).catch(() => {}) + void refresh() + } + + // Downloads run in the shared store, possibly finishing while this modal is + // closed — refresh the model list whenever one drops out (done/error/cancelled) + // while we're mounted, so "downloaded" flips without waiting for a remount. + const prevDownloadIdsRef = useRef>(new Set()) + useEffect(() => { + const prev = prevDownloadIdsRef.current + const current = new Set(Object.keys(downloads).filter((id) => downloads[id] !== undefined)) + let finished = false + for (const id of prev) if (!current.has(id)) finished = true + prevDownloadIdsRef.current = current + if (finished) void refresh() + }, [downloads, refresh]) + + const visible = models.filter((m) => inCategory(m, category)) + const counts = Object.fromEntries( + CATEGORIES.map((c) => [c.id, models.filter((m) => inCategory(m, c.id)).length]), + ) as Record + + return createPortal( +
{ if (e.target === e.currentTarget) onClose() }} + > +
+ +
+ + {/* Header */} +
+
+

Model library

+

+ Local models shared by the whole app — chat agent and extensions. +

+
+ +
+ + {/* Engine status */} +
+ {status === null ? ( +

+ Cannot reach the Modly API. + {/* The library no longer polls, so a backend that was still + * starting up needs a way back in short of reopening the modal. */} + +

+ ) : !status.binary_installed ? ( +
+

+ The inference engine is not installed. Modly will fetch the llama.cpp build matching this + machine ({status.has_nvidia_gpu ? 'NVIDIA GPU detected — CUDA build' : 'Vulkan/CPU build'}). +

+ {installing ? ( + + ) : ( + + )} +
+ ) : ( +

+ Engine installed{status.server.alive ? ` — ${status.server.model_id} loaded` : ''} +

+ )} + {(error || downloadError) && ( +

+ {error || downloadError} + +

+ )} +
+ + {/* Category tabs */} +
+ {CATEGORIES.map((c) => ( + + ))} +
+ + {/* Model list */} +
+ {visible.map((m) => { + const dl = downloads[m.id] + return ( +
+
+
+

+ {m.name} + {m.source === 'custom' && ( + custom + )} + {(m.tags ?? []).includes('cad') && ( + CAD + )} + {(m.tags ?? []).includes('vision') && ( + Vision + )} + {/* Size and VRAM say nothing about how well a model drives + * the agent — a 4B outscores a 20B here. The tooltip keeps + * a measured rate and an estimate visibly apart. */} + {(() => { + const grade = agentGrade(m) + return grade ? ( + + {grade.label} + + ) : null + })()} +

+

+ + {formatBytes(m.size_bytes)} + {m.quant ? ` · ${m.quant}` : ''} + {m.vram_estimate_mb ? ` · ~${(m.vram_estimate_mb / 1000).toFixed(1)} GB VRAM` : ''} + + {(() => { + const fit = vramFit(m.vram_estimate_mb, status?.vram_gb) + return fit ? ( + {fit.label} + ) : null + })()} +

+
+
+ {m.downloaded ? ( + <> + {localModel === m.id ? ( + Default + ) : ( + + )} + + + ) : dl ? ( + <> + {dl.paused ? ( + + ) : ( + + )} + + + ) : ( + + )} +
+
+ {m.description &&

{m.description}

} + {dl && } +
+ ) + })} + {visible.length === 0 && ( +

No models in this category.

+ )} +
+ + {/* Footer hint */} +
+

+ Custom models: drop any .gguf in {status?.models_dir ?? '~/.modly/llm/models'} — detected automatically. +

+
+
+
, + document.body, + ) +} diff --git a/src/shared/components/ui/agentGrade.test.mjs b/src/shared/components/ui/agentGrade.test.mjs new file mode 100644 index 00000000..3ee1aee4 --- /dev/null +++ b/src/shared/components/ui/agentGrade.test.mjs @@ -0,0 +1,53 @@ +import test from 'node:test' +import assert from 'node:assert/strict' +import { buildSync } from 'esbuild' +import { createRequire } from 'node:module' +import { mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +// Same bundling trick as vramFit.test.mjs — a pure helper, no React deps. +function loadModule() { + const outfile = join(mkdtempSync(join(tmpdir(), 'modly-agentgrade-test-')), 'agentGrade.cjs') + const require = createRequire(import.meta.url) + const result = buildSync({ + entryPoints: [resolve('src/shared/components/ui/agentGrade.ts')], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + }) + writeFileSync(outfile, result.outputFiles[0].text, 'utf8') + return require(outfile) +} + +const { agentGrade } = loadModule() + +test('a model without a tier gets no badge at all', () => { + assert.equal(agentGrade(undefined), null) + assert.equal(agentGrade({}), null) + assert.equal(agentGrade({ agent_tier: 'brilliant' }), null) // unknown tier, not a badge +}) + +test('a measured model shows its own score', () => { + const g = agentGrade({ agent_tier: 'excellent', agent_score: 0.98, agent_source: 'measured' }) + assert.equal(g.label, 'Agent: excellent (98%)') + assert.match(g.title, /Modly's tool-calling suite/) +}) + +test('an unmeasured model never borrows the credibility of a measurement', () => { + const g = agentGrade({ agent_tier: 'solid', agent_score: 0.9, agent_source: 'estimate' }) + assert.equal(g.label, 'Agent: solid') // no percentage + assert.match(g.title, /not measured in Modly/) +}) + +test('the note is carried into the tooltip', () => { + const g = agentGrade({ agent_tier: 'limited', agent_source: 'estimate', agent_note: 'older generation' }) + assert.match(g.title, /older generation/) + assert.equal(g.label, 'Agent: limited') +}) + +test('each tier gets its own colour', () => { + const classes = ['excellent', 'solid', 'limited'].map((t) => agentGrade({ agent_tier: t }).className) + assert.equal(new Set(classes).size, 3) +}) diff --git a/src/shared/components/ui/agentGrade.ts b/src/shared/components/ui/agentGrade.ts new file mode 100644 index 00000000..874b103d --- /dev/null +++ b/src/shared/components/ui/agentGrade.ts @@ -0,0 +1,44 @@ +/** + * How well a model drives the agent — the one thing the model list never said. + * + * Size and VRAM are already shown, and both are poor proxies: a 4B tops the + * tool-calling tests while a 20B sits below it. `agent_tier` comes from the + * catalog; when Modly's own eval suite has been run against the model, the + * measured pass rate is shown with it, and everything else is flagged as an + * estimate so the two are never confused. + */ + +export type AgentTier = 'excellent' | 'solid' | 'limited' + +export interface AgentGradeInput { + agent_tier?: string | null + agent_score?: number | null // 0..1, Modly's eval suite + agent_note?: string | null + agent_source?: string | null // 'measured' | 'estimate' +} + +export interface AgentGrade { + label: string + className: string + title: string +} + +const TIERS: Record = { + excellent: { label: 'Agent: excellent', className: 'border-emerald-500/30 bg-emerald-500/10 text-emerald-400' }, + solid: { label: 'Agent: solid', className: 'border-amber-500/30 bg-amber-500/10 text-amber-400' }, + limited: { label: 'Agent: limited', className: 'border-zinc-600/40 bg-zinc-600/10 text-zinc-400' }, +} + +export function agentGrade(m: AgentGradeInput | null | undefined): AgentGrade | null { + const tier = m?.agent_tier as AgentTier | undefined + if (!tier || !(tier in TIERS)) return null + const { label, className } = TIERS[tier] + + const measured = m?.agent_source === 'measured' && typeof m?.agent_score === 'number' + const score = measured ? `${Math.round((m!.agent_score as number) * 100)}% on Modly's tool-calling suite` : null + const title = [score ?? 'Estimated from published benchmarks — not measured in Modly', m?.agent_note] + .filter(Boolean) + .join(' · ') + + return { label: measured ? `${label} (${Math.round((m!.agent_score as number) * 100)}%)` : label, className, title } +} diff --git a/src/shared/components/ui/vramFit.test.mjs b/src/shared/components/ui/vramFit.test.mjs new file mode 100644 index 00000000..551fd69d --- /dev/null +++ b/src/shared/components/ui/vramFit.test.mjs @@ -0,0 +1,48 @@ +import test from 'node:test' +import assert from 'node:assert/strict' +import { buildSync } from 'esbuild' +import { createRequire } from 'node:module' +import { mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +// Bundle the pure vramFit helper (no React deps) into CJS — same approach as +// autoWire.test.mjs / preflight.test.mjs. +function loadModule() { + const outfile = join(mkdtempSync(join(tmpdir(), 'modly-vramfit-test-')), 'vramFit.cjs') + const require = createRequire(import.meta.url) + const result = buildSync({ + entryPoints: [resolve('src/shared/components/ui/vramFit.ts')], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + }) + writeFileSync(outfile, result.outputFiles[0].text, 'utf8') + return require(outfile) +} + +const { vramFit } = loadModule() + +test('returns null when the estimate or VRAM is unknown', () => { + assert.equal(vramFit(undefined, 12), null) + assert.equal(vramFit(4200, undefined), null) + assert.equal(vramFit(4200, null), null) + assert.equal(vramFit(0, 12), null) +}) + +test('"Fits" when the model sits at or under 90% of VRAM', () => { + // 12 GB → 12288 MB; 90% = 11059 MB + assert.equal(vramFit(4200, 12).label, 'Fits') + assert.equal(vramFit(11059, 12).label, 'Fits') +}) + +test('"Tight" between 90% and 100% of VRAM', () => { + assert.equal(vramFit(11060, 12).label, 'Tight') + assert.equal(vramFit(12288, 12).label, 'Tight') +}) + +test('"Won\'t fit" above 100% of VRAM', () => { + assert.equal(vramFit(13500, 12).label, "Won't fit") + assert.equal(vramFit(12289, 12).label, "Won't fit") +}) diff --git a/src/shared/components/ui/vramFit.ts b/src/shared/components/ui/vramFit.ts new file mode 100644 index 00000000..1f5c9bc0 --- /dev/null +++ b/src/shared/components/ui/vramFit.ts @@ -0,0 +1,20 @@ +/** + * "Does this model fit in VRAM?" verdict, computed from the model's estimated + * footprint against the detected VRAM. Only meaningful on GPUs we can measure + * (NVIDIA via nvidia-smi); returns null otherwise so no misleading badge shows + * on Vulkan/CPU where `vramGb` is unknown. + */ +export interface VramFit { + label: string + className: string +} + +export function vramFit(estimateMb?: number, vramGb?: number | null): VramFit | null { + if (!estimateMb || !vramGb) return null + const vramMb = vramGb * 1024 + if (estimateMb <= vramMb * 0.9) + return { label: 'Fits', className: 'border-emerald-500/30 bg-emerald-500/10 text-emerald-400' } + if (estimateMb <= vramMb) + return { label: 'Tight', className: 'border-amber-500/30 bg-amber-500/10 text-amber-400' } + return { label: "Won't fit", className: 'border-red-500/30 bg-red-500/10 text-red-400' } +} diff --git a/src/shared/services/llmDownloads.ts b/src/shared/services/llmDownloads.ts new file mode 100644 index 00000000..18de4412 --- /dev/null +++ b/src/shared/services/llmDownloads.ts @@ -0,0 +1,104 @@ +import { create } from 'zustand' +import { useAppStore } from '@shared/stores/appStore' + +// GGUF model downloads must survive the Model Library modal being closed and +// reopened — the backend already keeps downloading in the background once +// started (see api/routers/llm.py), it just needs a watcher that isn't tied to +// a component's lifecycle. This module-level store owns that watch, the same +// way agentChat.ts owns the chat's SSE stream outside React. + +export interface SseEvent { + percent?: number + status?: string + bytesDownloaded?: number + totalBytes?: number + error?: string + cancelled?: boolean + paused?: boolean +} + +/** Reads one SSE `data: {...}` frame at a time and calls `onEvent` for each. */ +export async function consumeSse(url: string, onEvent: (data: SseEvent) => void, signal?: AbortSignal): Promise { + const res = await fetch(url, { signal }) + if (!res.ok || !res.body) throw new Error(`HTTP ${res.status}`) + const reader = res.body.getReader() + const decoder = new TextDecoder() + let buf = '' + outer: for (;;) { + const { done, value } = await reader.read() + if (done) break + buf += decoder.decode(value, { stream: true }) + const parts = buf.split('\n\n') + buf = parts.pop() ?? '' + for (const part of parts) { + const line = part.split('\n').find((l) => l.startsWith('data: ')) + if (!line) continue + let data: SseEvent + try { data = JSON.parse(line.slice(6)) } catch { continue /* malformed frame */ } + onEvent(data) + // The stream stays open after an error frame — stop reading instead of + // spinning on a download the server has already given up on. + if (data.error) { void reader.cancel().catch(() => {}); break outer } + } + } +} + +interface LlmDownloadsStore { + downloads: Record + error: string | null + start: (modelId: string) => void + pause: (modelId: string) => Promise + cancel: (modelId: string) => Promise + dismissError: () => void +} + +// Model ids this renderer is currently watching. The backend is reconnect-safe +// (a GET while a download is already in flight attaches instead of restarting +// it), this just avoids opening a second redundant connection from this same +// store if start() is called twice in a row (e.g. StrictMode double-invoke). +const _watching = new Set() + +export const useLlmDownloadsStore = create((set) => ({ + downloads: {}, + error: null, + + start(modelId) { + if (_watching.has(modelId)) return + _watching.add(modelId) + const apiUrl = useAppStore.getState().apiUrl + + // Resuming from a paused entry: reset it to a live "connecting" state so the + // Resume button flips back to Pause immediately, before the first SSE frame. + set((s) => ({ downloads: { ...s.downloads, [modelId]: { percent: s.downloads[modelId]?.percent ?? 0, status: 'Starting…' } } })) + + let last: SseEvent | undefined + void consumeSse(`${apiUrl}/llm/download?model_id=${encodeURIComponent(modelId)}`, (e) => { + if (e.error) { set({ error: e.error }); return } + last = e + set((s) => ({ downloads: { ...s.downloads, [modelId]: e } })) + }) + .catch((e) => set({ error: e instanceof Error ? e.message : String(e) })) + .finally(() => { + _watching.delete(modelId) + // Keep the paused entry so a Resume button stays visible; the .part file + // is preserved server-side and start() resumes it. Any other terminal + // (done/cancelled/error) drops the row. + set((s) => ({ downloads: { ...s.downloads, [modelId]: last?.paused ? last : undefined } })) + }) + }, + + async pause(modelId) { + const apiUrl = useAppStore.getState().apiUrl + await fetch(`${apiUrl}/llm/download/pause?model_id=${encodeURIComponent(modelId)}`, { method: 'POST' }).catch(() => {}) + }, + + async cancel(modelId) { + const apiUrl = useAppStore.getState().apiUrl + await fetch(`${apiUrl}/llm/download/cancel?model_id=${encodeURIComponent(modelId)}`, { method: 'POST' }).catch(() => {}) + // A paused download has no live watcher to hit its .finally cleanup — drop + // the row here so Cancel visibly clears it. + set((s) => ({ downloads: { ...s.downloads, [modelId]: undefined } })) + }, + + dismissError: () => set({ error: null }), +})) diff --git a/src/shared/stores/agentStore.ts b/src/shared/stores/agentStore.ts index f6395c09..23b2cfc0 100644 --- a/src/shared/stores/agentStore.ts +++ b/src/shared/stores/agentStore.ts @@ -1,29 +1,155 @@ import { create } from 'zustand' -import { persist } from 'zustand/middleware' +import { persist, createJSONStorage } from 'zustand/middleware' export type ThinkingMode = 'auto' | 'on' | 'off' +export type ProviderId = 'local' | 'ollama' | 'openai' | 'anthropic' | 'mistral' | 'groq' | 'openrouter' | 'custom' + +export interface ExternalConfig { + apiKey: string + model: string + baseUrl?: string // only used by 'custom' +} + +export const PROVIDERS: Record = { + local: { label: 'Local (llama.cpp)', baseUrl: '' }, + // Ollama serves an OpenAI-compatible API — reuses models already pulled with it. + ollama: { label: 'Ollama', baseUrl: 'http://localhost:11434/v1', noKey: true }, + openai: { label: 'ChatGPT (OpenAI)', baseUrl: 'https://api.openai.com/v1' }, + anthropic: { label: 'Claude (Anthropic)', baseUrl: 'https://api.anthropic.com/v1' }, + mistral: { label: 'Mistral', baseUrl: 'https://api.mistral.ai/v1' }, + groq: { label: 'Groq', baseUrl: 'https://api.groq.com/openai/v1' }, + openrouter: { label: 'OpenRouter', baseUrl: 'https://openrouter.ai/api/v1' }, + custom: { label: 'Custom endpoint', baseUrl: '' }, +} + +export const DEFAULT_LOCAL_MODEL = 'qwen3-4b' + +// Catalog ids removed in the Qwen3/gpt-oss refresh → their closest replacement +const RETIRED_LOCAL_MODELS: Record = { + 'qwen2.5-3b': 'qwen3-4b', + 'qwen2.5-7b': 'qwen3-4b', + 'qwen2.5-14b': 'qwen3-14b', + 'llama-3.1-8b': 'qwen3-4b', + 'deepseek-r1-distill-qwen-7b': 'qwen3-14b', +} + +/** Resolve the base URL for a provider (custom uses its own field). */ +export function providerBaseUrl(provider: ProviderId, external: ExternalConfig | undefined): string { + if (provider === 'custom') return external?.baseUrl ?? '' + return PROVIDERS[provider].baseUrl +} + interface AgentSettings { - ollamaUrl: string - defaultModel: string - defaultThinking: ThinkingMode + provider: ProviderId + localModel: string // catalog id or custom: + external: Partial> + defaultThinking: ThinkingMode + + setProvider: (provider: ProviderId) => void + setLocalModel: (model: string) => void + setExternal: (provider: ProviderId, cfg: ExternalConfig) => void + setDefaultThinking: (mode: ThinkingMode) => void +} + +// ─── Secure persistence ──────────────────────────────────────────────────────── +// External provider API keys are the one sensitive field in this store. They're +// encrypted at rest via Electron's safeStorage (OS keychain/DPAPI/libsecret) — +// everything else (provider, localModel, defaultThinking) stays plain, it isn't +// a secret. The ciphertext itself still lives in localStorage as a hex string; +// only the plaintext key never touches disk unencrypted. + +type PersistedExternal = Partial> + +function hasSecureStore(): boolean { + return typeof window !== 'undefined' && !!window.electron?.secureStore +} + +async function transformApiKeys( + external: PersistedExternal | undefined, + transform: (key: string) => Promise, +): Promise { + if (!external || !hasSecureStore()) return external + const entries = await Promise.all( + Object.entries(external).map(async ([provider, cfg]) => { + if (!cfg?.apiKey) return [provider, cfg] as const + const key = await transform(cfg.apiKey) + // null = stored blob we can't decrypt here (different OS user/machine). + // Drop it: sending a ciphertext as an Authorization header just 401s, and + // keeping it in state would re-encrypt it on the next write. + return [provider, { ...cfg, apiKey: key ?? '' }] as const + }), + ) + return Object.fromEntries(entries) as PersistedExternal +} - setOllamaUrl: (url: string) => void - setDefaultModel: (model: string) => void - setDefaultThinking: (mode: ThinkingMode) => void +const secureAgentStorage = { + getItem: async (name: string): Promise => { + const raw = localStorage.getItem(name) + if (!raw) return raw + try { + const envelope = JSON.parse(raw) + if (envelope?.state?.external) { + envelope.state.external = await transformApiKeys(envelope.state.external, (k) => window.electron.secureStore.decrypt(k)) + } + return JSON.stringify(envelope) + } catch { + return raw + } + }, + setItem: async (name: string, value: string): Promise => { + try { + const envelope = JSON.parse(value) + if (envelope?.state?.external) { + envelope.state.external = await transformApiKeys(envelope.state.external, (k) => window.electron.secureStore.encrypt(k)) + } + localStorage.setItem(name, JSON.stringify(envelope)) + } catch { + localStorage.setItem(name, value) + } + }, + removeItem: async (name: string): Promise => { localStorage.removeItem(name) }, } export const useAgentStore = create()( persist( (set) => ({ - ollamaUrl: 'http://localhost:11434', - defaultModel: 'gemma4:e4b', + provider: 'local', + localModel: DEFAULT_LOCAL_MODEL, + external: {}, defaultThinking: 'auto', - setOllamaUrl: (url) => set({ ollamaUrl: url }), - setDefaultModel: (model) => set({ defaultModel: model }), - setDefaultThinking: (mode) => set({ defaultThinking: mode }), + setProvider: (provider) => set({ provider }), + setLocalModel: (model) => set({ localModel: model }), + setExternal: (provider, cfg) => set((s) => ({ external: { ...s.external, [provider]: cfg } })), + setDefaultThinking: (mode) => set({ defaultThinking: mode }), }), - { name: 'modly-agent-settings' }, + { + name: 'modly-agent-settings', + version: 3, + storage: createJSONStorage(() => secureAgentStorage), + // v0 stored { ollamaUrl, defaultModel } — drop them, keep only thinking. + // v2 retired the Qwen2.5/Llama3.1 catalog ids. + // v3 is a shape-less bump: the version change alone forces one re-persist + // through secureAgentStorage.setItem, which encrypts any legacy + // plaintext API key saved before safeStorage existed. Self-limiting — + // once stored as v3 it never re-runs, so it costs one write, not one per boot. + migrate: (persisted: unknown, version) => { + if (version === 0 && persisted && typeof persisted === 'object') { + const old = persisted as { defaultThinking?: ThinkingMode } + return { + provider: 'local' as ProviderId, + localModel: DEFAULT_LOCAL_MODEL, + external: {}, + defaultThinking: old.defaultThinking ?? 'auto', + } + } + const state = persisted as AgentSettings + if (version < 2 && state?.localModel && RETIRED_LOCAL_MODELS[state.localModel]) { + return { ...state, localModel: RETIRED_LOCAL_MODELS[state.localModel] } + } + return state + }, + }, ), ) diff --git a/src/shared/stores/llmModelsStore.react.test.mjs b/src/shared/stores/llmModelsStore.react.test.mjs new file mode 100644 index 00000000..b8c305f9 --- /dev/null +++ b/src/shared/stores/llmModelsStore.react.test.mjs @@ -0,0 +1,82 @@ +import test from 'node:test' +import assert from 'node:assert/strict' +import { createRequire } from 'node:module' +import { setupDom, loadModule, mount } from '../../../scripts/react-test-env.mjs' + +setupDom() +const React = createRequire(import.meta.url)('react') + +/** A fresh module per test: the catalog store caches across calls by design, so + * a shared instance would make each test depend on the ones before it. */ +const freshHook = () => loadModule('src/shared/stores/llmModelsStore.ts').useLlmModels + +/** Counts every request the hook makes, and answers with a fixed catalog. */ +function countingFetch() { + const calls = [] + globalThis.fetch = (url) => { + calls.push(String(url)) + return Promise.resolve({ json: () => Promise.resolve({ models: [{ id: 'qwen3-4b', downloaded: true }] }) }) + } + return calls +} + +test('the hook fetches the catalog once, however many times it renders', async () => { + const useLlmModels = freshHook() + const calls = countingFetch() + const Probe = () => { useLlmModels(); return null } + + const view = await mount(React.createElement(Probe)) + await view.rerender() + await view.rerender() + await view.flush() + + assert.equal(calls.length, 1) + await view.unmount() +}) + +test('refresh and the model list keep their identity across renders', async () => { + // The regression: `refresh` was rebuilt on every render, so any caller + // holding it in a dependency array re-ran its effect forever. + const useLlmModels = freshHook() + countingFetch() + const seen = [] + const Probe = () => { seen.push(useLlmModels()); return null } + + const view = await mount(React.createElement(Probe)) + await view.flush() + await view.rerender() + + const last = seen[seen.length - 1] + const previous = seen[seen.length - 2] + assert.equal(last.refresh, previous.refresh) + assert.equal(last.models, previous.models) + await view.unmount() +}) + +test('a consumer that refreshes from an effect settles instead of looping', async () => { + // Exactly the shape of ModelLibraryModal: an effect keyed on `refresh` that + // stores something in state. With an unstable `refresh` this rendered — and + // fetched — without end; the modal flickered for as long as it was open. + const useLlmModels = freshHook() + const calls = countingFetch() + let renders = 0 + + const Modal = () => { + const { refresh } = useLlmModels() + const [, setStatus] = React.useState(null) + renders++ + // Fails fast and says why: an unstable `refresh` makes this loop forever, + // and a test that hangs for a minute before dying explains nothing. + if (renders > 50) throw new Error('render loop — the effect keeps re-running') + React.useEffect(() => { void refresh(); setStatus({ checkedAt: renders }) }, [refresh]) + return null + } + + const view = await mount(React.createElement(Modal)) + await view.flush() + await view.flush() + + assert.equal(calls.length, 2) // the hook's own load, plus one forced refresh + assert.ok(renders < 10, `expected a handful of renders, got ${renders}`) + await view.unmount() +}) diff --git a/src/shared/stores/llmModelsStore.test.mjs b/src/shared/stores/llmModelsStore.test.mjs new file mode 100644 index 00000000..372d42ba --- /dev/null +++ b/src/shared/stores/llmModelsStore.test.mjs @@ -0,0 +1,106 @@ +import test from 'node:test' +import assert from 'node:assert/strict' +import { buildSync } from 'esbuild' +import { createRequire } from 'node:module' +import { mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' + +// Bundle the store with its real zustand dependency — same approach as +// workflowsStore.test.mjs. Only fetchModels() is exercised, so the React hook +// at the bottom of the module is never rendered. +function loadStore() { + const outfile = join(mkdtempSync(join(tmpdir(), 'modly-llmstore-test-')), 'llmModelsStore.cjs') + const require = createRequire(import.meta.url) + const result = buildSync({ + entryPoints: [resolve('src/shared/stores/llmModelsStore.ts')], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + }) + writeFileSync(outfile, result.outputFiles[0].text, 'utf8') + return require(outfile).useLlmModelsStore +} + +/** A /llm/models endpoint whose answers are released one at a time. */ +function deferredFetch() { + const calls = [] + globalThis.fetch = () => { + let release + const body = new Promise((r) => { release = r }) + calls.push({ release }) + return Promise.resolve({ json: () => body }) + } + return calls +} + +const model = (id, downloaded) => ({ + id, name: id, hf_filename: `${id}.gguf`, downloaded, source: 'catalog', +}) + +test('concurrent plain fetches share one request', async () => { + const useStore = loadStore() + const calls = deferredFetch() + + const a = useStore.getState().fetchModels('http://api') + const b = useStore.getState().fetchModels('http://api') + assert.equal(calls.length, 1) + + calls[0].release({ models: [model('qwen', false)] }) + await Promise.all([a, b]) + assert.equal(useStore.getState().models[0].downloaded, false) +}) + +test('a forced refresh re-fetches instead of joining the pending request', async () => { + const useStore = loadStore() + const calls = deferredFetch() + + // Mount-time fetch, still in flight when the download finishes. + const initial = useStore.getState().fetchModels('http://api') + const refreshed = useStore.getState().fetchModels('http://api', { force: true }) + + calls[0].release({ models: [model('qwen', false)] }) + await initial + assert.equal(calls.length, 2, 'force must issue its own request') + + calls[1].release({ models: [model('qwen', true)] }) + await refreshed + // Joining the pre-download request left `downloaded: false`, and the preflight + // kept refusing to run the workflow. + assert.equal(useStore.getState().models[0].downloaded, true) +}) + +test('a forced refresh still settles when the pending request fails', async () => { + const useStore = loadStore() + const calls = [] + globalThis.fetch = () => { + let settle + const p = new Promise((resolve, reject) => { settle = { resolve, reject } }) + calls.push(settle) + return p + } + + const initial = useStore.getState().fetchModels('http://api') + const refreshed = useStore.getState().fetchModels('http://api', { force: true }) + + calls[0].reject(new Error('offline')) + await initial + calls[1].resolve({ json: async () => ({ models: [model('qwen', true)] }) }) + await refreshed + + assert.equal(useStore.getState().models[0].downloaded, true) + assert.equal(useStore.getState().error, null) +}) + +test('a cached catalog is not re-fetched without force', async () => { + const useStore = loadStore() + const calls = deferredFetch() + + const first = useStore.getState().fetchModels('http://api') + calls[0].release({ models: [model('qwen', true)] }) + await first + + await useStore.getState().fetchModels('http://api') + assert.equal(calls.length, 1) +}) diff --git a/src/shared/stores/llmModelsStore.ts b/src/shared/stores/llmModelsStore.ts new file mode 100644 index 00000000..89da3ed1 --- /dev/null +++ b/src/shared/stores/llmModelsStore.ts @@ -0,0 +1,112 @@ +import { create } from 'zustand' +import { useCallback, useEffect, useMemo } from 'react' +import { useAppStore } from './appStore' + +export interface LlmModel { + id: string + name: string + description?: string + hf_filename: string + size_bytes?: number + quant?: string + vram_estimate_mb?: number + downloaded: boolean + source: 'catalog' | 'custom' + tags?: string[] + /** How well this model drives the agent — see components/ui/agentGrade.ts. + * `agent_score` is only shown when `agent_source` is 'measured', i.e. the + * model was actually run against Modly's own eval suite. */ + agent_tier?: 'excellent' | 'solid' | 'limited' + agent_score?: number + agent_source?: 'measured' | 'estimate' + agent_note?: string +} + +interface LlmModelsStore { + models: LlmModel[] + loading: boolean + error: string | null + fetchedApiUrl: string | null + fetchModels: (apiUrl: string, opts?: { force?: boolean }) => Promise +} + +// Module-level so concurrent callers (multiple nodes/components mounting at +// once) share one in-flight request instead of firing N identical fetches. +let inFlight: Promise | null = null +// Identifies the request currently owning `inFlight`, so a superseded one does +// not clear a newer forced refresh on its way out. +let inFlightId = 0 + +export const useLlmModelsStore = create((set, get) => ({ + models: [], + loading: false, + error: null, + fetchedApiUrl: null, + + async fetchModels(apiUrl, opts) { + const state = get() + if (!opts?.force && state.fetchedApiUrl === apiUrl && state.models.length > 0) return + // A forced refresh follows a mutation (download finished, model deleted), so + // it must not settle for a request issued BEFORE it: joining the in-flight + // one kept the pre-mutation `downloaded` flags, and the preflight went on + // reporting "…isn't downloaded" for a model that had just landed. + const pending = inFlight + if (pending && !opts?.force) return pending + + set({ loading: true, error: null }) + const id = ++inFlightId + const request = (async () => { + if (pending) { + await pending.catch(() => {}) + set({ loading: true, error: null }) // the request we waited on may have failed + } + try { + const res = await fetch(`${apiUrl}/llm/models`) + const data: { models?: LlmModel[] } = await res.json() + set({ models: data.models ?? [], fetchedApiUrl: apiUrl, loading: false, error: null }) + } catch (e) { + set({ error: e instanceof Error ? e.message : String(e), loading: false }) + } finally { + if (inFlightId === id) inFlight = null + } + })() + inFlight = request + return request + }, +})) + +/** + * Shared local-LLM catalog: fetched once per apiUrl and cached across every + * consumer (LLM node, extension param pickers, chat model picker, Settings…) + * instead of each component firing its own `/llm/models` request. + * + * `tag` mirrors the backend's own filter (`GET /llm/models?tag=`): custom + * GGUFs are always kept since their capabilities aren't known ahead of time. + */ +export function useLlmModels(tag?: string): { + models: LlmModel[] + loading: boolean + error: string | null + refresh: () => Promise +} { + const apiUrl = useAppStore((s) => s.apiUrl) + const models = useLlmModelsStore((s) => s.models) + const loading = useLlmModelsStore((s) => s.loading) + const error = useLlmModelsStore((s) => s.error) + const fetchModels = useLlmModelsStore((s) => s.fetchModels) + + useEffect(() => { void fetchModels(apiUrl) }, [apiUrl, fetchModels]) + + // Both memoised because callers put them in dependency arrays. A `refresh` + // rebuilt on every render made ModelLibraryModal's `useEffect(…, [refresh])` + // re-run on every render: each pass forced a /llm/models + /llm/status fetch, + // whose setState triggered the next one. The modal flickered and hammered the + // API for as long as it stayed open. + const filtered = useMemo( + () => (tag ? models.filter((m) => m.source === 'custom' || (m.tags ?? []).includes(tag)) : models), + [models, tag], + ) + const refresh = useCallback(() => fetchModels(apiUrl, { force: true }), [apiUrl, fetchModels]) + + return { models: filtered, loading, error, refresh } +} diff --git a/src/shared/types/electron.d.ts b/src/shared/types/electron.d.ts index 9ed0806e..8bce8426 100644 --- a/src/shared/types/electron.d.ts +++ b/src/shared/types/electron.d.ts @@ -209,6 +209,11 @@ declare global { get: () => Promise<{ modelsDir: string; workspaceDir: string; workflowsDir: string; extensionsDir: string; hfToken?: string }> set: (patch: { modelsDir?: string; workspaceDir?: string; workflowsDir?: string; extensionsDir?: string; hfToken?: string }) => Promise<{ modelsDir: string; workspaceDir: string; workflowsDir: string; extensionsDir: string; hfToken?: string }> } + /** decrypt returns null when the stored blob can't be decrypted here. */ + secureStore: { + encrypt: (plainText: string) => Promise + decrypt: (stored: string) => Promise + } cache: { clear: () => Promise<{ success: boolean; error?: string }> } From 9123d8b91919891efa3692cc37d446d509a8e722 Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 17:25:11 +0200 Subject: [PATCH 2/7] chore(deps): add jsdom for the React hook tests scripts/react-test-env.mjs imports jsdom, which no package declared, so llmModelsStore.react.test.mjs failed on any fresh install and npm test (and with it the pre-push hook) failed for everyone. --- package-lock.json | 495 +++++++++++++++++++++++++++++++++++++++++++++- package.json | 1 + 2 files changed, 495 insertions(+), 1 deletion(-) diff --git a/package-lock.json b/package-lock.json index 7c354c2b..d7b522f6 100644 --- a/package-lock.json +++ b/package-lock.json @@ -40,6 +40,7 @@ "eslint": "^9.17.0", "eslint-plugin-react-hooks": "^7.1.1", "globals": "^17.6.0", + "jsdom": "^29.1.1", "postcss": "^8.4.49", "tailwindcss": "^3.4.17", "typescript": "^5.7.2", @@ -59,6 +60,57 @@ "url": "https://github.com/sponsors/sindresorhus" } }, + "node_modules/@asamuzakjp/css-color": { + "version": "5.1.11", + "resolved": "https://registry.npmjs.org/@asamuzakjp/css-color/-/css-color-5.1.11.tgz", + "integrity": "sha512-KVw6qIiCTUQhByfTd78h2yD1/00waTmm9uy/R7Ck/ctUyAPj+AEDLkQIdJW0T8+qGgj3j5bpNKK7Q3G+LedJWg==", + "dev": true, + "license": "MIT", + "dependencies": { + "@asamuzakjp/generational-cache": "^1.0.1", + "@csstools/css-calc": "^3.2.0", + "@csstools/css-color-parser": "^4.1.0", + "@csstools/css-parser-algorithms": "^4.0.0", + "@csstools/css-tokenizer": "^4.0.0" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, + "node_modules/@asamuzakjp/dom-selector": { + "version": "7.1.1", + "resolved": "https://registry.npmjs.org/@asamuzakjp/dom-selector/-/dom-selector-7.1.1.tgz", + "integrity": "sha512-67RZDnYRc8H/8MLDgQCDE//zoqVFwajkepHZgmXrbwybzXOEwOWGPYGmALYl9J2DOLfFPPs6kKCqmbzV895hTQ==", + "dev": true, + "license": "MIT", + "dependencies": { + "@asamuzakjp/generational-cache": "^1.0.1", + "@asamuzakjp/nwsapi": "^2.3.9", + "bidi-js": "^1.0.3", + "css-tree": "^3.2.1", + "is-potential-custom-element-name": "^1.0.1" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, + "node_modules/@asamuzakjp/generational-cache": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/@asamuzakjp/generational-cache/-/generational-cache-1.0.1.tgz", + "integrity": "sha512-wajfB8KqzMCN2KGNFdLkReeHncd0AslUSrvHVvvYWuU8ghncRJoA50kT3zP9MVL0+9g4/67H+cdvBskj9THPzg==", + "dev": true, + "license": "MIT", + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, + "node_modules/@asamuzakjp/nwsapi": { + "version": "2.3.9", + "resolved": "https://registry.npmjs.org/@asamuzakjp/nwsapi/-/nwsapi-2.3.9.tgz", + "integrity": "sha512-n8GuYSrI9bF7FFZ/SjhwevlHc8xaVlb/7HmHelnc/PZXBD2ZR49NnN9sMMuDdEGPeeRQ5d0hqlSlEpgCX3Wl0Q==", + "dev": true, + "license": "MIT" + }, "node_modules/@babel/code-frame": { "version": "7.29.7", "resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.29.7.tgz", @@ -361,6 +413,159 @@ "node": ">=6.9.0" } }, + "node_modules/@bramus/specificity": { + "version": "2.4.2", + "resolved": "https://registry.npmjs.org/@bramus/specificity/-/specificity-2.4.2.tgz", + "integrity": "sha512-ctxtJ/eA+t+6q2++vj5j7FYX3nRu311q1wfYH3xjlLOsczhlhxAg2FWNUXhpGvAw3BWo1xBcvOV6/YLc2r5FJw==", + "dev": true, + "license": "MIT", + "dependencies": { + "css-tree": "^3.0.0" + }, + "bin": { + "specificity": "bin/cli.js" + } + }, + "node_modules/@csstools/color-helpers": { + "version": "6.1.2", + "resolved": "https://registry.npmjs.org/@csstools/color-helpers/-/color-helpers-6.1.2.tgz", + "integrity": "sha512-grhRy3OKmniaAEKXMjua5z/EODX0MSqBGjunw8+j/3HQjOnahs2AGhvEOIYVUWcU6ScApbhLhVrQTX8XqrMrow==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT-0", + "engines": { + "node": ">=20.19.0" + } + }, + "node_modules/@csstools/css-calc": { + "version": "3.4.3", + "resolved": "https://registry.npmjs.org/@csstools/css-calc/-/css-calc-3.4.3.tgz", + "integrity": "sha512-iex20d8CHVkyvg6B7UKV7uHnI2Bqo9g+EFfT9E0y+GvTvhZ/DwONJ+9aKb1dlqm0ZiGsL5RXjp0fCoJYnkeDjA==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT", + "engines": { + "node": ">=20.19.0" + }, + "peerDependencies": { + "@csstools/css-parser-algorithms": "^4.0.2", + "@csstools/css-tokenizer": "^4.0.2" + } + }, + "node_modules/@csstools/css-color-parser": { + "version": "4.2.6", + "resolved": "https://registry.npmjs.org/@csstools/css-color-parser/-/css-color-parser-4.2.6.tgz", + "integrity": "sha512-iiPQ3iRWwnJkeEn6RIu6SJPr7hYrLz6XZ9s/QZl+2/LI5KQVjpl2fdmDSZKuD4xP6GMmMPgHFFXg6k1Wkz0Trg==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT", + "dependencies": { + "@csstools/color-helpers": "^6.1.2", + "@csstools/css-calc": "^3.4.3" + }, + "engines": { + "node": ">=20.19.0" + }, + "peerDependencies": { + "@csstools/css-parser-algorithms": "^4.0.2", + "@csstools/css-tokenizer": "^4.0.2" + } + }, + "node_modules/@csstools/css-parser-algorithms": { + "version": "4.0.2", + "resolved": "https://registry.npmjs.org/@csstools/css-parser-algorithms/-/css-parser-algorithms-4.0.2.tgz", + "integrity": "sha512-40cSKyMvK+tq4qz6Awrlye2WGuOKt3FwPgtGg6KTfbHOWNw+Rk1rzbAtZnZ6IBhsY491HLRnDXwoyBAijmmILA==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT", + "engines": { + "node": ">=20.19.0" + }, + "peerDependencies": { + "@csstools/css-tokenizer": "^4.0.2" + } + }, + "node_modules/@csstools/css-syntax-patches-for-csstree": { + "version": "1.1.15", + "resolved": "https://registry.npmjs.org/@csstools/css-syntax-patches-for-csstree/-/css-syntax-patches-for-csstree-1.1.15.tgz", + "integrity": "sha512-J0u7HkVl2nzSlhsiTOp4AmwcUQ3D+mGEEKfBy/7To5/y7F2OHwyLrXfrhR0SMgr4p5Lo+eaMVSeai24zUcBIxA==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT-0", + "peerDependencies": { + "css-tree": "^3.2.1" + }, + "peerDependenciesMeta": { + "css-tree": { + "optional": true + } + } + }, + "node_modules/@csstools/css-tokenizer": { + "version": "4.0.2", + "resolved": "https://registry.npmjs.org/@csstools/css-tokenizer/-/css-tokenizer-4.0.2.tgz", + "integrity": "sha512-OoKoR0f76dCY666JlcbhmVTs2drYj1GUXZTYTcbUgJjh9Nv41aFfZ21bPQTERm5+L5cBDo466NltB2lplS5GBw==", + "dev": true, + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/csstools" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/csstools" + } + ], + "license": "MIT", + "engines": { + "node": ">=20.19.0" + } + }, "node_modules/@electron-internal/extract-zip": { "version": "1.0.5", "resolved": "https://registry.npmjs.org/@electron-internal/extract-zip/-/extract-zip-1.0.5.tgz", @@ -1288,6 +1493,24 @@ "node": "^18.18.0 || ^20.9.0 || >=21.1.0" } }, + "node_modules/@exodus/bytes": { + "version": "1.16.0", + "resolved": "https://registry.npmjs.org/@exodus/bytes/-/bytes-1.16.0.tgz", + "integrity": "sha512-IcpW84uEn3N7ETtNZMlxKhfl6Pec8rUNGOTBtWbK1FKhJxIFAptZyVrvVRVBimAJxJCgc3PxepxkdWWG4DVzfA==", + "dev": true, + "license": "MIT", + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + }, + "peerDependencies": { + "@noble/hashes": "^1.8.0 || ^2.0.0" + }, + "peerDependenciesMeta": { + "@noble/hashes": { + "optional": true + } + } + }, "node_modules/@humanfs/core": { "version": "0.19.1", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", @@ -3787,6 +4010,20 @@ "node": ">= 8" } }, + "node_modules/css-tree": { + "version": "3.2.1", + "resolved": "https://registry.npmjs.org/css-tree/-/css-tree-3.2.1.tgz", + "integrity": "sha512-X7sjQzceUhu1u7Y/ylrRZFU2FS6LRiFVp6rKLPg23y3x3c3DOKAwuXGDp+PAGjh6CSnCjYeAul8pcT8bAl+lSA==", + "dev": true, + "license": "MIT", + "dependencies": { + "mdn-data": "2.27.1", + "source-map-js": "^1.2.1" + }, + "engines": { + "node": "^10 || ^12.20.0 || ^14.13.0 || >=15.0.0" + } + }, "node_modules/cssesc": { "version": "3.0.0", "resolved": "https://registry.npmjs.org/cssesc/-/cssesc-3.0.0.tgz", @@ -3909,6 +4146,20 @@ "node": ">=12" } }, + "node_modules/data-urls": { + "version": "7.0.0", + "resolved": "https://registry.npmjs.org/data-urls/-/data-urls-7.0.0.tgz", + "integrity": "sha512-23XHcCF+coGYevirZceTVD7NdJOqVn+49IHyxgszm+JIiHLoB2TkmPtsYkNWT1pvRSGkc35L6NHs0yHkN2SumA==", + "dev": true, + "license": "MIT", + "dependencies": { + "whatwg-mimetype": "^5.0.0", + "whatwg-url": "^16.0.0" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, "node_modules/debug": { "version": "4.4.3", "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", @@ -3925,6 +4176,13 @@ } } }, + "node_modules/decimal.js": { + "version": "10.6.0", + "resolved": "https://registry.npmjs.org/decimal.js/-/decimal.js-10.6.0.tgz", + "integrity": "sha512-YpgQiITW3JXGntzdUmyUR1V812Hn8T1YVXhCu+wO3OpS4eU9l4YdD3qjyiKdV6mvV29zapkMeD390UVEf2lkUg==", + "dev": true, + "license": "MIT" + }, "node_modules/decompress-response": { "version": "6.0.0", "resolved": "https://registry.npmjs.org/decompress-response/-/decompress-response-6.0.0.tgz", @@ -4396,6 +4654,19 @@ "once": "^1.4.0" } }, + "node_modules/entities": { + "version": "8.1.0", + "resolved": "https://registry.npmjs.org/entities/-/entities-8.1.0.tgz", + "integrity": "sha512-kxL7msIffSuh9aaFAMD7rxAIuTRMAHMeBtgHW2yUdWw732ZNh4MehkF2gdjvtdmikkaIP9bFDDJOPlsvm7avrA==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">=20.19.0" + }, + "funding": { + "url": "https://github.com/fb55/entities?sponsor=1" + } + }, "node_modules/env-paths": { "version": "3.0.0", "resolved": "https://registry.npmjs.org/env-paths/-/env-paths-3.0.0.tgz", @@ -5352,6 +5623,19 @@ "dev": true, "license": "ISC" }, + "node_modules/html-encoding-sniffer": { + "version": "6.0.0", + "resolved": "https://registry.npmjs.org/html-encoding-sniffer/-/html-encoding-sniffer-6.0.0.tgz", + "integrity": "sha512-CV9TW3Y3f8/wT0BRFc1/KAVQ3TUHiXmaAb6VW9vtiMFf7SLoMd1PdAc4W3KFOFETBJUb90KatHqlsZMWV+R9Gg==", + "dev": true, + "license": "MIT", + "dependencies": { + "@exodus/bytes": "^1.6.0" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, "node_modules/http-cache-semantics": { "version": "4.2.0", "resolved": "https://registry.npmjs.org/http-cache-semantics/-/http-cache-semantics-4.2.0.tgz", @@ -5545,6 +5829,13 @@ "node": ">=0.12.0" } }, + "node_modules/is-potential-custom-element-name": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/is-potential-custom-element-name/-/is-potential-custom-element-name-1.0.1.tgz", + "integrity": "sha512-bCYeRA2rVibKZd+s2625gGnGF/t7DSqDs4dP7CrLA1m7jKWz6pps0LpYLJN8Q64HtmPKJ1hrN3nzPNKFEKOUiQ==", + "dev": true, + "license": "MIT" + }, "node_modules/is-promise": { "version": "2.2.2", "resolved": "https://registry.npmjs.org/is-promise/-/is-promise-2.2.2.tgz", @@ -5648,6 +5939,57 @@ "js-yaml": "bin/js-yaml.js" } }, + "node_modules/jsdom": { + "version": "29.1.1", + "resolved": "https://registry.npmjs.org/jsdom/-/jsdom-29.1.1.tgz", + "integrity": "sha512-ECi4Fi2f7BdJtUKTflYRTiaMxIB0O6zfR1fX0GXpUrf6flp8QIYn1UT20YQqdSOfk2dfkCwS8LAFoJDEppNK5Q==", + "dev": true, + "license": "MIT", + "dependencies": { + "@asamuzakjp/css-color": "^5.1.11", + "@asamuzakjp/dom-selector": "^7.1.1", + "@bramus/specificity": "^2.4.2", + "@csstools/css-syntax-patches-for-csstree": "^1.1.3", + "@exodus/bytes": "^1.15.0", + "css-tree": "^3.2.1", + "data-urls": "^7.0.0", + "decimal.js": "^10.6.0", + "html-encoding-sniffer": "^6.0.0", + "is-potential-custom-element-name": "^1.0.1", + "lru-cache": "^11.3.5", + "parse5": "^8.0.1", + "saxes": "^6.0.0", + "symbol-tree": "^3.2.4", + "tough-cookie": "^6.0.1", + "undici": "^7.25.0", + "w3c-xmlserializer": "^5.0.0", + "webidl-conversions": "^8.0.1", + "whatwg-mimetype": "^5.0.0", + "whatwg-url": "^16.0.1", + "xml-name-validator": "^5.0.0" + }, + "engines": { + "node": "^20.19.0 || ^22.13.0 || >=24.0.0" + }, + "peerDependencies": { + "canvas": "^3.0.0" + }, + "peerDependenciesMeta": { + "canvas": { + "optional": true + } + } + }, + "node_modules/jsdom/node_modules/lru-cache": { + "version": "11.5.3", + "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.5.3.tgz", + "integrity": "sha512-U4N8FgzmWxc8k1VH8Kr6lQg18U7Fjvby6wXHVRX/ZZ7IwWbRMgrRbP0Wrb5q5NVinryp4SQampHKdvtecItxUg==", + "dev": true, + "license": "BlueOak-1.0.0", + "engines": { + "node": "20 || >=22" + } + }, "node_modules/jsesc": { "version": "3.1.0", "resolved": "https://registry.npmjs.org/jsesc/-/jsesc-3.1.0.tgz", @@ -5877,6 +6219,13 @@ "node": ">= 0.4" } }, + "node_modules/mdn-data": { + "version": "2.27.1", + "resolved": "https://registry.npmjs.org/mdn-data/-/mdn-data-2.27.1.tgz", + "integrity": "sha512-9Yubnt3e8A0OKwxYSXyhLymGW4sCufcLG6VdiDdUGVkPhpqLxlvP5vl1983gQjJl3tqbrM731mjaZaP68AgosQ==", + "dev": true, + "license": "CC0-1.0" + }, "node_modules/merge2": { "version": "1.4.1", "resolved": "https://registry.npmjs.org/merge2/-/merge2-1.4.1.tgz", @@ -6336,6 +6685,19 @@ "node": ">=6" } }, + "node_modules/parse5": { + "version": "8.0.1", + "resolved": "https://registry.npmjs.org/parse5/-/parse5-8.0.1.tgz", + "integrity": "sha512-z1e/HMG90obSGeidlli3hj7cbocou0/wa5HacvI3ASx34PecNjNQeaHNo5WIZpWofN9kgkqV1q5YvXe3F0FoPw==", + "dev": true, + "license": "MIT", + "dependencies": { + "entities": "^8.0.0" + }, + "funding": { + "url": "https://github.com/inikulin/parse5?sponsor=1" + } + }, "node_modules/path-exists": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/path-exists/-/path-exists-4.0.0.tgz", @@ -7197,6 +7559,19 @@ "node": ">=11.0.0" } }, + "node_modules/saxes": { + "version": "6.0.0", + "resolved": "https://registry.npmjs.org/saxes/-/saxes-6.0.0.tgz", + "integrity": "sha512-xAg7SOnEhrm5zI3puOOKyy1OMcMlIJZYNJY7xLBwSze0UjhPLnWfj2GF2EpT0jmzaJKIWKHLsaSSajf35bcYnA==", + "dev": true, + "license": "ISC", + "dependencies": { + "xmlchars": "^2.2.0" + }, + "engines": { + "node": ">=v12.22.7" + } + }, "node_modules/scheduler": { "version": "0.21.0", "resolved": "https://registry.npmjs.org/scheduler/-/scheduler-0.21.0.tgz", @@ -7486,6 +7861,13 @@ "react": ">=17.0" } }, + "node_modules/symbol-tree": { + "version": "3.2.4", + "resolved": "https://registry.npmjs.org/symbol-tree/-/symbol-tree-3.2.4.tgz", + "integrity": "sha512-9QNk5KwDF+Bvz+PyObkmSYjI5ksVUYtjW7AU22r2NKcfLJcXp96hkDWU3+XndOsUb+AQ9QhfzfCT2O+CNWT5Tw==", + "dev": true, + "license": "MIT" + }, "node_modules/tailwindcss": { "version": "3.4.19", "resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-3.4.19.tgz", @@ -7728,6 +8110,26 @@ "url": "https://github.com/sponsors/jonschlinkert" } }, + "node_modules/tldts": { + "version": "7.4.16", + "resolved": "https://registry.npmjs.org/tldts/-/tldts-7.4.16.tgz", + "integrity": "sha512-QwBER5KMR86IIjpIiO7H/Z3IMJPsZ1A6RKPAqzTTgOyUQUSt9FdnKcqhTaJmkY6HVrgouZHZR0ncK5QxvmnQeg==", + "dev": true, + "license": "MIT", + "dependencies": { + "tldts-core": "^7.4.16" + }, + "bin": { + "tldts": "bin/cli.js" + } + }, + "node_modules/tldts-core": { + "version": "7.4.16", + "resolved": "https://registry.npmjs.org/tldts-core/-/tldts-core-7.4.16.tgz", + "integrity": "sha512-MDolfaSJtlSK5Y0A1xl3277ekubZwobpBjugknDizI9O5Rm60a1m8k4ICK+MRsCDzPygT81mp3BBf5RKDlFRfA==", + "dev": true, + "license": "MIT" + }, "node_modules/tmp": { "version": "0.2.7", "resolved": "https://registry.npmjs.org/tmp/-/tmp-0.2.7.tgz", @@ -7760,6 +8162,32 @@ "node": ">=8.0" } }, + "node_modules/tough-cookie": { + "version": "6.0.2", + "resolved": "https://registry.npmjs.org/tough-cookie/-/tough-cookie-6.0.2.tgz", + "integrity": "sha512-exgYmnmL/sJpR3upZfXG5PoatXQii55xAiXGXzY+sROLZ/Y+SLcp9PgJNI9Vz37HpQ74WvDcLT8eqm+kV3FzrA==", + "dev": true, + "license": "BSD-3-Clause", + "dependencies": { + "tldts": "^7.0.5" + }, + "engines": { + "node": ">=16" + } + }, + "node_modules/tr46": { + "version": "6.0.0", + "resolved": "https://registry.npmjs.org/tr46/-/tr46-6.0.0.tgz", + "integrity": "sha512-bLVMLPtstlZ4iMQHpFHTR7GAGj2jxi8Dg0s2h2MafAE4uSWF98FC/3MomU51iQAMf8/qDUbKWf5GxuvvVcXEhw==", + "dev": true, + "license": "MIT", + "dependencies": { + "punycode": "^2.3.1" + }, + "engines": { + "node": ">=20" + } + }, "node_modules/troika-three-text": { "version": "0.52.4", "resolved": "https://registry.npmjs.org/troika-three-text/-/troika-three-text-0.52.4.tgz", @@ -7925,8 +8353,8 @@ "version": "7.29.0", "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", "integrity": "sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==", + "devOptional": true, "license": "MIT", - "optional": true, "engines": { "node": ">=20.18.1" } @@ -8149,6 +8577,19 @@ "url": "https://github.com/sponsors/jonschlinkert" } }, + "node_modules/w3c-xmlserializer": { + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/w3c-xmlserializer/-/w3c-xmlserializer-5.0.0.tgz", + "integrity": "sha512-o8qghlI8NZHU1lLPrpi2+Uq7abh4GGPpYANlalzWxyWteJOCsr/P+oPBA49TOLu5FTZO4d3F9MnWJfiMo4BkmA==", + "dev": true, + "license": "MIT", + "dependencies": { + "xml-name-validator": "^5.0.0" + }, + "engines": { + "node": ">=18" + } + }, "node_modules/webcrypto-core": { "version": "1.9.2", "resolved": "https://registry.npmjs.org/webcrypto-core/-/webcrypto-core-1.9.2.tgz", @@ -8173,6 +8614,41 @@ "resolved": "https://registry.npmjs.org/webgl-sdf-generator/-/webgl-sdf-generator-1.1.1.tgz", "integrity": "sha512-9Z0JcMTFxeE+b2x1LJTdnaT8rT8aEp7MVxkNwoycNmJWwPdzoXzMh0BjJSh/AEFP+KPYZUli814h8bJZFIZ2jA==" }, + "node_modules/webidl-conversions": { + "version": "8.0.1", + "resolved": "https://registry.npmjs.org/webidl-conversions/-/webidl-conversions-8.0.1.tgz", + "integrity": "sha512-BMhLD/Sw+GbJC21C/UgyaZX41nPt8bUTg+jWyDeg7e7YN4xOM05YPSIXceACnXVtqyEw/LMClUQMtMZ+PGGpqQ==", + "dev": true, + "license": "BSD-2-Clause", + "engines": { + "node": ">=20" + } + }, + "node_modules/whatwg-mimetype": { + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/whatwg-mimetype/-/whatwg-mimetype-5.0.0.tgz", + "integrity": "sha512-sXcNcHOC51uPGF0P/D4NVtrkjSU2fNsm9iog4ZvZJsL3rjoDAzXZhkm2MWt1y+PUdggKAYVoMAIYcs78wJ51Cw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=20" + } + }, + "node_modules/whatwg-url": { + "version": "16.0.1", + "resolved": "https://registry.npmjs.org/whatwg-url/-/whatwg-url-16.0.1.tgz", + "integrity": "sha512-1to4zXBxmXHV3IiSSEInrreIlu02vUOvrhxJJH5vcxYTBDAx51cqZiKdyTxlecdKNSjj8EcxGBxNf6Vg+945gw==", + "dev": true, + "license": "MIT", + "dependencies": { + "@exodus/bytes": "^1.11.0", + "tr46": "^6.0.0", + "webidl-conversions": "^8.0.1" + }, + "engines": { + "node": "^20.19.0 || ^22.12.0 || >=24.0.0" + } + }, "node_modules/which": { "version": "2.0.2", "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", @@ -8221,6 +8697,16 @@ "dev": true, "license": "ISC" }, + "node_modules/xml-name-validator": { + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/xml-name-validator/-/xml-name-validator-5.0.0.tgz", + "integrity": "sha512-EvGK8EJ3DhaHfbRlETOWAS5pO9MZITeauHKJyb8wyajUfQUenkIg2MvLDTZ4T/TgIcm3HU0TFBgWWboAZ30UHg==", + "dev": true, + "license": "Apache-2.0", + "engines": { + "node": ">=18" + } + }, "node_modules/xmlbuilder": { "version": "15.1.1", "resolved": "https://registry.npmjs.org/xmlbuilder/-/xmlbuilder-15.1.1.tgz", @@ -8231,6 +8717,13 @@ "node": ">=8.0" } }, + "node_modules/xmlchars": { + "version": "2.2.0", + "resolved": "https://registry.npmjs.org/xmlchars/-/xmlchars-2.2.0.tgz", + "integrity": "sha512-JZnDKK8B0RCDw84FNdDAIpZK+JuJw+s7Lz8nksI7SIuU3UXJJslUthsi+uWBUYOwPFwW7W7PRLRfUKpxjtjFCw==", + "dev": true, + "license": "MIT" + }, "node_modules/y18n": { "version": "5.0.8", "resolved": "https://registry.npmjs.org/y18n/-/y18n-5.0.8.tgz", diff --git a/package.json b/package.json index 9ea8f639..a63727a2 100644 --- a/package.json +++ b/package.json @@ -51,6 +51,7 @@ "eslint": "^9.17.0", "eslint-plugin-react-hooks": "^7.1.1", "globals": "^17.6.0", + "jsdom": "^29.1.1", "postcss": "^8.4.49", "tailwindcss": "^3.4.17", "typescript": "^5.7.2", From 830f0b068fa20ebba50206af1f668c405206b8ce Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 17:25:11 +0200 Subject: [PATCH 3/7] feat(agent): keep the local LLM engine and models in an agent data folder The llama.cpp engine, GGUF models, logs and pool config lived in ~/.modly/llm on the system drive, apart from every other data folder the user picked at setup. They now go to an `agent` folder beside models/, extensions/, workflows/, workspace/ and dependencies/, passed to the API as MODLY_LLM_DIR. Setup writes /agent. Installs that predate it get the folder next to their models folder, created and pinned in settings.json at startup so moving the models folder later does not take the agent folder with it. --- api/services/llm_server.py | 10 ++- electron/main/ipc-handlers.ts | 1 + electron/main/python-bridge.ts | 8 +- electron/main/settings-store.test.mjs | 79 +++++++++++++++++++ electron/main/settings-store.ts | 44 +++++++++-- .../components/ui/ModelLibraryModal.tsx | 2 +- 6 files changed, 133 insertions(+), 11 deletions(-) create mode 100644 electron/main/settings-store.test.mjs diff --git a/api/services/llm_server.py b/api/services/llm_server.py index 0d4a7b23..213e40a1 100644 --- a/api/services/llm_server.py +++ b/api/services/llm_server.py @@ -1,9 +1,13 @@ """ Local LLM engine — manages a llama.cpp `llama-server` subprocess. -Everything lives under the per-user directory ~/.modly/llm/: - bin/ llama-server binary + DLLs (auto-downloaded from GitHub releases) - models/ GGUF files (catalog downloads + any custom .gguf the user drops in) +Everything lives under the agent directory (MODLY_LLM_DIR, set by Electron to +the `agent` folder beside models/, extensions/, … — ~/.modly/llm/ when the API +runs standalone): + bin/ llama-server binary + DLLs (auto-downloaded from GitHub releases) + models/ GGUF files (catalog downloads + any custom .gguf the user drops in) + logs/ one log per llama-server slot + config.json pool settings (max_models) Nothing is hardcoded to a machine: the binary variant is picked per-platform (CUDA if an NVIDIA driver is present, otherwise Vulkan, otherwise CPU) and diff --git a/electron/main/ipc-handlers.ts b/electron/main/ipc-handlers.ts index 352bae3f..01152735 100644 --- a/electron/main/ipc-handlers.ts +++ b/electron/main/ipc-handlers.ts @@ -314,6 +314,7 @@ export function setupIpcHandlers(pythonBridge: PythonBridge, getWindow: WindowGe workflowsDir: join(baseDir, 'workflows'), extensionsDir: join(baseDir, 'extensions'), dependenciesDir: join(baseDir, 'dependencies'), + agentDir: join(baseDir, 'agent'), }) }) diff --git a/electron/main/python-bridge.ts b/electron/main/python-bridge.ts index 82edb9ea..87aa2b6b 100644 --- a/electron/main/python-bridge.ts +++ b/electron/main/python-bridge.ts @@ -3,7 +3,7 @@ import { join } from 'path' import { app, BrowserWindow } from 'electron' import { existsSync, mkdirSync } from 'fs' import axios from 'axios' -import { getSettings } from './settings-store' +import { ensureAgentDir, getSettings } from './settings-store' import { getHfToken } from './hf-token' import { logger } from './logger' import { cleanPythonEnv, getVenvPythonExe } from './python-setup' @@ -65,6 +65,8 @@ export class PythonBridge { MODELS_DIR: this.resolveModelsDir(), WORKSPACE_DIR: this.resolveWorkspaceDir(), EXTENSIONS_DIR: this.resolveExtensionsDir(), + // llm_server.py keeps everything of the local LLM here: engine, GGUF models, logs, config. + MODLY_LLM_DIR: this.resolveAgentDir(), SELECTED_MODEL_ID: process.env['SELECTED_MODEL_ID'] ?? '', HUGGING_FACE_HUB_TOKEN: this.resolveHfToken(), HF_TOKEN: this.resolveHfToken(), @@ -246,6 +248,10 @@ export class PythonBridge { return s.extensionsDir } + private resolveAgentDir(): string { + return ensureAgentDir(app.getPath('userData')) + } + private resolveHfToken(): string { return getHfToken() // decrypted cache — settings.json holds the ciphertext } diff --git a/electron/main/settings-store.test.mjs b/electron/main/settings-store.test.mjs new file mode 100644 index 00000000..06a78fc1 --- /dev/null +++ b/electron/main/settings-store.test.mjs @@ -0,0 +1,79 @@ +/** + * The agent's folder (local LLM engine, GGUF models, logs) sits beside the + * other data folders. Installs set up before it existed have models/, + * extensions/, … under the base they picked at first run, so an unsaved + * agentDir must land next to those, not in userData. + */ +import test from 'node:test' +import assert from 'node:assert/strict' +import { buildSync } from 'esbuild' +import { createRequire } from 'node:module' +import { existsSync, mkdtempSync, readFileSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join, resolve } from 'node:path' +import vm from 'node:vm' + +function loadModule() { + const require = createRequire(import.meta.url) + const result = buildSync({ + entryPoints: [resolve('electron/main/settings-store.ts')], + bundle: true, + platform: 'node', + format: 'cjs', + write: false, + }) + const module = { exports: {} } + vm.runInNewContext(result.outputFiles[0].text, { module, exports: module.exports, require }) + return module.exports +} + +const { getSettings, setSettings, ensureAgentDir } = loadModule() + +test('a fresh install puts the agent folder in userData beside the others', () => { + const userData = mkdtempSync(join(tmpdir(), 'modly-settings-')) + const s = getSettings(userData) + assert.equal(s.agentDir, join(userData, 'agent')) + assert.equal(s.modelsDir, join(userData, 'models')) +}) + +test('an existing install gets the agent folder beside its chosen data folders', () => { + const userData = mkdtempSync(join(tmpdir(), 'modly-settings-')) + const base = join(userData, 'Documents', 'Modly') + writeFileSync(join(userData, 'settings.json'), JSON.stringify({ + modelsDir: join(base, 'models'), + workspaceDir: join(base, 'workspace'), + extensionsDir: join(base, 'extensions'), + })) + assert.equal(getSettings(userData).agentDir, join(base, 'agent')) +}) + +test('a saved agent folder is kept, and persists once any setting is written', () => { + const userData = mkdtempSync(join(tmpdir(), 'modly-settings-')) + const custom = join(userData, 'elsewhere', 'agent') + writeFileSync(join(userData, 'settings.json'), JSON.stringify({ agentDir: custom })) + assert.equal(getSettings(userData).agentDir, custom) + + setSettings(userData, { workflowsDir: join(userData, 'wf') }) + assert.equal(JSON.parse(readFileSync(join(userData, 'settings.json'), 'utf-8')).agentDir, custom) +}) + +test('an existing install gets its agent folder created and pinned at startup', () => { + const userData = mkdtempSync(join(tmpdir(), 'modly-settings-')) + const base = join(userData, 'Documents', 'Modly') + writeFileSync(join(userData, 'settings.json'), JSON.stringify({ modelsDir: join(base, 'models') })) + + assert.equal(ensureAgentDir(userData), join(base, 'agent')) + assert.equal(existsSync(join(base, 'agent')), true) + assert.equal(JSON.parse(readFileSync(join(userData, 'settings.json'), 'utf-8')).agentDir, join(base, 'agent')) + + // Pinned: moving the models folder afterwards leaves the agent folder where it is. + setSettings(userData, { modelsDir: join(userData, 'other-drive', 'models') }) + assert.equal(getSettings(userData).agentDir, join(base, 'agent')) +}) + +test('a fresh install gets its agent folder without settings.json being written', () => { + const userData = mkdtempSync(join(tmpdir(), 'modly-settings-')) + assert.equal(ensureAgentDir(userData), join(userData, 'agent')) + assert.equal(existsSync(join(userData, 'agent')), true) + assert.equal(existsSync(join(userData, 'settings.json')), false) +}) diff --git a/electron/main/settings-store.ts b/electron/main/settings-store.ts index c90599a5..a0279797 100644 --- a/electron/main/settings-store.ts +++ b/electron/main/settings-store.ts @@ -1,5 +1,5 @@ -import { join } from 'path' -import { readFileSync, writeFileSync, existsSync } from 'fs' +import { dirname, join } from 'path' +import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'fs' export interface AppSettings { modelsDir: string @@ -7,6 +7,8 @@ export interface AppSettings { workflowsDir: string extensionsDir: string dependenciesDir: string + /** Local LLM engine, GGUF models, logs and config used by the agent. */ + agentDir: string hfToken?: string } @@ -15,7 +17,7 @@ function settingsPath(userData: string): string { } export function getSettings(userData: string): AppSettings { - const defaults: AppSettings = { + const defaults: Omit = { modelsDir: join(userData, 'models'), workspaceDir: join(userData, 'workspace'), workflowsDir: join(userData, 'workflows'), @@ -24,7 +26,7 @@ export function getSettings(userData: string): AppSettings { } const file = settingsPath(userData) - if (!existsSync(file)) return defaults + if (!existsSync(file)) return withAgentDir(defaults) try { const saved = JSON.parse(readFileSync(file, 'utf-8')) as Record @@ -33,12 +35,42 @@ export function getSettings(userData: string): AppSettings { saved['workspaceDir'] = saved['outputsDir'] delete saved['outputsDir'] } - return { ...defaults, ...saved } + return withAgentDir({ ...defaults, ...saved }) } catch { - return defaults + return withAgentDir(defaults) } } +/** + * Installs set up before agentDir existed have their data folders under the base + * chosen at first run (/models, /extensions, …). Defaulting agentDir + * to userData would put the agent's multi-GB engine and models apart from all of + * them, so it defaults to a sibling of modelsDir instead. + */ +function withAgentDir(settings: Omit & { agentDir?: string }): AppSettings { + return { ...settings, agentDir: settings.agentDir || join(dirname(settings.modelsDir), 'agent') } +} + +/** + * Create the agent folder, and pin it in settings.json for installs that predate + * it. Until pinned it is derived from modelsDir, so moving the models folder in + * Settings → Storage would silently take the agent folder (engine, GGUF models) + * somewhere else. A fresh install has no settings.json yet: setup writes it. + */ +export function ensureAgentDir(userData: string): string { + const { agentDir } = getSettings(userData) + const file = settingsPath(userData) + if (existsSync(file)) { + let saved: Record | null = null + try { + saved = JSON.parse(readFileSync(file, 'utf-8')) as Record + } catch { /* unreadable: leave it as is */ } + if (saved && !saved['agentDir']) setSettings(userData, { agentDir }) + } + mkdirSync(agentDir, { recursive: true }) + return agentDir +} + export function setSettings(userData: string, patch: Partial): AppSettings { const updated = { ...getSettings(userData), ...patch } writeFileSync(settingsPath(userData), JSON.stringify(updated, null, 2), 'utf-8') diff --git a/src/shared/components/ui/ModelLibraryModal.tsx b/src/shared/components/ui/ModelLibraryModal.tsx index 81e6afe5..26e1d8d3 100644 --- a/src/shared/components/ui/ModelLibraryModal.tsx +++ b/src/shared/components/ui/ModelLibraryModal.tsx @@ -402,7 +402,7 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele {/* Footer hint */}

- Custom models: drop any .gguf in {status?.models_dir ?? '~/.modly/llm/models'} — detected automatically. + Custom models: drop any .gguf in {status?.models_dir ?? 'agent/models'} — detected automatically.

From dccca69e637d67976a9192a0bcb833ef282d4227 Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 18:11:05 +0200 Subject: [PATCH 4/7] feat(agent): engine status in settings and a model browser Settings > Agent shows whether the llama.cpp engine is installed, with an install button when it is missing, and lists the models in the agent folder with the selected one first. Browse opens the model library, now a wide dialog with two sections: Installed (everything in agent/models, Select / delete) and Suggested (catalog models not downloaded yet). Add picks a local .gguf and copies it into the agent models folder; picking and copying both run in the main process. The engine install moved out of the dialog into the settings section, and the shared SSE progress bar lives in its own component. --- electron/main/ipc-handlers.ts | 38 +- electron/preload/electron-api.ts | 7 + .../settings/components/AgentSection.tsx | 126 ++++-- .../components/ui/ModelLibraryModal.tsx | 390 ++++++++---------- src/shared/components/ui/SseProgressBar.tsx | 21 + src/shared/types/electron.d.ts | 4 + 6 files changed, 323 insertions(+), 263 deletions(-) create mode 100644 src/shared/components/ui/SseProgressBar.tsx diff --git a/electron/main/ipc-handlers.ts b/electron/main/ipc-handlers.ts index 01152735..e9cac040 100644 --- a/electron/main/ipc-handlers.ts +++ b/electron/main/ipc-handlers.ts @@ -1,8 +1,8 @@ import { ipcMain, BrowserWindow, Notification, dialog, app, shell } from 'electron' import { buildSync } from 'esbuild' import { autoUpdater } from 'electron-updater' -import { join } from 'path' -import { rm as rmAsync, readFile, writeFile, mkdir, readdir, rename, cp, symlink, lstat } from 'fs/promises' +import { basename, join } from 'path' +import { rm as rmAsync, readFile, writeFile, mkdir, readdir, rename, cp, symlink, lstat, copyFile } from 'fs/promises' import { existsSync, mkdirSync, readdirSync, statSync } from 'fs' import axios from 'axios' import * as tar from 'tar' @@ -1055,6 +1055,40 @@ export function setupIpcHandlers(pythonBridge: PythonBridge, getWindow: WindowGe return result.canceled ? null : result.filePaths[0] }) + // Add a local GGUF to the agent's models folder. Picking and copying both + // happen here, so the renderer never hands main an arbitrary path to copy. + ipcMain.handle('agent:addModel', async (): Promise<{ success: boolean; cancelled?: boolean; fileName?: string; error?: string }> => { + const win = getWindow() + if (!win) return { success: false, error: 'No window available' } + const result = await dialog.showOpenDialog(win, { + title: 'Add a model', + filters: [{ name: 'GGUF model', extensions: ['gguf'] }], + properties: ['openFile'], + }) + if (result.canceled || !result.filePaths[0]) return { success: false, cancelled: true } + + const src = result.filePaths[0] + const fileName = basename(src) + if (!fileName.toLowerCase().endsWith('.gguf')) return { success: false, error: 'Only .gguf files can be added.' } + + const modelsDir = join(getSettings(app.getPath('userData')).agentDir, 'models') + const dest = join(modelsDir, fileName) + if (existsSync(dest)) return { success: false, error: `"${fileName}" is already in your models.` } + + // Copied under a temporary name first: the API lists every *.gguf in the + // folder, so a multi-GB copy in progress would otherwise show up as a model. + const partial = `${dest}.part` + try { + await mkdir(modelsDir, { recursive: true }) + await copyFile(src, partial) + await rename(partial, dest) + return { success: true, fileName } + } catch (err) { + await rmAsync(partial, { force: true }).catch(() => {}) + return { success: false, error: String(err) } + } + }) + ipcMain.handle('fs:moveDirectory', async (_, { src, dest }: { src: string; dest: string }) => { try { await mkdir(dest, { recursive: true }) diff --git a/electron/preload/electron-api.ts b/electron/preload/electron-api.ts index a8c782ed..33944a57 100644 --- a/electron/preload/electron-api.ts +++ b/electron/preload/electron-api.ts @@ -111,6 +111,13 @@ export function createElectronApi(ipcRenderer: IpcRendererLike, webFrame: WebFra decrypt: (stored: string): Promise => ipcRenderer.invoke('secure:decrypt', stored) as Promise, }, + // Agent — local LLM models + agent: { + // Opens a file picker and copies the chosen .gguf into the agent's models folder. + addModel: (): Promise<{ success: boolean; cancelled?: boolean; fileName?: string; error?: string }> => + ipcRenderer.invoke('agent:addModel') as Promise<{ success: boolean; cancelled?: boolean; fileName?: string; error?: string }>, + }, + // Settings settings: { get: (): Promise<{ modelsDir: string; workspaceDir: string; workflowsDir: string; extensionsDir: string; hfToken?: string }> => diff --git a/src/areas/settings/components/AgentSection.tsx b/src/areas/settings/components/AgentSection.tsx index 99887122..da842eb2 100644 --- a/src/areas/settings/components/AgentSection.tsx +++ b/src/areas/settings/components/AgentSection.tsx @@ -4,7 +4,11 @@ import { type ThinkingMode, type ProviderId, type ExternalConfig, } from '@shared/stores/agentStore' import { useAppStore } from '@shared/stores/appStore' -import { ModelLibraryModal, type LlmModel } from '@shared/components/ui/ModelLibraryModal' +import { consumeSse, type SseEvent } from '@shared/services/llmDownloads' +import { SseProgressBar } from '@shared/components/ui/SseProgressBar' +import { ModelLibraryModal } from '@shared/components/ui/ModelLibraryModal' +import type { LlmModel } from '@shared/stores/llmModelsStore' +import { formatBytes } from '@shared/utils/format' // ─── Helpers ────────────────────────────────────────────────────────────────── @@ -18,15 +22,27 @@ function Field({ label, hint, children }: { label: string; hint?: string; childr ) } -function Group({ title, children }: { title: string; children: React.ReactNode }): JSX.Element { +function Group({ title, badge, children }: { title: string; badge?: React.ReactNode; children: React.ReactNode }): JSX.Element { return (
-

{title}

+
+

{title}

+ {badge} +
{children}
) } +function Badge({ tone, children }: { tone: 'ok' | 'warn' | 'muted'; children: React.ReactNode }): JSX.Element { + const cls = tone === 'ok' + ? 'bg-emerald-500/15 text-emerald-400' + : tone === 'warn' + ? 'bg-amber-500/15 text-amber-400' + : 'bg-zinc-700/40 text-zinc-400' + return {children} +} + const inputCls = 'bg-zinc-900 border border-zinc-700/60 rounded-lg px-3 py-2 text-[12.5px] text-zinc-200 focus:outline-none focus:border-zinc-500' // ─── Component ──────────────────────────────────────────────────────────────── @@ -38,8 +54,10 @@ export function AgentSection(): JSX.Element { } = useAgentStore() const apiUrl = useAppStore((s) => s.apiUrl) - // Local engine summary + // Local engine const [engineInstalled, setEngineInstalled] = useState(null) + const [engineInstall, setEngineInstall] = useState(null) + const [engineError, setEngineError] = useState(null) const [models, setModels] = useState([]) const [showLibrary, setShowLibrary] = useState(false) const [maxModels, setMaxModels] = useState('auto') @@ -59,7 +77,7 @@ export function AgentSection(): JSX.Element { try { const [s, m, c] = await Promise.all([ fetch(`${apiUrl}/llm/status`).then((r) => r.json()), - fetch(`${apiUrl}/llm/models`).then((r) => r.json()), + fetch(`${apiUrl}/llm/models?downloaded=true`).then((r) => r.json()), fetch(`${apiUrl}/llm/config`).then((r) => r.json()), ]) setEngineInstalled(Boolean(s.binary_installed)) @@ -69,10 +87,25 @@ export function AgentSection(): JSX.Element { setVramGb(c.vram_gb ?? null) } catch { setEngineInstalled(null) - setModels([]) } }, [apiUrl]) + async function installEngine() { + setEngineInstall({ percent: 0, status: 'Starting…' }) + setEngineError(null) + try { + await consumeSse(`${apiUrl}/llm/binary/install`, (e) => { + if (e.error) setEngineError(e.error) + else setEngineInstall(e) + }) + } catch (e) { + setEngineError(e instanceof Error ? e.message : String(e)) + } finally { + setEngineInstall(null) + void refreshLocal() + } + } + async function changeMaxModels(value: string) { setMaxModels(value) try { @@ -127,9 +160,6 @@ export function AgentSection(): JSX.Element { } } - const downloaded = models.filter((m) => m.downloaded) - const defaultEntry = models.find((m) => m.id === localModel) - const THINKING_OPTIONS: { value: ThinkingMode; label: string; desc: string }[] = [ { value: 'auto', label: 'Auto', desc: 'The model decides whether to think' }, { value: 'on', label: 'Enabled', desc: 'Forces thinking on every response' }, @@ -165,39 +195,65 @@ export function AgentSection(): JSX.Element { {provider === 'local' ? ( - -
-
- {engineInstalled === null ? ( -

Cannot reach the Modly API.

+ API unreachable + : engineInstalled ? Engine installed + : Engine missing + } + > + {engineInstalled === false && ( +
+ {engineInstall ? ( + ) : ( - <> -

- {defaultEntry ? defaultEntry.name : localModel} - default model -

-

- {!engineInstalled ? ( - Engine not installed - ) : downloaded.length === 0 ? ( - No model downloaded yet - ) : ( - {downloaded.length} model{downloaded.length > 1 ? 's' : ''} downloaded - )} -

- + )} + {engineError &&

{engineError}

}
+ )} + {engineInstalled === null && ( + )} + +
+
+ + +
+ {models.length === 0 ? ( +

No model yet — open Browse to add or download one.

+ ) : ( +
    + {/* The agent's selected model first, so it is what you see on closing Browse. */} + {[...models].sort((a, b) => Number(b.id === localModel) - Number(a.id === localModel)).map((m) => ( +
  • +
    + {m.name} + {m.id === localModel && Selected} +
    + {m.size_bytes ? {formatBytes(m.size_bytes)} : null} +
  • + ))} +
+ )}
-

- Browse models by category (General, Vision), download only what you need, and pick the chat default. -

-
- {event.status} - - {event.totalBytes ? `${formatBytes(event.bytesDownloaded)} / ${formatBytes(event.totalBytes)}` : `${event.percent ?? 0}%`} - -
-
-
-
-
- ) -} - // ─── Component ──────────────────────────────────────────────────────────────── export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Element { @@ -68,10 +32,9 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele const localModel = useAgentStore((s) => s.localModel) const setLocalModel = useAgentStore((s) => s.setLocalModel) - const [status, setStatus] = useState(null) - const [category, setCategory] = useState('all') - const [installing, setInstalling] = useState(null) - const [error, setError] = useState(null) + const [status, setStatus] = useState(null) + const [adding, setAdding] = useState(false) + const [error, setError] = useState(null) // The model list comes from the shared catalog store, so a download or delete // here immediately updates every other picker (chat, extension params, @@ -148,20 +111,16 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele // on every render and each pass forced another /llm/models + /llm/status. useEffect(() => { void refresh() }, [refresh]) - async function handleInstallEngine() { - setInstalling({ percent: 0, status: 'Starting…' }) + async function handleAdd() { + setAdding(true) setError(null) try { - await withAbort((signal) => consumeSse(`${apiUrl}/llm/binary/install`, (e) => { - if (!aliveRef.current) return - if (e.error) { setError(e.error); return } - setInstalling(e) - }, signal)) - } catch (e) { - if (aliveRef.current) setError(e instanceof Error ? e.message : String(e)) + const res = await window.electron.agent.addModel() + if (!aliveRef.current) return + if (res.error) setError(res.error) + if (res.success) void refresh() } finally { - if (aliveRef.current) setInstalling(null) - void refresh() + if (aliveRef.current) setAdding(false) } } @@ -183,10 +142,8 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele if (finished) void refresh() }, [downloads, refresh]) - const visible = models.filter((m) => inCategory(m, category)) - const counts = Object.fromEntries( - CATEGORIES.map((c) => [c.id, models.filter((m) => inCategory(m, c.id)).length]), - ) as Record + const installed = models.filter((m) => m.downloaded) + const suggested = models.filter((m) => !m.downloaded) return createPortal(
void }): JSX.Ele aria-modal="true" aria-label="Model library" tabIndex={-1} - className="relative w-[600px] max-w-[92vw] max-h-[85vh] rounded-2xl bg-zinc-900 border border-accent/20 shadow-2xl shadow-accent/5 overflow-hidden animate-slide-up-center flex flex-col focus:outline-none" + className="relative w-[960px] max-w-[94vw] max-h-[85vh] rounded-2xl bg-zinc-900 border border-zinc-700/60 shadow-[0_30px_60px_rgba(0,0,0,0.5)] overflow-hidden animate-slide-up-center flex flex-col focus:outline-none" > {/* Header */}
-

Model library

+

Models

Local models shared by the whole app — chat agent and extensions.

- +
+ + +
- {/* Engine status */} -
- {status === null ? ( -

+

+ {status === null && ( +

Cannot reach the Modly API. - {/* The library no longer polls, so a backend that was still - * starting up needs a way back in short of reopening the modal. */} + {/* The library does not poll, so a backend that was still starting + * up needs a way back in short of reopening the modal. */}

- ) : !status.binary_installed ? ( -
-

- The inference engine is not installed. Modly will fetch the llama.cpp build matching this - machine ({status.has_nvidia_gpu ? 'NVIDIA GPU detected — CUDA build' : 'Vulkan/CPU build'}). -

- {installing ? ( - - ) : ( - - )} -
- ) : ( -

- Engine installed{status.server.alive ? ` — ${status.server.model_id} loaded` : ''} -

)} {(error || downloadError) && ( -

+

{error || downloadError}

- {/* Category tabs */} -
- {CATEGORIES.map((c) => ( - - ))} -
+ {/* Installed first, then the catalog's suggestions not yet downloaded */} +
+ {([ + { title: 'Installed', items: installed, empty: 'No model installed yet — add a .gguf file or download a suggestion below.' }, + { title: 'Suggested', items: suggested, empty: 'Every suggested model is installed.' }, + ]).map((section) => ( +
+

+ {section.title} + {section.items.length} +

+ {section.items.length > 0 && ( +
+ {section.items.map((m) => { + const dl = downloads[m.id] + const tags = m.tags ?? [] + const isDefault = m.downloaded && localModel === m.id + // Size and VRAM say nothing about how well a model drives the + // agent — a 4B outscores a 20B here. The tooltip keeps a + // measured rate and an estimate visibly apart. + const grade = agentGrade(m) + const fit = vramFit(m.vram_estimate_mb, status?.vram_gb) + return ( +
+
+

{m.name}

+ {isDefault && ( + Default + )} +
- {/* Model list */} -
- {visible.map((m) => { - const dl = downloads[m.id] - return ( -
-
-
-

- {m.name} - {m.source === 'custom' && ( - custom - )} - {(m.tags ?? []).includes('cad') && ( - CAD - )} - {(m.tags ?? []).includes('vision') && ( - Vision - )} - {/* Size and VRAM say nothing about how well a model drives - * the agent — a 4B outscores a 20B here. The tooltip keeps - * a measured rate and an estimate visibly apart. */} - {(() => { - const grade = agentGrade(m) - return grade ? ( - - {grade.label} - - ) : null - })()} -

-

- - {formatBytes(m.size_bytes)} - {m.quant ? ` · ${m.quant}` : ''} - {m.vram_estimate_mb ? ` · ~${(m.vram_estimate_mb / 1000).toFixed(1)} GB VRAM` : ''} - - {(() => { - const fit = vramFit(m.vram_estimate_mb, status?.vram_gb) - return fit ? ( - {fit.label} - ) : null - })()} -

-
-
- {m.downloaded ? ( - <> - {localModel === m.id ? ( - Default - ) : ( - - )} - - - ) : dl ? ( - <> - {dl.paused ? ( - - ) : ( - +
+ {m.source === 'custom' && ( + custom + )} + {tags.includes('cad') && ( + CAD + )} + {tags.includes('vision') && ( + Vision + )} + {grade && ( + {grade.label} + )} + {fit && ( + {fit.label} + )} +
+ +

+ {formatBytes(m.size_bytes)} + {m.quant ? ` · ${m.quant}` : ''} + {m.vram_estimate_mb ? ` · ~${(m.vram_estimate_mb / 1000).toFixed(1)} GB VRAM` : ''} +

+ + {m.description && ( +

{m.description}

)} - - - ) : ( - - )} -
-
- {m.description &&

{m.description}

} - {dl && } -
- ) - })} - {visible.length === 0 && ( -

No models in this category.

- )} -
- {/* Footer hint */} -
-

- Custom models: drop any .gguf in {status?.models_dir ?? 'agent/models'} — detected automatically. -

+
+ {dl && } +
+ {m.downloaded ? ( + <> + {!isDefault && ( + + )} + + + ) : dl ? ( + <> + {dl.paused ? ( + + ) : ( + + )} + + + ) : ( + + )} +
+
+
+ ) + })} +
+ )} + {section.items.length === 0 && ( +

{section.empty}

+ )} +
+ ))}
, diff --git a/src/shared/components/ui/SseProgressBar.tsx b/src/shared/components/ui/SseProgressBar.tsx new file mode 100644 index 00000000..60f8596b --- /dev/null +++ b/src/shared/components/ui/SseProgressBar.tsx @@ -0,0 +1,21 @@ +import type { SseEvent } from '@shared/services/llmDownloads' +import { formatBytes } from '@shared/utils/format' + +/** Progress of an SSE-driven download or install (engine, GGUF models). */ +export function SseProgressBar({ event }: { event: SseEvent }): JSX.Element { + return ( +
+
+ {event.status} + + {event.totalBytes + ? `${formatBytes(event.bytesDownloaded ?? 0)} / ${formatBytes(event.totalBytes)}` + : `${event.percent ?? 0}%`} + +
+
+
+
+
+ ) +} diff --git a/src/shared/types/electron.d.ts b/src/shared/types/electron.d.ts index 9d25e666..c171b664 100644 --- a/src/shared/types/electron.d.ts +++ b/src/shared/types/electron.d.ts @@ -221,6 +221,10 @@ declare global { encrypt: (plainText: string) => Promise decrypt: (stored: string) => Promise } + agent: { + /** Opens a file picker and copies the chosen .gguf into the agent's models folder. */ + addModel: () => Promise<{ success: boolean; cancelled?: boolean; fileName?: string; error?: string }> + } cache: { clear: () => Promise<{ success: boolean; error?: string }> } From 5c62c72c277030a6151a042ac9dbb30a070e23d7 Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 19:21:20 +0200 Subject: [PATCH 5/7] feat(agent): redesign the Agent settings page and the model library Agent settings now use the Settings card kit in two columns: provider, local engine (status badge, install, selected model, simultaneous models) and thinking on the left; the MCP server setup on the right, with copy buttons and a tab per client. The shared Card gains an optional `aside` slot for header badges. The model library shows models as cards with a search box and filters (All, Fits my GPU, Vision, CAD), the card VRAM next to Add, and per-card tags, VRAM estimate and actions (Select, Download, Pause/Resume, Cancel, delete). --- .../settings/components/AgentSection.tsx | 563 ++++++++++-------- .../components/ui/ModelLibraryModal.tsx | 374 +++++++----- src/shared/ui/index.tsx | 19 +- 3 files changed, 544 insertions(+), 412 deletions(-) diff --git a/src/areas/settings/components/AgentSection.tsx b/src/areas/settings/components/AgentSection.tsx index da842eb2..9376ecb4 100644 --- a/src/areas/settings/components/AgentSection.tsx +++ b/src/areas/settings/components/AgentSection.tsx @@ -9,41 +9,113 @@ import { SseProgressBar } from '@shared/components/ui/SseProgressBar' import { ModelLibraryModal } from '@shared/components/ui/ModelLibraryModal' import type { LlmModel } from '@shared/stores/llmModelsStore' import { formatBytes } from '@shared/utils/format' +import { Section, Card, SegmentedControl } from '@shared/ui' // ─── Helpers ────────────────────────────────────────────────────────────────── -function Field({ label, hint, children }: { label: string; hint?: string; children: React.ReactNode }): JSX.Element { +/** A labelled block inside a Card, for controls that need the full width. */ +function Block({ label, description, hint, action, children }: { + label?: string + description?: string + hint?: string + action?: React.ReactNode + children?: React.ReactNode +}): JSX.Element { return ( -
- +
+ {(label || action) && ( +
+
+ {label &&

{label}

} + {description &&

{description}

} +
+ {action &&
{action}
} +
+ )} {children} - {hint &&

{hint}

} + {hint &&

{hint}

}
) } -function Group({ title, badge, children }: { title: string; badge?: React.ReactNode; children: React.ReactNode }): JSX.Element { +function Badge({ tone, children }: { tone: 'accent' | 'warn' | 'muted'; children: React.ReactNode }): JSX.Element { + const cls = tone === 'accent' + ? 'bg-accent/15 text-accent-light border-accent/30' + : tone === 'warn' + ? 'bg-amber-500/10 text-amber-400 border-amber-500/30' + : 'bg-zinc-800 text-zinc-400 border-zinc-700' + const dot = tone === 'accent' ? 'bg-accent-light' : tone === 'warn' ? 'bg-amber-400' : null return ( -
-
-

{title}

- {badge} -
+ + {dot && } {children} -
+ ) } -function Badge({ tone, children }: { tone: 'ok' | 'warn' | 'muted'; children: React.ReactNode }): JSX.Element { - const cls = tone === 'ok' - ? 'bg-emerald-500/15 text-emerald-400' - : tone === 'warn' - ? 'bg-amber-500/15 text-amber-400' - : 'bg-zinc-700/40 text-zinc-400' - return {children} +function CopyButton({ text }: { text: string }): JSX.Element { + const [copied, setCopied] = useState(false) + useEffect(() => { + if (!copied) return + const t = setTimeout(() => setCopied(false), 1500) + return () => clearTimeout(t) + }, [copied]) + return ( + + ) } -const inputCls = 'bg-zinc-900 border border-zinc-700/60 rounded-lg px-3 py-2 text-[12.5px] text-zinc-200 focus:outline-none focus:border-zinc-500' +function StepTitle({ n, children }: { n: number; children: React.ReactNode }): JSX.Element { + return ( +

+ {n}{children} +

+ ) +} + +const inputCls = 'w-full bg-zinc-800 border border-zinc-700 text-zinc-200 text-xs rounded-lg px-3 py-2 focus:outline-none focus:border-accent/50' +const secondaryBtnCls = 'px-2.5 py-1 rounded-md text-[11px] font-medium bg-zinc-800 hover:bg-zinc-700 text-zinc-300 transition-colors disabled:opacity-50' +const primaryBtnCls = 'px-3 py-1.5 rounded-lg bg-accent hover:bg-accent-dark text-white text-xs font-medium transition-colors' + +type McpClient = 'claude' | 'codex' | 'opencode' + +const MCP_CLIENTS: { value: McpClient; label: string; path: string; config: string }[] = [ + { + value: 'claude', + label: 'Claude Desktop', + path: '~/.config/claude/claude_desktop_config.json', + config: `{\n "mcpServers": {\n "modly": {\n "command": "modly-mcp"\n }\n }\n}`, + }, + { + value: 'codex', + label: 'Codex CLI', + path: '~/.codex/config.toml', + config: `[mcp_servers.modly]\ncommand = "modly-mcp"`, + }, + { + value: 'opencode', + label: 'OpenCode', + path: '~/.config/opencode/config.json', + config: `{\n "$schema": "https://opencode.ai/config.json",\n "mcp": {\n "modly": {\n "type": "local",\n "command": ["modly-mcp"]\n }\n }\n}`, + }, +] + +const MCP_INSTALL = 'npm install -g modly-cli-mcp' + +const THINKING_OPTIONS: { value: ThinkingMode; label: string; desc: string }[] = [ + { value: 'auto', label: 'Auto', desc: 'The model decides whether to think' }, + { value: 'on', label: 'Enabled', desc: 'Forces thinking on every response' }, + { value: 'off', label: 'Disabled', desc: 'Disables thinking (faster responses)' }, +] // ─── Component ──────────────────────────────────────────────────────────────── @@ -73,6 +145,8 @@ export function AgentSection(): JSX.Element { const [extTesting, setExtTesting] = useState(false) const [extResult, setExtResult] = useState<'ok' | 'error' | null>(null) + const [mcpClient, setMcpClient] = useState('claude') + const refreshLocal = useCallback(async () => { try { const [s, m, c] = await Promise.all([ @@ -160,270 +234,245 @@ export function AgentSection(): JSX.Element { } } - const THINKING_OPTIONS: { value: ThinkingMode; label: string; desc: string }[] = [ - { value: 'auto', label: 'Auto', desc: 'The model decides whether to think' }, - { value: 'on', label: 'Enabled', desc: 'Forces thinking on every response' }, - { value: 'off', label: 'Disabled', desc: 'Disables thinking (faster responses)' }, - ] - - const mcpConfigs = { - opencode: `{\n "$schema": "https://opencode.ai/config.json",\n "mcp": {\n "modly": {\n "type": "local",\n "command": ["modly-mcp"]\n }\n }\n}`, - codex: `[mcp_servers.modly]\ncommand = "modly-mcp"`, - claude: `{\n "mcpServers": {\n "modly": {\n "command": "modly-mcp"\n }\n }\n}`, - } + const selectedModel = models.find((m) => m.id === localModel) + const client = MCP_CLIENTS.find((c) => c.value === mcpClient) ?? MCP_CLIENTS[0] return ( -
-
-

Agent

-

Configure the LLM powering the chat — fully local by default.

-
+
+
- {/* Provider */} - - - - - - - {provider === 'local' ? ( - API unreachable - : engineInstalled ? Engine installed - : Engine missing - } - > - {engineInstalled === false && ( -
- {engineInstall ? ( - - ) : ( - - )} - {engineError &&

{engineError}

} -
- )} - {engineInstalled === null && ( - - )} + {/* ── Left column ── */} +
-
-
- - + )} + {engineError &&

{engineError}

} + + )} + {engineInstalled === null && ( + + + + )} - -
- { setKeyDraft(e.target.value); setExtResult(null) }} - placeholder={PROVIDERS[provider].noKey ? '' : 'sk-…'} - className={`${inputCls} flex-1`} - /> - } > - {extTesting ? 'Testing…' : 'Test'} - -
- {extResult === 'ok' && ( -

Connected — {extModels.length} model{extModels.length > 1 ? 's' : ''} available

- )} - {extResult === 'error' && ( -

- {provider === 'ollama' - ? 'Could not list models — is Ollama running?' - : `Could not list models — check the key${provider === 'custom' ? ' and URL' : ''}`} -

- )} -
- - - {extModels.length > 0 ? ( - - ) : ( - setExtModelDraft(e.target.value)} - placeholder={provider === 'anthropic' ? 'claude-sonnet-5' : provider === 'openai' ? 'gpt-5.2' : provider === 'ollama' ? 'qwen2.5:3b' : 'model name'} - className={inputCls} - /> - )} - - - - - )} + {selectedModel ? ( +
+ + + + +
+ {selectedModel.name} + + {[ + selectedModel.size_bytes ? formatBytes(selectedModel.size_bytes) : null, + selectedModel.quant, + selectedModel.vram_estimate_mb ? `~${(selectedModel.vram_estimate_mb / 1000).toFixed(1)} GB VRAM` : null, + ].filter(Boolean).join(' · ')} + +
+ In use +
+ ) : ( +

+ {models.length === 0 + ? 'No model yet — open Browse… to add or download one.' + : 'No model selected — open Browse… and select one.'} +

+ )} + - {/* Thinking */} - - -
- {THINKING_OPTIONS.map((opt) => ( -
-
- {([ - { label: 'Claude Desktop', key: 'claude' as const, hint: '~/.config/claude/claude_desktop_config.json' }, - { label: 'Codex CLI', key: 'codex' as const, hint: '~/.codex/config.toml' }, - { label: 'OpenCode', key: 'opencode' as const, hint: '~/.config/opencode/config.json' }, - ] as const).map(({ label, key, hint }) => ( -
-

{label} — {hint}

-
-
-                  {mcpConfigs[key]}
+        {/* ── Right column ── */}
+        
+ Control Modly from Claude Desktop, Codex or OpenCode. Community package by DrHepa.} + aside={Community} + > + + Install the package +
+ + ${MCP_INSTALL} + + +
+
+ + + Add it to your client +
+ ({ value, label }))} + ariaLabel="MCP client" + /> +
+
+
+

+ {client.label} + {client.path} +

+ +
+
+                  {client.config}
                 
-
-
- ))} + +
- +
{showLibrary && ( { setShowLibrary(false); void refreshLocal() }} /> )} -
+
) } diff --git a/src/shared/components/ui/ModelLibraryModal.tsx b/src/shared/components/ui/ModelLibraryModal.tsx index 099dc9e5..9c4ed475 100644 --- a/src/shared/components/ui/ModelLibraryModal.tsx +++ b/src/shared/components/ui/ModelLibraryModal.tsx @@ -5,6 +5,7 @@ import { useAgentStore } from '@shared/stores/agentStore' import { useLlmModels, type LlmModel } from '@shared/stores/llmModelsStore' import { useLlmDownloadsStore } from '@shared/services/llmDownloads' import { formatBytes as fmtBytes } from '@shared/utils/format' +import { SegmentedControl } from '@shared/ui' import { SseProgressBar } from './SseProgressBar' import { vramFit } from './vramFit' import { agentGrade } from './agentGrade' @@ -19,12 +20,51 @@ interface LlmStatus { vram_gb: number | null } +type Filter = 'all' | 'fits' | 'vision' | 'cad' + // ─── Helpers ────────────────────────────────────────────────────────────────── export function formatBytes(n?: number): string { return n ? fmtBytes(n) : '—' } +type Tone = 'accent' | 'warn' | 'muted' | 'outline' + +const TONES: Record = { + accent: 'bg-accent/15 text-accent-light border-accent/30', + warn: 'bg-amber-500/10 text-amber-400 border-amber-500/30', + muted: 'bg-zinc-800 text-zinc-400 border-zinc-700', + outline: 'bg-transparent text-accent-light border-accent/40', +} + +function Tag({ tone, colors, icon, title, children }: { + tone?: Tone + /** Explicit color classes, for verdicts that bring their own (vramFit). */ + colors?: string + icon?: React.ReactNode + title?: string + children: React.ReactNode +}): JSX.Element { + return ( + + {icon} + {children} + + ) +} + +const ICONS = { + check: , + warn: , + star: , + eye: , + cube: , + file: , +} + +const outlineBtnCls = 'inline-flex items-center gap-1.5 px-3.5 py-1.5 rounded-lg border border-accent/40 text-accent-light text-xs font-medium hover:bg-accent/10 hover:border-accent/60 transition-colors disabled:opacity-50' +const ghostBtnCls = 'px-3.5 py-1.5 rounded-lg border border-zinc-700 text-zinc-300 text-xs font-medium hover:text-white hover:border-zinc-500 transition-colors' + // ─── Component ──────────────────────────────────────────────────────────────── export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Element { @@ -35,6 +75,8 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele const [status, setStatus] = useState(null) const [adding, setAdding] = useState(false) const [error, setError] = useState(null) + const [query, setQuery] = useState('') + const [filter, setFilter] = useState('all') // The model list comes from the shared catalog store, so a download or delete // here immediately updates every other picker (chat, extension params, @@ -142,8 +184,121 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele if (finished) void refresh() }, [downloads, refresh]) - const installed = models.filter((m) => m.downloaded) - const suggested = models.filter((m) => !m.downloaded) + // VRAM is only known on NVIDIA (nvidia-smi); without it there is no fit verdict + // to show or filter on. + const vramGb = status?.vram_gb ?? null + + const filterOptions: { value: Filter; label: string }[] = [ + { value: 'all', label: 'All' }, + ...(vramGb ? [{ value: 'fits' as const, label: 'Fits my GPU' }] : []), + { value: 'vision', label: 'Vision' }, + { value: 'cad', label: 'CAD' }, + ] + + const q = query.trim().toLowerCase() + const matches = (m: LlmModel): boolean => { + const tags = m.tags ?? [] + if (filter === 'vision' && !tags.includes('vision')) return false + if (filter === 'cad' && !tags.includes('cad')) return false + if (filter === 'fits') { + const fit = vramFit(m.vram_estimate_mb, vramGb) + if (!fit || fit.label === "Won't fit") return false + } + return !q || m.name.toLowerCase().includes(q) || (m.description ?? '').toLowerCase().includes(q) + } + + const installed = models.filter((m) => m.downloaded && matches(m)) + const suggested = models.filter((m) => !m.downloaded && matches(m)) + const filtering = filter !== 'all' || q !== '' + + function renderCard(m: LlmModel): JSX.Element { + const dl = downloads[m.id] + const tags = m.tags ?? [] + const inUse = m.downloaded && localModel === m.id + const fit = vramFit(m.vram_estimate_mb, vramGb) + // Size and VRAM say nothing about how well a model drives the agent — a 4B + // outscores a 20B here. The tooltip keeps a measured rate and an estimate + // visibly apart. + const grade = agentGrade(m) + + return ( +
+
+

{m.name}

+ {inUse && ( + }>In use + )} +
+ +
+ {tags.includes('default') && Recommended} + {fit && {fit.label}} + {tags.includes('vision') && Vision} + {tags.includes('cad') && CAD} + {m.source === 'custom' && Custom} + {grade && {grade.label}} +
+ + {m.description && ( +

{m.description}

+ )} + +
+ {[m.size_bytes ? formatBytes(m.size_bytes) : null, m.quant].filter(Boolean).join(' · ')} + {m.vram_estimate_mb ? ( + + ~{(m.vram_estimate_mb / 1024).toFixed(1)}{vramGb ? ` / ${vramGb}` : ''} GB VRAM + + ) : null} +
+ + {dl && } + +
+ {m.downloaded ? ( + <> + {inUse ? ( + Used by the chat agent + ) : ( + + )} + + + ) : dl ? ( + <> + {dl.paused ? ( + + ) : ( + + )} + + + ) : ( + + )} +
+
+ ) + } return createPortal(
void }): JSX.Ele ref={dialogRef} role="dialog" aria-modal="true" - aria-label="Model library" + aria-label="Models" tabIndex={-1} - className="relative w-[960px] max-w-[94vw] max-h-[85vh] rounded-2xl bg-zinc-900 border border-zinc-700/60 shadow-[0_30px_60px_rgba(0,0,0,0.5)] overflow-hidden animate-slide-up-center flex flex-col focus:outline-none" + className="relative w-[1000px] max-w-[94vw] max-h-[88vh] rounded-2xl bg-zinc-900 border border-zinc-700/60 shadow-[0_30px_60px_rgba(0,0,0,0.5)] overflow-hidden animate-slide-up-center flex flex-col focus:outline-none" > {/* Header */} -
-
-

Models

-

- Local models shared by the whole app — chat agent and extensions. -

+
+
+
+

Models

+

+ Local models shared by the whole app — chat agent and extensions. +

+
+
+ {vramGb != null && ( + // self-stretch: as tall as the Add button beside it, whatever its padding. + + + + + {vramGb} GB VRAM + + )} + + +
-
- - + setQuery(e.target.value)} + placeholder="Search models" + className="w-full bg-zinc-800 border border-zinc-700 text-zinc-200 text-xs rounded-lg pl-8 pr-3 py-2 placeholder:text-zinc-500 focus:outline-none focus:border-accent/50" + /> +
+
-
-
{status === null && ( -

+

Cannot reach the Modly API. {/* The library does not poll, so a backend that was still starting * up needs a way back in short of reopening the modal. */} -

)} {(error || downloadError) && ( -

+

{error || downloadError}

{/* Installed first, then the catalog's suggestions not yet downloaded */} -
+
{([ - { title: 'Installed', items: installed, empty: 'No model installed yet — add a .gguf file or download a suggestion below.' }, - { title: 'Suggested', items: suggested, empty: 'Every suggested model is installed.' }, + { + title: 'Installed', + items: installed, + empty: filtering ? 'No installed model matches.' : 'No model installed yet — add a .gguf file or download a suggestion below.', + }, + { + title: 'Suggested', + items: suggested, + empty: filtering ? 'No suggested model matches.' : 'Every suggested model is installed.', + }, ]).map((section) => ( -
-

+
+

{section.title} {section.items.length}

- {section.items.length > 0 && ( -
- {section.items.map((m) => { - const dl = downloads[m.id] - const tags = m.tags ?? [] - const isDefault = m.downloaded && localModel === m.id - // Size and VRAM say nothing about how well a model drives the - // agent — a 4B outscores a 20B here. The tooltip keeps a - // measured rate and an estimate visibly apart. - const grade = agentGrade(m) - const fit = vramFit(m.vram_estimate_mb, status?.vram_gb) - return ( -
-
-

{m.name}

- {isDefault && ( - Default - )} -
- -
- {m.source === 'custom' && ( - custom - )} - {tags.includes('cad') && ( - CAD - )} - {tags.includes('vision') && ( - Vision - )} - {grade && ( - {grade.label} - )} - {fit && ( - {fit.label} - )} -
- -

- {formatBytes(m.size_bytes)} - {m.quant ? ` · ${m.quant}` : ''} - {m.vram_estimate_mb ? ` · ~${(m.vram_estimate_mb / 1000).toFixed(1)} GB VRAM` : ''} -

- - {m.description && ( -

{m.description}

- )} - -
- {dl && } -
- {m.downloaded ? ( - <> - {!isDefault && ( - - )} - - - ) : dl ? ( - <> - {dl.paused ? ( - - ) : ( - - )} - - - ) : ( - - )} -
-
-
- ) - })} + {section.items.length > 0 ? ( +
+ {section.items.map(renderCard)}
- )} - {section.items.length === 0 && ( -

{section.empty}

+ ) : ( +

{section.empty}

)}
))} diff --git a/src/shared/ui/index.tsx b/src/shared/ui/index.tsx index a9609995..a2c6c181 100644 --- a/src/shared/ui/index.tsx +++ b/src/shared/ui/index.tsx @@ -12,13 +12,22 @@ export function Section({ title, subtitle, children }: { title: string; subtitle ) } -export function Card({ title, description, children }: { title?: string; description?: string; children: React.ReactNode }): JSX.Element { +export function Card({ title, description, aside, children }: { + title?: string + description?: React.ReactNode + /** Rendered at the right of the header (status badge, …). */ + aside?: React.ReactNode + children: React.ReactNode +}): JSX.Element { return (
- {(title || description) && ( -
- {title &&

{title}

} - {description &&

{description}

} + {(title || description || aside) && ( +
+
+ {title &&

{title}

} + {description &&

{description}

} +
+ {aside &&
{aside}
}
)}
From 50ae1c4ac315e4f176ae6895629a5129a14d33a7 Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 19:35:20 +0200 Subject: [PATCH 6/7] fix(llm): verify the engine archive against its published sha256 The llama-server archive is downloaded from the llama.cpp GitHub release and its files are executed, but nothing checked the bytes. GitHub reports a sha256 digest for every release asset; the download is now hashed as it streams and refused on a mismatch. An asset without a digest (older uploads) is accepted as before. A failed or refused download no longer leaves its temp archive behind. --- api/services/llm_server.py | 61 +++++++++++++++++++++++---------- api/tests/test_llm_downloads.py | 60 ++++++++++++++++++++++++++++++++ 2 files changed, 103 insertions(+), 18 deletions(-) diff --git a/api/services/llm_server.py b/api/services/llm_server.py index 213e40a1..c41fb196 100644 --- a/api/services/llm_server.py +++ b/api/services/llm_server.py @@ -14,6 +14,7 @@ models are chosen by the user from the catalog. """ import contextlib +import hashlib import os import re import shutil @@ -342,27 +343,51 @@ def _download_asset(asset: dict, progress_cb: Callable[[dict], None], control_ch fd, tmp_name = tempfile.mkstemp(suffix=suffix) os.close(fd) tmp = Path(tmp_name) - with urlopen(Request(asset["browser_download_url"], headers=_UA), timeout=30) as resp: - total = int(resp.headers.get("Content-Length", 0)) or asset.get("size", 0) - done = 0 - last_emit = 0.0 - with open(tmp, "wb") as fh: - while chunk := resp.read(1 << 20): - control_check() - fh.write(chunk) - done += len(chunk) - now = time.monotonic() - if now - last_emit >= 0.5: - progress_cb({ - "status": f"Downloading {label} ({asset['name']})", - "bytesDownloaded": done, - "totalBytes": total, - "percent": round(done / total * 100) if total else 0, - }) - last_emit = now + sha256 = hashlib.sha256() + try: + with urlopen(Request(asset["browser_download_url"], headers=_UA), timeout=30) as resp: + total = int(resp.headers.get("Content-Length", 0)) or asset.get("size", 0) + done = 0 + last_emit = 0.0 + with open(tmp, "wb") as fh: + while chunk := resp.read(1 << 20): + control_check() + fh.write(chunk) + sha256.update(chunk) + done += len(chunk) + now = time.monotonic() + if now - last_emit >= 0.5: + progress_cb({ + "status": f"Downloading {label} ({asset['name']})", + "bytesDownloaded": done, + "totalBytes": total, + "percent": round(done / total * 100) if total else 0, + }) + last_emit = now + _verify_digest(asset, sha256.hexdigest()) + except BaseException: + tmp.unlink(missing_ok=True) # never leave a partial or rejected archive behind + raise return tmp +def _verify_digest(asset: dict, actual_sha256: str) -> None: + """Refuse an archive whose bytes differ from what GitHub published for it. + + The files extracted from it are executed (llama-server and its DLLs), so a + corrupted or tampered download must not get that far. GitHub reports a + `digest` ("sha256:") for every release asset; one without it (older + uploads) cannot be checked and is accepted as before.""" + expected = asset.get("digest") or "" + if not expected.startswith("sha256:"): + return + if actual_sha256.lower() != expected[len("sha256:"):].lower(): + raise RuntimeError( + f"Checksum mismatch for {asset['name']}: the download does not match the " + "published llama.cpp release. Try installing the engine again." + ) + + _LIB_SUFFIXES = (".so", ".dylib", ".metal") diff --git a/api/tests/test_llm_downloads.py b/api/tests/test_llm_downloads.py index 6450fa24..31d29915 100644 --- a/api/tests/test_llm_downloads.py +++ b/api/tests/test_llm_downloads.py @@ -1,7 +1,9 @@ +import hashlib import importlib import tempfile import unittest from pathlib import Path +from unittest import mock llm_server = importlib.import_module("services.llm_server") llm_router = importlib.import_module("routers.llm") @@ -57,5 +59,63 @@ def test_nothing_on_disk_is_not_an_error(self): self.assertEqual(llm_router._discard_incomplete(_VISION), []) +class _FakeResponse: + def __init__(self, body: bytes) -> None: + self._body = body + self.headers = {"Content-Length": str(len(body))} + + def read(self, _size: int) -> bytes: + chunk, self._body = self._body, b"" + return chunk + + def __enter__(self): + return self + + def __exit__(self, *_exc) -> None: + return None + + +class EngineDigestTests(unittest.TestCase): + """The engine archive's files are executed, so a download whose bytes differ + from the digest GitHub published for the asset must be refused — and must + not be left behind in the temp folder.""" + + BODY = b"llama-server archive bytes" + + def setUp(self) -> None: + self._tmp = tempfile.TemporaryDirectory() + tmp_dir = self._tmp.name + real_mkstemp = tempfile.mkstemp # llm_server.tempfile is this same module + patches = [ + mock.patch.object(llm_server, "urlopen", lambda *_a, **_k: _FakeResponse(self.BODY)), + mock.patch.object(llm_server.tempfile, "mkstemp", lambda suffix="": real_mkstemp(suffix=suffix, dir=tmp_dir)), + ] + for p in patches: + p.start() + self.addCleanup(p.stop) + + def tearDown(self) -> None: + self._tmp.cleanup() + + def _download(self, digest): + asset = {"name": "llama-bin-win-x64.zip", "browser_download_url": "https://example.invalid/a.zip"} + if digest is not None: + asset["digest"] = digest + return llm_server._download_asset(asset, lambda _msg: None, lambda: None, "engine") + + def test_matching_digest_keeps_the_archive(self): + path = self._download("sha256:" + hashlib.sha256(self.BODY).hexdigest()) + self.assertEqual(path.read_bytes(), self.BODY) + + def test_mismatching_digest_is_refused_and_removed(self): + with self.assertRaises(RuntimeError): + self._download("sha256:" + "0" * 64) + self.assertEqual(list(Path(self._tmp.name).iterdir()), []) + + def test_asset_without_digest_is_still_accepted(self): + path = self._download(None) + self.assertEqual(path.read_bytes(), self.BODY) + + if __name__ == "__main__": unittest.main() From b6f52602f72075ce1bcc42d4c150bbf90378b4f0 Mon Sep 17 00:00:00 2001 From: Lightning Pixel Date: Sat, 3 Oct 2026 19:35:21 +0200 Subject: [PATCH 7/7] fix(agent): keep code/CAD models out of the agent model selection The chat leaves code/CAD models out of its own picker, but the model library still offered Select on them, so a CAD coder could become the agent's chat model. Those cards now say they are for workflow nodes. Also drop LlmModelSelect, which nothing imports. --- src/shared/components/ui/LlmModelSelect.tsx | 62 ------------------- .../components/ui/ModelLibraryModal.tsx | 5 ++ 2 files changed, 5 insertions(+), 62 deletions(-) delete mode 100644 src/shared/components/ui/LlmModelSelect.tsx diff --git a/src/shared/components/ui/LlmModelSelect.tsx b/src/shared/components/ui/LlmModelSelect.tsx deleted file mode 100644 index 44bb1da1..00000000 --- a/src/shared/components/ui/LlmModelSelect.tsx +++ /dev/null @@ -1,62 +0,0 @@ -import { useLlmModels } from '@shared/stores/llmModelsStore' - -/** - * Dropdown of the shared local-LLM library, optionally filtered by category - * (`tag`). Catalog models that aren't downloaded yet stay visible (marked) so - * the user can pick ONE and download only that one in Settings → Agent. - * - * Shared by the LLM node and by extension params of type `llm-model`, so both - * show the same list, the same names and the same warning. - */ -/** A GGUF the user dropped in the models folder has no tags, so the API returns - * it for every category ("capabilities unknown" — api/routers/llm.py). Under a - * category filter that reads as an endorsement: a general-purpose model showed - * up as a CAD model in Text to CAD's picker with nothing to say otherwise. */ -function label(m: { source?: string }, tag?: string): string { - return tag && m.source === 'custom' ? ' — custom, not verified for this use' : '' -} - -export default function LlmModelSelect({ value, tag, disabled, className, onChange }: { - value: string - tag?: string - disabled?: boolean - className?: string - onChange: (v: string) => void -}) { - const { models } = useLlmModels(tag) - - const ready = models.filter((m) => m.downloaded) - const missing = models.filter((m) => !m.downloaded) - const selectedMissing = missing.some((m) => m.id === value) - - return ( -
- - {selectedMissing && ( -

- Download this model in Settings → Agent before running. -

- )} -
- ) -} diff --git a/src/shared/components/ui/ModelLibraryModal.tsx b/src/shared/components/ui/ModelLibraryModal.tsx index 9c4ed475..28d68999 100644 --- a/src/shared/components/ui/ModelLibraryModal.tsx +++ b/src/shared/components/ui/ModelLibraryModal.tsx @@ -215,6 +215,9 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele const dl = downloads[m.id] const tags = m.tags ?? [] const inUse = m.downloaded && localModel === m.id + // Code/CAD models are tools for workflow nodes, not chat models: the chat's + // own picker leaves them out, so they cannot become the agent's model here. + const nodeOnly = tags.some((t) => t === 'code' || t === 'cad') const fit = vramFit(m.vram_estimate_mb, vramGb) // Size and VRAM say nothing about how well a model drives the agent — a 4B // outscores a 20B here. The tooltip keeps a measured rate and an estimate @@ -264,6 +267,8 @@ export function ModelLibraryModal({ onClose }: { onClose: () => void }): JSX.Ele <> {inUse ? ( Used by the chat agent + ) : nodeOnly ? ( + For workflow nodes, not the chat ) : ( )}