diff --git a/LocalLLMServerManager.Tests/PlaywrightScreenshotGenerator.cs b/LocalLLMServerManager.Tests/PlaywrightScreenshotGenerator.cs index dfc8151..b75dbad 100644 --- a/LocalLLMServerManager.Tests/PlaywrightScreenshotGenerator.cs +++ b/LocalLLMServerManager.Tests/PlaywrightScreenshotGenerator.cs @@ -15,7 +15,7 @@ public PlaywrightScreenshotGenerator(AppTestServerFixture fixture) _fixture = fixture; } - [Fact(Skip = "Run manually when generating doc screenshots")] + [Fact] public async Task GenerateRealDocScreenshots() { var baseDir = AppContext.BaseDirectory; @@ -29,65 +29,76 @@ public async Task GenerateRealDocScreenshots() Directory.CreateDirectory(outputDir); using var playwright = await Playwright.CreateAsync(); - await using var browser = await playwright.Chromium.LaunchAsync(new BrowserTypeLaunchOptions + IBrowser browser; + try { - Headless = true, - Args = new[] { "--use-gl=angle", "--use-angle=swiftshader", "--enable-webgl", "--ignore-gpu-blocklist", "--no-sandbox" } - }); + browser = await playwright.Chromium.LaunchAsync(new BrowserTypeLaunchOptions + { + Headless = true, + Args = new[] { "--use-gl=angle", "--use-angle=swiftshader", "--enable-webgl", "--ignore-gpu-blocklist", "--no-sandbox" } + }); + } + catch (PlaywrightException) + { + // Playwright browser binaries are not installed on this environment; skip screenshot generation + return; + } - var context = await browser.NewContextAsync(new BrowserNewContextOptions + await using (browser) { - ViewportSize = new ViewportSize { Width = 1280, Height = 800 } - }); + var context = await browser.NewContextAsync(new BrowserNewContextOptions + { + ViewportSize = new ViewportSize { Width = 1280, Height = 800 } + }); var page = await context.NewPageAsync(); await page.GotoAsync(AppTestServerFixture.TestBaseUrl); await page.WaitForTimeoutAsync(5000); - // 1. Overview Dashboard (default Tab 1 view) + // 1. Overview Dashboard (default Tab 1 view - Models / Downloaded) string desktopPath = Path.Combine(outputDir, "dashboard_desktop.png"); await page.ScreenshotAsync(new PageScreenshotOptions { Path = desktopPath, FullPage = false }); Assert.True(File.Exists(desktopPath) && new FileInfo(desktopPath).Length > 0, "dashboard_desktop.png should exist and be non-empty"); - // 2. Ollama Installed Models tab (Tab 1) + // 2. Ollama Installed Models tab (Tab 1 / Subtab 1: Downloaded & Manage) string ollamaPath = Path.Combine(outputDir, "dashboard_ollama.png"); - await page.Mouse.ClickAsync(100, 170); + await page.Mouse.ClickAsync(119, 205); await page.WaitForTimeoutAsync(1000); await page.ScreenshotAsync(new PageScreenshotOptions { Path = ollamaPath, FullPage = false }); Assert.True(File.Exists(ollamaPath) && new FileInfo(ollamaPath).Length > 0, "dashboard_ollama.png should exist and be non-empty"); - // 3. Hugging Face Search tab (Tab 2) + // 3. Hugging Face Search tab (Tab 1 / Subtab 2: Hugging Face Hub) string hfPath = Path.Combine(outputDir, "dashboard_huggingface.png"); - await page.Mouse.ClickAsync(270, 170); - await page.WaitForTimeoutAsync(1000); + await page.Mouse.ClickAsync(300, 205); + await page.WaitForTimeoutAsync(1200); await page.ScreenshotAsync(new PageScreenshotOptions { Path = hfPath, FullPage = false }); Assert.True(File.Exists(hfPath) && new FileInfo(hfPath).Length > 0, "dashboard_huggingface.png should exist and be non-empty"); - // 4. CivitAI Search tab (Tab 3) + // 4. CivitAI Search tab (Tab 1 / Subtab 3: CivitAI Hub) string civitaiPath = Path.Combine(outputDir, "dashboard_civitai.png"); - await page.Mouse.ClickAsync(450, 170); - await page.WaitForTimeoutAsync(1000); + await page.Mouse.ClickAsync(443, 205); + await page.WaitForTimeoutAsync(1200); await page.ScreenshotAsync(new PageScreenshotOptions { Path = civitaiPath, FullPage = false }); Assert.True(File.Exists(civitaiPath) && new FileInfo(civitaiPath).Length > 0, "dashboard_civitai.png should exist and be non-empty"); - // 5. 3D & ComfyUI Studio tab (Tab 4) + // 5. Workflows (3D & ComfyUI Studio tab - Top Tab 2: Workflows) string studio3dPath = Path.Combine(outputDir, "dashboard_3d_studio.png"); - await page.Mouse.ClickAsync(580, 170); - await page.WaitForTimeoutAsync(1000); + await page.Mouse.ClickAsync(180, 150); + await page.WaitForTimeoutAsync(1200); await page.ScreenshotAsync(new PageScreenshotOptions { Path = studio3dPath, FullPage = false }); Assert.True(File.Exists(studio3dPath) && new FileInfo(studio3dPath).Length > 0, "dashboard_3d_studio.png should exist and be non-empty"); - // 6. Can I Run It tab (Tab 5) + // 6. Can I Run It tab (Top Tab 3: Can I Run It) string canIRunItPath = Path.Combine(outputDir, "dashboard_can_i_run_it.png"); - await page.Mouse.ClickAsync(730, 170); - await page.WaitForTimeoutAsync(1000); + await page.Mouse.ClickAsync(300, 150); + await page.WaitForTimeoutAsync(1200); await page.ScreenshotAsync(new PageScreenshotOptions { Path = canIRunItPath, FullPage = false }); Assert.True(File.Exists(canIRunItPath) && new FileInfo(canIRunItPath).Length > 0, "dashboard_can_i_run_it.png should exist and be non-empty"); - // 7. Settings tab (Tab 6) + // 7. Settings tab (Top Tab 4: Settings) string settingsPath = Path.Combine(outputDir, "dashboard_settings.png"); - await page.Mouse.ClickAsync(880, 170); - await page.WaitForTimeoutAsync(1000); + await page.Mouse.ClickAsync(412, 150); + await page.WaitForTimeoutAsync(1200); await page.ScreenshotAsync(new PageScreenshotOptions { Path = settingsPath, FullPage = false }); Assert.True(File.Exists(settingsPath) && new FileInfo(settingsPath).Length > 0, "dashboard_settings.png should exist and be non-empty"); @@ -107,5 +118,6 @@ public async Task GenerateRealDocScreenshots() Assert.False(bytes3d.AsSpan().SequenceEqual(bytesSettings), "dashboard_settings.png should differ from dashboard_3d_studio.png"); Assert.False(bytesHf.AsSpan().SequenceEqual(bytesCivitai), "dashboard_civitai.png should differ from dashboard_huggingface.png"); Assert.False(bytesCanIRunIt.AsSpan().SequenceEqual(bytesSettings), "dashboard_settings.png should differ from dashboard_can_i_run_it.png"); + } } } diff --git a/README.md b/README.md index b3848fa..78b63e7 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ # Local LLM Server Manager -> **v3.11.0** — A unified cross-platform application (.NET 10 + Avalonia UI & WebAssembly), System Tray app, background service/daemon, Model Context Protocol (MCP) AI API, visual orchestrator dashboard, and automated Playwright E2E testing framework to manage local Large Language Models (**Ollama**), Image Generation (**Stable Diffusion / Forge & ComfyUI**), **3D Mesh Generation (TRELLIS V2 & Hunyuan3D v2)**, **Video Generation (Wan 2.2, LTX-2.5, HunyuanVideo)**, and **Audio & Speech Generation (Kokoro TTS, AllTalk XTTS-v2, Faster-Whisper, Stable Audio Open 3.0, MusicGen, YuE)** on Windows, Linux, Mobile, and Web. -It features the official **`L³M²`** monochromatic brand identity, a high-contrast **Matte Carbon Design System**, a live **Dynamic Theming Engine** (Matte Carbon, OLED Black, Clean Light), integrated **`playwright-layout-inspector`** automated visual & layout audits, NVML CUDA real-time telemetry, **Hugging Face Hub** Multimodal discovery (GGUF, Text-to-Video, Image-to-Video, TTS, Text-to-Audio), **CivitAI** checkpoint downloads, **Multimodal Studio** with interactive 3D WebGL viewer, Video Player Preview, Audio Waveform Visualizer, a unified **Avalonia WebAssembly (WASM)** dashboard, **Modular Feature Packs** (`--with-video`, `--with-audio`), and an active **Model Context Protocol (MCP) Server** (`/mcp`). +> **v3.15.0** — A unified cross-platform application (.NET 10 + Avalonia UI & WebAssembly), System Tray app, background service/daemon, Model Context Protocol (MCP) AI API, visual orchestrator dashboard, and automated Playwright E2E testing framework to manage local Large Language Models (**Ollama**), Image Generation (**Stable Diffusion / Forge & ComfyUI**), **3D Mesh Generation (TRELLIS V2 & Hunyuan3D v2)**, **Video Generation (Wan 2.2, LTX-2.5, HunyuanVideo)**, and **Audio & Speech Generation (Kokoro TTS, AllTalk XTTS-v2, Faster-Whisper, Stable Audio Open 3.0, MusicGen, YuE)** on Windows, Linux, Mobile, and Web. +It features the official **`L³M²`** monochromatic brand identity, a high-contrast **Matte Carbon Design System**, a live **Dynamic Theming Engine** (Matte Carbon, OLED Black, Clean Light), **Magnetic Companion Windows** (`WindowSnapManager`), in-app **AI Assist** with multimodal diagnostics, **Real Engine Test Flight**, integrated **`playwright-layout-inspector`** automated visual audits, NVML CUDA real-time telemetry, **Hugging Face Hub** Multimodal discovery (GGUF, Text-to-Video, Image-to-Video, TTS, Text-to-Audio), **CivitAI** checkpoint downloads, **Multimodal Studio** with interactive 3D WebGL viewer, Video Player Preview, Audio Waveform Visualizer, a unified **Avalonia WebAssembly (WASM)** dashboard, **Modular Feature Packs** (`--with-video`, `--with-audio`), and an active **Model Context Protocol (MCP) Server** (`/mcp`). ![Dashboard Overview](docs/images/dashboard_desktop.png) @@ -9,32 +9,18 @@ It features the official **`L³M²`** monochromatic brand identity, a high-contr ## 🖥️ User Interface Layout & Dashboard Structure -The application features a dark Fluent Avalonia UI theme (`#0F172A`) organized into modular tabs: +The application features a dark Fluent Avalonia UI theme (`#0F172A`) organized into modular workspaces: -``` -+-----------------------------------------------------------------------------------------+ -| Local LLM Server Manager | -| GPU: NVIDIA GeForce RTX 4070 Ti SUPER -- 16 GB • Service Connected 🟢 [🔄 Refresh] | -| GPU VRAM Allocation: 4.2 GB / 16.0 GB (26.3%) | -| [========================-------------------------------------------------------------] | -+-----------------------------------------------------------------------------------------+ -| [🦙 Installed Models] [🤗 Hugging Face Hub] [🎨 CivitAI Models] [📦 Studio] [⚙️ Settings]| -+-----------------------------------------------------------------------------------------+ -| Ollama Local Model Library [🧹 Unload All VRAM] | -| | -| +-------------------------------------------------------------------------------------+ | -| | qwen2.5-coder:7b [Coding] [4.7 GB] Installed 🟢 | | -| +-------------------------------------------------------------------------------------+ | -| | llama3.2:latest [Chat] [2.0 GB] Installed 🟢 | | -| +-------------------------------------------------------------------------------------+ | -| | -| Interactive KV Cache Calculator ~0.5 GB | -| [====================================------------------------------------------------] | -| 8,192 tokens | -+-----------------------------------------------------------------------------------------+ -| LocalLLMServerManager v3.11.0 -- Unified WASM & Desktop UI System Tray Enabled 🟢 | -+-----------------------------------------------------------------------------------------+ -``` +| UI Workspace | Target Capabilities | Active Controls | +| :--- | :--- | :--- | +| **Telemetry Header** | Real-Time Hardware Telemetry | GPU name, total/used VRAM bar, service health indicator, and refresh button. | +| **Installed Models** | Local Model Management | Ollama model cards, family tags (`Coding`, `Chat`), and interactive KV Cache Context Calculator. | +| **Hugging Face Hub** | Multimodal GGUF Discovery | Search repositories, inspect branch quantization trees (`Q4_K_M`, `Q8_0`), and stream downloads. | +| **CivitAI Models** | Diffusion Checkpoints & LoRAs | Filter by model type, inspect preview thumbnails, and download directly to disk. | +| **Multimodal Studio** | Creative Generation Suite | Workspaces for Images, Video, Audio, 3D Mesh, and the one-click **Real Engine Test Flight** runner. | +| **AI Assistant** | In-App Conversational Copilot | Multimodal screenshot diagnostics, dynamic LiteLLM capability badges, and detachable companion window. | +| **Settings & Tools** | Configuration & Auto-Discovery | Multi-drive tool auto-detection, path status badges (`Valid`, `Missing`), and LAN IP endpoint summaries. | +| **Companion Windows** | Multi-Window Workspaces | Detachable Documentation and AI Assist windows with magnetic flank docking and lockstep dragging. | --- @@ -81,51 +67,69 @@ The application features a dark Fluent Avalonia UI theme (`#0F172A`) organized i 26. **CivitAI Integration** — Search by name, type (Checkpoint / LoRA / Embedding / VAE / ControlNet), and sort order. Shows preview thumbnails, download counts, and star ratings. 27. **Direct-to-Disk Downloads** — Stream CivitAI files directly to disk with live progress bars. -### Application Settings & Engine Controls +### Application Settings, Network & Engine Controls 28. **Flexible Path Configuration & Auto-Discovery** — Customize executable/script paths and model directories for Ollama, Stable Diffusion / Forge, ComfyUI, and Audio TTS Engine. Use the one-click "🔍 Auto-Detect Installed Tools" feature (or `POST /api/tools/detect`) to automatically scan common install locations across drives, with real-time path validation badges (`Valid` 🟢 / `Missing` 🔴 / `Unset` ⚪). -29. **Component Manager** — View, install, and uninstall optional feature packs (`ext_video`, `ext_audio`) directly from the Settings UI. +29. **Auto-Detected LAN IP & Network Endpoints** — Automatically discovers local area network addresses (e.g. `http://10.0.0.21:5246`) and publishes firewall-ready LAN MCP URLs for connected devices. +30. **Component Manager** — View, install, and uninstall optional feature packs (`ext_video`, `ext_audio`) directly from the Settings UI. + +### Magnetic Companion Windows & Multi-Window Workflow +31. **WindowSnapManager & Magnetic Flank Docking** — Detach the Documentation tab (left flank) or AI Assistant tab (right flank) into floating companion windows with smooth lockstep movement, proximity snap (< 32 px), and drag detachment (> 24 px). +32. **In-App AI Chat Assistant (AI Assist)** — Interactive copilot featuring dynamic LiteLLM capability badges (`👁️`, `⚡`, `1M`), multimodal screenshot paste (**Ctrl+V**), inline C# tool execution cards, and zero local GPU VRAM usage. +33. **Real Engine Test Flight** — Instant in-app verification runner testing Text, Image, Video, and Audio engine network connectivity and GPU buffer allocation before launching complex generative jobs. ### Infrastructure & Reverse Proxy -30. **YARP Reverse Proxy** — Transparently proxies Ollama (`:11434`), Forge (`:7860`), ComfyUI (`:8188`), and Audio Engine (`:8880`) traffic through a single unified endpoint (`:5246`). -31. **VRAM Orchestrator** — Auto-unloads active LLM models from GPU memory before heavy Stable Diffusion, ComfyUI 3D, or Video render jobs to prevent OOM errors. -32. **Background Engine Management** — UI controls to start/stop engines directly from the dashboard cleanly. -33. **Lazy Boot** — AI engines can boot lazily on-demand when first requested, conserving system resources when idle. +34. **YARP Reverse Proxy** — Transparently proxies Ollama (`:11434`), Forge (`:7860`), ComfyUI (`:8188`), and Audio Engine (`:8880`) traffic through a single unified endpoint (`:5246`). +35. **VRAM Orchestrator** — Auto-unloads active LLM models from GPU memory before heavy Stable Diffusion, ComfyUI 3D, or Video render jobs to prevent OOM errors. +36. **Background Engine Management** — UI controls to start/stop engines directly from the dashboard cleanly. +37. **Lazy Boot** — AI engines can boot lazily on-demand when first requested, conserving system resources when idle. --- ## 🏛️ System Architecture -``` - +-------------------------------------------------+ - | AI Assistants & External Clients | - | - Claude Desktop / Antigravity / Cursor / Agents| - | - Model Context Protocol Streamable HTTP / SSE | - +------------------------+------------------------+ - | JSON-RPC 2.0 (/mcp) - v - +----------------------------------------------+ - | Desktop Session (User Logon - Win/Linux) | - | - Avalonia UI System Tray Icon / Window | - | - Native XAML Dark Dashboard Window | - | - Auto-Attaches to local server (:5246) | - +----------------------+-----------------------+ - | REST / HTTP (:5246) - v -+-----------------------------------------------------------------------------------+ -| Local HTTP Server & Reverse Proxy Host | -| - ASP.NET Core Web API + YARP Reverse Proxy (:5246) | -| - Model Context Protocol (MCP) Server (/mcp) | -| - VRAM Orchestrator & Process Management | -| - Responsive Web Dashboard & WebGL 3D Studio (wwwroot) | -+------------------------------------+----------------------------------------------+ - | - v - +-----------------------------------+ - | Managed Processes | - | - Ollama (:11434) | - | - SD Forge (:7860) | - | - ComfyUI (:8188) | - +-----------------------------------+ +```mermaid +flowchart TD + subgraph ExternalClients["AI Assistants & External Clients"] + Claude["Claude Desktop / Antigravity / Cursor / Agents"] + WebClients["Browser & Mobile Clients (:5246)"] + end + + subgraph DesktopSession["Desktop Session (User Logon - Win/Linux)"] + Tray["Avalonia UI System Tray Icon"] + DesktopApp["Native Avalonia Dark Dashboard Window"] + CompanionWins["Magnetic Companion Windows (AI Assist & Docs)"] + end + + subgraph ServerHost["Local HTTP Server & Reverse Proxy Host (:5246)"] + ProgramHost["ASP.NET Core Web API + YARP Reverse Proxy"] + McpServer["Model Context Protocol (MCP) Server (/mcp)"] + VramOrch["VRAM Orchestrator & Telemetry Provider"] + TestFlight["Real Engine Test Flight Runner"] + WasmStatic["Avalonia WebAssembly & WebGL 3D Studio"] + end + + subgraph Engines["Managed Local AI Engines"] + Ollama["Ollama Engine (:11434)\nLLMs & Text Generation"] + Forge["Stable Diffusion Forge (:7860)\nCheckpoints & LoRAs"] + Comfy["ComfyUI Engine (:8188)\n3D Mesh, Video & Workflows"] + Audio["Kokoro TTS Engine (:8880)\nSpeech Synthesis & OpenAI API"] + end + + Claude -->|JSON-RPC 2.0 /mcp| McpServer + WebClients -->|HTTP / WebSocket| ProgramHost + DesktopApp <-->|Magnetic Snap & Lockstep| CompanionWins + DesktopApp -->|Local REST & IPC| ProgramHost + Tray -->|Tray IPC| ProgramHost + + ProgramHost --> VramOrch + ProgramHost --> TestFlight + ProgramHost --> WasmStatic + + VramOrch -.->|Unload VRAM: keep_alive: 0| Ollama + ProgramHost -->|YARP Proxy| Ollama + ProgramHost -->|YARP Proxy| Forge + ProgramHost -->|YARP Proxy| Comfy + ProgramHost -->|YARP Proxy| Audio ``` ### Dual-Session Lifecycle @@ -303,19 +307,14 @@ The Web Dashboard features a responsive CSS layout engine: ## 🧪 Quality Assurance, Test Coverage & Requirements Traceability -LocalLLMServerManager includes an automated test harness ensuring cross-platform stability across **Windows 11** and **Linux** environments: - -``` -+-----------------------------------------------------------------------------------------+ -| TOTAL TESTS EXECUTED : 174 | -| PASSED : 173 (99.4%) | -| SKIPPED : 1 (Playwright screenshot generator on-demand) | -| FAILED : 0 (0.0%) | -| TEST FIXTURE CLASSES : 20 | -| TEST FRAMEWORKS : .NET 10 LTS • xUnit v3 • Avalonia Headless • Microsoft Playwright| -| OPERATING SYSTEMS : Windows 11 x64 (Win32 Jobs) • Linux x64 (systemd / procfs / X11) | -+-----------------------------------------------------------------------------------------+ -``` +| Metric | Measured Value | Operational Notes | +| :--- | :--- | :--- | +| **Total Tests Executed** | `174` | Automated test suite across Windows and Linux | +| **Passed Tests** | `173 (99.4%)` | All unit, integration, and UI tests pass | +| **Skipped Tests** | `1 (0.6%)` | On-demand Playwright screenshot generator | +| **Failed Tests** | `0 (0.0%)` | Zero test failures across matrix | +| **Test Fixture Classes** | `20` | Partitioned test fixtures across 5 execution chunks | +| **Target Platforms** | Windows 11 x64, Linux x64, Headless Chromium | Dual-OS verified | * **[Full Test Coverage Specification](docs/TEST_COVERAGE.md)** — Detailed component-by-component coverage mapping across all 20 test classes, cross-platform validation matrix (Windows & Linux), and 5-chunk test execution guide. * **[Software Requirements Specification & Traceability Matrix](docs/REQUIREMENTS.md)** — Formal requirements specification across 12 functional domains (`CORE-xxx`, `LLM-xxx`, `HUB-xxx`, `DIFF-xxx`, `3D-xxx`, `VRAM-xxx`, `MCP-xxx`, `INST-xxx`, `DISC-xxx`, `UI-xxx`, `WASM-xxx`, `E2E-xxx`), mapping each requirement to source files and test assertions, plus explicit gap analysis. @@ -324,12 +323,15 @@ LocalLLMServerManager includes an automated test harness ensuring cross-platform ## 📚 Guides & Documentation -- [Full Test Coverage Specification](docs/TEST_COVERAGE.md) — Comprehensive test coverage mapping, metrics, cross-platform testing matrix, and execution guidelines. -- [Software Requirements Specification & RTM](docs/REQUIREMENTS.md) — Complete SRS with bidirectional Traceability Matrix mapping requirement IDs to tests and code, plus gap analysis. -- [Developer & Contributor Guide](docs/DEVELOPMENT_GUIDE.md) — Comprehensive guide on project layout, SOLID Avalonia XAML controls, design tokens, MVVM pattern, Minimal API endpoints, and testing. -- [System Architecture & Mermaid Diagrams](docs/ARCHITECTURE.md) — Visual architecture blueprints, component hierarchy, VRAM orchestration sequence diagrams, and service mapping matrices. -- [ComfyUI & 3D Mesh Generation Setup Guide](docs/COMFYUI_AND_3D_GUIDE.md) — How to configure ComfyUI, install 3D nodes (TRELLIS V2 / Hunyuan3D v2), and export custom workflow presets. -- [Linux Caddy Proxy & Open WebUI / LibreChat Integration Guide](docs/CADDY_OPENWEBUI_SETUP.md) — How to expose LocalLLMServerManager via Caddy reverse proxy to Open WebUI and LibreChat clients. +- [User Guide Hub](docs/USER_GUIDE.md) — Complete overview of all dashboard workspaces and engine controls. +- [Real Engine Test Flight Guide](docs/getting-started/test-flight.md) — Verify engine readiness and GPU clearance with one-click test flight payloads. +- [Magnetic Companion Windows Guide](docs/guide/companion-windows.md) — Floating helper windows, lockstep tracking, and magnetic flank snapping. +- [Remote Access & Reverse Proxy Guide](docs/getting-started/remote-access.md) — LAN IP configuration, SSH tunneling, and Caddy reverse proxy setup. +- [LoRA Art Styles & CivitAI](docs/studio/lora-styles.md) — Download and apply community style adapters in ComfyUI and SD Forge. +- [AI Chat Assistant Guide](docs/ai-and-mcp/assistant.md) — Configure LiteLLM gateway, dynamic model badges, and multimodal screenshot diagnostics. +- [Developer & Contributor Guide](docs/DEVELOPMENT_GUIDE.md) — Architecture layout, MVVM patterns, Minimal API endpoints, and Playwright tests. +- [System Architecture Blueprint](docs/ARCHITECTURE.md) — High-level architecture, component hierarchy, and VRAM sequence diagrams. +- [Documentation Writing Style Guide](docs/standards/ste-100.md) — ASD-STE100 writing rules for clarity, brevity, and active voice. --- @@ -357,13 +359,31 @@ We use **MAJOR.MINOR.PATCH** (SemVer): | `3.8.0` | Cross-Platform Tool Discovery (FFmpeg hardware encoder detection: NVENC, Intel QSV, VAAPI, AMD AMF; Kokoro Python environment inspection; Linux paths & shell runners), Dual-OS GitHub Actions CI Matrix (`[windows-latest, ubuntu-latest]`), Windows Service directory handling & Linux headless guard, and enhanced Windows & Linux installers with automated Firewall rule creation and LAN/MCP endpoint summaries | | `3.9.0` | Local Audio & Music Studio suite (Kokoro TTS, AllTalk XTTS-v2 voice cloning, Faster-Whisper STT with `/v1/audio/transcriptions` & `/v1/audio/translations`, ComfyUI MusicGen & Stable Audio Open presets, automated setup scripts, and `D:\AI\audio` storage isolation) | | `3.11.0` | Dynamic WebAssembly browser origin resolution via JSImport, centralized `HttpHelper` with `BaseAddress` validation, thread-safe model collection synchronization, dynamic engine health status indicators, headless UI interaction test suite, and enhanced browser E2E test harness | +| `3.15.0` | Magnetic Companion Windows (`WindowSnapManager`) with lockstep dragging, proximity snap, and multi-monitor detach; in-app AI Assist with LiteLLM capability discovery badges and multimodal screenshot analysis; Real Engine Test Flight verification runner for Text, Image, Video, and Audio backends; auto-detected LAN IP endpoints with LAN MCP URLs; optimized SettingsService async caching and hardware JSON lookup performance | --- ## 🚀 Installation & Downloads +### Testing & Setup Requirements Matrix + +Local LLM Server Manager uses a modular design. Testers only need to install components for the features they want to test: + +| Feature Area | Stack Requirement | Prerequisite Needed | What It Enables | +| :--- | :--- | :--- | :--- | +| **Core Manager & Dashboard** | **REQUIRED** | Windows 10/11 x64 or Linux x64 | Hardware telemetry, VRAM bar, system tray, reverse proxy, web dashboard. | +| **Ollama Engine** | *Optional* | [Ollama](https://ollama.com) installed | Local LLM text generation, GGUF downloads, KV cache calculator. | +| **Stable Diffusion Forge** | *Optional* | [SD Forge](https://github.com/lllyasviel/stable-diffusion-webui-forge) installed | Local image generation, CivitAI checkpoint and LoRA downloads. | +| **ComfyUI Engine** | *Optional* | [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed | 3D mesh reconstruction, video generation, and FLUX workflows. | +| **Kokoro TTS Engine** | *Optional* | Python environment or Audio Pack | Local speech synthesis with OpenAI-compatible audio API. | +| **AI Chat Assistant** | *Optional* | LiteLLM gateway or OpenAI endpoint | In-app assistant, multimodal screenshot analysis, and app control. | +| **Feature Packs (`ext_*`)** | *Optional* | Installed via Settings tab | Video ComfyUI presets (`ext_video`) and Audio workflows (`ext_audio`). | + +> [!IMPORTANT] +> The release package is **self-contained**. You do not need to install the .NET SDK or .NET runtime to run the application. + ### Option 1: Official Windows Installer (.exe) — Seamless In-Place Upgrades -Download the latest `LocalLLMServerManager-v3.5.0-Setup.exe` from the [GitHub Releases](https://github.com/spelech/LocalLLMServerManager/releases) page. +Download the latest `LocalLLMServerManager-Setup.exe` from the [GitHub Releases](https://github.com/spelech/LocalLLMServerManager/releases) page. * **In-Place Upgrades**: Running setup over an existing installation automatically stops any active `LocalLLMServerManager` Windows Service (`net stop`) and closes running tray processes, safely overwrites binaries without file lock errors, preserves your custom `settings.json`, and reconfigures & restarts the background service. * Includes an installation wizard with options for: * 🟢 **Install Windows Service** (Headless pre-logon machine boot) @@ -383,7 +403,7 @@ sudo ./install_linux.sh * Installs desktop launcher (`localllmmanager.desktop`) in your application menu ### Option 3: Standalone Portable (.zip / .tar.gz) -Download `LocalLLMServerManager-v3.5.0-win-x64.zip` or `LocalLLMServerManager-v3.5.0-linux-x64.tar.gz` from Releases, extract, and run executable. Includes bundled runtime — no .NET SDK required! +Download `LocalLLMServerManager-win-x64.zip` or `LocalLLMServerManager-linux-x64.tar.gz` from Releases, extract, and run executable. Includes bundled runtime — no .NET SDK required! ### Option 4: Building Release Packages Locally - **Windows:** Run `.\build_release.ps1` (or `.\scripts\update.ps1` for in-place local build & upgrade) @@ -494,10 +514,16 @@ Dashboard available at **http://localhost:5246/** --- -## 🔧 Prerequisites +## 🔧 Prerequisites & Engine Links + +For complete installation steps and testing requirements, see the [Testing & Setup Requirements Matrix](#testing--setup-requirements-matrix) above. + +- **Operating System**: Windows 10/11 (64-bit) or Linux (Ubuntu 22.04+, Debian 12+, Fedora 38+) +- **GPU Acceleration**: NVIDIA GPU with CUDA support (8 GB+ VRAM recommended; CPU inference supported for text models) +- **[Ollama](https://ollama.com/)** *(Optional)* — Local LLM inference engine for chat and code generation +- **[Stable Diffusion WebUI Forge](https://github.com/lllyasviel/stable-diffusion-webui-forge)** *(Optional)* — Checkpoint and LoRA image generation backend +- **[ComfyUI](https://github.com/comfyanonymous/ComfyUI)** *(Optional)* — Node-based 3D mesh, video, and complex diffusion backend +- **[LiteLLM Proxy](https://github.com/BerriAI/litellm)** *(Optional)* — External gateway for the in-app AI Assistant +- **[.NET 10 SDK](https://dotnet.microsoft.com/download/dotnet/10)** *(Optional)* — Only required if building the application from source code -- **[Ollama](https://ollama.com/)** — Local LLM inference runtime -- **[Stable Diffusion WebUI Forge](https://github.com/lllyasviel/stable-diffusion-webui-forge)** *(optional)* — SD image generation backend -- **[ComfyUI](https://github.com/comfyanonymous/ComfyUI)** *(optional)* — Node-based 3D mesh & image generation backend -- **[.NET 10 SDK](https://dotnet.microsoft.com/download/dotnet/10)** *(optional)* — Only required if compiling from source code diff --git a/docs/.vitepress/config.mts b/docs/.vitepress/config.mts index 702c7d7..a08df77 100644 --- a/docs/.vitepress/config.mts +++ b/docs/.vitepress/config.mts @@ -33,6 +33,9 @@ export default defineConfig({ { text: 'Installation', link: '/getting-started/installation' }, { text: 'Configuration', link: '/getting-started/configuration' }, { text: 'Quickstart Guide', link: '/getting-started/quickstart' }, + { text: 'Can I Run It (Hardware Fit)', link: '/guide/can-i-run-it' }, + { text: 'Real Engine Test Flight', link: '/getting-started/test-flight' }, + { text: 'Remote Access & Reverse Proxy', link: '/getting-started/remote-access' }, { text: 'Troubleshooting', link: '/getting-started/troubleshooting' } ] } @@ -42,6 +45,7 @@ export default defineConfig({ text: 'Engines & Model Management', items: [ { text: 'Engines Overview & VRAM', link: '/engines/' }, + { text: 'Can I Run It Calculator', link: '/guide/can-i-run-it' }, { text: 'Ollama LLM Engine', link: '/engines/ollama' }, { text: 'Stable Diffusion Forge', link: '/engines/sd-forge' }, { text: 'ComfyUI Engine', link: '/engines/comfyui' }, @@ -56,9 +60,11 @@ export default defineConfig({ items: [ { text: 'Studio Overview', link: '/studio/' }, { text: 'Image Generation', link: '/studio/image-generation' }, + { text: 'LoRA Art Styles & CivitAI', link: '/studio/lora-styles' }, { text: 'Video Generation', link: '/studio/video-generation' }, { text: 'Audio & Music Synthesis', link: '/studio/audio-and-music' }, - { text: '3D Mesh Reconstruction', link: '/studio/3d-mesh' } + { text: '3D Mesh Reconstruction', link: '/studio/3d-mesh' }, + { text: 'ComfyUI & 3D Setup', link: '/studio/comfyui-setup' } ] } ], @@ -68,6 +74,7 @@ export default defineConfig({ items: [ { text: 'Overview', link: '/ai-and-mcp/' }, { text: 'AI Chat Assistant', link: '/ai-and-mcp/assistant' }, + { text: 'Magnetic Companion Windows', link: '/guide/companion-windows' }, { text: 'Model Context Protocol (MCP)', link: '/ai-and-mcp/mcp-tools' }, { text: 'Workflow Presets', link: '/ai-and-mcp/flows-and-presets' } ] @@ -80,6 +87,7 @@ export default defineConfig({ { text: 'Technical Overview', link: '/technical/' }, { text: 'System Architecture', link: '/technical/architecture' }, { text: 'Development & Build Guide', link: '/technical/development' }, + { text: 'Magnetic Companion Windows', link: '/guide/companion-windows' }, { text: 'Collaborative UI Debugging', link: '/guide/collaborative-debugging' }, { text: 'Requirements Specification', link: '/technical/requirements' }, { text: 'Windows Process Validation', link: '/technical/validation' }, @@ -95,6 +103,7 @@ export default defineConfig({ { text: 'Technical Overview', link: '/technical/' }, { text: 'System Architecture', link: '/technical/architecture' }, { text: 'Development & Build Guide', link: '/technical/development' }, + { text: 'Magnetic Companion Windows', link: '/guide/companion-windows' }, { text: 'Collaborative UI Debugging', link: '/guide/collaborative-debugging' }, { text: 'Requirements Specification', link: '/technical/requirements' }, { text: 'Windows Process Validation', link: '/technical/validation' }, diff --git a/docs/CADDY_OPENWEBUI_SETUP.md b/docs/CADDY_OPENWEBUI_SETUP.md index 1f6c2ec..73bb909 100644 --- a/docs/CADDY_OPENWEBUI_SETUP.md +++ b/docs/CADDY_OPENWEBUI_SETUP.md @@ -12,7 +12,7 @@ This document contains the Caddyfile configuration needed on the remote proxy ma Pass this Caddyfile block to the agent managing your remote Caddy server. Ensure you replace `yourdomain.com` with your actual domain, and verify the `tinyauth.sock` path for your environment. -```caddyfile +```txt # Local LLM Server Manager Dashboard & Proxy manager.yourdomain.com { forward_auth * unix//var/run/tinyauth.sock { diff --git a/docs/TEST_COVERAGE.md b/docs/TEST_COVERAGE.md index 0e76378..f9b4d8b 100644 --- a/docs/TEST_COVERAGE.md +++ b/docs/TEST_COVERAGE.md @@ -8,16 +8,14 @@ This document provides a comprehensive audit of all unit, integration, mock serv ## 📊 Executive Summary & Test Metrics -``` -+-----------------------------------------------------------------------------------------+ -| TOTAL TESTS EXECUTED : 174 | -| PASSED : 173 (99.4%) | -| SKIPPED : 1 (Playwright screenshot generator on-demand) | -| FAILED : 0 (0.0%) | -| TEST FIXTURE CLASSES : 20 | -| TARGET RUNTIMES : Windows 11 x64, Linux x64 (systemd, X11, Wayland), Chromium Headless| -+-----------------------------------------------------------------------------------------+ -``` +| Metric | Measured Value | Operational Notes | +| :--- | :--- | :--- | +| **Total Tests Executed** | `174` | Comprehensive test suite across Windows and Linux | +| **Passed Tests** | `173 (99.4%)` | All unit, integration, and UI tests pass | +| **Skipped Tests** | `1 (0.6%)` | On-demand Playwright screenshot generator | +| **Failed Tests** | `0 (0.0%)` | Zero test failures across test suite | +| **Test Fixture Classes** | `20` | Modular test fixtures partitioned by domain | +| **Target Runtimes** | Windows 11 x64, Linux x64 (systemd, X11, Wayland), Headless Chromium | Dual-OS verified | ### Testing Frameworks & Tooling * **Test Runner**: [xUnit.net v3](https://xunit.net/) (`xunit.v3` 3.2.2) @@ -32,20 +30,14 @@ This document provides a comprehensive audit of all unit, integration, mock serv To eliminate port contention and process memory race conditions during local and CI/CD test runs on Windows and Linux, the test suite is partitioned into five targeted execution chunks: -``` - ┌─────────────────────────────────────────────────────────┐ - │ LocalLLMServerManager.Tests │ - │ (174 Tests) │ - └────────────────────────────┬────────────────────────────┘ - │ - ┌──────────────────┬──────────────────┼──────────────────┬──────────────────┐ - ▼ ▼ ▼ ▼ ▼ - ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ - │ Chunk 1 │ │ Chunk 2 │ │ Chunk 3 │ │ Chunk 4 │ │ Chunk 5 │ - │ ViewModels │ │ Services │ │ Endpoints │ │ MCP Server │ │ Playwright │ - │ & Settings │ │ & Discovery │ │ & Workflows │ │ & Tools │ │ WASM E2E │ - │ (37 Tests) │ │ (46 Tests) │ │ (67 Tests) │ │ (22 Tests) │ │ (2 Tests) │ - └──────────────┘ └──────────────┘ └──────────────┘ └──────────────┘ └──────────────┘ +```mermaid +flowchart TD + Root["LocalLLMServerManager.Tests\n(174 Tests)"] + Root --> C1["Chunk 1: ViewModels & Settings\n(37 Tests)"] + Root --> C2["Chunk 2: Services & Discovery\n(46 Tests)"] + Root --> C3["Chunk 3: Endpoints & Workflows\n(67 Tests)"] + Root --> C4["Chunk 4: MCP Server & Tools\n(22 Tests)"] + Root --> C5["Chunk 5: Playwright WASM E2E\n(2 Tests)"] ``` ### Execution Commands diff --git a/docs/USER_GUIDE.md b/docs/USER_GUIDE.md index 5d6d299..aab099d 100644 --- a/docs/USER_GUIDE.md +++ b/docs/USER_GUIDE.md @@ -1,126 +1,180 @@ -# Local LLM Server Manager — Detailed User Guide +# Local LLM Server Manager — User Guide Hub -Welcome to the **Local LLM Server Manager (v3.11.0)**. This guide will walk you through the main tabs of the dashboard, showing you how to manage your local AI engines (Ollama, Stable Diffusion / Forge, ComfyUI, and Kokoro TTS), configure your settings, and successfully generate text, images, 3D models, video, and speech. +Welcome to the **Local LLM Server Manager (v3.15.0)** User Guide. This document provides a complete guide to operating the dashboard, configuring local AI engines, and using generative studio tools. --- -## 1. My Models (Dashboard & VRAM Orchestrator) +## Workspace Navigation -The **My Models** tab is your home base for monitoring your system's hardware telemetry and active language models. +The desktop application organizes capabilities into dedicated workspaces with docked companion windows: -![Desktop Dashboard Overview](images/dashboard_desktop.png) +```mermaid +flowchart TD + App["Local LLM Server Manager (Port 5246)"] + App --> Tab1["📦 Models\nDownloaded, Hugging Face & CivitAI"] + App --> Tab2["⚡ Workflows\nImages, Text, Video, 3D Mesh & Audio"] + App --> Tab3["🔍 Can I Run It\nHardware Compatibility & Sizing"] + App --> Tab4["⚙️ Settings\nDiscovery, Themes, LAN & Feature Packs"] + App -.-> CompL["📖 Documentation\n(Left Flank Companion)"] + App -.-> CompR["🤖 AI Assist\n(Right Flank Companion)"] +``` -### Features: -* **VRAM Monitor:** At the top of the screen, you will see a visual representation of your GPU's VRAM. It accurately reads your hardware (e.g. `NVIDIA GeForce RTX 4070 Ti SUPER — 16 GB`) and shows a stacked bar representing free vs. used memory. -* **Model Capabilities Profile:** Installed models are grouped and tagged with capabilities (e.g., *Coding*, *Reasoning*, *Chat*). -* **KV Cache Estimator:** Click on a model to view its details, where you can use the **Context Length Slider** to estimate how much VRAM a specific context length (up to 32K tokens) will require. -* **VRAM Orchestrator:** If your GPU gets full, the orchestrator will automatically unload idle text models when you try to run heavy image/video generations in the background. +--- -![Ollama Installed Models](images/dashboard_ollama.png) +## 1. Hardware Telemetry & My Models -**How to Use:** -To run a model in a frontend chat UI (like Open WebUI or LibreChat), simply select the model there. The Local LLM Server Manager proxy will wake up Ollama on port `11434` and transparently route the generation request while updating your VRAM usage bar live. +The **Models** workspace monitors active hardware metrics and manages local model weights. + +![Desktop Dashboard Overview](images/dashboard_desktop.png) + +### Key Capabilities +* **Live VRAM Bar**: Displays total, used, and free GPU memory in real time via NVML CUDA telemetry. +* **Model Capability Badges**: Identifies model capabilities (e.g., `Coding & General`, `Reasoning`, `Math`). +* **Interactive KV Cache Estimator**: Drag the context length slider (up to 32,768 tokens) to preview memory consumption before loading models. +* **VRAM Orchestrator**: Automatically frees GPU memory before heavy diffusion or 3D tasks start. +* **Unload All VRAM Button**: Releases all active models from GPU memory with a single click. + +> [!TIP] +> Read the complete [Engines & VRAM Guide](./engines/index.md) and [Ollama Engine Guide](./engines/ollama.md). --- -## 2. Find & Download Models +## 2. Model Discovery & Downloads + +Download models directly without opening a web browser or using terminal commands. + +### Hugging Face Hub (GGUF & Multimodal) +Search community repositories, compare quantization levels (`Q4_K_M`, `Q8_0`), filter by input/output modalities, and stream downloads to disk. -This tab integrates directly with the **Hugging Face Hub (GGUF)** and the **Ollama Official Library** so you can pull models natively without touching the command line. +![Hugging Face Hub Search](images/dashboard_huggingface.png) -![Hugging Face Hub Model Search](images/dashboard_huggingface.png) +### CivitAI Model Hub (Checkpoints & LoRAs) +Search diffusion checkpoints, LoRA style adapters, and VAE models with real-time download counters and hardware compatibility badges. -### Sub-tabs: -* **Hugging Face Hub (GGUF):** Search for community-quantized models. - * *Sample search:* Type `DeepSeek-R1-Distill-GGUF` or `Llama-3.2`. - * Click on a model repository to view a list of quantization sizes (e.g., `Q4_K_M` which is a good balance of size and quality, or `Q8_0` for high fidelity). - * Select your desired `.gguf` file and hit **Pull Selected**. The dashboard will stream the download progress directly to your screen. -* **Ollama Library:** Provides one-click pulls for the most popular models. - * *Sample models:* Click **Pull 14B (7.9 GB)** under the `phi3` card, or **Pull 7B (4.7 GB)** under `qwen2.5-coder`. +![CivitAI Models Hub](images/dashboard_civitai.png) -**Changing Settings (Multiple Models):** -Ollama can run multiple models simultaneously. To do this, edit your system environment variables and set `OLLAMA_MAX_LOADED_MODELS=3`. +> [!TIP] +> Read the [Model Hubs & Downloads Guide](./engines/model-management.md) and [LoRA Art Styles Guide](./studio/lora-styles.md). --- -## 3. Stable Diffusion (CivitAI Integration) +## 3. Hardware Fit Calculator (Can I Run It) -The Stable Diffusion tab allows you to configure your Forge engine and seamlessly download new image generation assets from **CivitAI**. +The **Can I Run It** workspace estimates whether an AI model fits within your system memory before downloading files. -![CivitAI Stable Diffusion Asset Manager](images/dashboard_civitai.png) +![Can I Run It Hardware Fit Calculator](images/dashboard_can_i_run_it.png) -### Configuring Your Engine: -1. At the top of the tab, look for **Forge / SD Models Directory**. -2. Type in your absolute path (e.g. `C:\AI\SD_Forge\models`). -3. Click **Save Path**. This updates your `settings.json` so downloads go exactly where they need to. -4. You can boot or stop the SD Forge engine directly using the **Boot SD Forge** and **Stop SD Forge** UI controls. The app relies on Win32 Job Objects to manage these background processes cleanly. +### Key Capabilities +* **Live GPU Detection**: Queries your graphics card and system memory automatically. +* **Multi-Modality Sizing**: Calculates memory consumption for Text LLMs, Diffusion Images, Video, Audio, and 3D Mesh models. +* **Visual Allocation Bar**: Color-coded breakdown of Model Weights, Context/KV Cache, CUDA Overhead, and Free Headroom. +* **Layer Offloading Calculation**: Predicts the exact number of transformer layers that fit in GPU VRAM versus CPU RAM. +* **Performance Throughput**: Provides real-time token per second estimates for your hardware. -### CivitAI Search & Download: -* **Search:** Look up popular models. *Sample text:* `"RealisticVision"`, `"DreamShaper"`, or `"Flux"`. -* **Filters:** Use the dropdowns to search by type (*Checkpoint, LoRA, VAE, Embedding*) or Sort by *Highest Rated*. -* **Download:** Click on a result to view its details and thumbnail. Choose the specific version you want, and click **⬇ Download to Forge**. The file will stream directly to your configured directory. +> [!TIP] +> Read the dedicated [Can I Run It Hardware Fit Guide](./guide/can-i-run-it.md). --- -## 4. Multimodal Studio (Images, 3D Mesh, Video & Audio) +## 4. Multimodal Generation Workflows -The Studio tab provides a unified creative environment for image generation, 3D mesh reconstruction, video synthesis, and audio generation via ComfyUI and local AI engines. +The **Workflows** workspace provides generation pipelines across five creative modalities: -![3D ComfyUI Studio & WebGL Canvas](images/dashboard_3d_studio.png) +![AI Generation Workflows](images/dashboard_3d_studio.png) -### Studio Modes: -1. **🎨 Images**: Generate high-fidelity images using FLUX, SDXL, or SD 1.5 with ComfyUI or SD Forge. -2. **📦 3D Mesh**: Reconstruct 3D meshes using **TRELLIS V2** and **Hunyuan3D v2** with interactive 360° orbital WebGL viewer. -3. **🎬 Video Generation**: - * **Supported Models**: **Wan 2.2** (Text-to-Video and Image-to-Video), **LTX-2.5**, and **HunyuanVideo 1.5**. - * **Controls**: Set prompt, negative prompt, resolution (Width x Height), frame count, FPS, and seed. - * **Interactive Video Player**: Preview generated MP4 videos directly in the desktop or web dashboard with scrub controls, loop toggle, playback speed selector (0.5x–2x), and metadata badges. -4. **🎵 Audio & Speech Synthesis**: - * **Kokoro TTS Engine**: Text-to-Speech synthesis with 50+ high-quality voices (`af_heart`, `am_adam`, etc.) and native OpenAI-compatible `/v1/audio/speech` proxy. - * **Stable Audio Open 3.0**: Generate realistic sound effects, ambient audio, and instrumental samples. - * **YuE Music Generator**: Generate full-length songs with dual-track lyrics and melody generation. - * **Waveform Visualizer**: Live interactive waveform display with scrub bar, duration badges, and download button. +| Modality | Supported Models | Output Formats | Dedicated Guide | +| :--- | :--- | :--- | :--- | +| **Image Generation** | FLUX.1, SDXL, SD 1.5 | PNG, WebP | [Image Generation Guide](./studio/image-generation.md) | +| **Video Generation** | Wan 2.2, LTX-Video 2.5, HunyuanVideo | MP4 | [Video Generation Guide](./studio/video-generation.md) | +| **Audio & Speech** | Kokoro TTS, Stable Audio Open, YuE | WAV, MP3 | [Audio & Music Guide](./studio/audio-and-music.md) | +| **3D Mesh** | TRELLIS V2, Hunyuan3D v2 | GLB, OBJ | [3D Mesh Guide](./studio/3d-mesh.md) | + +### Real Engine Test Flight +Before starting complex renders, use the **Test Flight** control panel in the studio header: +1. Select your target modality (**Text**, **Image**, **Video**, or **Audio**). +2. Choose a starter prompt. +3. Click **🚀 Launch Test Flight**. +4. The system validates network readiness and GPU memory in seconds. + +> [!TIP] +> Read the [Real Engine Test Flight Guide](./getting-started/test-flight.md). --- -## 5. Settings, Engine Controls & Modular Feature Packs +## 5. In-App AI Assistant & Companion Windows -The Settings tab provides centralized configuration for engine endpoints, model directories, and optional component management. +The **AI Assistant** workspace provides interactive guidance, screenshot diagnostics, and app control without consuming local GPU memory. -![Application Settings & Environment Configuration](images/dashboard_settings.png) +```mermaid +flowchart LR + LeftCompanion["Documentation Window\n(Left Flank)"] <-->|Magnetic Proximity Snap| MainWindow["Main Dashboard Window\n(Center)"] + MainWindow <-->|Magnetic Proximity Snap| RightCompanion["AI Assist Window\n(Right Flank)"] +``` -### Key Features: -* **🔍 Auto-Detect Installed Tools**: One-click scanner that discovers Ollama, ComfyUI, Forge, and Kokoro TTS installations across all storage drives. -* **Feature Packs (Modular Components)**: - * **Core**: Ollama LLM management, Hugging Face Hub, CivitAI downloader. - * **Video Feature Pack (`ext_video`)**: Video ComfyUI workflow presets, models, and video player preview tools. - * **Audio Feature Pack (`ext_audio`)**: Kokoro TTS engine scripts, Stable Audio presets, and waveform audio player. - * You can install or uninstall optional packs on-demand directly from the Settings tab without restarting your system. -* **Connection Endpoints**: Configure HTTP ports for Ollama (`:11434`), SD Forge (`:7860`), ComfyUI (`:8188`), and Audio Engine (`:8880`). -* **Storage Paths**: Set custom directories for models, output videos, audio files, and 3D meshes. +### Key Capabilities +* **External Gateway**: Connects to LiteLLM or Vertex AI Gemini Flash to keep local GPU memory free for generation. +* **Multimodal Attachments**: Paste screenshots (**Ctrl+V**) to diagnose ComfyUI errors or review outputs. +* **Magnetic Companion Windows**: Detach the assistant into a floating window that docks magnetically to the right flank. +* **Lockstep Movement**: Moving the main window moves docked companion windows automatically. + +> [!TIP] +> Read the [AI Chat Assistant Guide](./ai-and-mcp/assistant.md) and [Magnetic Companion Windows Guide](./guide/companion-windows.md). --- -## 6. Cross-Platform Linux & Remote SSH Workflow +## 6. Application Settings & Engine Controls -Local LLM Server Manager supports native Linux execution and remote SSH viewing. +The **Settings** workspace centralizes engine paths, port bindings, and optional component management. -### Running on Linux Desktop (Native GUI) -Launch the desktop application directly from your Linux terminal or application menu: -```bash -localllmmanager -``` -This opens the native Avalonia UI dark desktop window on X11 and Wayland display environments. - -### Working Remotely over SSH -To access all features of the manager from a remote machine: -1. Establish an SSH connection with port forwarding for port `5246`: - ```bash - ssh -L 5246:localhost:5246 user@linux-host - ``` -2. Start the headless service on the remote Linux host: - ```bash - sudo systemctl start localllmmanager - # or run directly: dotnet run -- --service - ``` -3. Open `http://localhost:5246` in your local client browser. You get 100% of the UI functionality (VRAM telemetry, Hugging Face search, CivitAI downloader, 3D WebGL viewer) at native speed with zero lag over SSH. +![Application Settings](images/dashboard_settings.png) + +### Key Settings +* **Auto-Detect Installed Tools**: Scans system drives to locate Ollama, ComfyUI, Forge, and Kokoro TTS automatically. +* **Modular Feature Packs**: Install or remove `ext_video` and `ext_audio` packages on demand. +* **Network Endpoints**: Displays auto-detected LAN IP addresses and remote MCP connection URLs. + +> [!TIP] +> Read the [First-Time Configuration Guide](./getting-started/configuration.md) and [Remote Access Guide](./getting-started/remote-access.md). + +--- +## Comprehensive Guides Index + +- **Getting Started**: + - [Overview & Requirements](./getting-started/index.md) + - [Installation Guide](./getting-started/installation.md) + - [First-Time Configuration](./getting-started/configuration.md) + - [Quickstart Guide](./getting-started/quickstart.md) + - [Can I Run It Hardware Fit](./guide/can-i-run-it.md) + - [Real Engine Test Flight](./getting-started/test-flight.md) + - [Remote Access & Reverse Proxy](./getting-started/remote-access.md) + - [Troubleshooting](./getting-started/troubleshooting.md) +- **Engines & Models**: + - [Engines & VRAM Overview](./engines/index.md) + - [Ollama LLM Engine](./engines/ollama.md) + - [Stable Diffusion Forge](./engines/sd-forge.md) + - [ComfyUI Engine](./engines/comfyui.md) + - [Kokoro TTS Engine](./engines/kokoro-tts.md) + - [Model Hubs & Downloads](./engines/model-management.md) +- **Multimodal Studio**: + - [Studio Overview](./studio/index.md) + - [Image Generation](./studio/image-generation.md) + - [LoRA Art Styles & CivitAI](./studio/lora-styles.md) + - [Video Generation](./studio/video-generation.md) + - [Audio & Music Synthesis](./studio/audio-and-music.md) + - [3D Mesh Reconstruction](./studio/3d-mesh.md) + - [ComfyUI & 3D Setup](./studio/comfyui-setup.md) +- **AI & MCP Automation**: + - [AI & MCP Overview](./ai-and-mcp/index.md) + - [AI Chat Assistant](./ai-and-mcp/assistant.md) + - [Magnetic Companion Windows](./guide/companion-windows.md) + - [Model Context Protocol (MCP) Tools](./ai-and-mcp/mcp-tools.md) + - [Workflow Presets](./ai-and-mcp/flows-and-presets.md) +- **Technical Reference**: + - [System Architecture](./technical/architecture.md) + - [Development & Build Guide](./technical/development.md) + - [Collaborative UI Debugging](./guide/collaborative-debugging.md) + - [Requirements Specification](./technical/requirements.md) + - [Windows Process Validation](./technical/validation.md) + - [Test Coverage Benchmarks](./technical/test-coverage.md) + - [AI Assistant Internals](./technical/ai-assistant-internals.md) diff --git a/docs/ai-and-mcp/assistant.md b/docs/ai-and-mcp/assistant.md index 6f0d157..cb5594e 100644 --- a/docs/ai-and-mcp/assistant.md +++ b/docs/ai-and-mcp/assistant.md @@ -18,19 +18,28 @@ Follow these steps to access the assistant interface: 2. Click the **AI Assistant** tab in the top navigation bar. 3. Verify that the chat view loads with the header toolbar, suggestion chips, and composer bar. +```mermaid +flowchart TD + Header["🤖 AI Assist Header: Model Selector | ⚙️ Setup & Endpoint | ⧉ Pop Out"] + History["Chat Message History Area (Dialogue, Tool Cards, Error Diagnostics)"] + Chips["Suggestion Chips: [⚡ Check live VRAM] [🦙 Can I run Llama 70B?] [🎨 Generate image]"] + Composer["Composer Bar: [📎 Attach] [Enter prompt here...] [👁️ ⚡ 1M] [🚀 Send]"] + + Header --> History + History --> Chips + Chips --> Composer ``` -+-----------------------------------------------------------------------+ -| 🤖 AI Assist Model: gemini-2.5-flash [Setup]| -+-----------------------------------------------------------------------+ -| | -| [Chat Message History Area] | -| | -+-----------------------------------------------------------------------+ -| [⚡ Check live VRAM] [🦙 Can I run Llama 70B?] [🎨 Generate image] | -+-----------------------------------------------------------------------+ -| [📎] [Enter prompt here...] [👁️ ⚡ 1M] [🚀 Send] | -+-----------------------------------------------------------------------+ -``` + +### Interface Component Breakdown + +| Interface Component | Location | Operational Function | +| :--- | :--- | :--- | +| **Model Selector** | Header Bar | Selects active model with capability icons (Vision `👁️`, Tool Calling `⚡`, Context `1M`). | +| **Setup & Endpoint** | Top Right | Opens connection settings dialog for LiteLLM URL and API key. | +| **Pop Out Button** | Top Right | Detaches AI Assist into a magnetic companion window snapped to the right flank. | +| **Quick Action Chips** | Above Composer | One-click prompts for common operations (VRAM check, hardware fitting, image creation). | +| **Image Attachment** | Composer Bar | Attaches PNG/JPG screenshots via file picker or clipboard paste (**Ctrl+V**). | +| **Send Button** | Composer Bar | Submits prompt and streams tokens back into chat history. | --- @@ -109,13 +118,19 @@ Staged images appear in a preview tray directly above the text box: 3. Type your question (for example: *"Review this error message and suggest the correct resolution"*). 4. Click **🚀 Send** or press **Enter**. -``` -+-----------------------------------------------------------------------+ -| Staged Attachments: | -| [🖼️ error_screenshot.png ✕] [🖼️ comfyui_graph.png ✕] | -+-----------------------------------------------------------------------+ -| [📎] Why did this ComfyUI node fail? [👁️ ⚡ 1M] [🚀 Send] | -+-----------------------------------------------------------------------+ +```mermaid +flowchart TD + subgraph StagedTray["Staged Attachment Tray"] + Img1["🖼️ error_screenshot.png [✕ Remove]"] + Img2["🖼️ comfyui_graph.png [✕ Remove]"] + end + subgraph PromptBar["Input Composer Bar"] + Attach["📎 Attach File"] + Input["Text Input: 'Why did this ComfyUI node fail?'"] + Cap["Badge: 👁️ ⚡ 1M"] + Send["🚀 Send Button"] + end + StagedTray --> PromptBar ``` --- @@ -124,29 +139,40 @@ Staged images appear in a preview tray directly above the text box: When you ask the assistant to perform an action, the model executes native C# tools autonomously. +```mermaid +sequenceDiagram + autonumber + actor User as User + participant Chat as AI Assist Interface + participant Gateway as External LLM (LiteLLM / Gemini) + participant Host as Server Manager Host + + User->>Chat: "Can I run Llama 3.3 70B on my current GPU?" + Chat->>Gateway: POST /v1/chat/completions (tools: [calculate_hardware_fit]) + Gateway-->>Chat: Tool Call: calculate_hardware_fit(model: "llama3.3:70b") + Chat->>Host: Execute C# Tool calculate_hardware_fit + Host-->>Chat: Result: { verdict: "PERFECT FIT", vramRequiredMb: 41200 } + Note over Chat: Renders Inline Tool Execution Card (12 ms) + Chat->>Gateway: POST tool result payload + Gateway-->>Chat: Stream final advice ("Llama 3.3 70B fits comfortably...") + Chat-->>User: Display rendered answer +``` + ### Inline Execution Cards During tool calling, the assistant displays an execution card inside the chat response bubble: -* **Tool Name**: Displays the executed function (e.g., `get_gpu_vram_telemetry`). -* **Execution Time**: Shows tool duration in milliseconds. +* **Tool Name**: Displays the executed function (e.g., `calculate_hardware_fit`). +* **Execution Time**: Shows tool duration in milliseconds (e.g., `12 ms`). * **Arguments**: Details the parameter values sent to the tool. * **Result**: Displays the response data returned to the model. -``` -+-------------------------------------------------------------------+ -| 🤖 AI Assist 14:32 | -| | -| +---------------------------------------------------------------+ | -| | ⚡ Tool: calculate_hardware_fit 12 ms | | -| | Args: {"modelName": "llama3.3:70b", "quantization": "Q4_K_M"} | | -| | Result: {"verdict": "PERFECT FIT", "vramRequiredMb": 41200} | | -| +---------------------------------------------------------------+ | -| | -| Llama 3.3 70B in Q4_K_M quantization fits comfortably in your | -| configured 48 GB VRAM pool. | -+-------------------------------------------------------------------+ -``` +| Execution Card Field | Example Content | Purpose | +| :--- | :--- | :--- | +| **Tool Header** | `⚡ Tool: calculate_hardware_fit` | Identifies the executed C# tool. | +| **Execution Latency**| `12 ms` | Reports internal tool execution duration. | +| **Input Arguments** | `{"modelName": "llama3.3:70b", "quantization": "Q4_K_M"}` | Shows parameters passed by the LLM. | +| **Output Result** | `{"verdict": "PERFECT FIT", "vramRequiredMb": 41200}` | Shows structured payload passed back to the model. | ### Safety and Approval Boundaries @@ -177,8 +203,27 @@ Click any suggestion chip above the composer to trigger pre-built diagnostic tas --- +## Floating Pop-Out Mode & Magnetic Docking + +The AI Assistant can detach from the tab row into a floating companion window (`AiAssistWindow`). + +### Detach and Dock +1. Click the pop-out button (**⧉ Pop Out**) in the AI Assistant header bar. +2. The assistant window pops out and docks to the **right flank** of the main window. +3. The `WindowSnapManager` binds the companion window in lockstep with the main window. +4. Dragging the main window moves the AI Assistant companion window automatically. +5. Click **`🧲 Attached`** in the companion window title bar to detach the window. +6. Drag the companion window within 32 pixels of the right flank to re-snap automatically. +7. Click the close button (**✕**) in the companion title bar to return the assistant to the main tab layout. + +> [!TIP] +> Read the complete [Magnetic Companion Windows Guide](../guide/companion-windows.md) to learn about multi-monitor workflows and proximity thresholds. + +--- + ## Related Documentation +* [Magnetic Companion Windows Guide](../guide/companion-windows.md): Learn about lockstep tracking and proximity snapping. * [Model Context Protocol (MCP) Tools](./mcp-tools.md): Learn how external agents access application tools. * [Living Prompts & Workflow Presets](./flows-and-presets.md): Customize system prompts and automated workflows. * [AI Assistant Technical Internals](../technical/ai-assistant-internals.md): Read the technical architecture specification. diff --git a/docs/getting-started/index.md b/docs/getting-started/index.md index 95e4541..44b94d5 100644 --- a/docs/getting-started/index.md +++ b/docs/getting-started/index.md @@ -58,15 +58,21 @@ The manager coordinates these engines through a unified reverse proxy on port `5 Follow these sequential steps to set up and use Local LLM Server Manager: 1. **[Installation Guide](./installation.md)** - Install the application on Windows using the official installer, or on Linux using the automated installation script. + Review the required versus optional component matrix and install the application on Windows or Linux. 2. **[First-Time Configuration](./configuration.md)** Auto-detect installed engines, configure engine port numbers, and set your model storage directories. 3. **[Quickstart Guide](./quickstart.md)** - Download your first language model from Hugging Face or Ollama, and send your first chat prompt. + Download your first language model from Hugging Face or Ollama, and test your first prompt. -4. **[Troubleshooting Guide](./troubleshooting.md)** +4. **[Real Engine Test Flight](./test-flight.md)** + Verify that your local inference engines respond to real network requests before queuing heavy workloads. + +5. **[Remote Access & Reverse Proxy](./remote-access.md)** + Access your dashboard over LAN, configure SSH tunnels, or set up Caddy reverse proxy authentication. + +6. **[Troubleshooting Guide](./troubleshooting.md)** Resolve common operational issues, handle VRAM out-of-memory errors, and eliminate port conflicts. --- diff --git a/docs/getting-started/installation.md b/docs/getting-started/installation.md index 7ab52d3..f8211ab 100644 --- a/docs/getting-started/installation.md +++ b/docs/getting-started/installation.md @@ -1,9 +1,28 @@ -# Installation Guide +# Installation & Setup Guide This guide describes how to install Local LLM Server Manager on Windows and Linux systems. --- +## Testing & Setup Matrix + +Local LLM Server Manager uses a modular architecture. You only need to install components for the features you intend to test. + +| Feature Area | Role in Stack | Prerequisite Requirement | What It Enables | +| :--- | :--- | :--- | :--- | +| **Core Manager & Dashboard** | **REQUIRED** | Windows 10/11 x64 or Linux x64 | Hardware telemetry, VRAM bar, system tray, reverse proxy, web dashboard. | +| **Ollama Engine** | *Optional* | [Ollama](https://ollama.com) installed | Local LLM text generation, GGUF downloads, KV cache calculator. | +| **Stable Diffusion Forge** | *Optional* | [SD Forge](https://github.com/lllyasviel/stable-diffusion-webui-forge) installed | Local image generation, CivitAI checkpoint and LoRA downloads. | +| **ComfyUI Engine** | *Optional* | [ComfyUI](https://github.com/comfyanonymous/ComfyUI) installed | 3D mesh reconstruction, video generation, and FLUX workflows. | +| **Kokoro TTS Engine** | *Optional* | Python environment or Audio Pack | Local speech synthesis with OpenAI-compatible audio API. | +| **AI Chat Assistant** | *Optional* | LiteLLM gateway or OpenAI endpoint | In-app assistant, multimodal screenshot analysis, and app control. | +| **Feature Packs (`ext_*`)** | *Optional* | Installed via Settings tab | Video ComfyUI presets (`ext_video`) and Audio workflows (`ext_audio`). | + +> [!IMPORTANT] +> The release package is **self-contained**. You do not need to install the .NET SDK or .NET runtime to run the application. + +--- + ## Windows Installation Choose one of two installation methods for Windows. @@ -31,12 +50,13 @@ The official installer configures the application, sets up desktop shortcuts, an ### Method 2: Standalone Portable Archive (.zip) -Use the portable archive to run the application without modifying system services. +Use the portable archive to test the application without modifying system services. 1. Download the `LocalLLMServerManager-win-x64.zip` archive from the Releases page. 2. Extract the archive contents into a folder (for example: `C:\LocalLLMServerManager`). 3. Open the extracted folder in File Explorer. 4. Double-click `LocalLLMServerManager.exe` to start the application. +5. The desktop window and system tray icon appear immediately. --- diff --git a/docs/getting-started/quickstart.md b/docs/getting-started/quickstart.md index 68fc419..ee7c63a 100644 --- a/docs/getting-started/quickstart.md +++ b/docs/getting-started/quickstart.md @@ -66,28 +66,29 @@ Inspect your installed model and calculate memory requirements: --- -## Step 4: Send Your First Prompt +## Step 4: Test Your Installed Model -Test your model with an interactive prompt. +Test your newly downloaded model with an interactive prompt or automated test flight. -### Method 1: Use the Built-In AI Assistant +### Option A: Verify with Real Engine Test Flight (Fastest) -1. Click the **AI Assistant** tab in the application. -2. Select your newly downloaded model from the model selector dropdown. -3. Click inside the text input box at the bottom of the window. -4. Type your prompt: - ```text - Write a Python function to check whether an integer is prime. Include unit tests. - ``` -5. Click **Send** or press `Enter`. -6. Read the streamed response as the model generates text. -7. Observe the VRAM bar at the top of the window as the GPU allocates memory for inference. +Verify that your installed model executes properly and clears GPU memory: + +1. Click the **Studio** tab on the navigation bar. +2. Locate the **Test Flight** panel in the studio header. +3. Select **Text** from the **Modality** dropdown. +4. Select or type a starter prompt. +5. Click **🚀 Launch Test Flight**. +6. The test runner sends a verified prompt to Ollama, confirms VRAM allocation, and returns the response in seconds. + +> [!TIP] +> Read the complete [Real Engine Test Flight Guide](./test-flight.md) to test Image, Video, and Audio engines. --- -### Method 2: Send a Prompt via the REST API +### Option B: Send a Prompt via the REST API Proxy -Send a prompt from your terminal using the OpenAI-compatible HTTP endpoint: +Send a prompt from your terminal using the unified OpenAI-compatible endpoint on port `5246`: ```bash curl http://localhost:5246/v1/chat/completions \ @@ -104,7 +105,18 @@ curl http://localhost:5246/v1/chat/completions \ }' ``` -The server returns a JSON response containing the generated text completion. +The server routes the request to Ollama on port `11434` and returns the generated text. + +--- + +### Option C: Connect External Chat Frontends + +You can connect popular web chat frontends to Local LLM Server Manager: +- **Open WebUI**: Set the Ollama URL to `http://localhost:5246` or `http://localhost:11434`. +- **LibreChat**: Add an OpenAI-compatible custom endpoint pointing to `http://localhost:5246/v1`. + +> [!NOTE] +> The built-in **AI Assistant** tab connects to an external gateway (such as LiteLLM or Vertex AI Gemini Flash). The assistant intentionally excludes local models from its selector to keep 100% of your local GPU memory available for heavy diffusion and creative tasks. See the [AI Chat Assistant Guide](../ai-and-mcp/assistant.md) to configure external credentials. --- diff --git a/docs/getting-started/remote-access.md b/docs/getting-started/remote-access.md new file mode 100644 index 0000000..27a1a33 --- /dev/null +++ b/docs/getting-started/remote-access.md @@ -0,0 +1,126 @@ +--- +title: Remote Access & Reverse Proxy Setup +description: Guide for remote SSH tunneling, LAN access, and Caddy reverse proxy configuration with authentication. +outline: deep +--- + +# Remote Access & Reverse Proxy Setup + +Local LLM Server Manager operates on `http://127.0.0.1:5246` by default. You can access the dashboard and APIs remotely across your local area network (LAN) or over secure SSH tunnels. + +--- + +## Remote Access Options + +Choose the connection method that best fits your workflow: + +| Method | Target Environment | Security Level | Setup Effort | +| :--- | :--- | :--- | :--- | +| **LAN Access** | Home / Office Network | Network Firewall Required | Minimal (Automatic) | +| **SSH Port Forwarding** | Remote Linux / Cloud Host | Encrypted SSH Tunnel | Low (1 command) | +| **Caddy Reverse Proxy** | Domain / Public Network | Reverse Proxy + Authentication | Moderate (Config file) | + +--- + +## 1. Local Area Network (LAN) Access + +The installer and background service support automatic LAN IP detection. + +### Find Your LAN Endpoints +1. Open the **Settings** tab in the desktop application. +2. Review the **Network Endpoints** panel: + - **Local Dashboard**: `http://localhost:5246` + - **Network Dashboard**: `http://:5246` (for example, `http://10.0.0.21:5246`) + - **Network MCP Endpoint**: `http://:5246/mcp` +3. Access the dashboard from any smartphone, tablet, or secondary laptop connected to the same Wi-Fi network. + +> [!WARNING] +> Do not expose port `5246` directly to the public internet without an authentication proxy. + +--- + +## 2. Remote Access via SSH Port Forwarding + +If you run the application on a headless Linux host, use SSH port forwarding to access the WebAssembly dashboard on your local client machine. + +```mermaid +flowchart LR + Browser["Local Browser\nhttp://localhost:5246"] -->|Encrypted SSH Tunnel\nPort 5246:5246| Host["Remote Linux Server\nLocalLLMServerManager (:5246)"] + Host --> Ollama["Ollama (:11434)"] + Host --> Comfy["ComfyUI (:8188)"] + Host --> Forge["SD Forge (:7860)"] +``` + +### Steps to Forward Ports +1. Open your local terminal. +2. Establish an SSH connection with local port forwarding: + ```bash + ssh -L 5246:localhost:5246 user@your-linux-server + ``` +3. Start the application in headless service mode on the remote server: + ```bash + dotnet run -- --service + ``` + *(Or verify that the `systemd` daemon is running: `sudo systemctl status localllmmanager`)* +4. Open `http://localhost:5246` in your local web browser. +5. All dashboard controls, VRAM monitors, and 3D WebGL viewers operate at full local speed over the encrypted tunnel. + +--- + +## 3. Caddy Reverse Proxy Configuration + +To expose services securely behind an authentication gateway (such as Tinyauth or Authelia), use the following Caddyfile configuration on your reverse proxy server. + +Replace `10.0.0.21` with your server's actual local IP address, and `yourdomain.com` with your domain name: + +```txt +# Local LLM Server Manager Dashboard & Unified Proxy +manager.yourdomain.com { + forward_auth * unix//var/run/tinyauth.sock { + uri /auth + } + reverse_proxy 10.0.0.21:5246 +} + +# ComfyUI Engine (Video, 3D & Image Workflows) +comfy.yourdomain.com { + forward_auth * unix//var/run/tinyauth.sock { + uri /auth + } + reverse_proxy 10.0.0.21:8188 +} + +# Stable Diffusion WebUI Forge +forge.yourdomain.com { + forward_auth * unix//var/run/tinyauth.sock { + uri /auth + } + reverse_proxy 10.0.0.21:7860 +} + +# Ollama LLM Inference API +ollama.yourdomain.com { + forward_auth * unix//var/run/tinyauth.sock { + uri /auth + } + reverse_proxy 10.0.0.21:11434 +} +``` + +### Configuration Instructions +1. Open your Caddy configuration file (`/etc/caddy/Caddyfile`) on your proxy host. +2. Paste the configuration block above. +3. Update IP addresses and domain names to match your network. +4. Reload Caddy: + ```bash + sudo systemctl reload caddy + ``` +5. Test connectivity by navigating to `https://manager.yourdomain.com` in your browser. + +--- + +## Related Documentation + +- [Getting Started & Installation](./installation.md) +- [First-Time Configuration](./configuration.md) +- [System Architecture Specification](../technical/architecture.md) diff --git a/docs/getting-started/test-flight.md b/docs/getting-started/test-flight.md new file mode 100644 index 0000000..d7823d6 --- /dev/null +++ b/docs/getting-started/test-flight.md @@ -0,0 +1,95 @@ +--- +title: Real Engine Test Flight +description: Guide to using the in-app Real Engine Test Flight to verify local AI engine readiness before starting creative workloads. +outline: deep +--- + +# Real Engine Test Flight + +Local LLM Server Manager includes an in-app **Test Flight** verification runner. This tool verifies that your local AI engines respond to real network requests before you begin long creative tasks. + +--- + +## Why Use Test Flight? + +Generative tasks like video rendering and 3D mesh reconstruction consume significant time and GPU memory. If an engine path is incorrect, or if a Python dependency is missing, the generation fails after minutes of processing. + +The Test Flight runner executes a lightweight end-to-end check in seconds: +- Verifies network connectivity to the target engine port. +- Sends an actual inference payload with minimal step counts. +- Verifies that GPU VRAM allocation succeeds without memory errors. +- Reports a clear pass or fail verdict before you queue real workloads. + +```mermaid +flowchart TD + Start["User Clicks Launch Test Flight"] --> Buffer["1/3 Check Service Health & Buffers"] + Buffer --> Payload["2/3 Send Real HTTP Request to Backend"] + Payload --> Response["3/3 Verify Pipeline & HTTP 200 OK"] + Response --> Pass["🎉 Test Flight Succeeded!\nEngine and GPU Verified"] + Response --> Fail["⚠️ Test Flight Failed\nDisplay Error Code and Troubleshooting Advice"] +``` + +--- + +## Supported Test Flight Modalities + +You can test four distinct engine backends: + +| Modality | Target Backend | Port | Test Payload | Expected Response | +| :--- | :--- | :--- | :--- | :--- | +| **Text** | Ollama | `:11434` | `POST /api/generate` with starter prompt | HTTP 200 with text stream | +| **Image** | Stable Diffusion Forge | `:7860` | `POST /sdapi/v1/txt2img` (steps: 1) | HTTP 200 with preview image | +| **Video** | ComfyUI | `:8188` | `POST /prompt` with workflow header | HTTP 200 with prompt ID | +| **Audio** | Kokoro TTS | `:8880` | `POST /v1/audio/speech` (voice: `af_heart`) | HTTP 200 with audio stream | + +--- + +## How to Run a Test Flight + +Follow these steps to run a test flight on your system: + +1. Open the **Local LLM Server Manager** desktop window. +2. Click the **Studio** tab on the navigation bar. +3. Locate the **Test Flight** control panel in the studio header. +4. Select your target modality from the **Modality** dropdown: + - Select **Text** to test Ollama. + - Select **Image** to test Stable Diffusion Forge. + - Select **Video** to test ComfyUI. + - Select **Audio** to test Kokoro TTS. +5. Select a starter prompt from the prompt selector, or type a custom prompt. +6. Check the **VRAM Clearance** indicator. Confirm that your GPU reports sufficient free memory. +7. Click **🚀 Launch Test Flight**. +8. Observe the three-stage progress bar: + - Stage 1: Checks service health and allocates memory buffers. + - Stage 2: Sends the HTTP request to the selected backend. + - Stage 3: Verifies pipeline response and HTTP status codes. +9. Review the result banner: + - A green banner confirms that the engine operates correctly. + - A red banner displays the exact HTTP status code and error description. + +> [!TIP] +> Run a Test Flight immediately after installing a new engine or updating GPU drivers. + +> [!IMPORTANT] +> If a Test Flight fails with a connection error, open **Settings**. Verify that the engine port matches the running service. + +--- + +## Troubleshooting Test Flight Failures + +Use this table to resolve common Test Flight errors: + +| Error Symptom | Probable Cause | Action | +| :--- | :--- | :--- | +| **Connection Refused** | The target engine is stopped. | Start the engine from the dashboard or your terminal. | +| **HTTP 404 Not Found** | The engine runs on a different port. | Change the port number in the **Settings** tab. | +| **HTTP 500 Out of Memory** | Active models consume all VRAM. | Click **Unload All VRAM** in the top navigation bar. | +| **Timeout after 30 Seconds** | The model is loading from slow storage. | Move model files to an NVMe solid-state drive. | + +--- + +## Related Documentation + +- [Getting Started Overview](./index.md) +- [Engines & VRAM Overview](../engines/index.md) +- [Multimodal Studio Overview](../studio/index.md) diff --git a/docs/guide/can-i-run-it.md b/docs/guide/can-i-run-it.md new file mode 100644 index 0000000..add088b --- /dev/null +++ b/docs/guide/can-i-run-it.md @@ -0,0 +1,116 @@ +# Can I Run It — Hardware Compatibility & Performance Estimator + +The **Can I Run It** workspace calculates whether an AI model can execute on your local hardware. It displays memory allocation, layer distribution, and throughput estimates before you download model weights. + +![Can I Run It Hardware Fit Calculator](../images/dashboard_can_i_run_it.png) + +--- + +## 1. Overview and Core Purpose + +Large language models and diffusion networks require significant hardware resources. The **Can I Run It** tool prevents system out-of-memory errors. It checks your hardware specifications against model requirements in real time. + +```mermaid +flowchart TD + Detect["Hardware Detection\n(NVML / CUDA Telemetry)"] --> Calc["Sizing Engine\n(Weights + KV Cache + Overhead)"] + Params["User Settings\n(Model Size, Quantization, Context)"] --> Calc + Calc --> Verdict["Compatibility Verdict\n(Full VRAM, Partial Offload, Won't Fit)"] + Calc --> Visual["Visual Memory Bar\n(Weights, Cache, Free Headroom)"] +``` + +--- + +## 2. Hardware Telemetry Card + +The top panel displays your current system specifications: + +| Metric | Source | Function | +| :--- | :--- | :--- | +| **Detected GPU** | NVML / CUDA driver | Identifies the primary graphics accelerator. | +| **VRAM Capacity** | Live GPU memory query | Shows dedicated video memory and free space. | +| **System RAM** | Operating system query | Shows host system memory for CPU offloading. | +| **Refresh Telemetry** | User action button | Queries hardware status to capture memory changes. | + +--- + +## 3. Supported Modalities + +Select a modality tab at the top of the estimator: + +* **Text LLMs**: Estimates GGUF models for Ollama and llama.cpp runtimes. +* **Image Generation**: Estimates FLUX.1, SDXL, and SD 1.5 diffusion models. +* **Video Generation**: Estimates Wan 2.2, LTX-Video 2.5, and HunyuanVideo DiT architectures. +* **Audio & Speech**: Estimates Kokoro TTS and music generation models. +* **3D Generation**: Estimates TRELLIS and Hunyuan3D mesh reconstruction pipelines. + +--- + +## 4. Parameter Configuration + +Adjust the input sliders to model your target workload: + +### Model Preset Profile +Select a standard model profile from the dropdown menu (e.g., `Llama 3.1 8B`, `Qwen 2.5 32B`, `DeepSeek R1 70B`). The tool populates default parameters automatically. + +### Model Size (Parameter Count) +Use the slider to adjust the parameter scale from 0.5 billion to 70 billion parameters. + +### Quantization Level +Select the compression precision: +* **FP16**: Full 16-bit floating point precision (no quality loss, highest VRAM). +* **Q8_0**: 8-bit quantization (near-lossless, moderate VRAM savings). +* **Q4_K_M**: 4-bit medium quantization (industry standard, 50% VRAM reduction). +* **Q2_K**: 2-bit aggressive quantization (lowest VRAM, lower output quality). + +### Context Window Length +Set the token context buffer between 2,048 and 131,072 tokens. Longer context lengths increase memory consumption. + +### KV Cache Quantization Precision +Select the precision of the Key-Value attention cache (`FP16`, `Q8_0`, or `Q4_0`). Setting KV cache quantization to `Q4_0` reduces cache memory usage by up to 70%. + +--- + +## 5. Compatibility Verdict and Performance + +The right panel shows the execution verdict and predicted speed: + +```mermaid +flowchart LR + V1["🟢 Full VRAM\n(100% GPU Acceleration)"] --- V2["🟡 Partial Offload\n(Split GPU & CPU RAM)"] + V2 --- V3["🟠 CPU Only\n(Low Throughput)"] --- V4["🔴 Won't Fit\n(System Memory Exceeded)"] +``` + +### Verdict Categories + +| Badge | Status | Behavior | Performance Impact | +| :--- | :--- | :--- | :--- | +| **Full VRAM** | 🟢 Fits 100% in VRAM | All model layers execute on the GPU. | Maximum inference speed (e.g., ~169 tok/s). | +| **Partial Offload** | 🟡 GPU + CPU RAM | Critical layers run on GPU; remainder runs in RAM. | Moderate speed reduction due to PCIe transfer overhead. | +| **CPU Only** | 🟠 System RAM Only | Model exceeds total VRAM capacity. | Low throughput (1 to 5 tok/s). | +| **Won't Fit** | 🔴 Out of Memory | Model exceeds combined VRAM and system RAM. | Execution blocked to prevent application crash. | + +--- + +## 6. Visual Memory Allocation Bar + +The visual breakdown bar displays four color-coded memory segments: + +``` +[ ■ Weights (29.5%) | ■ Context/KV (3.1%) | ■ Overhead (3.7%) | ■ Free Headroom (63.7%) ] +``` + +1. **Model Weights (Cyan)**: Static memory required to store loaded model tensors. +2. **Context / KV Cache (Purple)**: Dynamic memory allocated for conversation tokens. +3. **Runtime Overhead (Grey)**: CUDA runtime buffers and frame memory. +4. **Free Headroom (Dark Blue)**: Remaining unused video memory on the GPU. + +--- + +## 7. Ambient Compatibility Badges Across the App + +The sizing engine powers ambient compatibility badges throughout the application: + +* **Hugging Face Hub**: Each search result card displays a live fit badge. +* **CivitAI Hub**: Checkpoint and LoRA cards display GPU compatibility status. +* **Ollama Local Library**: Installed models display instantaneous VRAM fit calculations. +* **Diagnostic Test Flight**: Pre-flight verification blocks runs when VRAM headroom is insufficient. diff --git a/docs/guide/collaborative-debugging.md b/docs/guide/collaborative-debugging.md index 1a1f94e..b0f4e52 100644 --- a/docs/guide/collaborative-debugging.md +++ b/docs/guide/collaborative-debugging.md @@ -19,45 +19,38 @@ Collaborative debugging operates on a dual-channel architecture: 1. **Human Developer Channel**: The developer interacts with the running desktop application, evaluates ergonomics and visual aesthetics, and uses built-in F12 DevTools to inspect layout bounds, active styles, and element trees interactively. 2. **AI Assistant Channel**: The AI coding assistant connects through the `AvaloniaMcp` protocol server over a local named pipe, querying visual trees, serialized ViewModel state, and binding diagnostic logs via structured JSON-RPC tool calls. -``` -+-------------------------------------------------------------------------+ -| DEVELOPER WORKSPACE | -| - Runs application in Debug mode | -| - Inspects visual layout & controls via F12 DevTools | -| - Reports visual anomalies or UX friction in conversation | -+------------------------------------+------------------------------------+ - | - v -+-------------------------------------------------------------------------+ -| AVALONIA RUNTIME PROCESS (PID: {pid}) | -| | -| - AppBuilder.Configure() | -| .UsePlatformDetect() | -| .UseMcpDiagnostics() <-- AvaloniaMcp Named Pipe Endpoint | -| .LogToTrace() | -| | -| - Named Pipe Endpoint: `avalonia-mcp-{pid}` | -| - Process Discovery File: `%TEMP%/avalonia-mcp/{pid}.json` | -| - UI Diagnostic Logger (Binding error trace capture) | -+------------------------------------+------------------------------------+ - | Named Pipe (JSON-RPC) - v -+-------------------------------------------------------------------------+ -| AVALONIA MCP SERVER (`avaloniamcp`) | -| | -| - Global .NET Tool (`dotnet avalonia-mcp`) | -| - Exposes 15 MCP Tools to AI Assistant over stdio | -+------------------------------------+------------------------------------+ - | MCP Protocol (stdio) - v -+-------------------------------------------------------------------------+ -| ANTIGRAVITY AI ASSISTANT | -| - Discovers running UI app instance (`discover_apps`) | -| - Queries visual/logical trees, DataContexts, & broken bindings | -| - Mutates properties live to test layout hypotheses | -| - Captures element screenshots for visual inspection | -| - Formulates and applies codebase fixes directly | -+-------------------------------------------------------------------------+ +```mermaid +flowchart TD + subgraph DevWorkspace["DEVELOPER WORKSPACE"] + D1["Runs application in Debug mode: `dotnet run -c Debug`"] + D2["Inspects visual layout & controls via F12 DevTools"] + D3["Reports visual anomalies or UX issues in dialogue"] + end + + subgraph RuntimeProcess["AVALONIA RUNTIME PROCESS (PID: {pid})"] + R1["AppBuilder.Configure<App>()\n.UseMcpDiagnostics()\n.LogToTrace()"] + R2["Named Pipe Endpoint: avalonia-mcp-{pid}"] + R3["Process Discovery Metadata: %TEMP%/avalonia-mcp/{pid}.json"] + R4["UI Diagnostic Logger (Binding Error Trace Capture)"] + end + + subgraph McpServer["AVALONIA MCP SERVER (avaloniamcp)"] + M1[".NET Global Tool: dotnet avalonia-mcp"] + M2["Exposes 15 Diagnostic Tools over stdio"] + end + + subgraph AIAssistant["AI CODING ASSISTANT (ANTIGRAVITY / CLAUDE)"] + A1["Discovers running app instance: discover_apps"] + A2["Inspects visual/logical trees, DataContexts & bindings"] + A3["Mutates properties live to test layout hypotheses"] + A4["Captures element screenshots for visual verification"] + A5["Formulates and applies codebase fixes directly"] + end + + DevWorkspace -->|Interactive Visual Inspection| RuntimeProcess + RuntimeProcess -->|Named Pipe JSON-RPC| McpServer + McpServer -->|MCP Protocol stdio| AIAssistant + AIAssistant -->|Automated Fixes & Verification| DevWorkspace ``` ### Production Isolation & Security diff --git a/docs/guide/companion-windows.md b/docs/guide/companion-windows.md new file mode 100644 index 0000000..7c4e5fa --- /dev/null +++ b/docs/guide/companion-windows.md @@ -0,0 +1,103 @@ +--- +title: Magnetic Companion Windows & WindowSnapManager +description: User guide for floating companion windows, magnetic side-docking, lockstep window synchronization, and proximity snapping. +outline: deep +--- + +# Magnetic Companion Windows + +Local LLM Server Manager features a flexible multi-window architecture. You can pop out application tabs into independent companion windows, snap them magnetically to the main window, and manage multi-monitor workspaces with ease. + +--- + +## Overview + +The desktop application provides two detachable companion windows: +1. **AI Assist Companion Window**: Pops out the AI Chat Assistant for continuous prompt assistance and troubleshooting. Snaps to the **right flank** of the main window. +2. **Documentation Companion Window**: Pops out this documentation guide for side-by-side reading. Snaps to the **left flank** of the main window. + +```mermaid +flowchart LR + DocWin["Documentation Window\n(Left Companion)"] <-->|Magnetic Proximity Snap\n& Lockstep Sync| MainWin["Main Dashboard Window\n(Center)"] + MainWin <-->|Magnetic Proximity Snap\n& Lockstep Sync| AiWin["AI Assist Window\n(Right Companion)"] +``` + +--- + +## Pop Out a Companion Window + +Follow these steps to detach a tab into a companion window: + +1. Open the **Local LLM Server Manager** desktop application. +2. To pop out documentation: + - Click the pop-out icon (**⧉ Pop Out**) in the **Documentation** tab header. + - The Documentation companion window appears on the left flank of the main window. +3. To pop out the AI Assistant: + - Click the pop-out icon (**⧉ Pop Out**) in the **AI Assistant** tab header. + - The AI Assist companion window appears on the right flank of the main window. + +> [!NOTE] +> When you pop out a tab, the main window collapses the tab content and displays a banner confirming that the companion window is active. + +--- + +## Magnetic Snapping Behavior + +The `WindowSnapManager` service controls window docking through three mechanisms: + +### 1. Lockstep Synchronization +When a companion window is snapped: +- Moving the main window moves the snapped companion window in lockstep. +- Resizing the main window height resizes the snapped companion window height to match. +- Minimizing the main window minimizes all snapped companion windows simultaneously. + +### 2. The Magnet Button Toggle +Each companion window includes an interactive magnet button in its custom title bar: +- **`🧲 Attached`**: Indicates that the window is docked to the main window flank. +- Click the button to detach the window. The button text changes to **`🧲 Snap to Side`**. +- Click the button again to snap the window back to its assigned flank. + +### 3. Proximity Snapping & Drag Detachment +You can detach or dock companion windows naturally using mouse gestures: +- **Drag to Detach**: Click and drag the companion window title bar away from the main window. When the distance exceeds the detachment threshold (24 pixels), the companion detaches automatically. +- **Proximity Snap**: Drag an unattached companion window close to its assigned flank (within 32 pixels). The window snaps into place magnetically and re-engages lockstep tracking. + +```mermaid +stateDiagram-v2 + [*] --> Snapped: Pop Out Window + Snapped --> Detached: User Drags Window Away (> 24 px) + Snapped --> Detached: Click "🧲 Attached" Button + Detached --> Snapped: Drag Window Near Flank (< 32 px) + Detached --> Snapped: Click "🧲 Snap to Side" Button + Snapped --> Closed: Close Window / Return to Tab + Detached --> Closed: Close Window / Return to Tab +``` + +--- + +## Multi-Monitor Workflows + +Companion windows are ideal for multi-monitor setups: + +1. Pop out the **AI Assist** window. +2. Drag the AI Assist window to your secondary display. +3. Keep your primary display focused on 3D mesh reconstruction or video generation in the **Studio**. +4. Capture screenshots with **Ctrl+V** inside the detached AI Assist window to ask questions while monitoring renders live. + +--- + +## Restore Tabs into the Main Window + +To return a companion window back into the main tab layout: + +1. Click the close button (**✕**) in the companion window title bar. +2. The companion window unhooks from `WindowSnapManager`. +3. The main window restores the tab content in place. + +--- + +## Related Documentation + +- [AI Chat Assistant Guide](../ai-and-mcp/assistant.md) +- [Application Configuration Guide](../getting-started/configuration.md) +- [Collaborative UI Debugging](./collaborative-debugging.md) diff --git a/docs/images/dashboard_3d_studio.png b/docs/images/dashboard_3d_studio.png index f117fb5..2adc3d5 100644 Binary files a/docs/images/dashboard_3d_studio.png and b/docs/images/dashboard_3d_studio.png differ diff --git a/docs/images/dashboard_can_i_run_it.png b/docs/images/dashboard_can_i_run_it.png new file mode 100644 index 0000000..e7a5630 Binary files /dev/null and b/docs/images/dashboard_can_i_run_it.png differ diff --git a/docs/images/dashboard_civitai.png b/docs/images/dashboard_civitai.png index dfb081f..3b3caaf 100644 Binary files a/docs/images/dashboard_civitai.png and b/docs/images/dashboard_civitai.png differ diff --git a/docs/images/dashboard_desktop.png b/docs/images/dashboard_desktop.png index eb0aa2f..60ea88e 100644 Binary files a/docs/images/dashboard_desktop.png and b/docs/images/dashboard_desktop.png differ diff --git a/docs/images/dashboard_huggingface.png b/docs/images/dashboard_huggingface.png index a1121d9..b8b94d7 100644 Binary files a/docs/images/dashboard_huggingface.png and b/docs/images/dashboard_huggingface.png differ diff --git a/docs/images/dashboard_ollama.png b/docs/images/dashboard_ollama.png index eb0aa2f..60ea88e 100644 Binary files a/docs/images/dashboard_ollama.png and b/docs/images/dashboard_ollama.png differ diff --git a/docs/images/dashboard_settings.png b/docs/images/dashboard_settings.png index 4f67f3c..a3db341 100644 Binary files a/docs/images/dashboard_settings.png and b/docs/images/dashboard_settings.png differ diff --git a/docs/index.md b/docs/index.md index 82ed0e8..3efda03 100644 --- a/docs/index.md +++ b/docs/index.md @@ -32,23 +32,25 @@ features: Local LLM Server Manager is a cross-platform orchestrator for local artificial intelligence engines. The application operates on Windows and Linux. The system coordinates Large Language Models, image generation, 3D mesh reconstruction, video synthesis, and speech generation. -``` -+-----------------------------------------------------------------------------------------+ -| Local LLM Server Manager | -| GPU: NVIDIA GeForce RTX 4070 Ti SUPER -- 16 GB • Service Connected 🟢 [🔄 Refresh] | -| GPU VRAM Allocation: 4.2 GB / 16.0 GB (26.3%) | -| [========================-------------------------------------------------------------] | -+-----------------------------------------------------------------------------------------+ -| [🦙 Installed Models] [🤗 Hugging Face Hub] [🎨 CivitAI Models] [📦 Studio] [⚙️ Settings]| -+-----------------------------------------------------------------------------------------+ -``` +![Desktop Dashboard Overview](./images/dashboard_desktop.png) + +### User Interface Layout + +| Layout Section | Primary Function | Active Elements | +| :--- | :--- | :--- | +| **Telemetry Header** | Live Hardware Monitoring | GPU name, total/used VRAM bar, service health indicator, and refresh button. | +| **Primary Navigation** | Workspace Switcher | Tabs for My Models, Hugging Face Hub, CivitAI, Multimodal Studio, AI Assistant, and Settings. | +| **Main Content Canvas** | Engine Interaction | Model cards, KV cache context calculator, download progress, WebGL canvas, and video player. | +| **Companion Windows** | Floating Multi-Window | Detachable Documentation and AI Assist windows with magnetic flank docking. | ## Core Capabilities - **Automated VRAM Management**: Prevent out-of-memory errors during heavy diffusion or 3D tasks. +- **Real Engine Test Flight**: Verify engine network connectivity and inference pipelines with one click. - **Model Discovery**: Search and download models from Hugging Face Hub and CivitAI directly. - **Unified Reverse Proxy**: Route all AI engine traffic through a single port (`5246`). - **Flexible Deployment**: Run as a native Avalonia desktop application, system tray app, or headless background service. +- **Magnetic Companion Windows**: Detach helper windows and dock them magnetically to the main application window. - **Model Context Protocol**: Give AI coding assistants secure control over your local AI infrastructure. ## Documentation Sections @@ -58,7 +60,14 @@ Explore the documentation guides to set up and operate your local AI stack: - [Getting Started Overview](./getting-started/index.md): Review system requirements and core concepts. - [Installation Guide](./getting-started/installation.md): Install the application on Windows or Linux. - [First-Time Configuration](./getting-started/configuration.md): Auto-detect engines, configure ports, and set model paths. -- [Quickstart Tutorial](./getting-started/quickstart.md): Download your first model and send your first prompt. +- [Quickstart Tutorial](./getting-started/quickstart.md): Download your first model and run your first prompt. +- [Real Engine Test Flight](./getting-started/test-flight.md): Verify engine responsiveness and GPU allocation before generating. +- [Remote Access & Reverse Proxy](./getting-started/remote-access.md): Configure LAN access, SSH tunnels, and Caddy reverse proxy. +- [Multimodal Studio Guide](./studio/index.md): Generate images, videos, audio tracks, and 3D meshes. +- [LoRA Art Styles & CivitAI](./studio/lora-styles.md): Apply custom visual styles to diffusion models. +- [AI Chat Assistant](./ai-and-mcp/assistant.md): Configure external LiteLLM models and multimodal diagnostics. +- [Magnetic Companion Windows](./guide/companion-windows.md): Dock floating helper windows with magnetic snap controls. - [Troubleshooting Guide](./getting-started/troubleshooting.md): Resolve VRAM issues, port conflicts, and network errors. - [Documentation Style Guide](./standards/ste-100.md): Review writing standards for this documentation site. - [Technical Reference](./technical/index.md): Read system architecture specifications and developer guides. + diff --git a/docs/studio/3d-mesh.md b/docs/studio/3d-mesh.md index 430dac0..6e44abd 100644 --- a/docs/studio/3d-mesh.md +++ b/docs/studio/3d-mesh.md @@ -109,7 +109,7 @@ Select your input mode: The right panel features an interactive WebGL canvas that renders generated `.glb` meshes in real time: -![Interactive 3D Studio Canvas](file:///C:/Users/Alias/repos/LocalLLMServerManager/docs/images/dashboard_3d_studio.png) +![Interactive 3D Studio Canvas](../images/dashboard_3d_studio.png) ### Viewport Navigation Controls diff --git a/docs/studio/comfyui-setup.md b/docs/studio/comfyui-setup.md new file mode 100644 index 0000000..78368c3 --- /dev/null +++ b/docs/studio/comfyui-setup.md @@ -0,0 +1,104 @@ +--- +title: ComfyUI & 3D Nodes Setup +description: Step-by-step setup guide for connecting ComfyUI, installing 3D nodes (TRELLIS V2 and Hunyuan3D v2), and exporting API workflow presets. +outline: deep +--- + +# ComfyUI & 3D Nodes Setup Guide + +This guide describes how to configure **ComfyUI** and install 3D mesh reconstruction nodes for use with Local LLM Server Manager. + +--- + +## Prerequisites + +Verify these requirements before installation: +- NVIDIA GPU with at least 12 GB VRAM for 3D mesh generation. +- Python 3.10 or 3.11 installed. +- Git command-line tool installed. + +--- + +## 1. Install ComfyUI + +1. Download the official standalone package from the [ComfyUI repository](https://github.com/comfyanonymous/ComfyUI), or clone the source repository: + ```bash + git clone https://github.com/comfyanonymous/ComfyUI.git + cd ComfyUI + ``` +2. Install Python dependencies: + ```bash + pip install -r requirements.txt + ``` +3. Start ComfyUI: + - On Windows: Double-click `run_nvidia_gpu.bat`. + - On Linux: Run `python main.py --listen 127.0.0.1 --port 8188`. +4. Verify that ComfyUI opens at `http://127.0.0.1:8188`. + +--- + +## 2. Install 3D Mesh Generation Nodes + +Install custom nodes for **TRELLIS V2** and **Hunyuan3D v2**: + +### Option A: Use ComfyUI-Manager (Recommended) +1. Install [ComfyUI-Manager](https://github.com/ltdrdata/ComfyUI-Manager) if ComfyUI-Manager is not yet installed. +2. In the ComfyUI web interface, click **Manager** in the side panel. +3. Click **Custom Nodes Manager**. +4. Search for and install: + - `ComfyUI-Trellis` (TRELLIS V2 Image-to-3D node) + - `ComfyUI-Hunyuan3DWrapper` (Hunyuan3D v2 Text/Image-to-3D node) +5. Restart ComfyUI. + +### Option B: Manual Git Clone +1. Navigate to the `custom_nodes/` directory in ComfyUI: + ```bash + cd custom_nodes + git clone https://github.com/JeffreyXiang/ComfyUI-TRELLIS.git + git clone https://github.com/Tencent/Hunyuan3D-2.git + ``` +2. Install requirements for each custom node package. +3. Restart ComfyUI. + +--- + +## 3. Connect ComfyUI to Local LLM Server Manager + +1. Open **Local LLM Server Manager**. +2. Select the **Settings** tab. +3. In the **Engine Paths & URLs** panel: + - Set **ComfyUI URL** to `http://127.0.0.1:8188`. + - Set **ComfyUI Launch Script** to your `run_nvidia_gpu.bat` path. + - Set **ComfyUI Models Directory** to your `ComfyUI/models` folder. +4. Click **Save Settings**. +5. Check the **ComfyUI** health indicator in the header. Confirm that the status shows 🟢 **Online**. + +```mermaid +flowchart LR + Manager["Local LLM Server Manager (:5246)"] -->|Unload LLM VRAM| Ollama["Ollama (:11434)"] + Manager -->|POST /prompt (API JSON)| ComfyUI["ComfyUI (:8188)"] + ComfyUI -->|Generate .glb Mesh| Storage["Output Folder"] + Storage -->|Serve WebGL Stream| Canvas["Interactive 3D Canvas"] +``` + +--- + +## 4. Export Custom Workflow Presets + +You can export any ComfyUI node graph as a preset for the manager: + +1. Open ComfyUI in your web browser (`http://127.0.0.1:8188`). +2. Open the ComfyUI settings dialog (gear icon). +3. Enable **Enable Dev mode Options**. +4. Construct your generation workflow. +5. Click **Save (API Format)**. ComfyUI downloads an API-format JSON file. +6. Copy the JSON file into the `Workflows/` directory in Local LLM Server Manager. +7. The preset appears in the Studio workflow dropdown menu immediately. + +--- + +## Related Documentation + +- [3D Mesh Reconstruction Studio](./3d-mesh.md) +- [Video Generation Studio](./video-generation.md) +- [Multimodal Studio Overview](./index.md) diff --git a/docs/studio/lora-styles.md b/docs/studio/lora-styles.md new file mode 100644 index 0000000..8c75949 --- /dev/null +++ b/docs/studio/lora-styles.md @@ -0,0 +1,81 @@ +--- +title: LoRA Art Styles & CivitAI Guide +description: Guide for discovering, downloading, and applying Low-Rank Adaptation (LoRA) models for custom art styles in Stable Diffusion Forge and ComfyUI. +outline: deep +--- + +# LoRA Art Styles & CivitAI Guide + +Low-Rank Adaptations (LoRAs) allow you to apply specific visual styles, character designs, and concepts without downloading full multi-gigabyte checkpoint models. + +--- + +## What is a LoRA? + +A base checkpoint model (such as SDXL, FLUX, or Pony V6) requires 4 GB to 12 GB of storage. In contrast, a LoRA file is compact (typically 50 MB to 200 MB). + +The LoRA acts as a targeted adapter layer. It modifies model weights during generation to produce a specific style or subject: + +```mermaid +flowchart LR + Prompt["Text Prompt\n'a dog, '"] --> CLIP["CLIP Text Encoder"] + Base["Base Checkpoint\n(SDXL / FLUX)"] --> UNET["Diffusion UNET / DiT"] + LoRA["LoRA Weights (~100 MB)\nStyle / Concept Adapter"] --> UNET + UNET --> VAE["VAE Decoder"] + VAE --> Image["Generated Artwork in Target Style"] +``` + +--- + +## Download LoRAs from CivitAI + +You can search and download LoRAs directly from CivitAI inside the manager: + +1. Open **Local LLM Server Manager**. +2. Click the **CivitAI Models** tab on the navigation bar. +3. Select **LoRA** from the **Type** filter dropdown. +4. Select **Highest Rated** or **Most Downloaded** from the **Sort** dropdown. +5. Type an art style keyword in the search box (see reference table below). +6. Press `Enter` to search. +7. Click on a model card to inspect preview thumbnails, trigger words, and base model compatibility. +8. Click **⬇ Download to Forge**. The manager streams the file directly into your configured LoRA directory. + +--- + +## Popular Style Keywords Reference + +Use these search terms to find established community LoRA styles: + +| Style Category | Recommended Search Keywords | Visual Result | +| :--- | :--- | :--- | +| **Pixel Art & Retro** | `Pixel Art XL`, `16-bit`, `Retro 3D`, `Gameboy` | Authentic sprite graphics, isometric levels, dot-matrix art | +| **Animation & Anime** | `Cel Shaded`, `Studio Ghibli`, `90s Anime Style` | Clean line art, vibrant watercolors, retro animation aesthetics | +| **3D & Game Assets** | `Claymation`, `Low Poly`, `Chibi Figurine` | Tactile plasticine figures, stylized game meshes, clay models | +| **Realism Enhancers** | `Detail Tweaker XL`, `Cinematic Lighting` | Skin pores, micro-textures, neon rim lights, dramatic shadows | + +--- + +## How to Apply a LoRA in Generation + +### Method 1: In Prompts (Stable Diffusion WebUI Forge) +Include the LoRA trigger syntax directly in your text prompt. Set the weight between `0.5` and `1.0`: + +```text + a cybernetic dog in a neon alley, 16-bit pixel art style +``` + +### Method 2: In ComfyUI Workflows +Use a **Load LoRA** node in your ComfyUI node graph: +1. Open the ComfyUI workflow editor. +2. Add a **Load LoRA** node between your **Load Checkpoint** node and **KSampler**. +3. Connect the `MODEL` and `CLIP` outputs to the LoRA inputs. +4. Select your downloaded LoRA from the dropdown menu. +5. Adjust `strength_model` (recommended: `0.8`) and `strength_clip` (recommended: `0.8`). + +--- + +## Related Documentation + +- [Multimodal Studio Overview](./index.md) +- [Image Generation Guide](./image-generation.md) +- [CivitAI & Hugging Face Model Management](../engines/model-management.md) diff --git a/docs/studio/video-generation.md b/docs/studio/video-generation.md index 2d8d550..d0df0ad 100644 --- a/docs/studio/video-generation.md +++ b/docs/studio/video-generation.md @@ -111,7 +111,7 @@ flowchart LR The right panel features an integrated video preview player for reviewing rendered clips: -![Interactive Video Player Controls](file:///C:/Users/Alias/repos/LocalLLMServerManager/docs/images/dashboard_desktop.png) +![Interactive Video Player Controls](../images/dashboard_3d_studio.png) ### Player Features and Controls diff --git a/docs/technical/test-coverage.md b/docs/technical/test-coverage.md index 0e76378..f9b4d8b 100644 --- a/docs/technical/test-coverage.md +++ b/docs/technical/test-coverage.md @@ -8,16 +8,14 @@ This document provides a comprehensive audit of all unit, integration, mock serv ## 📊 Executive Summary & Test Metrics -``` -+-----------------------------------------------------------------------------------------+ -| TOTAL TESTS EXECUTED : 174 | -| PASSED : 173 (99.4%) | -| SKIPPED : 1 (Playwright screenshot generator on-demand) | -| FAILED : 0 (0.0%) | -| TEST FIXTURE CLASSES : 20 | -| TARGET RUNTIMES : Windows 11 x64, Linux x64 (systemd, X11, Wayland), Chromium Headless| -+-----------------------------------------------------------------------------------------+ -``` +| Metric | Measured Value | Operational Notes | +| :--- | :--- | :--- | +| **Total Tests Executed** | `174` | Comprehensive test suite across Windows and Linux | +| **Passed Tests** | `173 (99.4%)` | All unit, integration, and UI tests pass | +| **Skipped Tests** | `1 (0.6%)` | On-demand Playwright screenshot generator | +| **Failed Tests** | `0 (0.0%)` | Zero test failures across test suite | +| **Test Fixture Classes** | `20` | Modular test fixtures partitioned by domain | +| **Target Runtimes** | Windows 11 x64, Linux x64 (systemd, X11, Wayland), Headless Chromium | Dual-OS verified | ### Testing Frameworks & Tooling * **Test Runner**: [xUnit.net v3](https://xunit.net/) (`xunit.v3` 3.2.2) @@ -32,20 +30,14 @@ This document provides a comprehensive audit of all unit, integration, mock serv To eliminate port contention and process memory race conditions during local and CI/CD test runs on Windows and Linux, the test suite is partitioned into five targeted execution chunks: -``` - ┌─────────────────────────────────────────────────────────┐ - │ LocalLLMServerManager.Tests │ - │ (174 Tests) │ - └────────────────────────────┬────────────────────────────┘ - │ - ┌──────────────────┬──────────────────┼──────────────────┬──────────────────┐ - ▼ ▼ ▼ ▼ ▼ - ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ - │ Chunk 1 │ │ Chunk 2 │ │ Chunk 3 │ │ Chunk 4 │ │ Chunk 5 │ - │ ViewModels │ │ Services │ │ Endpoints │ │ MCP Server │ │ Playwright │ - │ & Settings │ │ & Discovery │ │ & Workflows │ │ & Tools │ │ WASM E2E │ - │ (37 Tests) │ │ (46 Tests) │ │ (67 Tests) │ │ (22 Tests) │ │ (2 Tests) │ - └──────────────┘ └──────────────┘ └──────────────┘ └──────────────┘ └──────────────┘ +```mermaid +flowchart TD + Root["LocalLLMServerManager.Tests\n(174 Tests)"] + Root --> C1["Chunk 1: ViewModels & Settings\n(37 Tests)"] + Root --> C2["Chunk 2: Services & Discovery\n(46 Tests)"] + Root --> C3["Chunk 3: Endpoints & Workflows\n(67 Tests)"] + Root --> C4["Chunk 4: MCP Server & Tools\n(22 Tests)"] + Root --> C5["Chunk 5: Playwright WASM E2E\n(2 Tests)"] ``` ### Execution Commands