Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,14 @@ To recreate Linux/macOS launchers without reinstalling, run from the checkout:

Linux respects `XDG_DATA_HOME` and uses `xdg-user-dir DESKTOP` when available; otherwise it uses an existing `~/Desktop`. Disabled or missing desktops are skipped. These launchers start the server in the background and open your browser. Keep the checkout in place, or rerun the helper from its new location after moving it. If shortcut creation fails, installation still completes and the terminal launch scripts remain available.

On Linux, including NixOS, if your llama.cpp binaries run manually but the GUI reports missing runtime libraries, close the GUI and start it from the checkout with:

```bash
LLAMA_GUI_SKIP_LDD=1 ./mac_linux_start.sh
```

This opts out of the GUI's `ldd` checks, including custom-backend activation. It does not install missing libraries or bypass executable checks. Restart without the variable to restore library validation.

To build CUDA `llama.cpp` yourself on Linux, see `Linux_compile_toolkit/`.

## Install With Pinokio
Expand Down
20 changes: 20 additions & 0 deletions backend/services/llama_manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
import zipfile
from typing import Any, Callable, Iterable, Mapping, Optional

from ..config import parse_bool_env
from ..context import AppContext
from ..http import sanitize_error
from . import official_backends
Expand Down Expand Up @@ -423,12 +424,28 @@ def get_linux_runtime_probe_files(runtime_dir: pathlib.Path) -> list[pathlib.Pat
)


def _linux_runtime_validation_opt_out(current_platform: str) -> Optional[dict[str, Any]]:
"""Allow Linux users to bypass ldd when it misreports a working runtime."""
if current_platform.startswith("linux") and parse_bool_env(os.environ.get("LLAMA_GUI_SKIP_LDD")):
return {
"ok": True,
"checked": False,
"skip_reason": "LLAMA_GUI_SKIP_LDD",
"required_runtime_files": [],
"missing_runtime_files": [],
}
return None


def validate_runtime_dependencies(
ctx: AppContext, tools: Optional[Iterable[str]] = None
) -> dict[str, Any]:
current_platform = ctx.services.current_platform or sys.platform
if current_platform == "unknown":
current_platform = sys.platform
skipped = _linux_runtime_validation_opt_out(current_platform)
if skipped is not None:
return skipped
if current_platform != "darwin" and not current_platform.startswith("linux"):
return {
"ok": True,
Expand Down Expand Up @@ -702,6 +719,9 @@ def _validate_custom_runtime_dependencies(
current_platform = ctx.services.current_platform or sys.platform
if current_platform == "unknown":
current_platform = sys.platform
skipped = _linux_runtime_validation_opt_out(current_platform)
if skipped is not None:
return skipped
if current_platform != "darwin" and not current_platform.startswith("linux"):
return {
"ok": True,
Expand Down
2 changes: 1 addition & 1 deletion docs/directory.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,7 +84,7 @@
### Backend Capabilities

- Downloads `llama.cpp` releases from GitHub with SHA256 verification.
- Validates packaged runtime libraries with `otool` on macOS and `ldd` on Linux before launch, while preserving the local runtime-library search path.
- Validates packaged runtime libraries with `otool` on macOS and `ldd` on Linux before launch, while preserving the local runtime-library search path. `LLAMA_GUI_SKIP_LDD=1` opts out of Linux dependency probes for status, custom activation, preflight, and launch; skipped results report `checked: false` and `skip_reason: "LLAMA_GUI_SKIP_LDD"` and bypass the runtime-health cache. Executable and permission checks still apply.
- Runs `llama-server`, `llama-cli`, `llama-bench`, or `llama-perplexity` as a subprocess and streams stdout/stderr.
- Downloads the official WikiText-2 raw test file for Benchmarking clean perplexity runs.
- Handles preset, model file, and Hugging Face download APIs.
Expand Down
4 changes: 2 additions & 2 deletions docs/tests.md
Original file line number Diff line number Diff line change
Expand Up @@ -297,8 +297,8 @@ Backend tests use Python `unittest` and mostly exercise route/service logic with
- `test_web_fetch_transport.py`: real pinned HTTP/HTTPS connection and response parsing with only DNS/socket/TLS boundaries faked; covers destination pinning, Host/SNI, request targets, decoding, byte limits, redirect revalidation, cleanup, and sanitized transport failures.
- `test_server_baseline.py`: compatibility wrapper behavior, API dispatch, CORS, static asset versioning, and baseline server helpers.
- `test_diagnostics.py`: automatic startup capture for both entrypoints, main/worker exception tracebacks, native crash capture in an isolated child process, console preservation and unavailable-console support, idempotent setup, session retention, import side-effect isolation, and log-directory/disk failure isolation.
- `test_services.py`: service-level helpers for install specs, runtime validation, process/auth and active-runtime lifecycle, generation-bound health/stop behavior, downloads, file picker behavior, chat/search helpers, external-server registration (local-only validation, header-safe API keys, key never published or persisted, llama.cpp-aware probe identification, remembered-address round-tripping, unattended restore rules, runtime precedence), and HF validation.
- `test_custom_slots.py`: both fixed custom slots, persisted activation and official-build round trips, missing-tool/permission/runtime failures, Linux/macOS dependency fixtures, slot-specific executable and library paths, invalid/busy/running activation guards, official-download exclusions, status metadata, folder opening, and cleanup preservation. macOS dependency behavior uses fixtures on existing Linux/Windows runners; native macOS library loading still needs a manual check.
- `test_services.py`: service-level helpers for install specs, runtime validation, process/auth and active-runtime lifecycle, generation-bound health/stop behavior, downloads, file picker behavior, chat/search helpers, external-server registration (local-only validation, header-safe API keys, key never published or persisted, llama.cpp-aware probe identification, remembered-address round-tripping, unattended restore rules, runtime precedence), and HF validation. Linux `ldd` opt-out fixtures cover enabled/disabled values, tool/plugin probe skipping, cache behavior, and unchanged macOS validation.
- `test_custom_slots.py`: both fixed custom slots, persisted activation and official-build round trips, missing-tool/permission/runtime failures, Linux/macOS dependency fixtures, slot-specific executable and library paths, invalid/busy/running activation guards, official-download exclusions, status metadata, folder opening, and cleanup preservation. Linux `ldd` opt-out fixtures exercise status, activation, preflight, and launch for official and both custom installs while retaining executable guards and subprocess errors. NixOS launch and macOS library loading still need manual checks; fixtures run on the existing Linux/Windows runners.
- `test_review_regressions.py`: HF/API credential sanitization and quoted-argument rejection, live `/v1` fallback, external reconnect generations, installation rollback after grammar/permission/config failures, interrupted WikiText cleanup, and complete split-GGUF discovery/download/cancellation/overwrite handling. Frontend regression cases live in the existing launch-args, presets, Chat, and benchmark unit suites.
- `test_extracted_routes.py`: extracted route handlers and larger service flows, including preset secret scrubbing, launch preflight, active-runtime status, health/readiness, process launch/auth parsing, authoritative metrics/slots/chat targets, external chat-target registration and restore, HF download, tunnel, app update, and lifecycle routes.
- `test_docs_links.py`: documentation drift. Reads every git-tracked Markdown file and asserts its file references resolve — relative markdown links against the linking doc's directory, plus backtick-quoted repo-rooted paths (`docs/`, `ui/`, `backend/`, `tests/`, `scripts/`) against the repo root. Fenced code blocks are skipped, untracked `docs/design-docs/` is ignored, archived plan docs are exempt (existence-checked), and llama.cpp upstream paths colliding with the `tests/` prefix are allowlisted.
Expand Down
12 changes: 12 additions & 0 deletions docs/troubleshooting.md
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,18 @@ vulkaninfo --summary # Vulkan
rocminfo # ROCm / AMD kernel-driver access
```

If the binaries run manually but `ldd` reports missing libraries (for example on
NixOS), close the GUI and start it from the checkout with the Linux-only opt-out:

```bash
LLAMA_GUI_SKIP_LDD=1 ./mac_linux_start.sh
```

This skips the GUI's executable and ggml-plugin dependency probes for status,
custom-backend activation, preflight, and launch. Missing executables, permissions,
and actual runtime failures still apply; it does not supply missing libraries.
Restart without the variable to restore the checks.

Lemonade ROCm archives include user-space ROCm libraries, but the selected `gfx` target must match the GPU and the host still needs working AMD kernel-driver access. If model loading runs unusually long, the app keeps the process stoppable and adds a persistent warning directing you to the live process output.

## Antivirus / Defender quarantine
Expand Down
87 changes: 87 additions & 0 deletions tests/backend/test_custom_slots.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,9 @@

class CustomSlotsTests(unittest.TestCase):
def setUp(self):
opt_out = mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": ""})
opt_out.start()
self.addCleanup(opt_out.stop)
self.tmp = tempfile.TemporaryDirectory()
self.addCleanup(self.tmp.cleanup)
self.ctx = make_service_context(self.tmp.name)
Expand Down Expand Up @@ -124,6 +127,90 @@ def test_linux_validation_and_launch_environment_use_selected_slot(self):
self.assertEqual(env["LD_LIBRARY_PATH"], str(directory) + os.pathsep + "system-libs")
self.assertEqual(process_manager._fit_params_executable(self.ctx), directory / "llama-fit-params")

def test_linux_ldd_opt_out_allows_status_activation_preflight_and_launch(self):
self.ctx.services.current_platform = "linux"
self.ctx.services.binary_suffix = ""
self.ctx.services.normalize_llama_api_target = lambda host, port: {"host": host, "port": int(port)}
model = self.ctx.paths.root / "model.gguf"
model.write_text("model")
for backend in ("cpu", "custom", "custom-02"):
with self.subTest(backend=backend):
directory = self.write_build(backend)
(directory / "libggml-vulkan.so").write_text("plugin")
self.ctx.services.save_config({"backend": backend, "tag": "b123"})
fake_process = mock.Mock(pid=1234)
fake_process.poll.return_value = None
with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}), mock.patch.object(
llama_manager.os, "access", return_value=True
), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object(
process_manager.subprocess, "Popen", return_value=fake_process
) as popen, mock.patch.object(process_manager.threading, "Thread"):
if backend != "cpu":
activated = self.activate(backend)
self.assertTrue(activated.payload["ok"])
self.assertFalse(activated.payload["runtime_health"]["checked"])
self.assertEqual(activated.payload["runtime_health"]["skip_reason"], "LLAMA_GUI_SKIP_LDD")
response = DummyResponse()
status.get_status(Request("GET", "/api/status", "", {}), response, self.ctx)
self.assertTrue(response.payload["installed"])
self.assertFalse(response.payload["config_stale"])
self.assertFalse(response.payload["runtime_health"]["checked"])
args = ["-m", str(model)]
preflight = process_manager.preflight_launch(self.ctx, "llama-server", args, {})
self.assertTrue(preflight["ok"])
result = process_manager.launch_process(self.ctx, "llama-server", args)
self.assertEqual(result["pid"], 1234)
self.assertEqual(popen.call_args.args[0], [str(directory / "llama-server"), *args])
probe.assert_not_called()
fake_process.poll.return_value = 0

def test_linux_ldd_opt_out_keeps_missing_tool_and_permission_guards(self):
self.ctx.services.current_platform = "linux"
self.ctx.services.binary_suffix = ""
for backend in ("cpu", "custom", "custom-02"):
with self.subTest(backend=backend), mock.patch.dict(
os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}
), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object(
process_manager.subprocess, "Popen"
) as popen:
directory = self.write_build(backend)
self.ctx.services.save_config({"backend": backend, "tag": "b123"})
executable = directory / "llama-server"
executable.unlink()
result = process_manager.preflight_launch(self.ctx, "llama-server", [], {})
self.assertIn("not found", result["error"])
self.assertIn("not found", process_manager.launch_process(self.ctx, "llama-server", [])["error"])
if backend != "cpu":
self.assertEqual(self.activate(backend).payload["missing_required"], ["llama-server"])
response = DummyResponse()
status.get_status(Request("GET", "/api/status", "", {}), response, self.ctx)
self.assertFalse(response.payload["installed"])
executable.write_text("binary")
with mock.patch.object(llama_manager.os, "access", return_value=False):
result = process_manager.preflight_launch(self.ctx, "llama-server", [], {})
self.assertIn("not executable", result["error"])
self.assertIn("not executable", process_manager.launch_process(self.ctx, "llama-server", [])["error"])
if backend != "cpu":
self.assertEqual(len(self.activate(backend).payload["not_executable"]), 2)
probe.assert_not_called()
popen.assert_not_called()

def test_linux_ldd_opt_out_preserves_subprocess_launch_errors(self):
self.ctx.services.current_platform = "linux"
self.ctx.services.binary_suffix = ""
self.ctx.services.normalize_llama_api_target = lambda host, port: {"host": host, "port": int(port)}
self.write_build("cpu")
with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}), mock.patch.object(
llama_manager.os, "access", return_value=True
), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object(
process_manager.subprocess, "Popen", side_effect=OSError("runtime loader failed")
) as popen:
result = process_manager.launch_process(self.ctx, "llama-server", [])
self.assertEqual(result["error"], "runtime loader failed")
self.assertIsNone(self.ctx.state.process)
popen.assert_called_once()
probe.assert_not_called()

def test_macos_missing_runtime_and_permissions_leave_original_active(self):
self.ctx.services.current_platform = "darwin"
self.ctx.services.binary_suffix = ""
Expand Down
Loading
Loading