diff --git a/README.md b/README.md index fbf53b2..e0c9510 100644 --- a/README.md +++ b/README.md @@ -124,6 +124,14 @@ To recreate Linux/macOS launchers without reinstalling, run from the checkout: Linux respects `XDG_DATA_HOME` and uses `xdg-user-dir DESKTOP` when available; otherwise it uses an existing `~/Desktop`. Disabled or missing desktops are skipped. These launchers start the server in the background and open your browser. Keep the checkout in place, or rerun the helper from its new location after moving it. If shortcut creation fails, installation still completes and the terminal launch scripts remain available. +On Linux, including NixOS, if your llama.cpp binaries run manually but the GUI reports missing runtime libraries, close the GUI and start it from the checkout with: + +```bash +LLAMA_GUI_SKIP_LDD=1 ./mac_linux_start.sh +``` + +This opts out of the GUI's `ldd` checks, including custom-backend activation. It does not install missing libraries or bypass executable checks. Restart without the variable to restore library validation. + To build CUDA `llama.cpp` yourself on Linux, see `Linux_compile_toolkit/`. ## Install With Pinokio diff --git a/backend/services/llama_manager.py b/backend/services/llama_manager.py index 280cac1..780d8d9 100644 --- a/backend/services/llama_manager.py +++ b/backend/services/llama_manager.py @@ -18,6 +18,7 @@ import zipfile from typing import Any, Callable, Iterable, Mapping, Optional +from ..config import parse_bool_env from ..context import AppContext from ..http import sanitize_error from . import official_backends @@ -423,12 +424,28 @@ def get_linux_runtime_probe_files(runtime_dir: pathlib.Path) -> list[pathlib.Pat ) +def _linux_runtime_validation_opt_out(current_platform: str) -> Optional[dict[str, Any]]: + """Allow Linux users to bypass ldd when it misreports a working runtime.""" + if current_platform.startswith("linux") and parse_bool_env(os.environ.get("LLAMA_GUI_SKIP_LDD")): + return { + "ok": True, + "checked": False, + "skip_reason": "LLAMA_GUI_SKIP_LDD", + "required_runtime_files": [], + "missing_runtime_files": [], + } + return None + + def validate_runtime_dependencies( ctx: AppContext, tools: Optional[Iterable[str]] = None ) -> dict[str, Any]: current_platform = ctx.services.current_platform or sys.platform if current_platform == "unknown": current_platform = sys.platform + skipped = _linux_runtime_validation_opt_out(current_platform) + if skipped is not None: + return skipped if current_platform != "darwin" and not current_platform.startswith("linux"): return { "ok": True, @@ -702,6 +719,9 @@ def _validate_custom_runtime_dependencies( current_platform = ctx.services.current_platform or sys.platform if current_platform == "unknown": current_platform = sys.platform + skipped = _linux_runtime_validation_opt_out(current_platform) + if skipped is not None: + return skipped if current_platform != "darwin" and not current_platform.startswith("linux"): return { "ok": True, diff --git a/docs/directory.md b/docs/directory.md index ca7f61d..62fa536 100644 --- a/docs/directory.md +++ b/docs/directory.md @@ -84,7 +84,7 @@ ### Backend Capabilities - Downloads `llama.cpp` releases from GitHub with SHA256 verification. -- Validates packaged runtime libraries with `otool` on macOS and `ldd` on Linux before launch, while preserving the local runtime-library search path. +- Validates packaged runtime libraries with `otool` on macOS and `ldd` on Linux before launch, while preserving the local runtime-library search path. `LLAMA_GUI_SKIP_LDD=1` opts out of Linux dependency probes for status, custom activation, preflight, and launch; skipped results report `checked: false` and `skip_reason: "LLAMA_GUI_SKIP_LDD"` and bypass the runtime-health cache. Executable and permission checks still apply. - Runs `llama-server`, `llama-cli`, `llama-bench`, or `llama-perplexity` as a subprocess and streams stdout/stderr. - Downloads the official WikiText-2 raw test file for Benchmarking clean perplexity runs. - Handles preset, model file, and Hugging Face download APIs. diff --git a/docs/tests.md b/docs/tests.md index 03bcb42..aef205d 100644 --- a/docs/tests.md +++ b/docs/tests.md @@ -297,8 +297,8 @@ Backend tests use Python `unittest` and mostly exercise route/service logic with - `test_web_fetch_transport.py`: real pinned HTTP/HTTPS connection and response parsing with only DNS/socket/TLS boundaries faked; covers destination pinning, Host/SNI, request targets, decoding, byte limits, redirect revalidation, cleanup, and sanitized transport failures. - `test_server_baseline.py`: compatibility wrapper behavior, API dispatch, CORS, static asset versioning, and baseline server helpers. - `test_diagnostics.py`: automatic startup capture for both entrypoints, main/worker exception tracebacks, native crash capture in an isolated child process, console preservation and unavailable-console support, idempotent setup, session retention, import side-effect isolation, and log-directory/disk failure isolation. -- `test_services.py`: service-level helpers for install specs, runtime validation, process/auth and active-runtime lifecycle, generation-bound health/stop behavior, downloads, file picker behavior, chat/search helpers, external-server registration (local-only validation, header-safe API keys, key never published or persisted, llama.cpp-aware probe identification, remembered-address round-tripping, unattended restore rules, runtime precedence), and HF validation. -- `test_custom_slots.py`: both fixed custom slots, persisted activation and official-build round trips, missing-tool/permission/runtime failures, Linux/macOS dependency fixtures, slot-specific executable and library paths, invalid/busy/running activation guards, official-download exclusions, status metadata, folder opening, and cleanup preservation. macOS dependency behavior uses fixtures on existing Linux/Windows runners; native macOS library loading still needs a manual check. +- `test_services.py`: service-level helpers for install specs, runtime validation, process/auth and active-runtime lifecycle, generation-bound health/stop behavior, downloads, file picker behavior, chat/search helpers, external-server registration (local-only validation, header-safe API keys, key never published or persisted, llama.cpp-aware probe identification, remembered-address round-tripping, unattended restore rules, runtime precedence), and HF validation. Linux `ldd` opt-out fixtures cover enabled/disabled values, tool/plugin probe skipping, cache behavior, and unchanged macOS validation. +- `test_custom_slots.py`: both fixed custom slots, persisted activation and official-build round trips, missing-tool/permission/runtime failures, Linux/macOS dependency fixtures, slot-specific executable and library paths, invalid/busy/running activation guards, official-download exclusions, status metadata, folder opening, and cleanup preservation. Linux `ldd` opt-out fixtures exercise status, activation, preflight, and launch for official and both custom installs while retaining executable guards and subprocess errors. NixOS launch and macOS library loading still need manual checks; fixtures run on the existing Linux/Windows runners. - `test_review_regressions.py`: HF/API credential sanitization and quoted-argument rejection, live `/v1` fallback, external reconnect generations, installation rollback after grammar/permission/config failures, interrupted WikiText cleanup, and complete split-GGUF discovery/download/cancellation/overwrite handling. Frontend regression cases live in the existing launch-args, presets, Chat, and benchmark unit suites. - `test_extracted_routes.py`: extracted route handlers and larger service flows, including preset secret scrubbing, launch preflight, active-runtime status, health/readiness, process launch/auth parsing, authoritative metrics/slots/chat targets, external chat-target registration and restore, HF download, tunnel, app update, and lifecycle routes. - `test_docs_links.py`: documentation drift. Reads every git-tracked Markdown file and asserts its file references resolve — relative markdown links against the linking doc's directory, plus backtick-quoted repo-rooted paths (`docs/`, `ui/`, `backend/`, `tests/`, `scripts/`) against the repo root. Fenced code blocks are skipped, untracked `docs/design-docs/` is ignored, archived plan docs are exempt (existence-checked), and llama.cpp upstream paths colliding with the `tests/` prefix are allowlisted. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index 9faaf21..79dce2e 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -45,6 +45,18 @@ vulkaninfo --summary # Vulkan rocminfo # ROCm / AMD kernel-driver access ``` +If the binaries run manually but `ldd` reports missing libraries (for example on +NixOS), close the GUI and start it from the checkout with the Linux-only opt-out: + +```bash +LLAMA_GUI_SKIP_LDD=1 ./mac_linux_start.sh +``` + +This skips the GUI's executable and ggml-plugin dependency probes for status, +custom-backend activation, preflight, and launch. Missing executables, permissions, +and actual runtime failures still apply; it does not supply missing libraries. +Restart without the variable to restore the checks. + Lemonade ROCm archives include user-space ROCm libraries, but the selected `gfx` target must match the GPU and the host still needs working AMD kernel-driver access. If model loading runs unusually long, the app keeps the process stoppable and adds a persistent warning directing you to the live process output. ## Antivirus / Defender quarantine diff --git a/tests/backend/test_custom_slots.py b/tests/backend/test_custom_slots.py index cfe72c4..f2346c7 100644 --- a/tests/backend/test_custom_slots.py +++ b/tests/backend/test_custom_slots.py @@ -16,6 +16,9 @@ class CustomSlotsTests(unittest.TestCase): def setUp(self): + opt_out = mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": ""}) + opt_out.start() + self.addCleanup(opt_out.stop) self.tmp = tempfile.TemporaryDirectory() self.addCleanup(self.tmp.cleanup) self.ctx = make_service_context(self.tmp.name) @@ -124,6 +127,90 @@ def test_linux_validation_and_launch_environment_use_selected_slot(self): self.assertEqual(env["LD_LIBRARY_PATH"], str(directory) + os.pathsep + "system-libs") self.assertEqual(process_manager._fit_params_executable(self.ctx), directory / "llama-fit-params") + def test_linux_ldd_opt_out_allows_status_activation_preflight_and_launch(self): + self.ctx.services.current_platform = "linux" + self.ctx.services.binary_suffix = "" + self.ctx.services.normalize_llama_api_target = lambda host, port: {"host": host, "port": int(port)} + model = self.ctx.paths.root / "model.gguf" + model.write_text("model") + for backend in ("cpu", "custom", "custom-02"): + with self.subTest(backend=backend): + directory = self.write_build(backend) + (directory / "libggml-vulkan.so").write_text("plugin") + self.ctx.services.save_config({"backend": backend, "tag": "b123"}) + fake_process = mock.Mock(pid=1234) + fake_process.poll.return_value = None + with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}), mock.patch.object( + llama_manager.os, "access", return_value=True + ), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object( + process_manager.subprocess, "Popen", return_value=fake_process + ) as popen, mock.patch.object(process_manager.threading, "Thread"): + if backend != "cpu": + activated = self.activate(backend) + self.assertTrue(activated.payload["ok"]) + self.assertFalse(activated.payload["runtime_health"]["checked"]) + self.assertEqual(activated.payload["runtime_health"]["skip_reason"], "LLAMA_GUI_SKIP_LDD") + response = DummyResponse() + status.get_status(Request("GET", "/api/status", "", {}), response, self.ctx) + self.assertTrue(response.payload["installed"]) + self.assertFalse(response.payload["config_stale"]) + self.assertFalse(response.payload["runtime_health"]["checked"]) + args = ["-m", str(model)] + preflight = process_manager.preflight_launch(self.ctx, "llama-server", args, {}) + self.assertTrue(preflight["ok"]) + result = process_manager.launch_process(self.ctx, "llama-server", args) + self.assertEqual(result["pid"], 1234) + self.assertEqual(popen.call_args.args[0], [str(directory / "llama-server"), *args]) + probe.assert_not_called() + fake_process.poll.return_value = 0 + + def test_linux_ldd_opt_out_keeps_missing_tool_and_permission_guards(self): + self.ctx.services.current_platform = "linux" + self.ctx.services.binary_suffix = "" + for backend in ("cpu", "custom", "custom-02"): + with self.subTest(backend=backend), mock.patch.dict( + os.environ, {"LLAMA_GUI_SKIP_LDD": "1"} + ), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object( + process_manager.subprocess, "Popen" + ) as popen: + directory = self.write_build(backend) + self.ctx.services.save_config({"backend": backend, "tag": "b123"}) + executable = directory / "llama-server" + executable.unlink() + result = process_manager.preflight_launch(self.ctx, "llama-server", [], {}) + self.assertIn("not found", result["error"]) + self.assertIn("not found", process_manager.launch_process(self.ctx, "llama-server", [])["error"]) + if backend != "cpu": + self.assertEqual(self.activate(backend).payload["missing_required"], ["llama-server"]) + response = DummyResponse() + status.get_status(Request("GET", "/api/status", "", {}), response, self.ctx) + self.assertFalse(response.payload["installed"]) + executable.write_text("binary") + with mock.patch.object(llama_manager.os, "access", return_value=False): + result = process_manager.preflight_launch(self.ctx, "llama-server", [], {}) + self.assertIn("not executable", result["error"]) + self.assertIn("not executable", process_manager.launch_process(self.ctx, "llama-server", [])["error"]) + if backend != "cpu": + self.assertEqual(len(self.activate(backend).payload["not_executable"]), 2) + probe.assert_not_called() + popen.assert_not_called() + + def test_linux_ldd_opt_out_preserves_subprocess_launch_errors(self): + self.ctx.services.current_platform = "linux" + self.ctx.services.binary_suffix = "" + self.ctx.services.normalize_llama_api_target = lambda host, port: {"host": host, "port": int(port)} + self.write_build("cpu") + with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}), mock.patch.object( + llama_manager.os, "access", return_value=True + ), mock.patch.object(llama_manager.subprocess, "run") as probe, mock.patch.object( + process_manager.subprocess, "Popen", side_effect=OSError("runtime loader failed") + ) as popen: + result = process_manager.launch_process(self.ctx, "llama-server", []) + self.assertEqual(result["error"], "runtime loader failed") + self.assertIsNone(self.ctx.state.process) + popen.assert_called_once() + probe.assert_not_called() + def test_macos_missing_runtime_and_permissions_leave_original_active(self): self.ctx.services.current_platform = "darwin" self.ctx.services.binary_suffix = "" diff --git a/tests/backend/test_services.py b/tests/backend/test_services.py index bfcdd15..2284083 100644 --- a/tests/backend/test_services.py +++ b/tests/backend/test_services.py @@ -959,6 +959,11 @@ def test_handles_large_files(self): class RuntimeDependencyValidationTests(unittest.TestCase): + def setUp(self): + opt_out = mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": ""}) + opt_out.start() + self.addCleanup(opt_out.stop) + def make_runtime_context(self, tmpdir, platform_name="darwin"): from backend.context import AppContext, AppPaths, BackendServices @@ -1139,6 +1144,77 @@ def test_validate_linux_runtime_dependencies_degrades_without_ldd(self): self.assertFalse(result["checked"]) self.assertEqual(result["unchecked_tools"], ["llama-server"]) + def test_linux_ldd_opt_out_skips_tools_and_plugins_without_caching(self): + with tempfile.TemporaryDirectory() as tmp: + ctx = self.make_runtime_context(tmp, "linux") + for name in ("llama-cli", "llama-server", "libggml-vulkan.so"): + (ctx.paths.llama_bin / name).write_text("binary") + for value in ("1", "true", "yes", "on"): + with self.subTest(value=value), mock.patch.dict( + os.environ, {"LLAMA_GUI_SKIP_LDD": value} + ), mock.patch.object(llama_manager.subprocess, "run") as probe: + result = llama_manager.validate_runtime_dependencies(ctx) + self.assertTrue(result["ok"]) + self.assertFalse(result["checked"]) + self.assertEqual(result["skip_reason"], "LLAMA_GUI_SKIP_LDD") + self.assertEqual(result["required_runtime_files"], []) + self.assertEqual(result["missing_runtime_files"], []) + self.assertEqual(ctx.state.runtime_health_cache, {}) + probe.assert_not_called() + + def test_linux_ldd_opt_out_disabled_values_keep_dependency_validation(self): + with tempfile.TemporaryDirectory() as tmp: + ctx = self.make_runtime_context(tmp, "linux") + (ctx.paths.llama_bin / "llama-server").write_text("binary") + for value in ("", "0", "false", "no", "off", "invalid"): + ctx.state.clear_runtime_health_cache() + with self.subTest(value=value), mock.patch.dict( + os.environ, {"LLAMA_GUI_SKIP_LDD": value} + ), mock.patch.object( + llama_manager, "get_linux_missing_libraries", return_value=["libvulkan.so.1"] + ) as probe: + result = llama_manager.validate_runtime_dependencies(ctx, ["llama-server"]) + self.assertFalse(result["ok"]) + self.assertEqual(result["missing_runtime_files"], ["libvulkan.so.1"]) + probe.assert_called_once() + + def test_linux_ldd_opt_out_bypasses_cached_failure_without_replacing_it(self): + with tempfile.TemporaryDirectory() as tmp: + ctx = self.make_runtime_context(tmp, "linux") + (ctx.paths.llama_bin / "llama-server").write_text("binary") + with mock.patch.object( + llama_manager, "get_linux_missing_libraries", return_value=["libvulkan.so.1"] + ) as probe: + failed = llama_manager.validate_runtime_dependencies(ctx, ["llama-server"]) + self.assertFalse(failed["ok"]) + with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}): + skipped = llama_manager.validate_runtime_dependencies(ctx, ["llama-server"]) + self.assertTrue(skipped["ok"]) + self.assertFalse(skipped["checked"]) + self.assertEqual( + llama_manager.validate_runtime_dependencies(ctx, ["llama-server"]), failed + ) + probe.assert_called_once() + + def test_linux_ldd_opt_out_does_not_skip_macos_dependency_validation(self): + with tempfile.TemporaryDirectory() as tmp: + ctx = self.make_runtime_context(tmp, "darwin") + (ctx.paths.llama_bin / "llama-server").write_text("binary") + ctx.paths.llama_custom_bin.mkdir(parents=True) + (ctx.paths.llama_custom_bin / "llama-server").write_text("binary") + with mock.patch.dict(os.environ, {"LLAMA_GUI_SKIP_LDD": "1"}), mock.patch.object( + llama_manager, "get_macos_rpath_libraries", return_value=["libllama.0.dylib"] + ) as probe: + results = ( + llama_manager.validate_runtime_dependencies(ctx, ["llama-server"]), + llama_manager._validate_custom_runtime_dependencies(ctx, ["llama-server"]), + ) + for result in results: + self.assertFalse(result["ok"]) + self.assertTrue(result["checked"]) + self.assertEqual(result["missing_runtime_files"], ["libllama.0.dylib"]) + self.assertEqual(probe.call_count, 2) + def test_validate_macos_custom_runtime_checks_custom_bin_only(self): with tempfile.TemporaryDirectory() as tmp: ctx = self.make_runtime_context(tmp)