diff --git a/.clangd b/.clangd new file mode 100644 index 0000000..a26fa83 --- /dev/null +++ b/.clangd @@ -0,0 +1,20 @@ +# clangd config for cthreads (Cursor / VS Code clangd). +# Paths are relative to this file (repo root), except Vulkan SDK. + +CompileFlags: + Add: + - -std=c++17 + # Match CMake CTHREADS_GPU=ON so #ifdef CTHREADS_WITH_GPU paths analyze correctly. + - -DCTHREADS_WITH_GPU=1 + # Project includes (same as CMake target_include_directories for GPU builds). + - -Isrc/cthreads/cpp/headers + - -Isrc/cthreads/cpp/gpu/headers + # Vulkan headers from local SDK (VULKAN_SDK=C:\VulkanSDK\1.4.357.0). + # Update this -I if you install a newer SDK version. + - -IC:/VulkanSDK/1.4.357.0/Include + +Diagnostics: + UnusedIncludes: None + +Index: + Background: Build diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index b3579a1..c580442 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -36,3 +36,58 @@ jobs: - name: Pytest run: python -m pytest tests/ -q + + # Compile-smokes the Vulkan path (CTHREADS_GPU=ON). Live device tests still skip + # when no ICD is present; the goal is to catch header / glslang / link breakage. + test-gpu: + name: test-gpu (${{ matrix.os }}, py${{ matrix.python }}) + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + python: ["3.12"] + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python }} + + - name: Install Vulkan headers (Ubuntu) + if: runner.os == 'Linux' + run: | + sudo apt-get update + sudo apt-get install -y libvulkan-dev vulkan-tools + + - name: Install Vulkan SDK (Windows) + if: runner.os == 'Windows' + uses: humbletim/setup-vulkan-sdk@v1.2.1 + with: + vulkan-query-version: 1.3.296.0 + vulkan-components: Vulkan-Headers, Vulkan-Loader + vulkan-use-cache: true + + - name: Install CMake, Ninja, package (GPU ON, Linux) + if: runner.os == 'Linux' + run: | + python -m pip install -U pip + python -m pip install cmake ninja + export CMAKE_ARGS="-DCTHREADS_GPU=ON" + python -m pip install -e ".[test]" + + - name: Install CMake, Ninja, package (GPU ON, Windows) + if: runner.os == 'Windows' + shell: pwsh + run: | + python -m pip install -U pip + python -m pip install cmake ninja + $env:CMAKE_ARGS = "-DCTHREADS_GPU=ON" + python -m pip install -e ".[test]" + + - name: Assert GPU extension linked + run: | + python -c "from cthreads import _ext, gpu; assert hasattr(_ext, 'gpu'), 'CTHREADS_GPU build missing _ext.gpu'; print('available=', gpu.available())" + + - name: Pytest (GPU build; device tests skip if no ICD) + run: python -m pytest tests/ -q diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index e283859..aa1c4a7 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -2,6 +2,11 @@ name: Release # Production PyPI: only when you click Publish on a GitHub Release. # Manual "Run workflow" always targets TestPyPI, never PyPI. +# +# Artifacts: +# - cthreads CPU wheels + sdist (CTHREADS_GPU=OFF) +# - cthreads-gpu GPU wheels (CTHREADS_GPU=ON) +# Users: pip install cthreads OR pip install 'cthreads[gpu]' (pulls cthreads-gpu) on: release: @@ -28,6 +33,32 @@ jobs: python -m pip install -e ".[test]" python -m pytest tests/ -q + test-gpu: + name: test-gpu (ubuntu) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Install Vulkan headers + run: | + sudo apt-get update + sudo apt-get install -y libvulkan-dev + + - name: Install and test (GPU ON) + run: | + python -m pip install -U pip cmake ninja + export CMAKE_ARGS="-DCTHREADS_GPU=ON" + python -m pip install -e ".[test]" + python - <<'PY' + from cthreads import _ext + assert hasattr(_ext, "gpu"), "CTHREADS_GPU build missing _ext.gpu" + PY + python -m pytest tests/ -q + check-version: name: tag matches pyproject if: github.event_name == 'release' @@ -37,22 +68,34 @@ jobs: - uses: actions/setup-python@v5 with: python-version: "3.12" - - name: Compare release tag to project.version + - name: Compare release tag to project.version and [gpu] pin env: TAG: ${{ github.event.release.tag_name }} run: | python - <<'PY' import os import pathlib + import re import sys import tomllib tag = os.environ["TAG"].removeprefix("v") - ver = tomllib.loads(pathlib.Path("pyproject.toml").read_text(encoding="utf-8"))["project"]["version"] + text = pathlib.Path("pyproject.toml").read_text(encoding="utf-8") + ver = tomllib.loads(text)["project"]["version"] if tag != ver: print(f"GitHub tag version {tag!r} != pyproject.toml version {ver!r}") sys.exit(1) - print(f"version ok: {ver}") + m = re.search(r'gpu\s*=\s*\["cthreads-gpu==([^"]+)"\]', text) + if not m: + print("missing [project.optional-dependencies] gpu = [\"cthreads-gpu==...\"]") + sys.exit(1) + if m.group(1) != ver: + print( + f"optional gpu pin cthreads-gpu=={m.group(1)!r} " + f"!= project.version {ver!r}" + ) + sys.exit(1) + print(f"version ok: {ver} (gpu extra pin matches)") PY build-sdist: @@ -74,7 +117,7 @@ jobs: path: dist/*.tar.gz build-wheels: - name: wheels (${{ matrix.os }}) + name: wheels-cpu (${{ matrix.os }}) needs: [test] runs-on: ${{ matrix.os }} strategy: @@ -83,16 +126,70 @@ jobs: os: [ubuntu-latest, windows-latest] steps: - uses: actions/checkout@v4 - - name: Build wheels + - name: Build CPU wheels uses: pypa/cibuildwheel@v4.2.0 - uses: actions/upload-artifact@v4 with: name: cibw-wheels-${{ matrix.os }} path: wheelhouse/*.whl + build-wheels-gpu: + name: wheels-gpu (${{ matrix.os }}) + needs: [test-gpu] + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + steps: + - uses: actions/checkout@v4 + + - name: Install Vulkan headers (Ubuntu host; manylinux uses before-all) + if: runner.os == 'Linux' + run: | + sudo apt-get update + sudo apt-get install -y libvulkan-dev + + - name: Install Vulkan SDK (Windows) + if: runner.os == 'Windows' + uses: humbletim/setup-vulkan-sdk@v1.2.1 + with: + vulkan-query-version: 1.3.296.0 + vulkan-components: Vulkan-Headers, Vulkan-Loader + vulkan-use-cache: true + + - name: Retarget project name to cthreads-gpu + run: python scripts/retarget_gpu_wheel.py + + - name: Build GPU wheels + uses: pypa/cibuildwheel@v4.2.0 + env: + # Headers only at build time; loader is resolved at runtime. + CIBW_ENVIRONMENT: CMAKE_ARGS=-DCTHREADS_GPU=ON + CIBW_BEFORE_ALL_LINUX: | + set -eux + if command -v dnf >/dev/null 2>&1; then + dnf install -y vulkan-headers vulkan-loader-devel git + else + yum install -y vulkan-headers vulkan-loader-devel git + fi + # glslang FetchContent makes this slower than CPU wheels. + CIBW_BUILD_VERBOSITY: 1 + + - uses: actions/upload-artifact@v4 + with: + name: cibw-wheels-gpu-${{ matrix.os }} + path: wheelhouse/*.whl + publish-pypi: name: publish to PyPI - needs: [test, check-version, build-sdist, build-wheels] + needs: + - test + - test-gpu + - check-version + - build-sdist + - build-wheels + - build-wheels-gpu if: github.event_name == 'release' && github.event.action == 'published' runs-on: ubuntu-latest environment: @@ -106,11 +203,20 @@ jobs: pattern: cibw-* path: dist merge-multiple: true + - name: Show artifacts + run: ls -la dist + # Uploads both cthreads and cthreads-gpu wheels. Configure Trusted + # Publishing for BOTH PyPI projects (same workflow: release.yml). - uses: pypa/gh-action-pypi-publish@release/v1 publish-testpypi: name: publish to TestPyPI - needs: [test, build-sdist, build-wheels] + needs: + - test + - test-gpu + - build-sdist + - build-wheels + - build-wheels-gpu if: github.event_name == 'workflow_dispatch' runs-on: ubuntu-latest environment: @@ -124,6 +230,8 @@ jobs: pattern: cibw-* path: dist merge-multiple: true + - name: Show artifacts + run: ls -la dist - uses: pypa/gh-action-pypi-publish@release/v1 with: repository-url: https://test.pypi.org/legacy/ diff --git a/.gitignore b/.gitignore index 3be454b..06432db 100644 --- a/.gitignore +++ b/.gitignore @@ -17,6 +17,7 @@ dist/ *.dylib .cthreads_cache.json .pytest_cache/ +.vscode/ # CMake / MSVC litter - must never land in the source tree src/cthreads/cpp/CMakeCache.txt @@ -51,6 +52,7 @@ api/ # >>> cthreads (auto) __Thread__/ __Threadable__/ +__Gpu__/ .cthreads_cache.json cthreads_kernels.dll cthreads_kernels.so @@ -59,4 +61,6 @@ libcthreads_kernels.so libcthreads_kernels.dylib # <<< cthreads (auto) -todo.md \ No newline at end of file +todo.md +# demo artifacts +*.gif \ No newline at end of file diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..f98bfb4 --- /dev/null +++ b/LICENSE @@ -0,0 +1,127 @@ +MIT License + +Copyright (c) 2026 Tobias Karusseit + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. + +================================================================================ +Third-party notices +================================================================================ + +cthreads depends on (or may optionally link / redistribute) the following +third-party components. Their licenses apply to those components only. +Full license texts are available from the upstream projects. + +-------------------------------------------------------------------------------- +pybind11 +-------------------------------------------------------------------------------- +Used to bind the native `_ext` module to Python (FetchContent or system +package via CMake). + +License: BSD-3-Clause +Copyright (c) 2016 Wenzel Jakob , All rights reserved. +Homepage: https://github.com/pybind/pybind11 + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright notice, this + list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. + +3. Neither the name of the copyright holder nor the names of its contributors + may be used to endorse or promote products derived from this software + without specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE +DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE +FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL +DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR +SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, +OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + +-------------------------------------------------------------------------------- +Vulkan (Khronos Group) — optional, CTHREADS_GPU=ON +-------------------------------------------------------------------------------- +GPU builds use Vulkan headers from the Vulkan SDK / CMake Vulkan package. +At runtime, cthreads loads the system Vulkan loader (for example vulkan-1) +dynamically; the loader and ICD are provided by the platform / GPU vendor +and are not redistributed as part of cthreads by default. + +Vulkan-Headers / related Khronos materials are typically licensed under the +Apache License, Version 2.0. +Homepage: https://github.com/KhronosGroup/Vulkan-Headers +License reference: https://www.apache.org/licenses/LICENSE-2.0 + +-------------------------------------------------------------------------------- +Khronos glslang — GPU GLSL -> SPIR-V (CTHREADS_GPU=ON) +-------------------------------------------------------------------------------- +When built with CTHREADS_GPU=ON, cthreads FetchContent-vendors and statically +links Khronos glslang into `_ext` to implement `_ext.gpu.compile_glsl`. +This is the same compiler engine Google shaderc wraps. End users of a GPU +wheel do not need glslc or the Vulkan SDK shader tools. + +License: Apache License, Version 2.0 (with BSD-style components in the tree) +Homepage: https://github.com/KhronosGroup/glslang +License reference: https://www.apache.org/licenses/LICENSE-2.0 + +Upstream license text is copied at build time into: + + cthreads/gpu/third_party_notices/ + +(see README.md there). Redistribute that directory with any binary that +includes the GPU extension. + +-------------------------------------------------------------------------------- +scikit-build-core +-------------------------------------------------------------------------------- +Used as the Python build backend to drive CMake (build-time dependency; not +part of the runtime import of cthreads). + +License: Apache License, Version 2.0 +Homepage: https://github.com/scikit-build/scikit-build-core +License reference: https://www.apache.org/licenses/LICENSE-2.0 + +-------------------------------------------------------------------------------- +Apache License, Version 2.0 (summary for Apache-licensed deps above) +-------------------------------------------------------------------------------- +You may reproduce and distribute copies of Apache-2.0 works under the terms +of that license. A copy of the full license text is available at: + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the Apache License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the Apache License for the specific language governing permissions and +limitations under the License. + +-------------------------------------------------------------------------------- +Python +-------------------------------------------------------------------------------- +cthreads is designed to run on CPython. The Python interpreter and standard +library are governed by the Python Software Foundation License. +Homepage: https://www.python.org/ diff --git a/ReadMe.md b/ReadMe.md index 8be2df4..7394ac5 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -11,8 +11,9 @@ Use `@Thread` on functions/methods and `@Threadable` on classes. The whitelist c ### Docs -- [Install](./docs/install.md) +- [Install](./docs/install.md) (includes GPU / Vulkan notes) - [Guides](./docs/index.md) +- [GPU guides (0.2.0)](./docs/guide/gpu/README.md) - [Release (GitHub / PyPI)](./docs/release.md) - [Math & linalg](./docs/guide/math_and_linalg.md) - [Compiler notes](./docs/COMPILER.md) @@ -31,6 +32,8 @@ Published wheels (Linux / Windows x86_64) and the sdist are on [PyPI](https://py ```bash pip install cthreads +# Vulkan GPU (@Gpu) support: +pip install "cthreads[gpu]" ``` You still need a C++ compiler for the first `thread(...)` (user kernels). On Linux, wheels include a prebuilt `_ext`; CMake is only required if you install from the sdist or develop from source. @@ -56,6 +59,7 @@ Annotate what should become a native kernel: - **`@Thread`** - functions / methods compiled to C++ - **`@Threadable`** - classes compiled to C++ structs (shared state across kernels) +- **`@Gpu`** - functions compiled to Vulkan compute (lists of scalars; see [GPU guides](./docs/guide/gpu/README.md)) ## Supported types @@ -180,3 +184,27 @@ result = await job Signature: `cthreads.thread(fn, *args, force: bool = False, **kwargs) -> Job`. ---- + +# GPU (`@Gpu`, from 0.2.0) + +Vulkan compute kernels use the same "annotate then launch" idea on a separate backend: + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + +x = [1.0, 2.0, 3.0, 4.0] +y = [10.0, 20.0, 30.0, 40.0] +gpu(saxpy, len(x), 2.0, x, y).join() +``` + +Full guides (concepts, best practices, examples): [docs/guide/gpu/README.md](./docs/guide/gpu/README.md). +Install / drivers: [docs/install.md](./docs/install.md#gpu-vulkan-compute). + +---- diff --git a/docs/API.md b/docs/API.md index c5de60a..9e46321 100644 --- a/docs/API.md +++ b/docs/API.md @@ -473,3 +473,36 @@ assert job.result() == 10 | `d.get(k, default)` / `d.pop(k, default)` | Bare `d.get(k)` / `d.pop(k)` (needs Optional / exceptions) | | Bare-name receivers: `xs.append(v)` | Nested receivers: `self.items.append(v)` (not yet) | | Annotated locals + name/attr assign | `xs[i] = v` / `d[k] = v` plain assign (not yet) | + +--- + +## 11. GPU (`cthreads.gpu`, 0.2.0) + +Public Vulkan compute path. Full narrative docs: [guide/gpu/README.md](./guide/gpu/README.md). +Compact API: [guide/gpu/api.md](./guide/gpu/api.md). + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + +gpu(saxpy, len(x), 2.0, x, y).join() +``` + +| Piece | Role | +|-------|------| +| `@Gpu` | Mark + register a device kernel (`-> None`) | +| `gpu(fn, *args)` | Compile if needed, launch, return `GpuJob` | +| `GpuJob.join(download=True)` | Wait; download list args by default | +| `GpuArena` | Keep lists resident across launches | +| `GlobalIdx` / friends | Invocation indexes | +| `cthreads.sync.__sync_threads` / `Barrier.arrive_and_wait()` | Workgroup barrier inside `@Gpu` | + +GPU types are narrower than CPU: scalars and `list` of scalars only. No mid-run +Python observe. Shared memory is planned for **0.2.1**. + diff --git a/docs/STYLE.md b/docs/STYLE.md index 3b6f514..647a2ec 100644 --- a/docs/STYLE.md +++ b/docs/STYLE.md @@ -77,7 +77,9 @@ def example_fn() -> None: 1. the use of special utf characters is not permitted ```latex -Exmaple: +Example: — should be - or depending on ctx ,. etc. → should be -> ``` + +2. `from __future__ import annotations` is only permitted iff its used to avoid import errors, improve import performance, with typechecking, or to avoid any other error. Otherwise explicit imports are preffered to ensure easy dependency maintenace \ No newline at end of file diff --git a/docs/concepts.md b/docs/concepts.md index 6acfbd1..a59403c 100644 --- a/docs/concepts.md +++ b/docs/concepts.md @@ -87,5 +87,6 @@ One important caveat: during the run, Python-side memory is **not** live-updated | Sync / locks / state | [guide/sync.md](./guide/sync.md), [sync_state_docs](./sync_state_docs.md) | | Jobs / async | [guide/jobs.md](./guide/jobs.md) | | Math and linalg | [guide/math_and_linalg.md](./guide/math_and_linalg.md) | +| GPU (`@Gpu` / `gpu()`, 0.2.0) | [guide/gpu/README.md](./guide/gpu/README.md) | | Compiler details | [COMPILER.md](./COMPILER.md) | | API surface | [README](../README.md), [API.md](./API.md) | diff --git a/docs/gpu_future_cpu_to_gpu.md b/docs/gpu_future_cpu_to_gpu.md new file mode 100644 index 0000000..72befc1 --- /dev/null +++ b/docs/gpu_future_cpu_to_gpu.md @@ -0,0 +1,113 @@ +# Future work: CPU kernels calling GPU + +**Status:** Planned after the public Python GPU package (landed in **0.2.0**: +`@Gpu`, `gpu()` / `GpuJob`, marshal, launch, join, writeback, `GpuArena`, +workgroup barriers). Shared memory is tracked separately for **0.2.1**. + +This note records the intended design so we do not bolt on a second launch stack later. Write it as if the GPU runtime and Python `gpu()` path already exist; this feature only adds a native caller on top. + +--- + +## Idea + +GPU work is started from **Python** today (`gpu(fn, *args) -> GpuJob`). + +A **CPU** `@Thread` kernel (native C++ in `kernels.dll`) should be able to launch the same GPU work and wait on it, without going back through Python for every dispatch. + +Example intent (shape, not final API): + +```text +@Thread +def step(...): + # CPU work + gpu_job = launch_gpu(some_gpu_fn, ...) # C++ handle + # more CPU work overlapping GPU + gpu_job.join() +``` + +Same GpuPack / descriptor / pipeline / writeback machinery as `gpu()`. Only the **caller** changes: Python host vs native CPU kernel. + +--- + +## Why this is a separate step + +1. Python `gpu()` is the place that proves the permanent launch types and memory model. +2. Native CPU->GPU needs a stable **C++ `GpuJob` handle** and a native launch entry that mirrors Python marshal. +3. Overlapping CPU threads calling GPU will stress **transfer** and **queue submit** concurrency. That is easier to size once the single-caller path is solid. + +--- + +## Assumptions (GPU system already done) + +Before this work, the GPU package already provides: + +- `Context`, TransferEngine (or an engine pool API), `memory::`, `GpuPack` +- Descriptor and pipeline path, `GpuJob.join`, writeback into caller-owned buffers / objects +- Kernel cache keyed by symbol (layout, pipeline, SPIR-V) +- Inflight state owned by each job (pack, descriptor set, fence) +- Python `gpu()` as a thin wrapper over that C++ surface + +--- + +## What this needs + +### A. C++ job handle (GPU side) + +This may already exist for Python bindings. Confirm it is first-class native, not Python-only. + +- Type like `cthreads::gpu::GpuJob`. +- API roughly: create/launch from packed args or from a native marshal helper; `join()`; optional `result()`; destroy / RAII. +- Safe to store and `join` from a CPU `@Thread` worker (OS thread; no GIL required for the wait itself). +- Overlapping jobs: two launches of the same GPU symbol get distinct inflight rows (symbol + inflight index), shared **kernel cache** entry. + +### B. Native launch entry (GPU side) + +Something CPU code can call without pybind, for example: + +```text +GpuJob launch(symbol_or_pipeline_key, GpuPack&& pack, /* writeback plan */) +``` + +or a small C ABI used by generated CPU kernels (same spirit as `Fn__call`). + +Marshal from C++ values into `GpuPack` (lists/scalars already in native memory). Do not go through `cthreads.marshal` / ctypes. + +### C. Transfer concurrency (GPU side) + +CPU threads will upload/download without Python sequencing. + +- Prefer a **pool of TransferEngines** (or a checkout API) on `Context`, not one global engine held for the whole job lifetime without a plan. +- Mutex around a single engine is a minimum; a pool is the better fit once CPU->GPU is real. +- Compute queue submit may need a mutex if multiple host threads submit at once. + +### D. Changes to the cthreads CPU / codegen system + +- Allow CPU kernels to **call into** the GPU runtime (link against `_ext` GPU symbols or a shared `cthreads_gpu` API exported for kernels). +- Codegen / validator: a controlled way to express "launch this `@Gpu` from `@Thread`" (name binding, arity, types). Reject anything that implies mid-run Python sync on the GPU job. +- Job lifetime: CPU `SpawnedKernel` may own or await a child `GpuJob`. Define destroy order (GPU join before CPU pack free if they share buffers). Prefer **separate** packs unless explicitly aliased. +- Pools: a CPU pool worker launching GPU must not assume the GIL; completion is fence-based like Python `join`. +- Optional: register GPU child jobs for debugging / `Job` trees. Not required for the first version of this feature. + +### E. What not to invent + +- A second descriptor/pipeline stack for "native only." +- A public `DeviceBuffer` type. +- GPU `__sync_state` / mid-run Python observe. +- Replacing Python `gpu()`. It stays the host entry; native launch is an additional caller. + +--- + +## Suggested sequencing + +1. Confirm `GpuJob` is a complete C++ type used by Python bindings. +2. Add TransferEngine pool / checkout if not already there. +3. Add native `launch` + C++ marshal-into-GpuPack for supported types. +4. Extend CPU codegen/validator for an approved call shape. +5. Tests: CPU `@Thread` launches `@Gpu`, overlaps, joins; two CPU workers launch GPU concurrently. +6. Docs: call rules, lifetime, no GIL assumptions. + +--- + +## Relation to other later work + +This is an **additive** product feature on top of the finished Python GPU path. It is not a rewrite of pack, descriptors, pipelines, or launch. Track it separately from atomics, MoltenVK, and shared IR (those stay in the Later list in `todo.md`). diff --git a/docs/guide/gpu/README.md b/docs/guide/gpu/README.md new file mode 100644 index 0000000..48f6203 --- /dev/null +++ b/docs/guide/gpu/README.md @@ -0,0 +1,65 @@ +# GPU guides (cthreads 0.2.0) + +These guides cover the **public** Vulkan compute path: `@Gpu` kernels, `gpu()` launch, +`GpuArena` residency, and workgroup barriers. They assume you know Python and are +comfortable with the CPU side of cthreads (`@Thread`, jobs, join). They do **not** +assume prior GPU or Vulkan experience. + +Internals for contributors (Vulkan substrate, C++ modules) live under +[docs/vk_guide](../../vk_guide/README.md) and [docs/internals/gpu](../../internals/gpu/README.md). + +## Version note + +| Release | GPU surface | +|---------|-------------| +| **0.2.0** | `@Gpu` / `gpu()` / `GpuJob`, index builtins, list writeback, `GpuArena`, workgroup `__sync_threads` / `Barrier.arrive_and_wait()` | +| **0.2.1 (planned)** | Workgroup shared memory (tiles / cooperative reductions) | + +Shared memory is intentionally out of scope for 0.2.0. Barriers are documented so the +API is stable; cooperative tile patterns become useful once shared memory lands. + +## Reading order + +| Start here | Why | +|------------|-----| +| [concepts.md](./concepts.md) | Host vs device, workgroups, writeback, what a GPU kernel is | +| [quickstart.md](./quickstart.md) | First end-to-end `saxpy` | +| [kernels.md](./kernels.md) | `@Gpu` rules, types, language subset | +| [indexes.md](./indexes.md) | `GlobalIdx` / `ThreadIdx` / dispatch sizing | +| [launch.md](./launch.md) | `prepare`, `gpu()`, `GpuJob.join` | +| [arena.md](./arena.md) | Keep lists on device across launches | +| [sync.md](./sync.md) | Workgroup barriers (and what they are not) | +| [best_practices.md](./best_practices.md) | Patterns that stay correct and fast | +| [examples.md](./examples.md) | Worked examples | +| [errors.md](./errors.md) | Probe API, exceptions, troubleshooting | +| [api.md](./api.md) | Compact API reference | + +## Install and probe + +| Command | Result | +|---------|--------| +| `pip install cthreads` | CPU wheel | +| `pip install "cthreads[gpu]"` | CPU package + **`cthreads-gpu`** (GPU `_ext`) | +| `pip install cthreads-gpu` | GPU wheel only | + +See [install.md](../../install.md#gpu-vulkan-compute) and [release.md](../../release.md). + +Always probe before launching: + +```python +from cthreads import gpu + +if not gpu.available(): + raise SystemExit("No usable Vulkan compute device for cthreads.gpu") +print(gpu.device_name()) +``` + +## Product rules (locked) + +1. **Python types only** on the default path: scalars and `list[...]` of scalars. No public `DeviceBuffer` type for ordinary kernels. +2. **Launch then join.** There is no mid-run Python observe of GPU state (unlike CPU `__sync_state`). +3. **Results are list writeback.** Kernels are `-> None`. Scalars are inputs only. +4. **Workgroup barriers are workgroup-local.** They do not synchronize the whole launch grid. + +CPU `@Thread` and GPU `@Gpu` are separate backends. The same process can use both; +a CPU kernel cannot yet launch a GPU kernel (see [gpu_future_cpu_to_gpu.md](../../gpu_future_cpu_to_gpu.md)). diff --git a/docs/guide/gpu/api.md b/docs/guide/gpu/api.md new file mode 100644 index 0000000..29ce665 --- /dev/null +++ b/docs/guide/gpu/api.md @@ -0,0 +1,159 @@ +# GPU API reference (0.2.0) + +Compact reference for the public `cthreads.gpu` surface and related sync entry +points. Narrative guides: [README.md](./README.md). + +# Contents + +- [Package imports](#package-imports) +- [`@Gpu`](#gpu) +- [Launch](#launch) +- [Indexes](#indexes) +- [GpuArena](#gpuarena) +- [Probe / lifecycle](#probe--lifecycle) +- [Errors](#errors) +- [Sync barriers](#sync-barriers) +- [Kernel language (summary)](#kernel-language-summary) + +# Package imports + +```python +from cthreads.gpu import ( + Gpu, + gpu, + prepare, + compile, + GpuJob, + GpuArena, + GlobalIdx, + ThreadIdx, + BlockIdx, + BlockDim, + GridDim, + available, + device_name, + init, + shutdown, +) + +from cthreads.sync import Barrier, __sync_threads +``` + +# `@Gpu` + +```python +@Gpu +def kernel(...) -> None: ... + +@Gpu(log=True) +def kernel_logged(...) -> None: ... +``` + +- Validates GPU type allowlist and registers the function. +- Requires GPU availability at decorate time. +- Does not launch. + +Allowed parameter types: `int`, `float`, `bool`, `list[int]`, `list[float]`, +`list[bool]`. Return must be `None`. + +# Launch + +### `prepare(force: bool = False) -> dict` + +Ensure Vulkan is usable and compile registered `@Gpu` kernels. + +### `compile(force: bool = False) -> dict` + +Emit / refresh SPIR-V artifacts for registered kernels. + +### `gpu(fn, *args, force: bool = False) -> GpuJob` + +Launch a `@Gpu` function. Positional args only. Auto-prepares when needed. + +### `GpuJob` + +| Method | Signature | Notes | +|--------|-----------|-------| +| `join` | `join(download: bool = True) -> None` | Wait; download ref lists when `download` is true | +| `result` | `result() -> None` | Always `None` | + +# Indexes + +Markers with `.x` / `.y` / `.z`: + +| Name | Meaning | +|------|---------| +| `GlobalIdx` | Global invocation id (prefer for 1D maps) | +| `ThreadIdx` | Id within workgroup | +| `BlockIdx` | Workgroup id | +| `BlockDim` | Workgroup size | +| `GridDim` | Number of workgroups | + +Default workgroup size X: **64**. `group_count_x` defaults to `ceil(n / 64)` with +`n` from parameter `n` or longest list length. + +# GpuArena + +```python +with GpuArena() as arena: + arena.bind(x=x, y=y) + gpu(fn, n, x, y).join(download=False) + arena.sync() # or arena.sync("y") +# arena.release() on exit +``` + +| Method | Role | +|--------|------| +| `bind(**lists)` | Upload / register lists by keyword slot name | +| `sync(*names)` | Download slots into Python lists | +| `release()` | Destroy buffers; clear registrations | +| `names()` | Bound slot names | + +Residency is keyed by list object identity + length. + +# Probe / lifecycle + +| Function | Returns | Raises | +|----------|---------|--------| +| `available()` | `bool` | No (soft fail) | +| `device_name()` | `str` | Mapped GPU errors / not built | +| `init()` | `None` | Mapped GPU errors / not built | +| `shutdown()` | `None` | Soft if not built | + +# Errors + +`CThreadsGPUError` and subclasses: + +- `VulkanNotBuiltError` +- `VulkanLoaderNotFound` +- `VulkanNoDevice` +- `VulkanInitFailed` +- `VulkanOutOfMemory` +- `GpuInvalidArgument` +- `GpuUseAfterDestroy` +- `GPUNotAvailable` + +Details: [errors.md](./errors.md). + +# Sync barriers + +Inside `@Gpu` only: + +```python +__sync_threads() +Barrier.arrive_and_wait() +``` + +Workgroup scope. `Barrier(...)` construction inside `@Gpu` raises `TypeError`. +Guide: [sync.md](./sync.md). + +# Kernel language (summary) + +Supported: annotated locals, `if`/`else`, `while`, `for i in range(...)`, early +`return`, list indexing, arithmetic/comparisons, `sqrt` / `floor` / `int(...)`, +barrier calls. + +Not supported: list `for-in`, rich Python types, valued returns, keyword +`gpu(...)` args, mid-run Python observe. + +Full rules: [kernels.md](./kernels.md). diff --git a/docs/guide/gpu/arena.md b/docs/guide/gpu/arena.md new file mode 100644 index 0000000..26b29b6 --- /dev/null +++ b/docs/guide/gpu/arena.md @@ -0,0 +1,158 @@ +# GpuArena (resident lists) + +`GpuArena` keeps named Python lists bound to process-wide device buffers so repeated +`gpu()` launches can skip alloc/upload when you pass the **same list objects**. + +# Contents + +- [When to use it](#when-to-use-it) +- [Basic pattern](#basic-pattern) +- [Option B: same list objects](#option-b-same-list-objects) +- [API](#api) +- [Host edits between launches](#host-edits-between-launches) +- [Length and type stability](#length-and-type-stability) +- [Release and context managers](#release-and-context-managers) +- [Pitfalls](#pitfalls) + +# When to use it + +| Situation | Recommendation | +|-----------|----------------| +| One or few launches | Default `gpu(...).join()` is enough | +| Many launches over the same arrays | `GpuArena` + `join(download=False)` + occasional `sync()` | +| Need Python to read results every iteration | Default join download, or `sync()` each time (you pay the transfer) | + +Arena does not change kernel source. It only changes how launch finds buffers. + +# Basic pattern + +```python +from cthreads.gpu import Gpu, GlobalIdx, GpuArena, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + +x = [1.0] * 1_000_000 +y = [0.0] * 1_000_000 +n = len(x) +a = 1.0001 + +with GpuArena() as arena: + arena.bind(x=x, y=y) + for _ in range(200): + gpu(saxpy, n, a, x, y).join(download=False) + arena.sync() # download x and y into the Python lists +``` + +# Option B: same list objects + +Launch recognizes residency by **object identity** (`id(list)`) plus length checks: + +1. `arena.bind(x=x, ...)` uploads and registers `x`. +2. Later `gpu(..., x, ...)` looks up `id(x)`. +3. If found and length still matches, launch **borrows** the resident buffer + instead of allocating a fresh one. + +Consequences: + +- Pass the **same** list instance you bound. A copy (`x2 = list(x)`) is a different + object and will not hit the resident path. +- Do not replace the list variable with a new list of the same values without + rebinding. + +# API + +### `GpuArena()` + +Creates a session with a unique id. Prefer `with GpuArena() as arena:`. + +### `arena.bind(**named_lists)` + +```python +arena.bind(positions=pos, velocities=vel) +``` + +- Keyword names are slot names inside the arena (`"positions"`, ...). +- Values must be non-empty `list` objects (`int` / `float` / `bool` elements). +- Element kind is inferred from the first element (`bool` before `int`). +- Rebinding the same slot name with the same list, length, and kind re-uploads. +- Rebinding with a different shape destroys and recreates the device buffer. +- A given list object cannot be bound under two different slots/arenas at once. + +Returns `self` for chaining: `GpuArena().bind(x=x).bind(y=y)` is valid, though one +`bind(x=x, y=y)` call is clearer. + +### `arena.sync(*names)` + +Downloads device bytes into the bound Python lists. + +```python +arena.sync() # all slots +arena.sync("y") # one slot by bind name +arena.sync("x", "y") +``` + +### `arena.release()` + +Destroys device buffers owned by this arena and clears identity registrations. +Called automatically from `__exit__`. + +### `arena.names()` / `name in arena` + +Introspection helpers for bound slot names. + +# Host edits between launches + +There is no proxy that watches Python list mutations. If you change host list +contents in Python between launches and need the device to see them: + +```python +# host updated y in Python +arena.bind(y=y) # re-upload that slot +gpu(kernel, n, x, y).join(download=False) +``` + +If only the device mutates the lists, re-bind is unnecessary; just launch again. + +# Length and type stability + +If a bound list's length changes, launch raises `GpuInvalidArgument` and asks you +to bind again. Shrinking/growing arrays mid-session means: rebuild the list, bind +again, continue. + +Element kind must keep matching the kernel parameter (`list[float]` vs inferred +kind from the host list). + +# Release and context managers + +```python +arena = GpuArena() +try: + arena.bind(x=x) + gpu(fn, n, x).join(download=False) + arena.sync("x") +finally: + arena.release() +``` + +After `release()`, further `bind` / `sync` raise `GpuInvalidArgument`. + +# Pitfalls + +| Pitfall | Symptom | Fix | +|---------|---------|-----| +| `join()` with default download every iteration | Slow loop; residency benefits muted | `join(download=False)` + `sync` when needed | +| New list each iteration | Misses resident lookup | Reuse bound objects | +| Read Python lists without `sync` | Stale host values | `arena.sync()` or downloading join | +| Empty list at bind | `GpuInvalidArgument` | Bind non-empty lists | +| Two arenas bind the same list | `GpuInvalidArgument` | One bind owner at a time | + +## See also + +- [launch.md](./launch.md#join-and-download) +- [best_practices.md](./best_practices.md) +- [examples.md](./examples.md#resident-multi-pass-loop) diff --git a/docs/guide/gpu/best_practices.md b/docs/guide/gpu/best_practices.md new file mode 100644 index 0000000..1e36cfe --- /dev/null +++ b/docs/guide/gpu/best_practices.md @@ -0,0 +1,171 @@ +# GPU best practices + +Practical guidance for correct, maintainable, and efficient use of `cthreads.gpu` +in 0.2.0. Pair with [concepts.md](./concepts.md) and the worked samples in +[examples.md](./examples.md). + +# Contents + +- [Choose CPU or GPU deliberately](#choose-cpu-or-gpu-deliberately) +- [Keep kernels regular](#keep-kernels-regular) +- [Always bounds-check](#always-bounds-check) +- [Prefer SoA lists](#prefer-soa-lists) +- [Name an explicit `n`](#name-an-explicit-n) +- [Minimize host round-trips](#minimize-host-round-trips) +- [Phase on the host for global ordering](#phase-on-the-host-for-global-ordering) +- [Treat barriers carefully](#treat-barriers-carefully) +- [Respect float32](#respect-float32) +- [Fail fast on availability](#fail-fast-on-availability) +- [Keep orchestration in Python](#keep-orchestration-in-python) +- [Testing habits](#testing-habits) +- [Performance expectations](#performance-expectations) + +# Choose CPU or GPU deliberately + +| Prefer `@Gpu` when | Prefer `@Thread` when | +|--------------------|------------------------| +| Large arrays, similar work per index | Irregular control flow / sparse work | +| Throughput matters more than latency of tiny n | Small n where launch overhead dominates | +| Outputs fit list writeback | You need return values or rich types | +| No mid-run Python observe | You need `__sync_state` / locks / TBuffer | + +A slow GPU path is often a small problem size, accidental per-iteration download, +or an algorithm that needs atomics/shared memory that 0.2.0 does not provide yet. + +# Keep kernels regular + +Write kernels so most invocations do the same kind of work: + +- Same loop trip counts when possible. +- Avoid large `if` trees that split the workgroup into many divergent paths. +- Push rare special cases to a separate pass or to the CPU. + +Regular code maps cleanly to SIMD-style GPU execution. + +# Always bounds-check + +```python +i: int = GlobalIdx.x +if i >= n: + return +``` + +Dispatch is rounded up to workgroup multiples. Skipping the guard is undefined for +your arrays' logical length. + +# Prefer SoA lists + +Use parallel arrays: + +```python +px: list[float] +py: list[float] +vx: list[float] +vy: list[float] +``` + +rather than objects or nested structures. The GPU allowlist is scalar lists only. +SoA also keeps memory access patterns predictable. + +# Name an explicit `n` + +```python +def kernel(n: int, ..., out: list[float]) -> None: +``` + +An `int` parameter named `n` drives `group_count_x` inference. Keep `n` equal to +the logical length you intend to process, and keep list lengths consistent with it. + +# Minimize host round-trips + +Default launch uploads and downloads lists. For iterative algorithms: + +1. `GpuArena.bind(...)` once. +2. Loop with `gpu(...).join(download=False)`. +3. `arena.sync()` when Python must read results. + +Re-bind (or upload via bind) when the **host** changes list contents that the +device must see. + +# Phase on the host for global ordering + +If step B must see every index's result from step A across the whole problem: + +```text +gpu(step_a).join(download=False) +gpu(step_b).join(download=False) +``` + +Do not expect `__sync_threads()` to provide that guarantee. Barriers are +workgroup-local ([sync.md](./sync.md)). + +# Treat barriers carefully + +- Use `__sync_threads()` or `Barrier.arrive_and_wait()` only inside `@Gpu`. +- Do not construct `Barrier(...)` in device code. +- In 0.2.0, barriers are forward-compatible API; shared-memory tile recipes arrive + in 0.2.1. Prefer host multi-pass for real algorithms until then. + +# Respect float32 + +GPU `float` is 32-bit. CPU `@Thread` `float` is double-backed. Mixing CPU and GPU +passes can accumulate different rounding. If you need higher precision end-to-end, +validate tolerances explicitly or keep sensitive reductions on CPU for now. + +# Fail fast on availability + +At process start: + +```python +from cthreads import gpu + +if not gpu.available(): + # choose CPU fallback or exit with a clear message + ... +``` + +Decorating `@Gpu` when the path is unavailable raises. Probe first in applications +that must run on machines without Vulkan compute. + +# Keep orchestration in Python + +Good split of responsibility: + +- **Python:** allocate lists, choose launches, arena lifetime, I/O, plotting, tests. +- **`@Gpu`:** arithmetic and indexed reads/writes over buffers. + +Avoid trying to rebuild a full application framework inside the shader language +subset. + +# Testing habits + +1. Compare GPU outputs to a pure-Python or `@Thread` reference on modest `n`. +2. Include non-multiples of 64 for `n` to exercise bounds checks. +3. Test arena loops with a final `sync` and assert against a single-launch baseline. +4. Assert `gpu.available()` or skip in environments without a device (CI often has + no GPU). + +# Performance expectations + +What usually helps: + +- Larger `n` (hundreds of thousands to millions of elements). +- Resident buffers for multi-pass loops. +- Fewer, thicker kernels rather than tiny launches in a Python loop without arena. + +What usually does not match CUDA driver-enqueue folklore: + +- Expecting nanosecond host enqueue with full list marshal every call. +- Tiny `n` where compile/upload dominate. + +Measure with your data sizes; microbenchmarks on `n=16` rarely predict production. + +## Checklist + +- [ ] `available()` checked +- [ ] `-> None` and allowed types only +- [ ] `GlobalIdx` + `i >= n` guard +- [ ] Explicit `n` when possible +- [ ] Arena for hot multi-pass loops +- [ ] Global phases via multiple launches, not workgroup barriers alone +- [ ] Tolerances aware of float32 diff --git a/docs/guide/gpu/concepts.md b/docs/guide/gpu/concepts.md new file mode 100644 index 0000000..4c02daa --- /dev/null +++ b/docs/guide/gpu/concepts.md @@ -0,0 +1,194 @@ +# GPU concepts + +This page builds a mental model for `cthreads.gpu` without requiring prior GPU +experience. If you already know CUDA or Vulkan compute, skim for cthreads-specific +rules (writeback, lists-only, no mid-run Python sync). + +# Contents + +- [What a GPU kernel is](#what-a-gpu-kernel-is) +- [Host and device](#host-and-device) +- [Many workers, one program](#many-workers-one-program) +- [Workgroups and the grid](#workgroups-and-the-grid) +- [How data moves](#how-data-moves) +- [Writeback vs residency](#writeback-vs-residency) +- [CPU `@Thread` vs GPU `@Gpu`](#cpu-thread-vs-gpu-gpu) +- [What 0.2.0 does not include](#what-020-does-not-include) +- [Glossary](#glossary) + +# What a GPU kernel is + +A **GPU kernel** is a short function that the GPU runs many times in parallel. +You write it once in Python with `@Gpu`. cthreads compiles it to a compute shader +(GLSL, then SPIR-V) and dispatches it on the Vulkan device. + +The usual pattern is **one invocation per array element** (or per particle, pixel, +and so on): + +```python +from cthreads.gpu import Gpu, GlobalIdx + +@Gpu +def scale(n: int, factor: float, data: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + data[i] = data[i] * factor +``` + +Invocation `i` only touches `data[i]`. Thousands of invocations can run at once. +That is the main reason GPUs help on large, regular loops. + +# Host and device + +| Role | Where | In cthreads | +|------|-------|-------------| +| **Host** | CPU + Python | Build lists, call `gpu(...)`, call `join()` | +| **Device** | GPU | Run the compiled `@Gpu` body | + +The host does **not** see intermediate GPU memory while the kernel runs. After +`join()` (by default), list arguments that the kernel mutated are copied back into +the same Python list objects you passed in. + +```text +Python lists --upload--> GPU buffers --kernel--> GPU buffers --download--> Python lists + (launch) (join, default) +``` + +This is close to the CPU pack / writeback model in [concepts.md](../../concepts.md), +with one important difference: on GPU there is **no** mid-run `sync_state` into +Python. Observation happens on join (or via `GpuArena.sync` for resident buffers). + +# Many workers, one program + +Think of each parallel copy of the kernel as an **invocation** (CUDA often says +"thread"; GLSL says "invocation"). Every invocation runs the **same** function body. +They differ only by their index builtins (`GlobalIdx.x`, and so on). + +Implications: + +1. **Branch divergence.** If half the invocations take an `if` branch and half do + not, hardware still has to schedule both paths. Prefer uniform control flow when + you can. +2. **Bounds checks.** Dispatch size is often rounded up to a multiple of the + workgroup size. Extra invocations must early-return: `if i >= n: return`. +3. **No shared Python state.** Invocations do not share Python objects. They share + GPU buffers (your lists) according to the memory model of the shader. + +# Workgroups and the grid + +Invocations are grouped into **workgroups** (CUDA: blocks). A launch is a **grid** +of workgroups. + +```text +grid (all workgroups) + workgroup 0: invocations 0 .. local_size-1 + workgroup 1: invocations local_size .. 2*local_size-1 + ... +``` + +Default `local_size_x` in cthreads is **64**. Launch chooses how many workgroups to +run so that `group_count_x * 64` covers your problem size (see [indexes.md](./indexes.md)). + +Why this matters: + +- **`__sync_threads()` / `Barrier.arrive_and_wait()`** wait only for invocations + **inside one workgroup**. They do not wait for the whole grid. +- Algorithms that need every element of a large array to "meet" after a step need + either another `gpu()` launch (host-side phase) or future shared-memory tile + patterns inside a workgroup. + +```mermaid +flowchart LR + subgraph grid [Launch grid] + WG0[Workgroup 0] + WG1[Workgroup 1] + WG2[Workgroup 2] + end + Host[Python gpu then join] --> grid + WG0 -.->|barrier only here| WG0 +``` + +# How data moves + +### Scalars + +Parameters typed as `int`, `float`, or `bool` are packed into a small storage +buffer and read by the shader. They are **not** written back to Python on join. +Treat them as inputs (counts, coefficients, flags). + +### Lists + +Parameters typed as `list[int]`, `list[float]`, or `list[bool]` become storage +buffers. On a default launch: + +1. Host bytes are uploaded. +2. The kernel reads and/or writes elements. +3. On `join()`, mutated lists are downloaded into the **same** Python list objects + (`pass_as="ref"`). + +Empty lists are not useful for binding: arena bind requires a non-empty list to +infer element type, and a zero-length problem usually means "do not launch." + +### Binding layout (intuition) + +Scalars share one buffer; each list gets its own. You do not manage bindings by +hand. The rule of thumb: **pass everything the kernel needs as annotated +parameters**. + +# Writeback vs residency + +| Mode | When to use | Host traffic | +|------|-------------|--------------| +| Default `gpu(...).join()` | One-shot or infrequent launches | Upload + download each launch | +| `GpuArena` + `join(download=False)` | Many launches over the same lists | Upload on `bind` / re-upload when you change host data; download on `arena.sync()` | + +Residency does not change kernel code. You still pass the **same list objects** to +`gpu()`. See [arena.md](./arena.md). + +# CPU `@Thread` vs GPU `@Gpu` + +| | `@Thread` (CPU) | `@Gpu` (GPU) | +|--|-----------------|--------------| +| Backend | Native C++ workers | Vulkan compute | +| Return values | Yes | Always `-> None` | +| Mid-run Python sync | `sync_state` / `__sync_state` | Not available | +| Rich types | `dict`, `str`, `@Threadable`, sync types | Scalars + `list` of scalars | +| Parallelism model | One job ~ one OS thread (or pool) | Many invocations per launch | +| Barrier | Host `Barrier(parties)` | Workgroup `__sync_threads` / `Barrier.arrive_and_wait()` | + +Use **CPU** when the work is irregular, needs Python-visible mid-run state, or +needs richer types. Use **GPU** when you have large, regular element-wise or +stencil-like passes over arrays. + +# What 0.2.0 does not include + +Documented so expectations stay accurate: + +- Workgroup **shared memory** arrays (planned for 0.2.1) +- Device **atomics** in the dialect +- Grid-wide barriers inside one kernel +- Public buffer objects / explicit Vulkan handles on the default path +- Launching `@Gpu` from inside `@Thread` (future design note: + [gpu_future_cpu_to_gpu.md](../../gpu_future_cpu_to_gpu.md)) +- macOS / MoltenVK as a first-class target in this release line + +# Glossary + +| Term | Meaning in cthreads | +|------|---------------------| +| **Invocation** | One parallel run of the `@Gpu` body (one index) | +| **Workgroup** | Group of invocations that can use a workgroup barrier together | +| **Grid** | All workgroups in one `gpu()` launch | +| **Host** | Python / CPU side | +| **Device** | GPU side | +| **Writeback** | Copy list buffers into Python lists on join | +| **Residency** | Keep buffers on device across launches (`GpuArena`) | +| **SPIR-V** | Binary shader format Vulkan consumes | +| **SSBO** | Storage buffer for kernel parameters / lists | + +## See also + +- [quickstart.md](./quickstart.md) +- [best_practices.md](./best_practices.md) +- [CPU concepts](../../concepts.md) diff --git a/docs/guide/gpu/errors.md b/docs/guide/gpu/errors.md new file mode 100644 index 0000000..2f25e0d --- /dev/null +++ b/docs/guide/gpu/errors.md @@ -0,0 +1,123 @@ +# GPU errors and troubleshooting + +Probe helpers, exception types, and common failure modes for `cthreads.gpu`. + +# Contents + +- [Probe API](#probe-api) +- [Exception types](#exception-types) +- [When `@Gpu` raises at decorate time](#when-gpu-raises-at-decorate-time) +- [Common symptoms](#common-symptoms) +- [Build notes](#build-notes) + +# Probe API + +```python +from cthreads import gpu + +gpu.available() # bool: loader + compute device usable +gpu.device_name() # str: raises if not built / init fails +gpu.init() # explicit Vulkan init +gpu.shutdown() # tear down device; next prepare/gpu recompiles +``` + +`available()` is the soft probe: it returns `False` instead of raising when the +path cannot start. Prefer it for feature detection. + +`device_name()` and `init()` raise mapped errors on failure. + +`shutdown()` releases the native shader cache with the device and marks the +runtime so the next `prepare()` / `gpu()` walks the registry again. + +# Exception types + +All of the following subclass `CThreadsGPUError` (except where noted). + +| Type | Typical cause | +|------|----------------| +| `VulkanNotBuiltError` | `_ext` compiled without `CTHREADS_GPU` | +| `VulkanLoaderNotFound` | `vulkan-1.dll` / `libvulkan.so.1` missing | +| `VulkanNoDevice` | Loader present, no compute-capable device | +| `VulkanInitFailed` | Instance/device creation or missing entry points | +| `VulkanOutOfMemory` | GPU memory allocation failed | +| `GpuInvalidArgument` | Bad sizes, arena misuse, join flags, dtype mismatch | +| `GpuUseAfterDestroy` | Use after native resource destroy | +| `GPUNotAvailable` | Generic "GPU path not usable" from Python helpers | + +Import: + +```python +from cthreads.gpu import ( + CThreadsGPUError, + GPUNotAvailable, + GpuInvalidArgument, + VulkanLoaderNotFound, + VulkanNotBuiltError, + VulkanNoDevice, +) +``` + +Native errors are mapped through `_map_error` based on message prefixes. + +# When `@Gpu` raises at decorate time + +`@Gpu` requires `available()` to be true. On a machine without Vulkan compute, +importing a module that eagerly decorates kernels can fail at import time. + +Patterns: + +1. Probe in `__main__` and import GPU modules only then. +2. Or document that the application requires a GPU. +3. For libraries, delay decoration / registration until the caller opts in. + +# Common symptoms + +| Symptom | Likely cause | What to try | +|---------|--------------|-------------| +| `VulkanNotBuiltError` | Extension built without GPU | `pip install "cthreads[gpu]"` or rebuild with `-DCTHREADS_GPU=ON` | +| `available()` is False | No loader, no device, or CPU-only build | Update GPU drivers; confirm Vulkan ICD; install `cthreads[gpu]` / rebuild with GPU ON | +| `VulkanLoaderNotFound` | Runtime library missing | Install/repair GPU drivers; on Linux install `vulkan-icd-loader` + vendor ICD | +| `TypeError` on decorate | Unsupported annotation | Scalars and `list` of scalars only | +| `TypeError` from `gpu()` | Missing `@Gpu` or bad arity | Check decorator and positional args | +| Wrong / partial results | Missing `i >= n` guard | Add bounds check | +| Stale Python lists in arena loop | No `sync`, `download=False` | Call `arena.sync()` | +| `GpuInvalidArgument` length changed | Mutated list length after bind | Rebind after resize | +| Barrier `TypeError` for `Barrier(n)` | Constructor in `@Gpu` | Use `Barrier.arrive_and_wait()` or `__sync_threads()` | +| `__sync_threads` RuntimeError | Called from host Python | Only inside compiled `@Gpu` bodies | +| GLSL / SPIR-V compile error | Unsupported statement or math | Simplify body; stick to documented subset | + +# Build notes + +End users of GPU-enabled wheels need **GPU drivers** with a Vulkan ICD. They do +not need the LunarG SDK to *run*. + +Contributors building from source with GPU enabled: + +```powershell +# PowerShell +$env:CMAKE_ARGS="-DCTHREADS_GPU=ON" +pip install -e ".[test]" +``` + +```bash +# bash +export CMAKE_ARGS="-DCTHREADS_GPU=ON" +pip install -e ".[test]" +``` + +Also valid: + +```bash +pip install -e . --config-settings=cmake.define.CTHREADS_GPU=ON +``` + +`find_package(Vulkan)` requires Vulkan headers (SDK) on the build machine. Runtime +still loads the loader dynamically. + +Full install context: [install.md](../../install.md#gpu-vulkan-compute). + +## See also + +- [quickstart.md](./quickstart.md) +- [api.md](./api.md) +- Contributor Vulkan notes: [vk_guide/03-sdk-runtime-drivers.md](../../vk_guide/03-sdk-runtime-drivers.md) diff --git a/docs/guide/gpu/examples.md b/docs/guide/gpu/examples.md new file mode 100644 index 0000000..1b665d8 --- /dev/null +++ b/docs/guide/gpu/examples.md @@ -0,0 +1,367 @@ +# GPU examples + +Worked examples for `cthreads.gpu` 0.2.0. Each sample is self-contained. Adapt +sizes and names to your project. These are documentation samples, not a shipped +demo suite. + +# Contents + +- [Probe device](#probe-device) +- [Fill a buffer](#fill-a-buffer) +- [SAXPY](#saxpy) +- [Clamp / activation-style map](#clamp--activation-style-map) +- [Pairwise combine into an output list](#pairwise-combine-into-an-output-list) +- [Integer histogram bins (no atomics)](#integer-histogram-bins-no-atomics) +- [Stencil 1D (host multi-pass safe pattern)](#stencil-1d-host-multi-pass-safe-pattern) +- [Resident multi-pass loop](#resident-multi-pass-loop) +- [Two-phase pipeline with arena](#two-phase-pipeline-with-arena) +- [CPU reference check](#cpu-reference-check) +- [Graceful CPU fallback sketch](#graceful-cpu-fallback-sketch) +- [Workgroup barrier call shape](#workgroup-barrier-call-shape) + +# Probe device + +```python +from cthreads import gpu + +def main() -> None: + if not gpu.available(): + print("cthreads.gpu is not usable on this machine") + return + gpu.init() + print("device:", gpu.device_name()) + +if __name__ == "__main__": + main() +``` + +# Fill a buffer + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def fill(n: int, value: float, out: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + out[i] = value + +def main() -> None: + n = 10_000 + out = [0.0] * n + gpu(fill, n, 3.14, out).join() + assert out[0] == 3.14 + assert out[n - 1] == 3.14 + +if __name__ == "__main__": + main() +``` + +# SAXPY + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + +def main() -> None: + x = [1.0, 2.0, 3.0, 4.0, 5.0] + y = [10.0, 20.0, 30.0, 40.0, 50.0] + gpu(saxpy, len(x), 2.0, x, y).join() + print(y) # [12.0, 24.0, 36.0, 48.0, 60.0] + +if __name__ == "__main__": + main() +``` + +# Clamp / activation-style map + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def clamp01(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + v: float = x[i] + if v < 0.0: + v = 0.0 + if v > 1.0: + v = 1.0 + y[i] = v + +def main() -> None: + x = [-1.0, 0.25, 2.0] + y = [0.0, 0.0, 0.0] + gpu(clamp01, len(x), x, y).join() + print(y) # [0.0, 0.25, 1.0] + +if __name__ == "__main__": + main() +``` + +# Pairwise combine into an output list + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def hadamard(n: int, a: list[float], b: list[float], out: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + out[i] = a[i] * b[i] + +def main() -> None: + a = [1.0, 2.0, 3.0, 4.0] + b = [2.0, 2.0, 2.0, 2.0] + out = [0.0] * len(a) + gpu(hadamard, len(a), a, b, out).join() + print(out) # [2.0, 4.0, 6.0, 8.0] + +if __name__ == "__main__": + main() +``` + +# Integer histogram bins (no atomics) + +Without atomics, invocations must not contend on the same output slot. A safe +teaching pattern is **one output per input** (category id), then aggregate on the +host -- or give each invocation a private region. Example: map values to bin ids +on GPU, count on CPU. + +```python +import math +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def value_to_bin( + n: int, + n_bins: int, + lo: float, + inv_width: float, + x: list[float], + bins: list[int], +) -> None: + i: int = GlobalIdx.x + if i >= n: + return + # floor((x - lo) * inv_width), clamped into [0, n_bins-1] + t: float = (x[i] - lo) * inv_width + b: int = int(math.floor(t)) + if b < 0: + b = 0 + if b >= n_bins: + b = n_bins - 1 + bins[i] = b + +def main() -> None: + x = [0.1, 0.4, 0.9, 1.2] + n_bins = 4 + lo = 0.0 + hi = 1.0 + inv_width = float(n_bins) / (hi - lo) + bin_of = [0] * len(x) + gpu(value_to_bin, len(x), n_bins, lo, inv_width, x, bin_of).join() + + counts = [0] * n_bins + for b in bin_of: + counts[b] += 1 + print(bin_of, counts) + +if __name__ == "__main__": + main() +``` + +When device atomics land in a later release, in-kernel histogram accumulation +becomes more direct. Until then, keep contested reductions on the host or design +conflict-free writes. + +# Stencil 1D (host multi-pass safe pattern) + +Neighbor reads from a **read-only** input list into a separate output list avoid +hazards inside one launch: + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def smooth(n: int, src: list[float], dst: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + left: float = src[i] + right: float = src[i] + if i > 0: + left = src[i - 1] + if i + 1 < n: + right = src[i + 1] + dst[i] = 0.25 * left + 0.5 * src[i] + 0.25 * right + +def main() -> None: + src = [0.0, 1.0, 0.0, 1.0, 0.0] + dst = [0.0] * len(src) + gpu(smooth, len(src), src, dst).join() + print(dst) + +if __name__ == "__main__": + main() +``` + +For iterative smoothing, swap roles across launches (or use an arena and two +buffers) rather than reading and writing the same list unsafely in one pass. + +# Resident multi-pass loop + +```python +from cthreads.gpu import Gpu, GlobalIdx, GpuArena, gpu + +@Gpu +def damp(n: int, factor: float, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = y[i] * factor + +def main() -> None: + y = [1.0] * 100_000 + n = len(y) + with GpuArena() as arena: + arena.bind(y=y) + for _ in range(50): + gpu(damp, n, 0.99, y).join(download=False) + arena.sync("y") + print(y[0]) + +if __name__ == "__main__": + main() +``` + +# Two-phase pipeline with arena + +```python +from cthreads.gpu import Gpu, GlobalIdx, GpuArena, gpu + +@Gpu +def square(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] * x[i] + +@Gpu +def add_const(n: int, c: float, y: list[float], z: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + z[i] = y[i] + c + +def main() -> None: + x = [1.0, 2.0, 3.0, 4.0] + y = [0.0] * len(x) + z = [0.0] * len(x) + n = len(x) + with GpuArena() as arena: + arena.bind(x=x, y=y, z=z) + gpu(square, n, x, y).join(download=False) + gpu(add_const, n, 1.0, y, z).join(download=False) + arena.sync("z") + print(z) # [2.0, 5.0, 10.0, 17.0] + +if __name__ == "__main__": + main() +``` + +# CPU reference check + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def scale(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + +def ref_scale(a: float, x: list[float]) -> list[float]: + return [a * v for v in x] + +def main() -> None: + x = [float(i) for i in range(1000)] + y = [0.0] * len(x) + a = 1.5 + gpu(scale, len(x), a, x, y).join() + expect = ref_scale(a, x) + for i in range(len(x)): + err = abs(y[i] - expect[i]) + if err > 1e-5: + raise AssertionError(f"mismatch at {i}: {y[i]} vs {expect[i]}") + print("ok") + +if __name__ == "__main__": + main() +``` + +# Graceful CPU fallback sketch + +```python +from cthreads import gpu as gpu_api +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def scale_gpu(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + +def scale_cpu(a: float, x: list[float], y: list[float]) -> None: + for i in range(len(x)): + y[i] = a * x[i] + +def scale(a: float, x: list[float], y: list[float]) -> None: + if gpu_api.available(): + gpu(scale_gpu, len(x), a, x, y).join() + else: + scale_cpu(a, x, y) +``` + +Note: defining `@Gpu` requires availability at decorate time. For binaries that +must import on GPU-less hosts, gate the decorated definitions behind +`if gpu_api.available():` or keep GPU kernels in a module imported only after the +probe succeeds. + +# Workgroup barrier call shape + +Illustrates the legal call forms. Without shared memory, this is primarily an API +shape example; see [sync.md](./sync.md). + +```python +from cthreads.sync import Barrier, __sync_threads +from cthreads.gpu import Gpu, GlobalIdx + +@Gpu +def barrier_shape(n: int, data: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + data[i] = data[i] + 1.0 + __sync_threads() + # Same device sync: + Barrier.arrive_and_wait() + data[i] = data[i] * 1.0 +``` + +## See also + +- [best_practices.md](./best_practices.md) +- [arena.md](./arena.md) +- [sync.md](./sync.md) diff --git a/docs/guide/gpu/indexes.md b/docs/guide/gpu/indexes.md new file mode 100644 index 0000000..5832aad --- /dev/null +++ b/docs/guide/gpu/indexes.md @@ -0,0 +1,134 @@ +# Indexes and dispatch + +How each invocation knows which element to process, and how cthreads chooses how +many workgroups to launch. + +# Contents + +- [The five builtins](#the-five-builtins) +- [Prefer `GlobalIdx` for 1D kernels](#prefer-globalidx-for-1d-kernels) +- [Workgroup size](#workgroup-size) +- [How `group_count_x` is inferred](#how-group_count_x-is-inferred) +- [Bounds checks](#bounds-checks) +- [2D and 3D axes](#2d-and-3d-axes) +- [Mapping to GLSL names](#mapping-to-glsl-names) + +# The five builtins + +Import from `cthreads.gpu`: + +```python +from cthreads.gpu import ( + ThreadIdx, # index inside the workgroup + BlockIdx, # which workgroup + BlockDim, # workgroup size + GridDim, # number of workgroups + GlobalIdx, # global index across the grid +) +``` + +Each exposes `.x`, `.y`, and `.z` axis markers. In 0.2.0 the public launch path is +effectively **1D in X** for sizing (`group_count_y` / `group_count_z` default to 1), +so most kernels only need `.x`. + +Relationship (same as CUDA 1D): + +```text +GlobalIdx.x == BlockIdx.x * BlockDim.x + ThreadIdx.x +``` + +# Prefer `GlobalIdx` for 1D kernels + +Element-wise kernels almost always look like this: + +```python +from cthreads.gpu import Gpu, GlobalIdx + +@Gpu +def axpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] +``` + +Use `ThreadIdx` / `BlockIdx` when you intentionally structure work per workgroup +(for example future shared-memory tiles). Today, without shared memory, most +user code can stay on `GlobalIdx`. + +# Workgroup size + +Default `local_size_x` is **64**. That is the number of invocations in one +workgroup along X. + +Implications: + +- Launch count is rounded up: for `n = 100`, you get enough workgroups that + `group_count_x * 64 >= 100`, so some invocations have `i >= n`. +- Workgroup barriers wait for those 64 (or fewer active) peers in one group, not + for all `n` elements. + +# How `group_count_x` is inferred + +When you call `gpu(fn, *args)`, cthreads sets `group_count_x` if metadata does not +already fix it: + +1. If there is an `int` parameter named **`n`**, use that value as the problem size. +2. Otherwise use the length of the longest list argument. +3. Then `group_count_x = ceil(n / local_size_x)` (at least 1). + +Recommendation: **pass an explicit `n: int`** as the first scalar and keep list +lengths consistent with `n`. That makes dispatch intent obvious in the signature. + +```python +@Gpu +def fill(n: int, value: float, out: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + out[i] = value + +out = [0.0] * 1000 +gpu(fill, len(out), 1.0, out).join() +``` + +# Bounds checks + +Always guard: + +```python +i: int = GlobalIdx.x +if i >= n: + return +``` + +Without the guard, padded invocations can write past the logical end of your +arrays. This is one of the most common GPU correctness bugs for newcomers; it is +mechanical once you treat dispatch rounding as normal. + +# 2D and 3D axes + +`.y` and `.z` exist on the builtins for dialect completeness and for future +multi-dimensional launches. The high-level `gpu()` path currently fills +`group_count_y = 1` and `group_count_z = 1` when unset. Prefer 1D indexing with +`GlobalIdx.x` unless you have a clear reason to use other axes and matching +launch metadata. + +# Mapping to GLSL names + +| cthreads | GLSL built-in | +|----------|----------------| +| `ThreadIdx` | `gl_LocalInvocationID` | +| `BlockIdx` | `gl_WorkGroupID` | +| `BlockDim` | `gl_WorkGroupSize` | +| `GridDim` | `gl_NumWorkGroups` | +| `GlobalIdx` | `gl_GlobalInvocationID` | + +You do not write GLSL by hand for ordinary kernels; the table is for readers who +already know Vulkan/GLSL or are debugging lowered shaders. + +## See also + +- [concepts.md](./concepts.md#workgroups-and-the-grid) +- [sync.md](./sync.md) - barriers are workgroup-scoped +- [examples.md](./examples.md) diff --git a/docs/guide/gpu/kernels.md b/docs/guide/gpu/kernels.md new file mode 100644 index 0000000..dbfd6ea --- /dev/null +++ b/docs/guide/gpu/kernels.md @@ -0,0 +1,194 @@ +# `@Gpu` kernels + +How to author device functions: types, language subset, math helpers, and how +decoration differs from launch. + +# Contents + +- [Decorating](#decorating) +- [Allowed types](#allowed-types) +- [Return type and outputs](#return-type-and-outputs) +- [Locals and annotations](#locals-and-annotations) +- [Language subset](#language-subset) +- [Math and casts](#math-and-casts) +- [Imports inside kernels](#imports-inside-kernels) +- [What is rejected](#what-is-rejected) +- [Decorate vs launch vs call-as-Python](#decorate-vs-launch-vs-call-as-python) + +# Decorating + +```python +from cthreads.gpu import Gpu, GlobalIdx + +@Gpu +def fill(n: int, value: float, out: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + out[i] = value +``` + +(`GlobalIdx` is documented in [indexes.md](./indexes.md).) + +Optional logging of the assigned device name: + +```python +@Gpu(log=True) +def fill_logged(n: int, value: float, out: list[float]) -> None: + ... +``` + +Decoration: + +1. Checks that the GPU path is available (`gpu.available()`). +2. Validates annotations against the GPU allowlist. +3. Builds launch metadata on the function (`__gpu_kernel_meta__`). +4. Registers the function for later SPIR-V compile. + +Decoration does **not** upload data or run the kernel. + +# Allowed types + +| Annotation | Role | Notes | +|------------|------|-------| +| `int` | Scalar input | Packed into the scalar storage buffer | +| `float` | Scalar input | Device `float` (32-bit). CPU `@Thread` uses double; GPU follows GLSL float | +| `bool` | Scalar input | Stored as 32-bit on the device packing path | +| `list[int]` | Buffer (writeback) | Same Python list updated on join | +| `list[float]` | Buffer (writeback) | Same | +| `list[bool]` | Buffer (writeback) | Same | + +Not supported on the GPU path (raise at meta build / decorate): + +- `str`, `dict[...]` +- `@Threadable` classes +- `cthreads.sync` lock / event / TBuffer annotations as kernel parameters +- Nested lists (`list[list[float]]`) +- `set`, optional types, `Any` + +Structure-of-arrays is the intended style: parallel lists of scalars +(`pos_x`, `pos_y`, `vel_x`, ...) rather than a list of particle objects. + +# Return type and outputs + +Every `@Gpu` function must be annotated `-> None`. + +Outputs are **in-place list mutations**. After `join()`, inspect those lists on +the host. Scalar parameters are not written back. + +```python +@Gpu +def add_scaled( + n: int, + alpha: float, + src: list[float], + dst: list[float], +) -> None: + i: int = GlobalIdx.x + if i >= n: + return + dst[i] = dst[i] + alpha * src[i] +``` + +# Locals and annotations + +Locals must use annotated assignment, same discipline as `@Thread`: + +```python +i: int = GlobalIdx.x +tmp: float = x[i] * a +ok: bool = i < n +``` + +Bare `i = GlobalIdx.x` is not part of the supported subset. + +# Language subset + +Supported control flow (compiled to GLSL): + +| Construct | Support | +|-----------|---------| +| `if` / `else` | Yes | +| `while` | Yes (no `while`/`else`) | +| `for i in range(...)` | Yes (`range` with 1-3 args; no keywords) | +| `return` | Early exit only (`return;` / bare `return`) | +| `break` / `continue` | Via the shared flow helpers (same spirit as CPU) | +| Expression statements that are calls | Yes (math, barriers) | + +Not supported: + +- `for x in some_list:` (no list iteration) +- `for`/`else`, `while`/`else` +- Rebinding an existing name as a `for` target +- Arbitrary Python calls, comprehensions, generators, `try`/`except`, classes, nested `def` + +Index expressions on list parameters (`y[i] = ...`) are the primary memory API. + +# Math and casts + +The GPU CallPlugin path currently lowers: + +| Python form | Device | +|-------------|--------| +| `sqrt(x)` or `math.sqrt(x)` | GLSL `sqrt` | +| `floor(x)` or `math.floor(x)` | GLSL `floor` (toward -inf) | +| `int(x)` | GLSL `int` (truncate toward zero) | + +Example: + +```python +import math +from cthreads.gpu import Gpu, GlobalIdx + +@Gpu +def cell_of(n: int, inv_h: float, x: list[float], cell: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + c: float = math.floor(x[i] * inv_h) + cell[i] = int(c) +``` + +Richer math will grow over time; do not assume the full `math` module is available +inside `@Gpu`. Prefer the helpers above or express formulas with `+ - * /` and +comparisons. + +# Imports inside kernels + +You may import names that the translator recognizes (`math` for `math.sqrt` / +`math.floor`, sync barrier names, index builtins). The compiled kernel does not +run Python imports on the device; matching is by AST shape at compile time. + +Typical host-side imports for authoring: + +```python +from cthreads.gpu import Gpu, GlobalIdx, BlockDim, ThreadIdx +from cthreads.sync import Barrier, __sync_threads +``` + +# What is rejected + +| Pattern | Why | +|---------|-----| +| `Barrier(4)` inside `@Gpu` | Host-style construction; use `Barrier.arrive_and_wait()` | +| Valued `return x` | Kernels are void; use list writeback | +| Keyword args to `gpu(fn, x=...)` | Not supported yet; positional only | +| Mid-run Python reads of lists | No GPU `sync_state`; join or `arena.sync` | +| Assuming `float` is IEEE double | Device float is 32-bit | + +# Decorate vs launch vs call-as-Python + +| Action | Effect | +|--------|--------| +| `@Gpu` on a function | Validate + register | +| Direct `fn(*args)` in Python | Runs the Python source as ordinary Python (useful for tiny checks; not the device path) | +| `gpu(fn, *args)` | Compile if needed, upload, dispatch, return `GpuJob` | +| `job.join()` | Wait (+ download lists by default) | + +For production device work, always go through `gpu(...)`. + +## See also + +- [indexes.md](./indexes.md) +- [launch.md](./launch.md) +- [sync.md](./sync.md) diff --git a/docs/guide/gpu/launch.md b/docs/guide/gpu/launch.md new file mode 100644 index 0000000..6a47531 --- /dev/null +++ b/docs/guide/gpu/launch.md @@ -0,0 +1,143 @@ +# Launch and jobs + +Public entry points: `prepare`, `compile`, `gpu`, and `GpuJob`. + +# Contents + +- [Imports](#imports) +- [`prepare` and `compile`](#prepare-and-compile) +- [`gpu(fn, *args)`](#gpufn-args) +- [`GpuJob`](#gpujob) +- [Join and download](#join-and-download) +- [Force rebuild](#force-rebuild) +- [Await](#await) +- [Lifecycle diagram](#lifecycle-diagram) + +# Imports + +```python +from cthreads.gpu import Gpu, gpu, prepare, compile, GpuJob +# or: +from cthreads import gpu as gpu_mod +gpu_mod.prepare() +``` + +The package `cthreads.gpu` re-exports the launch helpers. The probe helpers +`available`, `device_name`, `init`, and `shutdown` live there as well +([errors.md](./errors.md)). + +# `prepare` and `compile` + +| Call | Role | +|------|------| +| `compile(force=False)` | Walk registered `@Gpu` functions and emit SPIR-V / cache artifacts | +| `prepare(force=False)` | Ensure Vulkan is available, then `compile` | + +`gpu()` calls `prepare` automatically on first launch in the process (or when +`force=True`). Explicit `prepare()` is useful to fail fast at startup: + +```python +from cthreads.gpu import prepare + +prepare() # raises GPUNotAvailable if the device path is unusable +``` + +`prepare(force=True)` shuts down and re-inits Vulkan, then force-rebuilds GPU +units. Use after intentional device reset scenarios, not on every frame. + +# `gpu(fn, *args)` + +```python +job = gpu(saxpy, n, a, x, y) +``` + +Requirements: + +- `fn` must be decorated with `@Gpu`. +- Arguments are **positional only** (no keyword args yet). +- Arity must match the kernel signature. +- GPU must be available. + +Behavior: + +1. Prepare/compile if this process has not prepared yet. +2. Build ordered values from metadata. +3. Infer `group_count_x` when needed ([indexes.md](./indexes.md)). +4. Attach residency metadata for arena-bound lists ([arena.md](./arena.md)). +5. Launch via the native path and return a started `GpuJob`. + +# `GpuJob` + +`GpuJob` subclasses the CPU `Job` handle with GPU-specific join behavior. + +| Method | Behavior | +|--------|----------| +| `start()` | Already called by `gpu()`; safe to think of the job as running | +| `join(download=True)` | Wait for the GPU fence; optionally download ref lists | +| `result()` | Always `None` (void kernels) | +| `await job` | Async wait (same idea as CPU jobs) | + +There is no GPU equivalent of CPU `job.sync_state()` for mid-run Python observe. + +# Join and download + +```python +job.join() # wait + download list args (default) +job.join(download=False) # wait only; lists stay on device / unsynced to Python +``` + +Use `download=False` together with `GpuArena`: + +- Avoids host round-trips on every iteration of a multi-launch loop. +- Python list contents are **stale** until `arena.sync()` (or a later join with + download enabled on a non-resident path). + +If you pass `download=False` on a build without residency support, cthreads +raises `GpuInvalidArgument` with a rebuild hint. + +# Force rebuild + +```python +gpu(saxpy, n, a, x, y, force=True) +``` + +Forces re-prepare (Vulkan re-init + recompile). Useful while iterating on kernel +source during development. Avoid in tight production loops. + +# Await + +```python +import asyncio +from cthreads.gpu import gpu + +async def run(): + job = gpu(saxpy, n, a, x, y) + await job + # y updated after await completes (default download) + +asyncio.run(run()) +``` + +# Lifecycle diagram + +```mermaid +sequenceDiagram + participant Py as Python host + participant RT as cthreads.gpu + participant Dev as Vulkan device + + Py->>RT: gpu(fn, args) + RT->>RT: prepare if needed + RT->>Dev: upload buffers + dispatch + RT-->>Py: GpuJob (started) + Py->>RT: join(download=True) + RT->>Dev: wait fence + Dev-->>RT: download ref lists + RT-->>Py: Python lists updated +``` + +## See also + +- [arena.md](./arena.md) +- [quickstart.md](./quickstart.md) +- [CPU jobs](../jobs.md) - shared job vocabulary where it applies diff --git a/docs/guide/gpu/quickstart.md b/docs/guide/gpu/quickstart.md new file mode 100644 index 0000000..722a6f8 --- /dev/null +++ b/docs/guide/gpu/quickstart.md @@ -0,0 +1,118 @@ +# GPU quickstart + +End-to-end path from install probe to a working element-wise kernel. For the mental +model behind the steps, see [concepts.md](./concepts.md). + +# Contents + +- [1. Confirm GPU availability](#1-confirm-gpu-availability) +- [2. Write a `@Gpu` kernel](#2-write-a-gpu-kernel) +- [3. Launch and join](#3-launch-and-join) +- [4. Read results](#4-read-results) +- [5. Optional: keep data on device](#5-optional-keep-data-on-device) +- [Common first mistakes](#common-first-mistakes) + +# 1. Confirm GPU availability + +```python +from cthreads import gpu + +print(gpu.available()) +if gpu.available(): + print(gpu.device_name()) +``` + +If `available()` is `False`, stop here and check +[install.md](../../install.md#gpu-vulkan-compute) and [errors.md](./errors.md). +Decorating `@Gpu` or calling `gpu()` raises `GPUNotAvailable` when the path is not +usable. + +# 2. Write a `@Gpu` kernel + +Rules that matter for the first kernel: + +- Annotate every parameter and use `-> None`. +- Use only `int`, `float`, `bool`, and `list` of those scalars. +- Introduce locals with annotated assignment (`i: int = ...`). +- Index with `GlobalIdx.x` for 1D element-wise work. +- Guard out-of-range invocations. + +```python +from cthreads.gpu import Gpu, GlobalIdx, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + """ + y[i] = a * x[i] + y[i] for i in 0 .. n-1 + """ + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] +``` + +Calling `saxpy(...)` as ordinary Python still runs the Python body. Device +execution happens only through `gpu(saxpy, ...)`. + +# 3. Launch and join + +```python +x: list[float] = [1.0, 2.0, 3.0, 4.0] +y: list[float] = [10.0, 20.0, 30.0, 40.0] +n: int = len(x) +a: float = 2.0 + +job = gpu(saxpy, n, a, x, y) +job.join() +``` + +What `gpu()` does on first use in the process: + +1. Ensures Vulkan is usable. +2. Compiles registered `@Gpu` functions to SPIR-V (cached afterward). +3. Uploads arguments and records a compute dispatch. +4. Returns a `GpuJob` that is already started. + +`join()` waits for the GPU fence and, by default, downloads ref lists into the +Python lists you passed. + +# 4. Read results + +```python +print(y) # [12.0, 24.0, 36.0, 48.0] +``` + +There is no scalar return value: `job.result()` is always `None`. Outputs live in +the list arguments you marked for writeback (all lists today). + +# 5. Optional: keep data on device + +When the same lists are reused for many launches, bind them once: + +```python +from cthreads.gpu import GpuArena + +with GpuArena() as arena: + arena.bind(x=x, y=y) + for _ in range(100): + gpu(saxpy, n, a, x, y).join(download=False) + arena.sync() # download into x and y when you need Python to see values +``` + +Details: [arena.md](./arena.md). + +# Common first mistakes + +| Mistake | What happens | Fix | +|---------|--------------|-----| +| Missing `if i >= n: return` | Out-of-range writes / undefined behavior on padded invocations | Always bounds-check | +| Expecting `job.result()` to hold an array | Always `None` | Read the list args after join | +| Using `dict` / `@Threadable` / `str` args | `TypeError` at decorate or compile | Stick to scalars and scalar lists | +| Calling `__sync_threads()` from plain Python | `RuntimeError` | Only inside `@Gpu` bodies | +| Assuming a barrier syncs the whole array | Only one workgroup waits | Use multiple launches for global phases | + +## Next + +- [kernels.md](./kernels.md) - full language and type rules +- [indexes.md](./indexes.md) - how `GlobalIdx` relates to workgroups +- [examples.md](./examples.md) - more complete samples diff --git a/docs/guide/gpu/sync.md b/docs/guide/gpu/sync.md new file mode 100644 index 0000000..0d3d9f9 --- /dev/null +++ b/docs/guide/gpu/sync.md @@ -0,0 +1,154 @@ +# GPU sync (workgroup barriers) + +Device-side rendezvous for invocations **inside one workgroup**. This is not the +CPU `Barrier(parties)` class, and it is not a whole-grid barrier. + +# Contents + +- [Why barriers exist](#why-barriers-exist) +- [API: two names, one lowering](#api-two-names-one-lowering) +- [Scope: workgroup only](#scope-workgroup-only) +- [CPU Barrier vs GPU barrier](#cpu-barrier-vs-gpu-barrier) +- [What you can do in 0.2.0](#what-you-can-do-in-020) +- [Shared memory (0.2.1)](#shared-memory-021) +- [Rejected forms](#rejected-forms) +- [Host-side phases instead of grid barriers](#host-side-phases-instead-of-grid-barriers) + +# Why barriers exist + +Invocations in a workgroup can run ahead of each other. When algorithm step B must +see results that step A wrote into **memory shared by that workgroup**, every +invocation in the group must reach a meeting point first. That meeting point is a +**barrier**. + +On Vulkan/GLSL this is a shader builtin (`barrier` plus a shared-memory memory +barrier), not a host lock and not a spinloop you write by hand. + +# API: two names, one lowering + +```python +from cthreads.sync import Barrier, __sync_threads +from cthreads.gpu import Gpu, GlobalIdx, ThreadIdx + +@Gpu +def example(n: int, data: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + # ... writes that peers in this workgroup must observe ... + __sync_threads() + # equivalent: + # Barrier.arrive_and_wait() + # ... reads of those writes ... +``` + +| Form | Audience | GPU effect | +|------|----------|------------| +| `__sync_threads()` | CUDA-familiar | Workgroup barrier + shared memory barrier | +| `Barrier.arrive_and_wait()` | Same idea, method-shaped name | Identical lowering | + +Both are compile-only on the GPU path. Calling `__sync_threads()` from ordinary +Python raises `RuntimeError`. + +# Scope: workgroup only + +```text +Workgroup 0 invocations <--> barrier waits here only +Workgroup 1 invocations <--> separate barrier domain +``` + +A barrier does **not**: + +- Wait for all workgroups in the launch +- Make Python see updated lists mid-run +- Replace a second `gpu()` launch when your algorithm needs a global phase + +Default workgroup size is 64 along X. At most those peers participate in one +barrier instance. + +# CPU Barrier vs GPU barrier + +| | CPU `cthreads.sync.Barrier` | GPU forms above | +|--|----------------------------|-----------------| +| Construction | `Barrier(parties)` then instance `arrive_and_wait()` | No construction in `@Gpu` | +| Parties | Explicit count you choose | Implicit: workgroup size | +| Machine | OS threads / host mutex | GPU shader builtin | +| Import package | Same `cthreads.sync` | Same `cthreads.sync` | + +Sharing the import surface is intentional. Sharing the implementation is not: +the compiler backend chooses the lowering. + +# What you can do in 0.2.0 + +Barriers are available so cooperative patterns can be authored against a stable +API. **Without workgroup shared memory**, the practical uses inside a single +kernel are limited: there is no user-declared shared array to stage tiles into +yet. You can still: + +- Call the barrier forms (they lower correctly). +- Structure multi-pass algorithms as **multiple `gpu()` launches** on the host + (global phases). +- Use residency (`GpuArena`) so those phases do not thrash host memory. + +# Shared memory (0.2.1) + +Planned follow-up: declare workgroup-local shared storage and use barriers between +produce/consume steps inside the group (classic tiled reductions, stencils, and +so on). Until that lands, prefer host multi-pass for algorithms that need +cross-element staging beyond ordinary list buffers. + +# Rejected forms + +Inside `@Gpu`: + +```python +Barrier(64) # TypeError: construction not valid inside @Gpu +Barrier() # same +b = Barrier(64) # not a GPU pattern +b.arrive_and_wait() # instance path is CPU-oriented +``` + +Use only: + +```python +__sync_threads() +Barrier.arrive_and_wait() +``` + +# Host-side phases instead of grid barriers + +Example: two global steps over the same arrays. + +```python +from cthreads.gpu import Gpu, GlobalIdx, GpuArena, gpu + +@Gpu +def step_a(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] * 2.0 + +@Gpu +def step_b(n: int, y: list[float], z: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + z[i] = y[i] + 1.0 + +n = len(x) +with GpuArena() as arena: + arena.bind(x=x, y=y, z=z) + gpu(step_a, n, x, y).join(download=False) + gpu(step_b, n, y, z).join(download=False) + arena.sync("z") +``` + +Each `gpu()` launch completes before the next begins when you `join` in between. +That is the supported way to get **grid-wide** ordering in 0.2.0. + +## See also + +- [concepts.md](./concepts.md#workgroups-and-the-grid) +- [CPU sync guide](../sync.md) +- [best_practices.md](./best_practices.md) diff --git a/docs/guide/sync.md b/docs/guide/sync.md index de9f0c9..f927fcd 100644 --- a/docs/guide/sync.md +++ b/docs/guide/sync.md @@ -477,3 +477,4 @@ handle.destroy() * [concepts.md](../concepts.md) - pack / writeback overview * [sync_state_docs.md](../sync_state_docs.md) - bridge, TLS, by-ref packs * [math_and_linalg.md](./math_and_linalg.md) - arrays (separate from TBuffer) +* [gpu/sync.md](./gpu/sync.md) - GPU workgroup barriers (`__sync_threads` / `Barrier.arrive_and_wait` inside `@Gpu`) diff --git a/docs/index.md b/docs/index.md index 2965aa7..a964973 100644 --- a/docs/index.md +++ b/docs/index.md @@ -2,13 +2,17 @@ | description | link | |-|-| -| **Install**: Python, C++ compiler, CMake (including CMake/Ninja in a venv) | [link](./install.md) | +| **Install**: Python, C++ compiler, CMake (including CMake/Ninja in a venv); GPU / Vulkan notes | [link](./install.md) | | Important ``concepts for beginners``. CThreads ``tricks``, ``best practices`` and an ``introduction to multithreading`` | [link](./concepts.md) | | `@Thread` / `@Threadable`: types, kernel language subset, structs, methods, constructor | [link](./guide/thread_and_threadable.md) | | Indepth guide on ``Thread Pools`` for professional thread management | [link](./guide/pools.md) | | `cthreads.sync`: Guides on how to use the native `sync` module (includes **`LOCKS`** and **`state synchronization`**). Explains how `C++` and `Python` interact in the threaded environment and how to ensure memory validity | [link](./guide/sync.md) | | Guide to the internal **`linalg`** and **`math`** modules for high performance math tasks. Includes: `tensor (cthreads.Array)`, `cmath` and `python stdlib math` | [link](./guide/math_and_linalg.md) | | How to use `cthreads.Job` for custom thread handling and `async` applications | [link](./guide/jobs.md) | +| **GPU (0.2.0):** `@Gpu` / `gpu()` / `GpuArena` / workgroup barriers - concepts, quickstart, best practices, examples | [link](./guide/gpu/README.md) | | `cthreads documentation` | [link](./COMPILER.md) | | **Release**: GitHub Actions, TestPyPI, PyPI trusted publishing | [link](./release.md) | | End-to-end example (`@Thread` / `@Threadable` through codegen) | [link](./Example.md) | +| **Vulkan / GPU contributor guide** (substrate, not the first user path) | [link](./vk_guide/README.md) | +| **Internals:** GPU C++ modules (Context -> launch/join) | [link](./internals/gpu/README.md) | +| **Future:** CPU `@Thread` launching GPU | [link](./gpu_future_cpu_to_gpu.md) | diff --git a/docs/install.md b/docs/install.md index 6579dbf..0a664c9 100644 --- a/docs/install.md +++ b/docs/install.md @@ -20,6 +20,8 @@ python -m venv .venv # activate the venv, then: python -m pip install -U pip python -m pip install cthreads +# optional Vulkan GPU build of _ext: +# python -m pip install "cthreads[gpu]" ``` Check: @@ -159,12 +161,76 @@ Calling `thread(..., force=True)` while kernels are still loaded raises. Unload | Editable import finds Python but not `_ext` | Re-run `pip install -e .` with the venv active so the post-build copy lands beside `cthreads/`. | | `LoadLibrary` error **4551** on Windows | Smart App Control blocking the unsigned `cthreads_kernels.dll` compiled in your project. Turn SAC off or use WSL/Linux. See [release.md](./release.md). | +## GPU (Vulkan compute) + +From **0.2.0**, cthreads can run `@Gpu` kernels on a Vulkan compute device when the +native extension includes GPU support and the machine has a working Vulkan ICD +(normally installed with your GPU drivers). + +### End users (PyPI) + +| Install | `_ext` contents | +|---------|-----------------| +| `pip install cthreads` | CPU only (`CTHREADS_GPU` off) | +| `pip install "cthreads[gpu]"` | Pulls **`cthreads-gpu`** (GPU built into `_ext`) | +| `pip install cthreads-gpu` | Same GPU build; import path remains `cthreads` | + +Pip extras cannot change the files inside one wheel, so GPU support is a second +project (`cthreads-gpu`) that provides the same `cthreads` package with Vulkan +code linked in. Prefer one of the GPU installs above when you need `@Gpu`. + +```python +from cthreads import gpu + +if gpu.available(): + print(gpu.device_name()) +else: + print("GPU path not usable (no Vulkan ICD / device); CPU @Thread still works") +``` + +You need GPU **drivers** with a Vulkan ICD. You do **not** need the LunarG SDK +only to run. User guides: [guide/gpu/README.md](./guide/gpu/README.md). + +### Contributors (editable / source) + +Default editable builds leave `CTHREADS_GPU` **OFF**. The `[gpu]` extra installs the +PyPI `cthreads-gpu` dependency and does **not** flip CMake for a local editable +build. To compile GPU into your local `_ext`: + +```powershell +# PowerShell +$env:CMAKE_ARGS="-DCTHREADS_GPU=ON" +pip install -e ".[test]" +``` + +```bash +export CMAKE_ARGS="-DCTHREADS_GPU=ON" +pip install -e ".[test]" +``` + +Or: + +```bash +pip install -e . --config-settings=cmake.define.CTHREADS_GPU=ON +``` + +Building with `CTHREADS_GPU=ON` needs Vulkan **headers** (LunarG SDK or distro +`libvulkan-dev`). Runtime still loads the loader dynamically; end users need +drivers, not the SDK. + +Release packaging (CPU + GPU wheels): [release.md](./release.md). +More detail: [vk_guide/03-sdk-runtime-drivers.md](./vk_guide/03-sdk-runtime-drivers.md) +and [guide/gpu/errors.md](./guide/gpu/errors.md). + ## Publishing to PyPI -Wheels and the sdist are built on GitHub Actions when a GitHub Release is published. End users install with `pip install cthreads` (see [From PyPI](#from-pypi-recommended) above). Maintainer walkthrough: [release.md](./release.md). +Wheels and the sdist are built on GitHub Actions when a GitHub Release is published. +End users: `pip install cthreads` or `pip install "cthreads[gpu]"`. Maintainer +walkthrough: [release.md](./release.md). ## Next - [README](../README.md) - `@Thread` / `@Threadable` and first `thread(...)` / `join` / `await` - [concepts](./concepts.md) - GIL, pack / writeback, rules +- [GPU guides](./guide/gpu/README.md) - `@Gpu`, `gpu()`, arena, barriers - [Guides](./index.md) - pools, sync, jobs, math diff --git a/docs/internals/gpu/README.md b/docs/internals/gpu/README.md new file mode 100644 index 0000000..304512c --- /dev/null +++ b/docs/internals/gpu/README.md @@ -0,0 +1,87 @@ +# GPU internals (C++ Vulkan path) + +This folder documents the C++ GPU modules under `src/cthreads/cpp/gpu/`. It is +written for contributors who know C++ (and maybe Python) but may not know Vulkan. + +**Application authors** should use the product guides first: +[docs/guide/gpu/README.md](../../guide/gpu/README.md). Those cover `@Gpu`, +`gpu()`, `GpuArena`, and barriers for **0.2.0**. This folder is the permanent +substrate those APIs call. + +Related reading: + +- User GPU guides: [docs/guide/gpu/README.md](../../guide/gpu/README.md) +- Contributor Vulkan tutorial: [docs/vk_guide/README.md](../../vk_guide/README.md) +- Future work (CPU `@Thread` calling GPU): [docs/gpu_future_cpu_to_gpu.md](../../gpu_future_cpu_to_gpu.md) +- Build flag: CMake `CTHREADS_GPU=ON` compiles these sources into `cthreads._ext` + +## What problem this stack solves + +CPU `@Thread` kernels run as normal operating system threads calling generated C++. GPU work is different. The GPU is a separate processor with its own memory. You cannot hand it a Python list pointer and expect a shader to read it. You must: + +1. Open a connection to a Vulkan-capable GPU (the [Context](./context.md)). +2. Allocate GPU memory and copy host bytes in and out ([memory](./memory.md)). +3. Group one launch's buffers into a GpuPack ([pack](./pack.md)). +4. Tell the shader which binding number maps to which buffer ([descriptors](./descriptors.md)). +5. Build a reusable compute pipeline from SPIR-V bytes ([shader](./shader.md)). +6. Launch: record bind+dispatch, submit with a fence, then `join` writeback ([module](./module.md)). + +## Mental model in one picture + +```text +Python / tests + | + v +Context (loader, device, queue, function pointers, TransferEngine) + | + +-- memory:: create/destroy GpuBuffer, upload/download via staging + | + +-- pack:: GpuPack (scalar SSBO + list SSBOs) + | + descriptor pool/set/update (same namespace, descriptors.hpp) + | + +-- shader:: ShaderCacheEntry (module, layouts, pipeline) + | ShaderCache (symbol -> entry) + | + +-- launch:: launch_gpu_kernel -> SpawnedGpuKernel::join (writeback) +``` + +## Binding convention + +cthreads packs shader arguments like this: + +| Binding | Contents | +|-|-| +| 0 | One storage buffer for all scalars (`int`, `float`, `bool`, POD fields) | +| 1 | First list argument | +| 2 | Second list argument | +| ... | More lists | + +There are no buffer device addresses (raw GPU pointers) stuffed inside the scalar struct. The descriptor set is the wiring table. + +## Module index + +| Doc | Namespace / types | Source | +|-|-|-| +| [Context](./context.md) | `cthreads::gpu::Context`, `TransferEngine` | `headers/context.hpp`, `impl/context.cpp` | +| [Memory](./memory.md) | `cthreads::gpu::memory` | `headers/memory.hpp`, `impl/memory.cpp` | +| [Pack](./pack.md) | `cthreads::gpu::pack` (GpuPack create/upload/download) | `headers/pack.hpp`, `impl/pack.cpp` | +| [Descriptors](./descriptors.md) | `cthreads::gpu::pack` (pool/set/update) | `headers/descriptors.hpp`, `impl/descriptors.cpp` | +| [Shader](./shader.md) | `cthreads::gpu::shader` | `headers/shader.hpp`, `shader_cache.hpp`, `impl/shader.cpp`, `shader_cache.cpp` | +| [Module / launch](./module.md) | `SpawnedGpuKernel`, `launch_gpu_kernel` | `headers/module.hpp`, `impl/module.cpp` | + +## Build and availability + +GPU code is compiled only when `CTHREADS_GPU` is on. The extension still loads `vulkan-1` (or `libvulkan.so.1`) dynamically at runtime. A machine without a Vulkan loader or GPU fails cleanly through Python `cthreads.gpu.available()` rather than crashing the import of CPU-only wheels. + +## What is intentionally not here yet + +Product Python (`@Gpu` / `gpu()` / list writeback / arena / workgroup barriers) +lives in `src/cthreads/python/cthreads/gpu/` and is documented for users in +[guide/gpu](../../guide/gpu/README.md). Remaining substrate / dialect gaps include: + +- Workgroup shared memory (planned product **0.2.1**) +- Device atomics in the dialect +- Threadable / nested object marshal on GPU +- Inflight job store (header stub only) +- Dummy SSBOs for empty list slots in `update_descriptors` +- CPU `@Thread` launching `@Gpu` ([gpu_future_cpu_to_gpu.md](../../gpu_future_cpu_to_gpu.md)) diff --git a/docs/internals/gpu/context.md b/docs/internals/gpu/context.md new file mode 100644 index 0000000..26eeaa2 --- /dev/null +++ b/docs/internals/gpu/context.md @@ -0,0 +1,169 @@ +# Context and TransferEngine + +Source: `src/cthreads/cpp/gpu/headers/context.hpp`, `src/cthreads/cpp/gpu/impl/context.cpp`. + +Namespace: `cthreads::gpu`. + +This document explains the process-wide Vulkan connection that every other GPU helper needs. If you have never used Vulkan, start here. + +## Why a Context exists + +Vulkan is not a single library call like "run this kernel." It is a layered API: + +1. An operating system library (the Vulkan loader) finds GPU drivers. +2. You create an instance (your application's connection to the loader). +3. You pick a physical device (a GPU the driver can see). +4. You create a logical device (your opened session on that GPU). +5. You get a queue (a submission port where you send work). +6. You resolve dozens of function pointers by name, because cthreads loads the loader dynamically instead of linking `vulkan-1` into every wheel. + +The `Context` struct holds all of that state for the whole process. There is one Context, accessed through `context()`, guarded by a mutex for `init` / `shutdown` / `available`. + +## Public functions + +### `void init()` + +Opens the loader, creates instance and device, resolves entry points, marks `ready = true`, and creates the TransferEngine and LaunchEngine. Throws typed-style error strings (for example `cthreads.gpu.VulkanLoaderNotFound`) that Python maps into exceptions in `cthreads.gpu`. + +Call this when you need the GPU. Python `cthreads.gpu.init()` ends up here. + +### `void shutdown()` + +Destroys children first, then parents: + +1. LaunchEngine (free fences, then command pool) +2. TransferEngine (staging buffer, fence, command pool) +2. Shader cache entries (pipelines and layouts) +3. Logical device +4. Instance +5. Unload the loader library +6. Clear all function pointers and handles + +Safe to call if the Context was never initialized. + +### `bool available()` + +Tries to ensure the Context is ready without throwing to the caller for "no GPU" style probes. Used by Python `available()` and by tests that skip when there is no device. + +### `const std::string& device_name()` + +Returns the human-readable GPU name from Vulkan device properties. Requires a ready Context. + +### `Context& context()` + +Returns the singleton. Most C++ helpers take `Context&` explicitly so ownership and testing stay clear. + +## Struct `Context` fields (grouped) + +### Loader and bootstrap + +- `loader_module`: operating system handle to `vulkan-1.dll` / `libvulkan.so.1`. +- `vkGetInstanceProcAddr`: the bootstrap function used to look up almost every other Vulkan function by name string. + +### Instance and device entry points + +Examples: `vkCreateInstance`, `vkEnumeratePhysicalDevices`, `vkCreateDevice`, `vkGetDeviceQueue`, `vkDestroyDevice`. + +These are stored as typed function pointers (`PFN_vk...`). A pointer-to-function (PFN) is simply a C function pointer with the Vulkan signature. cthreads assigns them during init after the loader is open. + +### Buffer and memory entry points + +Used by [memory](./memory.md): create/destroy buffers, allocate/free device memory, map host-visible memory, query memory properties. + +### Command, copy, and sync entry points + +Used by transfers and launch: command pools, command buffers, `vkCmdCopyBuffer`, fences, `vkQueueSubmit`, wait/reset fences. + +### Shader and pipeline entry points + +Used by [shader](./shader.md): create/destroy shader modules, descriptor set layouts, pipeline layouts, and compute pipelines. + +### Descriptor pool and update entry points + +Used by [descriptors](./descriptors.md): create/destroy descriptor pools, allocate/free sets, `vkUpdateDescriptorSets`. + +### Compute dispatch entry points + +Used by [module / launch](./module.md): `vkCmdBindPipeline`, `vkCmdBindDescriptorSets`, `vkCmdDispatch`, `vkCmdPipelineBarrier`. + +### Opaque handles + +- `instance`: connection to the loader for this application. +- `physical_device`: the chosen GPU. +- `device`: the logical device (opened GPU session). +- `queue`: compute queue used to submit copies and dispatches. +- `queue_family`: index of the queue family that supports compute (needed when creating command pools). +- `device_name`: string name for logging and Python. +- `ready`: true only after a fully successful init. + +### TransferEngine and mutex + +See the next section. `transfer_engine_mutex` serializes use of the single shared engine. + +### LaunchEngine and mutex + +`LaunchEngine` owns the process-lifetime command pool used by `launch_gpu_kernel`. Jobs checkout a command buffer + fence, submit under `launch_engine_mutex`, wait their own fence on join, then return the CB/fence to free lists. Overlapping jobs are supported; one shared fence is not. + +## Technical terms + +- Vulkan loader: system library that discovers Installable Client Drivers (ICDs), which are the vendor GPU drivers. +- Instance: application-level Vulkan object. Not the GPU itself. +- Physical device: one GPU as seen by the driver. +- Logical device: your opened handle to use that GPU. +- Queue: port where the CPU submits recorded command buffers. +- Queue family: group of queues with the same capabilities (graphics, compute, transfer). +- Dynamic loading: open the DLL/SO at runtime and resolve symbols by name, instead of linking at build time. +- `VK_NULL_HANDLE`: sentinel meaning "no object." + +## TransferEngine + +### Purpose + +Uploading and downloading buffers needs a short GPU copy: host-visible staging memory to or from a device-local buffer. Creating a brand new command pool and fence for every copy would be slow and wasteful. The TransferEngine keeps reusable machinery on the Context for the process lifetime. + +### Fields + +- `command_pool`: Vulkan command pool for the compute queue family. Command buffers for copies are allocated from here and freed after each wait. +- `fence`: CPU-waitable fence. Created signaled so the first reset/wait path treats it as idle. After each copy, the CPU waits on this fence. +- `staging`: optional host-visible `GpuBuffer`. Empty until the first upload or download grows it. It only grows; it never shrinks until Context shutdown. + +### Lifecycle + +- Created in `init_transfer_engine` after the device is ready. +- Destroyed in `shutdown_transfer_engine` before the logical device is destroyed. +- Staging is grown by `memory::ensure_staging` under the transfer engine mutex. + +### Concurrency + +There is one engine today. All upload/download paths hold `transfer_engine_mutex` for the whole transfer (staging grow, memcpy, GPU copy, wait). That avoids races on the shared pool, fence, staging buffer, and queue submit. + +A future pool of engines is discussed for CPU threads launching GPU work. See [gpu_future_cpu_to_gpu.md](../../gpu_future_cpu_to_gpu.md). + +## Key Vulkan calls during init (story order) + +1. Load the loader library (`LoadLibrary` / `dlopen`). +2. Resolve `vkGetInstanceProcAddr`, then `vkCreateInstance`. +3. `vkCreateInstance` builds the instance. +4. Resolve instance-level functions. +5. `vkEnumeratePhysicalDevices` lists GPUs; pick one with a compute queue family (prefer discrete when possible). +6. `vkCreateDevice` opens the logical device; `vkGetDeviceQueue` gets the queue. +7. Resolve device-level functions (buffers, commands, shaders, descriptors). +8. Create TransferEngine pool and fence. +9. Create LaunchEngine command pool (CB/fence free lists grow on demand). +9. Set `ready = true`. + +## Key Vulkan calls during shutdown (story order) + +1. Wait on the transfer fence if needed; destroy staging, fence, command pool. +2. Clear the shader cache (destroy pipelines and layouts). +3. `vkDestroyDevice`. +4. `vkDestroyInstance`. +5. Unload the loader; null all pointers. + +## Relationship to other modules + +Every helper in memory, pack, descriptors, and shader takes a `Context&` and expects `ready == true` plus the entry points it needs. If init failed or shutdown already ran, those helpers throw `VulkanInitFailed`-style errors. + +## Python surface today + +`cthreads.gpu` exposes `init`, `shutdown`, `available`, and `device_name`, plus typed errors. It does not yet expose packs, shaders, or jobs. Those will wrap the C++ types documented in the sibling pages. diff --git a/docs/internals/gpu/descriptors.md b/docs/internals/gpu/descriptors.md new file mode 100644 index 0000000..0ed88e0 --- /dev/null +++ b/docs/internals/gpu/descriptors.md @@ -0,0 +1,118 @@ +# Descriptors (pack namespace) + +Source: `src/cthreads/cpp/gpu/headers/descriptors.hpp`, `src/cthreads/cpp/gpu/impl/descriptors.cpp`. + +Namespace: `cthreads::gpu::pack` (same as GpuPack; separate files for clarity). + +Depends on: [Context](./context.md), [Pack](./pack.md), [Shader](./shader.md) (for set layouts and `ShaderCacheEntry`). + +## Purpose in plain language + +A compute shader does not receive C++ pointers. It declares numbered bindings, for example "binding 0 is my scalar block" and "binding 1 is list x." On the CPU you own `VkBuffer` handles inside a `GpuPack`. Descriptors are the table that connects those two worlds: + +```text +shader binding 0 -> pack.scalar_buffer +shader binding 1 -> pack.container_slots[0] +shader binding 2 -> pack.container_slots[1] +... +``` + +There are two layers: + +1. Descriptor set layout: the schema (which binding numbers exist and that each is a storage buffer). Created once per kernel in `shader::create_entry` and stored on `ShaderCacheEntry`. +2. Descriptor set: one filled-in instance of that schema for one launch. Created from a pool, then updated to point at this launch's pack buffers. + +## Technical terms + +- Descriptor: one slot in the table (for us, always a storage buffer reference). +- Descriptor set layout: immutable schema of bindings for a pipeline. +- Descriptor set: concrete table instance you bind before dispatch. +- Descriptor pool: allocator that vends descriptor sets (similar in spirit to a memory pool). +- Storage buffer descriptor: a descriptor type that points at a `VkBuffer` used as an SSBO. +- `vkUpdateDescriptorSets`: Vulkan call that writes buffer handles into a set. +- Binding number: integer the shader and the CPU agree on (0 for scalars, then lists). + +## Struct `DescriptorPool` + +Fields: + +- `pool`: Vulkan `VkDescriptorPool` handle. +- `max_sets`: how many sets the pool was created to hold. +- `binding_count`: how many storage buffer descriptors each set needs (must match the shader cache entry). + +The pool is created with `VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT` so individual sets can be returned with `free_set` when a job finishes. + +## Function `create_pool` + +Creates a pool for compute sets that follow the binding convention. + +Parameters: + +- `binding_count`: storage buffers per set (1 + number of list slots). +- `max_sets`: capacity. + +Internally it declares one pool size entry of type `STORAGE_BUFFER` with `descriptorCount = binding_count * max_sets`, then calls `vkCreateDescriptorPool`. + +## Function `destroy_pool` + +Destroys the Vulkan pool and clears the struct. All sets allocated from the pool become invalid. Prefer freeing live sets first when jobs still hold them. + +## Function `allocate_set` + +Allocates one `VkDescriptorSet` from the pool using a `VkDescriptorSetLayout` (normally `ShaderCacheEntry::set_layout`). The set is empty until `update_descriptors` runs. + +Key call: `vkAllocateDescriptorSets`. + +## Function `free_set` + +Returns a set to the pool via `vkFreeDescriptorSets`. Safe no-op if the set handle is already null. After success, the caller's set handle is set to `VK_NULL_HANDLE`. + +## Function `update_descriptors` (binding count overload) + +Writes the pack into the set. + +Rules: + +- `binding_count` must equal `1 + pack.container_slots.size()`. +- Binding 0 uses `pack.scalar_buffer`. +- Binding `i` for `i >= 1` uses `pack.container_slots[i - 1]`. +- Every written binding must have a non-null buffer and non-zero size. Empty pack slots are not supported yet. + +For each binding it builds a `VkDescriptorBufferInfo` (buffer, offset 0, range = size) and a `VkWriteDescriptorSet` of type `STORAGE_BUFFER`, then calls `vkUpdateDescriptorSets` once for all writes. + +## Function `update_descriptors` (entry overload) + +Convenience wrapper that uses `entry.binding_count` from a `ShaderCacheEntry`. + +## Who calls these + +The launch path (`launch_gpu_kernel` today via tests; later `gpu()`): + +```text +ShaderCache.get(symbol) -> entry +create_pool / allocate_set(entry.set_layout) +update_descriptors(set, entry, pack) +record CB: barrier -> bind pipeline -> bind set -> dispatch +vkQueueSubmit(..., fence) +// join: wait fence -> download ref lists -> release +``` + +The shader cache never updates descriptors. It only stores the layout schema. + +## Key Vulkan calls summary + +| Call | Role | +|-|-| +| `vkCreateDescriptorPool` | Create the allocator | +| `vkDestroyDescriptorPool` | Destroy the allocator | +| `vkAllocateDescriptorSets` | Get one set matching a layout | +| `vkFreeDescriptorSets` | Return a set to the pool | +| `vkUpdateDescriptorSets` | Point bindings at `VkBuffer`s | + +## Current limitation: empty lists + +`GpuPack` allows empty container slots without a buffer. `update_descriptors` throws if a required binding has a null buffer, because standard Vulkan does not accept a null buffer descriptor without special null-descriptor features. Until a process-wide dummy SSBO exists, test kernels should use non-empty buffers for every binding they declare. + +## What comes after update + +Descriptors alone do not run the shader. [Module / launch](./module.md) records the command buffer, submits with a fence, and `join` downloads ref lists into Python. diff --git a/docs/internals/gpu/memory.md b/docs/internals/gpu/memory.md new file mode 100644 index 0000000..0b10a09 --- /dev/null +++ b/docs/internals/gpu/memory.md @@ -0,0 +1,140 @@ +# Memory helpers + +Source: `src/cthreads/cpp/gpu/headers/memory.hpp`, `src/cthreads/cpp/gpu/impl/memory.cpp`. + +Namespace: `cthreads::gpu::memory`. + +Depends on: [Context and TransferEngine](./context.md). + +This module owns one idea: a contiguous byte region on the GPU (`GpuBuffer`), plus helpers to create it, destroy it, and copy bytes between host memory and device-local storage. + +## Host memory versus device memory + +- Host memory is ordinary process RAM. C++ `memcpy` and Python buffers live here. +- Device-local memory is GPU memory that shaders prefer. The CPU usually cannot keep a permanent pointer into it. +- Host-visible memory is a special GPU allocation the CPU can map. It is often slower for heavy shader traffic, so cthreads uses it only as a temporary staging mirror. + +cthreads's rule: shader-facing data lives in device-local buffers. Host traffic always goes through staging plus a GPU copy. + +## Technical terms + +- Storage buffer (SSBO): a buffer a compute shader can read and write through a descriptor binding. +- Staging buffer: host-visible buffer used only as the CPU side of an upload or download. +- `VkBuffer`: opaque Vulkan handle describing a buffer resource (size and usage). It is not a raw C pointer to bytes. +- `VkDeviceMemory`: opaque handle for an allocated memory slab. You bind a buffer to memory before using it. +- Memory type: one of the driver-exposed categories (device-local, host-visible, host-coherent, and so on). +- Fence: a GPU timeline object the CPU can wait on until submitted work finishes. +- Command buffer: a recorded list of GPU commands (for transfers, usually a single copy). +- Command pool: allocator that owns command buffers for one queue family. + +## Enum `BufferKind` + +One create API, two property sets. + +### `BufferKind::Staging` + +- Usage: transfer source and transfer destination. +- Memory: host-visible and host-coherent. +- After create, the buffer is mapped and `GpuBuffer::mapped` points at CPU-writable bytes. + +### `BufferKind::DeviceLocal` + +- Usage: storage buffer plus transfer source and destination (so shaders and copies both work). +- Memory: device-local. +- Never persistently mapped (`mapped` stays null). + +## Struct `GpuBuffer` + +Fields: + +- `buffer`: `VkBuffer` handle, or `VK_NULL_HANDLE` if empty. +- `memory`: `VkDeviceMemory` bound to that buffer, or null handle if empty. +- `size`: caller-facing byte count requested at create time. +- `mapped`: CPU pointer for staging only. +- `kind`: staging or device-local. + +Ownership: the struct owns the Vulkan objects until `destroy_buffer` runs. Moving the struct moves the handles; it does not clone GPU bytes. + +## Function `find_memory_type` + +Vulkan returns a bitmask of legal memory types for a new buffer (`type_bits`). You also request property flags (for example host-visible). This helper walks the physical device's memory types and returns the first index that is allowed by the bitmask and has every requested property. + +Used internally by `create_buffer`. You rarely call it from higher layers. + +## Function `create_buffer` + +Creates a `GpuBuffer` of the given kind and size. + +Steps in plain language: + +1. Reject size 0 and missing entry points. +2. Choose usage flags and memory properties from `BufferKind`. +3. Call `vkCreateBuffer` to create the buffer object. +4. Query memory requirements with `vkGetBufferMemoryRequirements`. +5. Pick a memory type with `find_memory_type`. +6. Call `vkAllocateMemory`, then `vkBindBufferMemory` at offset 0. +7. For staging, call `vkMapMemory` and store the pointer in `mapped`. + +Throws on failure. On partial failure it cleans up objects it already created. + +## Function `destroy_buffer` + +Unmaps staging memory if needed, destroys the `VkBuffer`, frees `VkDeviceMemory`, and clears the struct. Safe on an already-empty buffer. + +## Function `upload_buffer` + +Copies host bytes into a device-local `GpuBuffer`. + +Path: + +```text +host pointer + -> memcpy into TransferEngine staging (grown if needed) + -> GPU vkCmdCopyBuffer staging -> device-local + -> wait on TransferEngine fence +``` + +The whole path holds `context.transfer_engine_mutex` so two threads cannot share the engine's staging, pool, or fence unsafely. + +Requirements: + +- Target buffer must be device-local and non-null. +- `data` non-null; `size` in `(0, buffer.size]`. + +## Function `download_buffer` + +Opposite direction: + +```text +device-local + -> GPU copy into TransferEngine staging + -> wait + -> memcpy staging -> host pointer +``` + +Same mutex and validation rules as upload. + +## Internal helper `copy_buffer_and_wait` + +Not part of the public header. Records a one-time command buffer that copies `size` bytes from one `VkBuffer` to another, submits it on the Context queue with the TransferEngine fence, waits, then frees the command buffer. + +Important Vulkan calls: + +- `vkAllocateCommandBuffers` / `vkFreeCommandBuffers` +- `vkBeginCommandBuffer` / `vkEndCommandBuffer` +- `vkCmdCopyBuffer` +- `vkResetFences` / `vkQueueSubmit` / `vkWaitForFences` + +Caller must already hold the transfer engine mutex. + +## Internal helper `ensure_staging` + +Grows the TransferEngine staging buffer so it is at least `size` bytes. Never shrinks until Context shutdown. Caller must hold the transfer engine mutex. + +## Why not map device-local buffers? + +Many discrete GPUs cannot give the CPU a fast permanent pointer to device-local memory. Staging plus an explicit GPU copy is the portable model and matches how real engines move data. + +## How pack uses this + +[Pack](./pack.md) creates one device-local `GpuBuffer` for scalars and one per non-empty list. Upload and download helpers on the pack call `upload_buffer` / `download_buffer` for each region. diff --git a/docs/internals/gpu/module.md b/docs/internals/gpu/module.md new file mode 100644 index 0000000..2af5a4f --- /dev/null +++ b/docs/internals/gpu/module.md @@ -0,0 +1,56 @@ +# Launch path (`SpawnedGpuKernel`) + +Sources: + +- `src/cthreads/cpp/gpu/headers/module.hpp` +- `src/cthreads/cpp/gpu/impl/module.cpp` + +Namespace: `cthreads::gpu`. + +Depends on: [Context](./context.md), [Pack](./pack.md), [Descriptors](./descriptors.md), [Shader](./shader.md). + +## Purpose + +`launch_gpu_kernel` is the GPU analogue of CPU `spawn_from_meta`: build a `GpuPack`, wire descriptors, record bind+dispatch, submit with a fence, and return a job handle. `SpawnedGpuKernel::join` waits on that fence, downloads ref lists into the same Python objects, then releases Vulkan state. There is no OS worker thread and no mid-run `sync_state`. + +Public `gpu()` / `@Gpu` (later) will call these same types. Product pybind: +`_ext.gpu.launch_gpu_kernel` + `_ext.gpu.GpuJob`. Tests register smoke SPIR-V +via `_ext.gpu.testing.register_smoke_saxpy`, then launch on the product path. + +## Technical terms + +- Fence: CPU waits until the submitted dispatch has finished. +- Writeback: copy device list SSBOs back into the kept Python `list` objects (`pass_as` ref). +- Per-job command buffer + fence: checked out from Context `LaunchEngine` for the job lifetime; returned on join (supports overlapping launches). The command **pool** is process-lifetime on Context. + +## Struct `SpawnedGpuKernel` + +Owns per-launch GPU handles plus writeback inputs: + +| Field | Role | +|-|-| +| `pack` | Device-local scalar + list SSBOs | +| `descriptor_pool` / `descriptor_set` | Wired to this pack | +| `command_pool` / `command_buffer` | Recorded dispatch | +| `fence` | Signals when submit completes | +| `values_keep` | Python args kept alive for list writeback | +| `writeback_lists` | Plan of ref list slots to download on join | + +## Function `launch_gpu_kernel` + +Takes `meta` + `ordered_values` (same shape as the docstring on `module.hpp`). Registers nothing in the shader cache; the symbol must already exist. Returns `shared_ptr` without waiting. + +## Method `join` + +1. `vkWaitForFences` on the job fence +2. Compute → transfer barrier (TransferEngine) +3. For each ref list: `download_container` → fill the kept `py::list` in place +4. `release_inflight` (free CB/pool/set/fence/pack, drop `values_keep`) + +Value scalars are not written back. Threadable/schema marshal is later. + +## Testing + +`register_smoke_saxpy` puts committed saxpy SPIR-V in ShaderCache (test-only). +Pytest drives product `launch_gpu_kernel` + `GpuJob.join` and asserts `y`: +`tests/unit/test_gpu_shader.py::test_live_launch_saxpy_product_path`. diff --git a/docs/internals/gpu/pack.md b/docs/internals/gpu/pack.md new file mode 100644 index 0000000..d021ca0 --- /dev/null +++ b/docs/internals/gpu/pack.md @@ -0,0 +1,103 @@ +# GpuPack + +Source: `src/cthreads/cpp/gpu/headers/pack.hpp`, `src/cthreads/cpp/gpu/impl/pack.cpp`. + +Namespace: `cthreads::gpu::pack`. + +Depends on: [Context](./context.md), [Memory](./memory.md). + +Descriptor pool and update helpers live in the same namespace but in separate files. See [Descriptors](./descriptors.md). + +## Purpose + +A `GpuPack` is the per-launch bag of GPU buffers for one compute dispatch under the binding convention: + +- One device-local scalar storage buffer (binding 0), or none if there are no scalar bytes. +- One device-local storage buffer per list/container argument (bindings 1..N). + +The pack does not know Python argument names or std430 field offsets. Marshal and codegen decide how to flatten scalars into the scalar blob and which list is slot `i`. The pack only owns buffers and moves bytes. + +## Technical terms + +- Binding convention: one scalar SSBO plus one SSBO per list, with no buffer device addresses inside the scalar block. +- SSBO: storage buffer object; a shader-readable and writable buffer bound through descriptors. +- Container slot: one list-like argument inside the pack. +- `numel`: number of elements in a slot. Zero means "no Vulkan buffer for this slot." +- std430: a GLSL/SPIR-V memory layout rule for how fields pack in a storage buffer. Marshal must match it; this module only stores opaque bytes. + +## Struct `ContainerSpec` + +Create-time size description for one list slot. + +- `elem_bytes`: bytes per element (4 for `float` or 32-bit `int`). +- `numel`: element count. If zero, create keeps an empty slot and does not call `vkCreateBuffer` with size 0. + +## Struct `ContainerSlot` + +- `buffer`: device-local `GpuBuffer` when `numel > 0`; empty handles otherwise. +- `spec`: the `ContainerSpec` used at create time (also used for upload size checks). + +## Struct `GpuPack` + +- `scalar_buffer`: device-local blob for all packed scalars. Empty if `scalar_bytes` was 0 at create. +- `container_slots`: vector in binding order for bindings 1..N. + +Returning a `GpuPack` by value moves handles only. It does not clone GPU memory. + +## Function `create_gpu_pack` + +Allocates the pack on a ready Context. + +Behavior: + +- If `scalar_bytes > 0`, create a device-local scalar buffer of that size. +- For each container spec, append a slot. If `numel > 0`, require `elem_bytes > 0` and create a device-local buffer of `elem_bytes * numel`. If `numel == 0`, keep a slot with no buffer. + +Zero-size Vulkan buffers are never created. + +## Upload helpers + +### `upload_scalars` + +Copies host bytes into `pack.scalar_buffer` via [memory upload](./memory.md). + +### `upload_container` + +Copies host bytes into one non-empty slot. Size must match the slot's byte size. + +### `upload_containers` + +Uploads every non-empty slot from parallel host pointers. Empty slots are skipped. + +## Download helpers + +### `download_scalars` / `download_container` / `download_containers` + +Mirror the upload helpers in the device-to-host direction. Results land in caller-owned host memory. `SpawnedGpuKernel::join` uses these for ref-list writeback (see [Module / launch](./module.md)). + +## Function `destroy_gpu_pack` + +Destroys every non-null buffer through `memory::destroy_buffer` and clears the pack. Safe on an already-empty pack. Does not shut down the Context. + +## Empty lists + +Empty lists are first-class in the pack: the slot exists so binding indices stay stable, but there is no `VkBuffer`. Descriptor update currently rejects null buffers (see [Descriptors](./descriptors.md)). A future dummy SSBO may fill empty bindings; until then, smoke launches should use non-empty lists for every binding the shader declares. + +## How this fits the launch path + +```text +create_gpu_pack +upload_scalars / upload_containers +allocate descriptor set + update_descriptors(pack) // descriptors.hpp +record CB: barrier + bind pipeline/set + dispatch // module.cpp +vkQueueSubmit(..., fence) +join: wait fence -> barrier -> download_* into kept Python lists -> release +destroy_gpu_pack +``` + +## Testing today + +`_ext.gpu.testing` exposes pack round-trip helpers (float and int packs) and +`register_smoke_saxpy` (SPIR-V cache only). Launch/join use product +`_ext.gpu.launch_gpu_kernel`. Pytest: `tests/unit/test_gpu_pack.py`, +`tests/unit/test_gpu_shader.py`. diff --git a/docs/internals/gpu/shader.md b/docs/internals/gpu/shader.md new file mode 100644 index 0000000..4628132 --- /dev/null +++ b/docs/internals/gpu/shader.md @@ -0,0 +1,134 @@ +# Shader cache and create_entry + +Sources: + +- `src/cthreads/cpp/gpu/headers/shader_cache.hpp` +- `src/cthreads/cpp/gpu/headers/shader.hpp` +- `src/cthreads/cpp/gpu/impl/shader_cache.cpp` +- `src/cthreads/cpp/gpu/impl/shader.cpp` + +Namespace: `cthreads::gpu::shader`. + +Depends on: [Context](./context.md). Used by: [Descriptors](./descriptors.md) (set layout), [Module / launch](./module.md). + +## Purpose + +Building a compute pipeline from SPIR-V is expensive. Doing it on every launch would waste time. The shader cache stores, per kernel symbol, the reusable Vulkan objects that stay identical across launches: + +- Shader module (SPIR-V wrapped for Vulkan) +- Descriptor set layout (binding convention schema) +- Pipeline layout (how sets attach to the pipeline) +- Compute pipeline (compiled program ready to bind) +- Binding count metadata + +Per-launch objects (GpuPack buffers, descriptor sets, fences) are not stored here. + +## Access rights + +- Writers: `ShaderRegistry::register_spirv` only (friend of `ShaderCache`; also + exposed as `_ext.gpu.register_shader`). Tests use the same writer. +- Everyone else: `get` returns a const reference. +- Context shutdown: `clear` destroys all Vulkan objects, then empties the map. + +Entries are not mutated in place after insert. Duplicate `add` of the same key throws. + +## Technical terms + +- SPIR-V: binary intermediate language for shaders. Vulkan drivers consume SPIR-V, not GLSL text, at runtime. +- Shader module: Vulkan object created from SPIR-V bytes (`VkShaderModule`). +- Compute pipeline: prepared compute program plus layout (`VkPipeline` with compute bind point). +- Pipeline layout: declares which descriptor set layouts (and optional push constants) a pipeline uses. +- Descriptor set layout: schema of bindings; see [Descriptors](./descriptors.md). +- Entry point name: function name inside the shader. cthreads uses `"main"`. +- Push constants: tiny values pushed in the command buffer without a buffer object. Not used in create_entry today; scalars live in binding 0. + +## Struct `ShaderCacheEntry` + +Move-only. Copying would duplicate Vulkan handles and double-destroy them. + +Fields: + +- `shader_module`: may remain non-null after pipeline create (kept for simplicity; `clear` destroys it). +- `set_layout`: storage buffer bindings `0 .. binding_count-1` (scalars then lists). +- `pipeline_layout`: layout used when creating the pipeline. +- `pipeline`: compute pipeline handle. +- `binding_count`: number of storage buffer bindings (at least 1). + +## Class `ShaderCache` + +Process-wide singleton via `getInstance()`. + +### `add(key, entry)` (private until registry friend exists) + +Moves the entry into an internal `unordered_map`. Returns a const reference to the map node. Throws if the key already exists. + +### `get(key)` + +Returns a const reference to an existing entry. Throws if missing. + +### `clear(context)` + +Destroys every entry's Vulkan objects through `destroy_entry`, then clears the map. Called from Context shutdown before the logical device is destroyed. + +### Destructor + +Only drops the map. Shutdown must have already cleared handles, because static destruction order versus Context is undefined. + +## Function `create_entry` + +Declared in `shader.hpp`, implemented in `shader.cpp`. + +Builds a complete `ShaderCacheEntry` from SPIR-V words and a binding count. Does not insert into the cache; the caller (registry or test) calls `add`. + +### Parameters + +- `context`: ready Context with create and destroy entry points. +- `spirv`: pointer to SPIR-V code as `uint32_t` words. +- `spirv_word_count`: number of words (byte size divided by 4). +- `binding_count`: storage buffer bindings (`>= 1`). + +### Steps + +1. Validate ready device, non-empty SPIR-V, binding count, and create PFNs. +2. `vkCreateShaderModule` from the SPIR-V bytes. +3. Build `binding_count` layout bindings, each `STORAGE_BUFFER`, compute stage, then `vkCreateDescriptorSetLayout`. +4. `vkCreatePipelineLayout` with that single set layout and no push constant ranges. +5. `vkCreateComputePipelines` with stage compute, module, entry `"main"`, and the pipeline layout. +6. On any failure after partial success, call `destroy_entry` and throw. + +### Return + +Owned `ShaderCacheEntry`. Move it into `ShaderCache::add`, or destroy it with `destroy_entry` if you abandon it. + +## Function `destroy_entry` + +Destroys pipeline, pipeline layout, set layout, and shader module in that order (children before parents), then nulls handles. Used by `ShaderCache::clear` and by fail paths in `create_entry`. + +If the device handle is already gone, it only nulls fields (shutdown edge case). + +## Key Vulkan calls + +| Call | Role | +|-|-| +| `vkCreateShaderModule` | Wrap SPIR-V | +| `vkCreateDescriptorSetLayout` | Binding convention schema | +| `vkCreatePipelineLayout` | Attach set layout to pipeline interface | +| `vkCreateComputePipelines` | Compile compute pipeline | +| Matching `vkDestroy*` | Tear down in `destroy_entry` / `clear` | + +## Where SPIR-V comes from + +Issue 3 feeds committed SPIR-V (or library-built smoke shaders) under `_ext.gpu.testing`. Later, `@Gpu` compilation will produce SPIR-V and call the same `create_entry` path. + +## What create_entry does not do + +- Allocate descriptor sets or pools +- Point bindings at a GpuPack +- Record dispatch +- Register the entry in the cache (caller must `add`) + +Those are separate steps on the launch timeline. + +## Relationship to the cache key + +The cache key is a string symbol (stable kernel name). Long term, content hashing of SPIR-V may be stored inside the entry for invalidation. Today the contract is: one symbol maps to one immutable entry for the process lifetime after `add`. diff --git a/docs/quickstart.md b/docs/quickstart.md index 9638210..e456d40 100644 --- a/docs/quickstart.md +++ b/docs/quickstart.md @@ -1,2 +1,17 @@ # Quickstart +## CPU (`@Thread`) + +See the package [README](../ReadMe.md) for a first `@Thread` / `thread(...)` / +`join` example, then [concepts.md](./concepts.md) for the pack / writeback model. + +## GPU (`@Gpu`, 0.2.0) + +Start here: [guide/gpu/quickstart.md](./guide/gpu/quickstart.md). + +Hub (concepts, best practices, examples, API): [guide/gpu/README.md](./guide/gpu/README.md). + +## Install + +[install.md](./install.md) covers PyPI, editable builds, compilers, and Vulkan +driver notes for GPU. diff --git a/docs/release.md b/docs/release.md index 697ab99..08ff70b 100644 --- a/docs/release.md +++ b/docs/release.md @@ -11,11 +11,39 @@ PyPI upload is **not** tied to merges into `main`. Broken PRs cannot publish. | Artifact | Built on | Notes | |----------|----------|--------| | Source dist (sdist) | Ubuntu | Users with a compiler can build `_ext` themselves | -| `manylinux` wheels | Ubuntu | Python 3.10–3.13, x86_64 | -| Windows wheels | `windows-latest` | x86_64; kernel DLL is still compiled on the user’s machine | +| `cthreads` wheels | Ubuntu + Windows | CPU `_ext` (`CTHREADS_GPU=OFF`), Python 3.10-3.13 x86_64 | +| `cthreads-gpu` wheels | Ubuntu + Windows | Same import path `cthreads`, `_ext` built with `CTHREADS_GPU=ON` | macOS wheels are skipped for now (CMake enables AVX2 on non-MSVC; Apple Silicon would fail). +### CPU vs GPU install + +Pip extras cannot swap binary contents of one project name. GPU builds are therefore a +**second PyPI project**: + +```bash +pip install cthreads # CPU wheel +pip install "cthreads[gpu]" # installs cthreads + cthreads-gpu (GPU _ext) +pip install cthreads-gpu # GPU wheel only (also provides import cthreads) +``` + +Keep `project.version` and the `gpu = ["cthreads-gpu==..."]` pin equal on every release +(the Release workflow checks this). + +### Trusted Publishing (one-time, both projects) + +Add a trusted publisher for **`cthreads`** and another for **`cthreads-gpu`**: + +| Field | Value | +|-------|--------| +| Owner | your GitHub user or org | +| Repository | `CThreads` | +| Workflow | `release.yml` | +| Environment | `pypi` or `testpypi` | + +Do this on both https://pypi.org and https://test.pypi.org. One `publish` job uploads +all wheels; each wheel goes to the project matching its metadata name. + ## One-time setup ### 1. GitHub environments @@ -41,18 +69,20 @@ No API token is stored in GitHub. PyPI trusts this repo + workflow. 1. Sign in at [https://test.pypi.org](https://test.pypi.org) 2. Account settings -> **Publishing** (or create the pending project) -3. Add a **trusted publisher**: +3. Add a **trusted publisher** for **`cthreads`**: - Owner: your GitHub user or org - Repository: `CThreads` (the repo name on GitHub) - Workflow: `release.yml` - Environment: `testpypi` +4. Add the same trusted publisher for **`cthreads-gpu`** **PyPI** (same fields, production): 1. Sign in at [https://pypi.org](https://pypi.org) -2. Add a trusted publisher with environment **`pypi`** and workflow **`release.yml`** +2. Add a trusted publisher with environment **`pypi`** and workflow **`release.yml`** for **`cthreads`** +3. Add the same trusted publisher again for the **`cthreads-gpu`** project -If the project name `cthreads` is not registered yet, use PyPI’s **pending publisher** / first-upload flow for that name. +If the project name `cthreads` / `cthreads-gpu` is not registered yet, use PyPI's **pending publisher** / first-upload flow for that name. Exact labels in the PyPI UI change occasionally; look for **Trusted publishers** / **Publishing**. @@ -75,13 +105,16 @@ Then `main` cannot merge red tests. ```bash python -m pip install -U pip python -m pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ cthreads +# GPU build: +python -m pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ "cthreads[gpu]" ``` `--extra-index-url` is only needed if TestPyPI cannot see some dependency (cthreads currently has none). ## Production release -1. Set `version` in `pyproject.toml` (must match the tag without the leading `v`). +1. Set `version` in `pyproject.toml` **and** the matching pin + `gpu = ["cthreads-gpu==X.Y.Z"]` under `[project.optional-dependencies]`. 2. GitHub -> **Releases -> Draft a new release**. 3. Tag `v0.1.0` (or whatever matches `project.version`). Target `main`. 4. Click **Publish release**. @@ -96,6 +129,8 @@ PyPI versions cannot be overwritten. If `0.1.0` is bad, yank it and ship `0.1.1` ```bash python -m pip install cthreads +# or with Vulkan GPU support built into _ext: +python -m pip install "cthreads[gpu]" ``` Windows 11 with Smart App Control on may still fail to **load** `cthreads_kernels.dll` (`LoadLibrary` 4551). That is a Windows policy, not a missing wheel. See [install.md](./install.md). @@ -104,6 +139,6 @@ Windows 11 with Smart App Control on may still fail to **load** `cthreads_kernel | File | Trigger | Publishes? | |------|---------|------------| -| `.github/workflows/ci.yml` | PR and push to `main` | No | -| `.github/workflows/release.yml` | GitHub Release published | PyPI | +| `.github/workflows/ci.yml` | PR and push to `main` | No (CPU tests + GPU-ON compile smoke) | +| `.github/workflows/release.yml` | GitHub Release published | PyPI (`cthreads` + `cthreads-gpu`) | | `.github/workflows/release.yml` | Actions -> Run workflow | TestPyPI only | diff --git a/docs/vk_guide/00-read-me-first.md b/docs/vk_guide/00-read-me-first.md new file mode 100644 index 0000000..0c24347 --- /dev/null +++ b/docs/vk_guide/00-read-me-first.md @@ -0,0 +1,93 @@ +# 00 — Read me first + +## What cthreads is building + +cthreads already turns typed Python (`@Thread`) into **CPU** native kernels: + +```text +Python args -> pack (C++ struct) -> run on OS thread -> writeback -> same Python objects +``` + +The GPU path follows the **same product idea** with a different middle: + +```text +Python args -> GpuPack (Vulkan buffers) -> run SPIR-V compute -> writeback -> same Python objects +``` + +Users still pass `list` / `int` / `float`. They do **not** learn a second buffer type. +Vulkan stays inside `_ext`. + +## What Vulkan is (intuition) + +Vulkan is a **low-level remote control for the GPU**. + +- The API does not hide a high-level "draw a triangle" call. +- Callers create memory, bind it, record commands, submit them to a queue, and wait until done. +- The driver does less magic. The application owns lifetimes, synchronization, and layouts. + +That is why Vulkan feels verbose. Every step that OpenGL/CUDA hid becomes a function call. +The upside: portable compute on NVIDIA / AMD / Intel with one API, and predictable behavior. + +## What this project deliberately ignores + +These topics are **out of scope** for the cthreads compute backend: + +| Topic | Why skip | +|-------|----------| +| Swapchain / windows / present | Not drawing to the screen | +| Render passes / framebuffers | Graphics only | +| Images / textures / samplers | Buffer-first compute path | +| Graphics pipelines / vertex input | Compute only | +| Multi-GPU / sparse memory | Out of scope for now | +| Buffer device address / bindless | Rejected for the current design | +| Ray tracing | Out of scope | + +If a random tutorial spends chapters on "hello triangle," skim or skip those sections. +The cthreads "hello world" is: **upload floats, run compute (when landed), download floats**. + +## How Vulkan compares to things contributors might know + +### vs writing normal C++ + +| C++ | Vulkan | +|-----|--------| +| `new[]` / `malloc` | `vkAllocateMemory` + choose memory type | +| `memcpy` to a buffer the CPU owns | `memcpy` only works if memory is **host-visible** | +| Call a function | Record commands, then **submit** to a queue (async GPU) | +| Function returns when done | GPU may still be running; **fence** or similar to wait | + +### vs CUDA (if familiar) + +| CUDA | cthreads Vulkan path | +|------|----------------------| +| `cudaMalloc` | device-local `VkBuffer` + memory | +| `cudaMemcpy` | staging buffer + `vkCmdCopyBuffer` | +| `<<>>` kernel launch | `vkCmdDispatch` in a command buffer | +| CUDA toolkit for end users | **Drivers only** for end users; SDK headers for *building* GPU-enabled `_ext` | + +### vs OpenGL compute + +OpenGL hides a lot of binding and sync. Vulkan makes bind points and barriers explicit. +Same idea (SSBO = storage buffer), more paperwork. + +## The one sentence to remember + +**Vulkan objects are handles to driver resources; almost nothing happens until a command buffer is submitted to a queue; CPU and GPU run out of sync unless the CPU waits.** + +## Locked decisions in this project (so generic tutorials do not confuse contributors) + +1. **Binding convention pack:** one scalar SSBO + one SSBO per `list`. +2. **Device-local** data for shaders; **staging** for CPU copies. +3. **Descriptors** bind buffers by binding index (not pointers in the scalar struct). +4. **Launch then join** — no mid-run Python `__sync_state` on GPU. +5. **Dynamic load** `vulkan-1.dll` / `libvulkan.so.1` — do not hard-link for default CPU wheels. + +If a blog says "just map the SSBO and memcpy," that is a shortcut cthreads does **not** use as the production list path. + +## Suggested study rhythm + +1. Read 01-03 for intuition (no code required). +2. Read 04-05 while looking at `gpu/headers/context.hpp` and `gpu/impl/context.cpp`. +3. Read 06-08 while looking at `gpu/headers/memory.hpp` and `gpu/impl/memory.cpp`. +4. Read 09-12 before working on pipelines, descriptors, or `@Gpu` emit. +5. Keep 14 glossary open while coding. diff --git a/docs/vk_guide/01-cthreads-gpu-big-picture.md b/docs/vk_guide/01-cthreads-gpu-big-picture.md new file mode 100644 index 0000000..892d034 --- /dev/null +++ b/docs/vk_guide/01-cthreads-gpu-big-picture.md @@ -0,0 +1,115 @@ +# 01 — cthreads GPU big picture + +## The CPU path + +A typical typed kernel looks like: + +```python +@Thread +def add(n: int, x: list[float], y: list[float]) -> None: + i: int = 0 + while i < n: + y[i] = x[i] + y[i] + i = i + 1 + +job = thread(add, 4, [1,2,3,4], [10,10,10,10]).start() +job.join() +``` + +Roughly: + +1. **Marshal** copies Python values into a C++ **pack** (args struct). +2. A worker thread runs the compiled C++ body on that pack. +3. **Writeback** copies mutated pack fields into the **same** Python lists. + +Scalars live as fields in the pack. Lists live as native containers owned by / pointed from the pack. + +## The GPU path (same story, different storage) + +Same Python. Different backend: + +1. **GPU marshal** creates a **GpuPack**: + - one **device-local** buffer holding all scalars as a `std430` struct + - one **device-local** buffer per list (tight array of floats/ints) +2. Host bytes go into those buffers via **staging + GPU copy** (upload). +3. A **compute shader** reads/writes those buffers through **descriptor bindings**. +4. After the GPU finishes (`join`): **download** + writeback into the same Python objects. + +```text + CPU world GPU world + ----------------- -------------------- +Python list[float] --upload--> staging --copy--> device-local SSBO +Python int/float --upload--> staging --copy--> scalar SSBO (struct) + | + compute shader + | +Python list[float] <--writeback-- staging <--copy-- device-local SSBO +``` + +## Why not one giant buffer for everything? + +cthreads uses this binding convention: + +| Piece | Where it lives | +|-------|----------------| +| Scalars (`n`, `a`, …) | One small SSBO interpreted as a C-like struct | +| Each `list[...]` | Its own SSBO | + +Reasons: + +- Matches how compute shaders usually declare `buffer` blocks (one binding per array). +- Easy to reuse the same list buffer across multiple dispatches in a batch. +- Layout for arrays stays simple (tight packed floats). +- Scalar writeback is one download of a small blob. + +The design **rejects** putting GPU pointers inside the scalar struct (that needs buffer device address). Bindings connect the shader to buffers instead. + +## Job contract (important product rule) + +On CPU, `__sync_state` can mirror pack -> Python mid-run. + +On the GPU path: **only launch then wait**. + +- No observing Python lists while the shader runs. +- Results appear after `GpuJob.join()` (fence wait + download + writeback). + +That matches how GPUs want to work: record work, submit, wait once. + +## Layers of the GPU stack + +| Layer | Responsibility | +|-------|----------------| +| Context | Talk to Vulkan: instance, device, queue, entry points | +| Memory helpers | Allocate buffers, staging upload/download | +| GpuPack | Scalar SSBO + per-list SSBOs; marshal/writeback | +| Launch path | Descriptors, pipeline, dispatch, `GpuJob.join` | +| `@Gpu` / `gpu()` | Compile and run user kernels (same typing model as CPU) | +| Workloads / packaging | Real numeric steps, docs, CI, capability gates | + +## Mental link: pack vs GpuPack + +| CPU pack | GpuPack | +|----------|---------| +| One C++ struct in process memory | Several `VkBuffer`s on the device | +| Field `int n` | Bytes at offset 0 in scalar SSBO | +| Field `vector x` | Separate SSBO whose bytes are the floats | +| Kernel gets C++ references | Shader gets bindings 0, 1, 2, … | + +Same **roles**. Different **implementation**. + +## What "done" looks like for a library user + +```python +from cthreads import Gpu, gpu + +@Gpu +def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + ... + +x = [1.0, 2.0, 3.0, 4.0] +y = [10.0, 10.0, 10.0, 10.0] +gpu(saxpy, 4, 2.0, x, y).join() +# y is updated in place, like a CPU Thread job +``` + +No `DeviceBuffer` in that story. Vulkan is an implementation detail of `_ext`. diff --git a/docs/vk_guide/02-mental-model.md b/docs/vk_guide/02-mental-model.md new file mode 100644 index 0000000..0de4d2c --- /dev/null +++ b/docs/vk_guide/02-mental-model.md @@ -0,0 +1,143 @@ +# 02 — Mental model: objects, handles, and "nothing runs until submit" + +## Vulkan is an object graph + +Almost everything you create is an **object**: + +- Instance +- Physical device (the GPU hardware the loader knows about) +- Logical device (your open connection to that GPU) +- Queue (a port you submit work to) +- Buffer (a typed "array of bytes" object) +- Device memory (the actual allocated slab) +- Command pool / command buffer (recorded GPU instructions) +- Fence (CPU-side "GPU finished" signal) +- Shader module, pipeline, descriptor set (launch path) + +Objects form a tree of ownership. Typical rule: + +**Destroy children before parents.** Example: destroy buffers and command pools before destroying the logical device; destroy the device before the instance. + +## Handles are not pointers you dereference + +When Vulkan returns a `VkBuffer`, you get an opaque **handle** (often an integer-sized id). + +You cannot do: + +```cpp +buffer->data[i] = 1.0f; // NOT how Vulkan works +``` + +You talk to objects only through API functions: + +```cpp +vkDestroyBuffer(device, buffer, nullptr); +``` + +`VK_NULL_HANDLE` means "no object" (like a null pointer, but for handles). + +In our code, `Context` stores many handles and many **function pointers** (`PFN_vkCreateBuffer`, …) because we load Vulkan dynamically. + +## Explicit means: you fill structs + +Vulkan APIs almost always take a `*CreateInfo` struct: + +```cpp +VkBufferCreateInfo info{}; +info.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO; +info.size = 1024; +info.usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT; +... +vkCreateBuffer(device, &info, nullptr, &buffer); +``` + +Rules of thumb: + +1. Zero the struct (`{}` in C++). +2. Set `sType` to the matching enum (tells the driver which struct this is). +3. Set only the fields you need; leave the rest zero/null. +4. Check the `VkResult` return (`VK_SUCCESS` or throw). + +This pattern repeats everywhere. Once you recognize it, Vulkan stops feeling like "a million APIs" and starts feeling like "the same paperwork for every object." + +## CPU time vs GPU time + +This is the biggest intuition jump. + +```text +CPU thread GPU +--------- --- +record commands into CB +submit CB to queue ------------> (maybe starts later) +do other CPU work execute copies / dispatches +vkWaitForFences <--------------- signals fence when done +``` + +If you forget to wait, you might download a buffer the GPU has not finished writing. +That is a classic bug class (CPU/GPU race). + +The upload/download helpers wait on a fence after the copy so the CPU memcpy of staging memory is safe. + +## Queues are submission ports + +A GPU exposes **queue families** (groups of queues with capabilities): + +- graphics +- compute +- transfer +- present (for windows — we ignore) + +cthreads picks a family that supports **compute** (and also uses it for buffer copies). +We create one logical device requesting that family, then call `vkGetDeviceQueue` to get `VkQueue queue`. + +All work we care about goes: **record -> vkQueueSubmit(queue, ...)**. + +## Two kinds of "memory" people confuse + +1. **Host memory** — normal RAM your C++ `memcpy` can touch. +2. **Device memory** — memory the GPU allocates through Vulkan (`VkDeviceMemory`). + +Some device memory is **host-visible**: the driver can give you a CPU pointer via `vkMapMemory` (staging). +Some is **device-local**: fast for the GPU, **not** for persistent CPU poking (our SSBOs). + +A `VkBuffer` alone is not enough. You must: + +1. Create the buffer object (describes size/usage). +2. Ask for memory requirements. +3. Allocate matching `VkDeviceMemory`. +4. **Bind** memory to the buffer. + +Until bind succeeds, the buffer cannot store data. + +## Layers of our stack (keep this map) + +```text +Python cthreads.gpu / future gpu() + | + v +pybind _ext.gpu + | + v +GpuPack (binding convention) + | + v +memory:: create/upload/download + | + v +Context (instance/device/queue/PFNs) + | + v +vulkan-1.dll / libvulkan.so.1 (loader) + | + v +GPU driver (ICD) +``` + +## Validation layers (optional while learning) + +The Vulkan SDK can enable **validation layers**: extra checks that print readable errors when you misuse the API. + +For shipping cthreads to users we do **not** require layers. +While developing, turning them on is highly recommended (see SDK docs for `VK_LAYER_KHRONOS_validation`). + +We may gate this behind something like `CTHREADS_VK_VALIDATE=1` later. Not required to understand Context and memory. diff --git a/docs/vk_guide/03-sdk-runtime-drivers.md b/docs/vk_guide/03-sdk-runtime-drivers.md new file mode 100644 index 0000000..bf6be77 --- /dev/null +++ b/docs/vk_guide/03-sdk-runtime-drivers.md @@ -0,0 +1,120 @@ +# 03 — SDK vs runtime vs drivers + +This chapter clears the most common confusion: **"Do end users need the Vulkan SDK?"** + +**Short answer for cthreads:** + +- **End users:** GPU drivers only (they typically already have `vulkan-1.dll` / `libvulkan.so.1`). +- **Contributors building with `CTHREADS_GPU=ON`:** Vulkan SDK (headers) + drivers. +- **Default CPU wheels do not hard-link** `vulkan-1.lib`. + +## Three different things + +### 1. GPU driver (required to *run* Vulkan apps) + +NVIDIA / AMD / Intel install a **Vulkan Installable Client Driver (ICD)** with the graphics driver. + +That is what makes games and compute apps find a GPU. + +Sanity checks on Windows: + +- Device Manager shows the GPU. +- `vulkaninfo` (from SDK tools) lists the GPU if the runtime works. + +### 2. Vulkan loader (`vulkan-1.dll` / `libvulkan.so.1`) + +The **loader** is a small library that: + +- finds ICDs +- exposes `vkGetInstanceProcAddr` +- dispatches calls to the right driver + +On Windows it is typically `vulkan-1.dll` (comes with the driver / runtime). +On Linux it is `libvulkan.so.1` (package like `vulkan-icd-loader` + vendor ICD). + +The Context bootstrap does: + +```text +LoadLibrary("vulkan-1.dll") / dlopen("libvulkan.so.1") +-> get vkGetInstanceProcAddr +-> resolve every other function by name +``` + +So the **process must find that DLL/SO at runtime**. Linking the import library at build time is optional; we chose **not** to require it for default CPU builds. + +### 3. Vulkan SDK (LunarG) — mainly for *developers* + +The SDK gives you: + +| Piece | Use | +|-------|-----| +| Headers (`vulkan/vulkan.h`) | Compile C++ that mentions `VkBuffer`, `PFN_vkCreateInstance`, … | +| `vulkaninfo` | Diagnose devices/extensions | +| Validation layers | Catch API misuse while developing | +| Shader tools (`glslangValidator`, etc.) | Compile GLSL -> SPIR-V (we may use shaderc in-library later) | + +**End users of cthreads do not install the SDK** just to `pip install` a CPU wheel. +**Contributors** install the SDK so `find_package(Vulkan)` can see headers when `CTHREADS_GPU=ON`. + +## How this shows up in our CMake + +When `CTHREADS_GPU=ON`: + +- CMake runs `find_package(Vulkan REQUIRED)` — needs SDK/headers on the machine building `_ext`. +- We compile `context.cpp`, `memory.cpp`, … +- We define `CTHREADS_WITH_GPU=1`. +- We still **do not** need to link `Vulkan::Vulkan` if we resolve symbols dynamically (our design). + +When `CTHREADS_GPU=OFF`: + +- GPU sources are not compiled. +- Python `cthreads.gpu` soft-fails (`available()` false / `VulkanNotBuiltError`). + +## Practical Windows setup (contributors) + +1. Install recent GPU drivers. +2. Install [LunarG Vulkan SDK](https://vulkan.lunarg.com/). +3. Open **x64 Native Tools** / VS Dev Cmd. +4. Build: + +```bat +set CMAKE_ARGS=-DCTHREADS_GPU=ON +pip install -e . -v +``` + +Look for CMake status: `cthreads GPU: ON (Vulkan)`. + +5. Smoke: + +```python +from cthreads import gpu +print(gpu.available(), gpu.device_name() if gpu.available() else None) +``` + +## What can go wrong + +| Symptom | Likely cause | +|---------|----------------| +| `VulkanLoaderNotFound` | No `vulkan-1.dll` on PATH / not installed with driver | +| `VulkanNoDevice` | Loader ok but no compute-capable ICD / bad driver | +| `missing vkDestroyInstance` style bugs | Resolved instance functions with null instance (we fixed this) | +| CMake cannot find Vulkan | SDK not installed or env not visible to the build | +| Tiny wheel / no GPU log | `CTHREADS_GPU` not actually ON (wrong shell env, or cached CMake) | + +## Mental picture + +```text +Your C++ (_ext) + | dynamic LoadLibrary + v +vulkan-1.dll (LOADER) + | finds + v +nvoglv64.dll / amdvlk / igvx64 ... (ICD / DRIVER) + | + v +GPU hardware +``` + +Headers from the SDK are only a **compile-time** description of that API. +They are not the DLL. diff --git a/docs/vk_guide/04-instance-device-queue.md b/docs/vk_guide/04-instance-device-queue.md new file mode 100644 index 0000000..94647f4 --- /dev/null +++ b/docs/vk_guide/04-instance-device-queue.md @@ -0,0 +1,145 @@ +# 04 — Instance, physical device, logical device, queue + +This is the "plug the GPU in" chapter. In cthreads this lives in `Context`. + +## The four names people mix up + +| Name | What it is | Analogy | +|------|------------|---------| +| **Loader** | DLL/SO that finds drivers | "Phone book + switchboard" | +| **Instance** (`VkInstance`) | Your app's connection to Vulkan | "Logged into the Vulkan system" | +| **Physical device** (`VkPhysicalDevice`) | One GPU the loader can see | "The hardware card" | +| **Logical device** (`VkDevice`) | Your open session on that GPU | "File handle / open connection" | +| **Queue** (`VkQueue`) | Submission port for work | "Inbox where GPU jobs go" | + +You always need: loader -> instance -> pick physical device -> create logical device -> get queue. + +## Step by step (what `init()` does) + +### Step A — Open the loader + +Windows: `LoadLibraryA("vulkan-1.dll")` +Linux: `dlopen("libvulkan.so.1", …)` + +Then get **one** symbol by name: `vkGetInstanceProcAddr`. + +Everything else is resolved through that function. + +### Step B — Resolve `vkCreateInstance` (global) + +Some functions can be queried with a **null instance**: + +- `vkCreateInstance` +- (and a few enumerate-instance helpers we do not need yet) + +```text +get_fn(NULL, "vkCreateInstance") +``` + +### Step C — Create the instance + +You fill `VkApplicationInfo` (app name, engine name, API version) and `VkInstanceCreateInfo`. + +We request **Vulkan 1.1** (`VK_API_VERSION_1_1`). That is enough for our compute path. + +Out parameter: `c.instance`. + +### Step D — Resolve instance-level functions + +**Important rule (easy to get wrong):** + +After you have an instance, resolve functions like: + +- `vkDestroyInstance` +- `vkEnumeratePhysicalDevices` +- `vkGetPhysicalDeviceProperties` +- `vkCreateDevice` +- … + +with **`c.instance`**, not `NULL`. + +If you pass `NULL`, many loaders return nullptr for those names. That is a common failure mode (`missing vkDestroyInstance` / similar). + +### Step E — Enumerate physical devices + +Vulkan uses a two-call pattern constantly: + +1. Call with `nullptr` data pointer to get **count**. +2. Allocate a vector of that size. +3. Call again to fill the vector. + +```text +enumerate(instance, &count, nullptr) +vector devices(count) +enumerate(instance, &count, devices.data()) +``` + +### Step F — Pick a device + queue family + +For each physical device: + +1. Read properties (name, discrete vs integrated). +2. Read queue family properties. +3. Find a family whose flags include `VK_QUEUE_COMPUTE_BIT`. +4. Prefer `VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU` (score higher). + +Store: + +- `physical_device` +- `queue_family` (index) +- `device_name` (string for Python) + +### Step G — Create the logical device + +You request queues: + +```text +VkDeviceQueueCreateInfo: family index = queue_family, count = 1, priority = 1.0 +VkDeviceCreateInfo: those queue infos +vkCreateDevice(physical_device, ...) +``` + +Out: `c.device`. + +### Step H — Get the queue handle + +```text +vkGetDeviceQueue(device, queue_family, 0, &queue) +``` + +Queue index `0` means "first queue of that family." + +### Step I — Mark ready + +`c.ready = true`. + +On any failure: destroy what you created and unload the loader (`shutdown_unlocked`). + +## Thread safety in our Context + +`init` / `shutdown` take a mutex. Double-checked locking: + +- Fast path: if `ready`, return. +- Lock, check again, then initialize once. + +`available()` tries `init()` and returns false on failure (no throw). +`device_name()` tries `init()` and may throw mapped errors. + +## Shutdown order + +1. Destroy logical device (also invalidates queues). +2. Destroy instance. +3. Clear physical device handle. +4. `FreeLibrary` / `dlclose` the loader. +5. Null all function pointers so a late call cannot jump into unloaded code. + +Command pools, fences, and buffers must be destroyed **before** destroying the logical device. + +## What the Context layer does *not* do + +- No buffers +- No command pools +- No shaders +- No descriptors + +It only proves: **we can talk to a compute-capable GPU and print its name.** diff --git a/docs/vk_guide/05-dynamic-loading.md b/docs/vk_guide/05-dynamic-loading.md new file mode 100644 index 0000000..593c524 --- /dev/null +++ b/docs/vk_guide/05-dynamic-loading.md @@ -0,0 +1,84 @@ +# 05 — Dynamic loading and `PFN_*` function pointers + +## Why we do not link `vulkan-1.lib` by default + +If you link the Vulkan import library: + +- Every machine that imports `_ext` needs the loader present **or** load fails at process start. +- CPU-only users would pay a Vulkan dependency they do not need. + +cthreads design: + +- Default build: no Vulkan. +- `CTHREADS_GPU=ON`: compile GPU code against **headers only**, resolve symbols at runtime. + +## What a `PFN_` is + +Vulkan headers define typedefs like: + +```cpp +typedef VkResult (VKAPI_PTR *PFN_vkCreateBuffer)(VkDevice, const VkBufferCreateInfo*, ...); +``` + +So `PFN_vkCreateBuffer` means: "pointer to a function with that signature." + +In `Context` we store: + +```cpp +PFN_vkCreateBuffer vkCreateBuffer = nullptr; +``` + +After init: + +```cpp +context.vkCreateBuffer(device, &info, nullptr, &buffer); +``` + +That is a normal indirect call through a function pointer. + +## The bootstrap chain + +```text +1. LoadLibrary / dlopen +2. GetProcAddress / dlsym("vkGetInstanceProcAddr") +3. vkGetInstanceProcAddr(instance_or_null, "vkCreateInstance") +4. create instance +5. vkGetInstanceProcAddr(real_instance, "vkCreateBuffer") // etc. +``` + +The Context helper `get_fn(context, instance, "name")`: + +1. Calls `vkGetInstanceProcAddr`. +2. Throws if null (`cthreads.gpu.VulkanInitFailed: missing …`). +3. Casts to the typed `PFN_*`. + +## Global vs instance vs device level (practical rules) + +You do not need the full spec table. Use these project rules: + +1. **Before instance exists:** only resolve truly global entry points (`vkCreateInstance`, …) with `VK_NULL_HANDLE`. +2. **After instance exists:** resolve everything else we need with `c.instance`. +3. **After device exists:** we still resolve device commands via `vkGetInstanceProcAddr(instance, name)` (loader trampoline). That matches our code. (Advanced apps often use `vkGetDeviceProcAddr`; not required for our path.) + +## Why so many pointers on `Context`? + +Buffer, command, and fence functions are loaded once during Context init and cleared on shutdown. + +Groups in `context.hpp`: + +1. Instance / device bootstrap +2. Buffer + memory +3. Command pool / command buffer / copy / fence / submit + +When adding a new Vulkan call in `memory.cpp`, first check: is its `PFN_` on `Context` and loaded in `create_instance_and_device`? If not, add it there. + +## Errors contributors will see + +| Message prefix | Meaning | +|----------------|---------| +| `VulkanLoaderNotFound` | DLL/SO missing | +| `VulkanInitFailed: missing X` | `get_fn` returned null for name `X` | +| `VulkanNoDevice` | No compute-capable GPU | +| `VulkanInitFailed: vkCreate… failed` | Driver rejected create | + +Python maps these string prefixes to exception types in `cthreads.gpu.frontend.errors`. diff --git a/docs/vk_guide/06-buffers-and-memory.md b/docs/vk_guide/06-buffers-and-memory.md new file mode 100644 index 0000000..d1c0321 --- /dev/null +++ b/docs/vk_guide/06-buffers-and-memory.md @@ -0,0 +1,142 @@ +# 06 — Buffers and device memory + +This is the core of the memory layer (before copies). + +## Two objects, one job + +To store N bytes the GPU can use, you need **both**: + +1. **`VkBuffer`** — describes a buffer resource: size, usage flags, sharing mode. +2. **`VkDeviceMemory`** — the actual allocated memory slab. + +Then you **bind** them: + +```text +vkBindBufferMemory(device, buffer, memory, offset=0) +``` + +Until bind succeeds, the buffer is an empty shell. + +In our API this pair is `GpuBuffer`: + +```text +GpuBuffer { + buffer, memory, size, mapped, kind +} +``` + +## Usage flags (what the buffer is allowed to do) + +When creating a buffer you set `VkBufferUsageFlags`. Think of them as permissions. + +| Flag | Meaning for us | +|------|----------------| +| `TRANSFER_SRC` | May be source of a GPU copy | +| `TRANSFER_DST` | May be destination of a GPU copy | +| `STORAGE_BUFFER` | May be bound as an SSBO for compute shaders | + +cthreads buffer kinds: + +**Staging** + +```text +TRANSFER_SRC | TRANSFER_DST +``` + +**DeviceLocal** (shader data) + +```text +STORAGE_BUFFER | TRANSFER_SRC | TRANSFER_DST +``` + +Why transfer bits on device-local? Because upload/download copy through them. + +## Memory properties (where the bytes live) + +After `vkCreateBuffer`, call `vkGetBufferMemoryRequirements`: + +- `size` — how many bytes to allocate (may be larger than you asked; alignment) +- `alignment` +- `memoryTypeBits` — bitmask of legal memory type indices + +Then look at the physical device's memory types (`vkGetPhysicalDeviceMemoryProperties`). + +Each type has flags like: + +| Flag | Meaning | +|------|---------| +| `HOST_VISIBLE` | CPU can map it and read/write | +| `HOST_COHERENT` | No manual flush/invalidate needed for visibility | +| `DEVICE_LOCAL` | Lives in GPU-friendly memory (often VRAM) | + +### `find_memory_type` in cthreads + +```text +for i in 0 .. memoryTypeCount-1: + if bit i not set in type_bits: skip + if (type.flags & requested) == requested: return i +throw if none +``` + +Requested properties: + +- Staging: `HOST_VISIBLE | HOST_COHERENT` +- DeviceLocal: `DEVICE_LOCAL` + +## Mapping (CPU pointer into device memory) + +`vkMapMemory` returns a `void*` the CPU can `memcpy` into. + +**Only do this for host-visible memory.** + +cthreads rules: + +- Staging: map once at create, keep `GpuBuffer.mapped` until destroy. +- DeviceLocal: **never** persistently map. Upload goes through staging. + +Unmap with `vkUnmapMemory` before freeing memory. + +## Create algorithm (what `create_buffer` does) + +```text +1. Validate context.ready, size > 0 +2. Choose usage + memory property flags from BufferKind +3. vkCreateBuffer +4. vkGetBufferMemoryRequirements +5. find_memory_type(...) +6. vkAllocateMemory (allocationSize = mem_reqs.size) +7. vkBindBufferMemory(..., offset 0) +8. If Staging: vkMapMemory -> mapped +9. Store caller size in GpuBuffer.size +``` + +On failure: destroy whatever was created (buffer/memory) before throwing. + +## Destroy algorithm + +```text +if mapped: unmap +destroy buffer +free memory +zero the GpuBuffer fields +``` + +Empty buffer (already null handles): no-op. + +## Why size 0 is rejected + +Vulkan implementations often dislike zero-sized buffers. +Empty Python lists are handled by **GpuPack** (no buffer / skip), not by creating a 0-byte `VkBuffer`. + +## Sharing mode + +cthreads uses `VK_SHARING_MODE_EXCLUSIVE`: one queue family owns the buffer. +We only use one compute/transfer family, so exclusive is correct and simpler. + +## Common mistakes + +1. Creating a buffer and forgetting to allocate/bind memory. +2. Mapping device-local memory that is not host-visible. +3. Using `GpuBuffer.size` wrong vs `mem_reqs.size` (we store the **caller** size; allocation may be larger — copies use caller size). +4. Destroying the device while buffers still exist. +5. Assuming `list` memory in Python is the GPU buffer — it is not; marshal must copy. diff --git a/docs/vk_guide/07-staging-upload-download.md b/docs/vk_guide/07-staging-upload-download.md new file mode 100644 index 0000000..cc4235a --- /dev/null +++ b/docs/vk_guide/07-staging-upload-download.md @@ -0,0 +1,96 @@ +# 07 — Staging, upload, and download (final memory model) + +## The problem + +Shaders want **device-local** buffers (fast). +The CPU wants to `memcpy` from Python lists (host RAM). + +Those are often **different** memory heaps. You cannot always map the SSBO and pretend it is a C array. + +## The solution in cthreads + +```text +UPLOAD: + host pointer + --memcpy--> staging (host-visible, mapped) + --GPU copy--> device-local SSBO + +DOWNLOAD: + device-local SSBO + --GPU copy--> staging (mapped) + --memcpy--> host pointer +``` + +This is the permanent design for list/scalar marshal traffic. + +## What `upload_buffer` does (in `memory.cpp` today) + +Given a **device-local** `GpuBuffer`: + +1. Check kind, null data, size bounds. +2. `create_buffer(..., Staging)` of `size` bytes. +3. `memcpy(staging.mapped, data, size)`. +4. Record/submit `vkCmdCopyBuffer(staging -> device)` and wait on a fence. +5. `destroy_buffer(staging)`. + +`download_buffer` reverses the copy direction, then `memcpy` out of staging. + +## One-shot command pool vs TransferEngine + +In the current memory implementation, each upload/download creates a temporary: + +- command pool +- command buffer +- fence + +then destroys them after wait. + +That is **correct** and matches the final *API*. + +A later **TransferEngine** on `Context` can reuse one pool/fence/scratch staging for speed. +That is an optimization of the same path, not a new memory model. + +## Barriers (intuition for later) + +GPUs reorder work. Sometimes you need a **pipeline barrier** so "copy finished" happens before "shader reads." + +For CPU-side round-trips (copy then CPU wait then CPU read), a fence wait is enough because the CPU does not read device memory until the GPU signaled. + +When a shader runs in the same command buffer after a copy, insert a barrier between copy and dispatch. See the pipelines chapter. + +## Why not host-visible SSBOs only? + +It works on some integrated GPUs and for tiny demos. +It is the wrong long-term model for: + +- discrete GPUs (VRAM vs sysmem) +- large lists +- matching how production Vulkan compute is written + +cthreads does not ship a "map the SSBO forever" production path. + +## End-to-end round-trip (what tests will prove) + +```text +floats = [1, 2, 3, 4] +create device-local buffer (16 bytes) +upload_buffer(buf, floats) +download_buffer(buf, out) +assert out == floats +destroy +``` + +Same idea for a small scalar struct blob. + +## How this becomes GpuPack + +```text +GpuPack: + scalars: GpuBuffer DeviceLocal # bytes of std430 struct + lists[]: GpuBuffer DeviceLocal # one per list arg + +upload pack: upload each piece +download pack: download each piece into Python objects +``` + +Staging is an implementation detail inside `upload_buffer` / `download_buffer` (or the transfer engine). Users never see it. diff --git a/docs/vk_guide/08-commands-fences.md b/docs/vk_guide/08-commands-fences.md new file mode 100644 index 0000000..fe4833e --- /dev/null +++ b/docs/vk_guide/08-commands-fences.md @@ -0,0 +1,114 @@ +# 08 — Command buffers, queues, and fences + +## Why command buffers exist + +The GPU does not execute your C++ line by line. +You **record** a list of GPU commands, then **submit** that list to a queue. + +Think of a command buffer as a recipe card: + +```text +Begin + CopyBuffer A -> B + (later) Bind pipeline + (later) Dispatch +End +Submit to queue +``` + +Recording is cheap CPU work. Executing happens on the GPU, possibly later. + +## Objects involved + +| Object | Role | +|--------|------| +| `VkCommandPool` | Allocator/owner of command buffers for one queue family | +| `VkCommandBuffer` | The recorded recipe | +| `VkQueue` | Where recipes are submitted | +| `VkFence` | CPU-waitable "this submit finished" flag | + +## Record / submit / wait pattern (copy) + +This is exactly what our `copy_buffer_and_wait` helper does: + +```text +1. vkCreateCommandPool (family = context.queue_family) +2. vkAllocateCommandBuffers (one primary buffer) +3. vkBeginCommandBuffer (ONE_TIME_SUBMIT) +4. vkCmdCopyBuffer(src, dst, region) +5. vkEndCommandBuffer +6. vkCreateFence +7. vkQueueSubmit(queue, cmd, fence) +8. vkWaitForFences(..., UINT64_MAX) +9. destroy fence, free command buffer, destroy pool +``` + +### Primary vs secondary + +cthreads uses **primary** command buffers (can be submitted directly). +Secondary buffers are for advanced recording reuse; ignore for now. + +### One-time submit + +`VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT` means: we will record, submit once, then throw away / reset. +Perfect for copies. + +## Fences vs semaphores (only fences matter for us now) + +| Primitive | Who waits | Typical use | +|-----------|-----------|-------------| +| **Fence** | CPU | `join`, wait for copy before memcpy | +| **Semaphore** | GPU | Sync between queue submits / present (graphics) | + +cthreads launch/wait uses **fences**. +Semaphores can be ignored until multi-queue or swapchains (not part of this compute path). + +## Reset and reuse (launch hot-path) + +Instead of destroy/recreate every time: + +- Keep a pool with `RESET` flag or reset individual buffers. +- `vkResetFences` before reuse. +- Re-record and submit again. + +Reusing pools and cached pipelines is how launches stay cheap. + +## Queue submit is asynchronous + +After `vkQueueSubmit` returns `VK_SUCCESS`, the GPU might still be working. +Only after the fence signals is it safe to: + +- destroy the buffers you copied from/to (if you are done with them) +- `memcpy` from staging after a download copy +- tell Python the job finished + +`GpuJob.join()` waits on the fence for the dispatch submit. + +## Ordering: destroy vs in-flight work + +Never destroy a buffer or pool while the GPU still has commands referencing it. +Wait for fences first, then destroy. + +The upload helper waits, then destroys staging — correct order. + +## Transient pools + +`VK_COMMAND_POOL_CREATE_TRANSIENT_BIT` hints that command buffers are short-lived. +cthreads uses that for one-shot copies. + +## How this connects to compute dispatch (preview) + +A full kernel launch command buffer looks like: + +```text +Begin + (optional) barriers after uploads + Bind pipeline + Bind descriptor set (buffers) + Push constants (optional) + vkCmdDispatch(groups_x, groups_y, groups_z) +End +Submit + fence +``` + +Same machinery as copy; different commands inside. diff --git a/docs/vk_guide/09-descriptors-ssbo.md b/docs/vk_guide/09-descriptors-ssbo.md new file mode 100644 index 0000000..468a859 --- /dev/null +++ b/docs/vk_guide/09-descriptors-ssbo.md @@ -0,0 +1,100 @@ +# 09 — Descriptors and SSBOs (how shaders see buffers) + +The memory layer creates buffers. The launch path lets shaders read them. This chapter is the bridge. + +## The problem descriptors solve + +You have device-local buffers on the CPU/C++ side (`VkBuffer` handles). +A shader is SPIR-V running on the GPU. It cannot see your C++ variables. + +You need a **binding table**: "shader binding 0 is this buffer, binding 1 is that buffer." + +That table is built with: + +- Descriptor set layout (the schema: binding i is a storage buffer) +- Descriptor pool + descriptor set (an instance of that schema) +- Descriptor writes (`vkUpdateDescriptorSets`) pointing at actual `VkBuffer`s +- `vkCmdBindDescriptorSets` before dispatch + +## SSBO = storage buffer + +In GLSL compute: + +```glsl +layout(set = 0, binding = 1, std430) buffer XBlock { + float data[]; +} x; +``` + +That means: + +- descriptor set 0 +- binding 1 +- storage buffer +- std430 packing +- unsized array of floats at the end of the block + +On the C++ side, binding 1's descriptor must reference the `VkBuffer` that holds those floats. + +## cthreads binding convention + +| Binding | Contents | +|---------|----------| +| 0 | Scalar SSBO (`std430` struct: `n`, `a`, …) | +| 1 | First list SSBO | +| 2 | Second list SSBO | +| … | More lists | + +Saxpy example: + +```glsl +layout(set=0, binding=0, std430) buffer Scalars { int n; float a; } scalars; +layout(set=0, binding=1, std430) buffer X { float data[]; } x; +layout(set=0, binding=2, std430) buffer Y { float data[]; } y; +``` + +No pointers inside `Scalars`. Bindings do the wiring. + +## Descriptor types we care about + +| Vulkan type | GLSL idea | +|-------------|-----------| +| `STORAGE_BUFFER` | `buffer { ... }` (read/write) | +| `UNIFORM_BUFFER` | `uniform` block (we avoid for mutable pack) | + +cthreads uses **storage buffers for scalars and lists** (binding convention = all SSBO). + +## Lifecycle sketch (launch path) + +```text +1. Create descriptor set layout (bindings 0..N as STORAGE_BUFFER) +2. Create pipeline layout (includes that set layout) +3. Create compute pipeline (shader module + layout) +4. Create descriptor pool; allocate one set +5. For each launch: + write descriptors to point at this GpuPack's buffers + record: bind pipeline, bind set, dispatch +6. On shutdown: destroy sets/pool/pipeline/layout/module (order matters) +``` + +## Why update descriptors per launch? + +Because each job has its own GpuPack buffers (or at least different list buffers). +The layout stays cached; the **buffer handles** inside the set change. + +CB reuse and pipeline caching optimize recording/submit; the descriptor model stays. + +## Ranges + +When writing a descriptor for a buffer you specify offset + range (often `VK_WHOLE_SIZE` or `buffer.size`). +For our tightly packed arrays, whole buffer / exact size both work if the buffer was created for that array only. + +## Mental model + +```text +Shader source says: binding 1 is float array x +Descriptor set says: binding 1 -> VkBuffer handle of list x +Dispatch runs: loads/stores hit that memory +``` + +If descriptors are wrong, you get garbage or crashes — validation layers help. diff --git a/docs/vk_guide/10-std430-layouts.md b/docs/vk_guide/10-std430-layouts.md new file mode 100644 index 0000000..d031190 --- /dev/null +++ b/docs/vk_guide/10-std430-layouts.md @@ -0,0 +1,93 @@ +# 10 — `std430` layouts (scalar structs that match C++) + +## Why layout rules exist + +The CPU uploads a blob of bytes for scalars. +The shader interprets those bytes as a struct. + +If padding differs, `n` and `a` get scrambled. Silent corruption — nasty to debug. + +GLSL storage buffers default to **`std430`** packing when you write: + +```glsl +layout(..., std430) buffer Scalars { ... }; +``` + +Your host C++ struct must match **std430**, not "whatever the C++ compiler naturally did" unless you carefully match. + +## Rules of thumb (std430) + +For members we use in v1: + +| Type | Size | Alignment | +|------|------|-----------| +| `int` / `uint` / `float` | 4 | 4 | +| `bool` in GLSL | treat carefully; prefer `int` as 0/1 on host | 4 | +| `vec2` | 8 | 8 | +| `vec3` | 12 | **16** (align like vec4) | +| `vec4` | 16 | 16 | + +Arrays of `float`/`int` in an SSBO are tightly packed (stride 4). + +Structs get padding so each member starts at a multiple of its alignment. +The struct's overall alignment is the max of its members (roughly). + +## Easy case (saxpy scalars) + +GLSL: + +```glsl +layout(set=0, binding=0, std430) buffer Scalars { + int n; + float a; +} scalars; +``` + +Host: + +```cpp +struct ScalarPack { + int32_t n; // offset 0 + float a; // offset 4 +}; +// sizeof == 8, no surprise padding +``` + +Use fixed-width types (`int32_t`) so Windows LLP64 and Linux agree. + +## Hard case (why vec3 bites) + +```glsl +float a; +vec3 v; +``` + +`v` must start at offset 16, not 4. Bytes 4..15 are padding. + +Host must insert the same padding or use `alignas`. + +**For the current cthreads GPU path:** prefer scalars that are int/float/bool-as-int. Avoid vec3 in the scalar blob until a layout emitter matches std430 automatically. + +## Lists are not inside the scalar struct + +Do not put `float x[];` inside the scalar block when you also need `y[]`. +Unsized arrays in std430 must be last, and you only get one. + +That is another reason cthreads uses **separate list SSBOs**. + +## How codegen will help later + +The `@Gpu` emitter should emit: + +1. GLSL `std430` scalar block from kernel meta field order. +2. Matching host mirror struct / byte offsets for marshal. + +Until then, hand-written smoke tests use a tiny fixed struct both sides agree on. + +## Checklist when adding a scalar field + +1. Append field to GLSL block in the same order. +2. Append field to host struct with matching type width. +3. Recompute offsets; watch for padding. +4. Bump any cached shader hash (old SPIR-V would mismatch). +5. Upload `sizeof(host_struct)` bytes (include trailing padding if required by std430). diff --git a/docs/vk_guide/11-spirv-pipelines-dispatch.md b/docs/vk_guide/11-spirv-pipelines-dispatch.md new file mode 100644 index 0000000..320514d --- /dev/null +++ b/docs/vk_guide/11-spirv-pipelines-dispatch.md @@ -0,0 +1,121 @@ +# 11 — SPIR-V, pipelines, and compute dispatch + +## Shader languages in our stack + +| Form | Role | +|------|------| +| GLSL compute (human-written or emitted) | Source you can read | +| SPIR-V | Binary IR Vulkan drivers consume | +| `VkShaderModule` | Vulkan object wrapping SPIR-V bytes | +| `VkPipeline` (compute) | Compiled shader + layout ready to bind | + +Flow: + +```text +GLSL -> (shaderc / glslang at library build or runtime) -> SPIR-V bytes + -> vkCreateShaderModule + -> vkCreateComputePipelines +``` + +Users never run `glslangValidator` themselves. Either: + +- we commit `.spv` files, or +- we compile with shaderc inside `_ext` when `CTHREADS_GPU=ON`. + +## Compute shader shape (intuition) + +```glsl +#version 450 +layout(local_size_x = 64) in; // workgroup size + +layout(set=0, binding=0, std430) buffer Scalars { int n; float a; } scalars; +layout(set=0, binding=1, std430) buffer X { float data[]; } x; +layout(set=0, binding=2, std430) buffer Y { float data[]; } y; + +void main() { + uint i = gl_GlobalInvocationID.x; + if (i >= uint(scalars.n)) return; + y.data[i] = scalars.a * x.data[i] + y.data[i]; +} +``` + +### Workgroups and local size + +- `local_size_x = 64` means each workgroup runs 64 invocations. +- `vkCmdDispatch(groupCountX, 1, 1)` launches that many workgroups. +- Global id roughly: `groupIndex * local_size + localIndex`. + +For `n` elements: + +```text +groups = ceil(n / 64) +vkCmdDispatch(groups, 1, 1) +``` + +cthreads may also support serial-looking `@Gpu` bodies later (`local_size=1`). The product API stays launch/wait; the emitter chooses a parallelism model and documents it. + +## Pipeline layout + +Connects: + +- descriptor set layouts (which bindings exist) +- push constant ranges (optional) + +Even if you use a scalar SSBO, you might later mirror hot scalars into push constants for speed. Writeback truth remains the scalar SSBO. + +## Creating a compute pipeline (conceptual) + +```text +VkShaderModuleCreateInfo <- SPIR-V code +vkCreateShaderModule + +VkPipelineShaderStageCreateInfo stage: COMPUTE, module, entry "main" +VkComputePipelineCreateInfo: stage + pipelineLayout +vkCreateComputePipelines +``` + +Pipelines should be cached by shader hash so launches do not rebuild every time. + +## Dispatch command buffer (full story) + +```text +Begin command buffer + // ensure uploads visible to compute (barrier) when needed + vkCmdBindPipeline(COMPUTE, pipeline) + vkCmdBindDescriptorSets(... pack's set ...) + vkCmdDispatch(groupsX, 1, 1) +End +vkQueueSubmit(..., fence) +``` + +`GpuJob.join()`: + +```text +vkWaitForFences +download GpuPack -> Python writeback +``` + +## Barriers in one sentence + +A **pipeline barrier** tells the GPU: finish these earlier commands' memory effects before later commands use the data. + +Example: copy into `y` finished before compute reads/writes `y`. + +CPU-side round-trips can rely on fence waits. +In-GPU sequences need barriers between copy and dispatch (and between dispatches in a batch). + +## What to skip in graphics tutorials + +When reading vkguide / vulkan-tutorial, skip: + +- swapchain +- render pass +- framebuffer +- vertex buffers for drawing + +Keep: + +- buffer creation +- command buffers +- descriptors +- compute pipeline / dispatch (some tutorials bury this) diff --git a/docs/vk_guide/12-gpupack-marshal.md b/docs/vk_guide/12-gpupack-marshal.md new file mode 100644 index 0000000..7c2c1a2 --- /dev/null +++ b/docs/vk_guide/12-gpupack-marshal.md @@ -0,0 +1,90 @@ +# 12 - GpuPack and marshal (binding convention end-to-end) + +## Definition + +**GpuPack** is the GPU analogue of the CPU args pack for one launch: + +```text +GpuPack { + GpuBuffer scalars; // DeviceLocal, std430 blob + vector lists; // DeviceLocal, one per list arg +} +``` + +Optional host-side metadata: + +- dtype per list (float vs int) +- element counts +- pointer/identity of Python objects for writeback + +## Build from Python args (marshal / smoke) + +Example call: + +```text +n: int = 4 +a: float = 2.0 +x: list[float] len 4 +y: list[float] len 4 +``` + +Steps: + +1. Build host `ScalarPack { n, a }` bytes. +2. `scalars = create_buffer(sizeof(ScalarPack), DeviceLocal)`. +3. `upload_buffer(scalars, &host_scalars, sizeof)`. +4. For `x`: create device-local buffer `4*sizeof(float)`; upload from a temporary C array filled from the Python list (or from a contiguous buffer you built while iterating). +5. Same for `y`. +6. Store Python object references for writeback. + +Empty list rule: **do not** create a 0-size VkBuffer; store "no buffer / count 0". + +## After compute (join) + +1. Wait fence. +2. `download_buffer` scalars -> host struct -> write Python ints/floats if they were mutable outputs (usually scalars are inputs; returns go in scalar blob too). +3. `download_buffer` each list -> update the **same** Python list objects in place (like CPU writeback). +4. Destroy pack buffers when the job is done (or retain if you design persistent buffers later — v1 can destroy per job). + +## Comparison table + +| Concern | CPU | GPU | +|---------|-----|-----| +| Where scalars live | Fields in C++ pack | Scalar SSBO bytes | +| Where lists live | `vector` / container in pack | Per-list SSBO | +| Fill | `pack_params` trampolines | `upload_buffer` | +| Run | call C++ function | dispatch SPIR-V | +| Sync mid-run | optional `__sync_state` | **not in v1** | +| Finish | writeback | download + writeback | + +## What marshal must guarantee + +1. Host scalar blob matches shader `std430` block. +2. List buffers contain tightly packed elements of the declared dtype. +3. Descriptor bindings match the emitter's binding assignment. +4. Download size matches what was uploaded (or the mutated full buffer). + +## Internal test API vs public API + +Tests may use an internal helper such as `_ext.gpu.roundtrip_lists(...)`. +That is **not** a user-facing `DeviceBuffer`. + +Public surface stays: + +```python +from cthreads import gpu +gpu.available() +# later: gpu(fn, *args).join() +``` + +## Lifetime vs Context + +```text +Context process lifetime (init/shutdown) +TransferEngine process lifetime (optional reuse) +GpuPack per job (or per batch) +GpuBuffer owned by pack / staging temps +``` + +Destroy packs before shutting down Context. +Destroy transfer engine resources before destroying the device. diff --git a/docs/vk_guide/13-map-to-our-code.md b/docs/vk_guide/13-map-to-our-code.md new file mode 100644 index 0000000..ffe83dd --- /dev/null +++ b/docs/vk_guide/13-map-to-our-code.md @@ -0,0 +1,72 @@ +# 13 — Map this guide to our repository + +## Files (GPU) + +| Path | Role | +|------|------| +| `src/cthreads/cpp/gpu/headers/context.hpp` | `Context` handles + all `PFN_*` | +| `src/cthreads/cpp/gpu/impl/context.cpp` | Load loader, init/shutdown, resolve entry points | +| `src/cthreads/cpp/gpu/headers/memory.hpp` | `BufferKind`, `GpuBuffer`, memory API | +| `src/cthreads/cpp/gpu/impl/memory.cpp` | find/create/destroy/upload/download + TransferEngine | +| `src/cthreads/cpp/gpu/headers/pack.hpp` | `GpuPack` create/upload/download | +| `src/cthreads/cpp/gpu/impl/pack.cpp` | Pack helpers | +| `src/cthreads/cpp/gpu/headers/descriptors.hpp` | Descriptor pool/set/update (namespace `pack`) | +| `src/cthreads/cpp/gpu/impl/descriptors.cpp` | Descriptor helpers | +| `src/cthreads/cpp/gpu/headers/shader.hpp` / `shader_cache.hpp` | `create_entry`, `ShaderCache` | +| `src/cthreads/cpp/gpu/impl/shader.cpp` / `shader_cache.cpp` | Pipeline build + cache | +| `src/cthreads/cpp/gpu/headers/module.hpp` | `SpawnedGpuKernel`, `launch_gpu_kernel` | +| `src/cthreads/cpp/gpu/impl/module.cpp` | Launch + join writeback | +| `src/cthreads/cpp/gpu/testing/` | Pack roundtrip + saxpy smoke SPIR-V | +| `src/cthreads/cpp/bindings/gpu_module.cpp` | pybind `cthreads._ext.gpu` | +| `src/cthreads/cpp/bindings/gpu_testing_module.cpp` | Test-only `_ext.gpu.testing` | +| `src/cthreads/cpp/bindings/module.cpp` | `#ifdef CTHREADS_WITH_GPU` calls `bind_gpu` | +| `src/cthreads/cpp/CMakeLists.txt` | `CTHREADS_GPU` option + sources | +| `src/cthreads/python/cthreads/gpu/` | Python façade + errors | +| `tests/unit/test_gpu_*.py` | Context / pack / shader+launch tests | + +Newcomer-oriented C++ module docs: [docs/internals/gpu/README.md](../internals/gpu/README.md). + +## How to study with the code open + +### Pass 1 — Context + +1. Read guide 03-05. +2. Open `context.hpp` — read every PFN comment. +3. Open `context.cpp` — follow `open_loader` -> `create_instance_and_device` -> `shutdown_unlocked`. +4. Run `gpu.available()` / `device_name()` with `CTHREADS_GPU=ON`. + +### Pass 2 — Memory + +1. Read guide 06-08. +2. Open `memory.hpp` — `BufferKind` + `GpuBuffer`. +3. Walk `create_buffer` and `upload_buffer` in `memory.cpp` line by line. +4. Mentally simulate uploading 4 floats. + +### Pass 3 — Pack, descriptors, shaders, launch + +1. Read guide 09-12. +2. Open `pack.hpp` / `descriptors.hpp` / `shader_cache.hpp` / `module.hpp`. +3. Trace product `launch_gpu_kernel` after `testing.register_smoke_saxpy`. +4. Run `tests/unit/test_gpu_shader.py::test_live_launch_saxpy_product_path`. + +## Build flag reminder + +```bat +set CMAKE_ARGS=-DCTHREADS_GPU=ON +pip install -e . +``` + +Wipe `build/` if CMake cached GPU off. + +## Python error mapping + +C++ throws `std::runtime_error` with prefixes like: + +```text +cthreads.gpu.VulkanLoaderNotFound: ... +cthreads.gpu.VulkanInitFailed: ... +cthreads.gpu.VulkanNoDevice: ... +cthreads.gpu.GpuInvalidArgument: ... +``` + +`cthreads.gpu` catches and raises typed exceptions. Keep prefixes stable when adding new errors. diff --git a/docs/vk_guide/14-glossary.md b/docs/vk_guide/14-glossary.md new file mode 100644 index 0000000..ae6df62 --- /dev/null +++ b/docs/vk_guide/14-glossary.md @@ -0,0 +1,51 @@ +# 14 — Glossary + +| Term | Meaning in this project | +|------|-------------------------| +| **Vulkan** | Low-level cross-vendor GPU API | +| **Loader** | `vulkan-1.dll` / `libvulkan.so.1` — finds drivers, exports `vkGetInstanceProcAddr` | +| **ICD** | Installable Client Driver — vendor Vulkan implementation inside the GPU driver | +| **SDK** | LunarG Vulkan SDK — headers and tools for *developers* | +| **Instance** | App-wide Vulkan connection (`VkInstance`) | +| **Physical device** | A GPU enumerated by the loader | +| **Logical device** | Opened GPU connection (`VkDevice`) | +| **Queue family** | Group of queues with shared capabilities (compute/graphics/…) | +| **Queue** | Submission port (`VkQueue`) | +| **Handle** | Opaque id for a Vulkan object (`VkBuffer`, …) | +| **PFN_*** | Typed function pointer to a Vulkan entry point | +| **Buffer (`VkBuffer`)** | Buffer resource object (needs memory bound) | +| **Device memory** | Allocated GPU memory slab (`VkDeviceMemory`) | +| **Host-visible** | Memory the CPU can map | +| **Host-coherent** | Mapped writes/reads without manual flush | +| **Device-local** | GPU-fast memory; not our persistently mapped SSBO path | +| **Staging buffer** | Host-visible buffer used only for upload/download traffic | +| **Map** | Get a CPU pointer into host-visible device memory | +| **SSBO** | Storage buffer — shader-readable/writable buffer block | +| **Descriptor** | Binding that tells a shader which buffer is at binding i | +| **std430** | Packing rules for storage buffer structs/arrays | +| **Command pool** | Owns command buffers for a queue family | +| **Command buffer** | Recorded list of GPU commands | +| **Submit** | Give a command buffer to a queue for execution | +| **Fence** | CPU-waitable completion signal for a submit | +| **Semaphore** | GPU-side sync (we mostly ignore for now) | +| **Barrier** | GPU ordering/visibility constraint between commands | +| **SPIR-V** | Portable shader IR Vulkan consumes | +| **Shader module** | Vulkan object holding SPIR-V | +| **Pipeline (compute)** | Prepared compute shader + layout | +| **Dispatch** | Launch compute workgroups (`vkCmdDispatch`) | +| **Workgroup / local size** | Group of invocations that run together | +| **GpuPack** | Per-launch scalar SSBO + list SSBOs (binding convention) | +| **Marshal** | Copy Python values into native/GPU pack storage | +| **Writeback** | Copy native/GPU results into the same Python objects | +| **Join** | Wait for GPU job completion then writeback | + +## Abbreviations + +| Short | Full | +|-------|------| +| H2D | Host to device (upload) | +| D2H | Device to host (download) | +| SSBO | Shader Storage Buffer Object | +| GIPA | `vkGetInstanceProcAddr` | +| BDA | Buffer device address (rejected for v1) | +| AoS | Array of structures (Threadable lists later) | diff --git a/docs/vk_guide/15-checklist.md b/docs/vk_guide/15-checklist.md new file mode 100644 index 0000000..4769ed1 --- /dev/null +++ b/docs/vk_guide/15-checklist.md @@ -0,0 +1,44 @@ +# 15 — Checklist: concepts contributors should be able to explain + +Use this after reading. If a concept cannot be explained in plain language, revisit that chapter. + +## Foundations + +- [ ] Why cthreads GPU uses normal Python types, not a public DeviceBuffer +- [ ] Difference between loader, SDK, and GPU driver +- [ ] Why we dynamic-load Vulkan instead of linking for default wheels +- [ ] What `VK_NULL_HANDLE` means + +## Context layer + +- [ ] Instance vs physical device vs logical device vs queue +- [ ] Why instance-level functions must be resolved with a real instance +- [ ] How we pick a compute queue family and prefer discrete GPUs +- [ ] Shutdown order (device before instance before FreeLibrary) + +## Memory / transfer layer + +- [ ] Why a buffer needs both `VkBuffer` and `VkDeviceMemory` +- [ ] What `find_memory_type` does with `type_bits` and property flags +- [ ] Staging vs device-local (`BufferKind`) +- [ ] Why device-local SSBOs are not persistently mapped +- [ ] Upload path: memcpy staging -> GPU copy -> device-local +- [ ] Download path reverse +- [ ] Why size 0 buffers are rejected +- [ ] What a fence wait guarantees after `vkQueueSubmit` + +## Launch path / GpuPack / shaders + +- [ ] What an SSBO is in GLSL +- [ ] cthreads binding 0 = scalars, 1..N = lists convention +- [ ] What std430 padding is and why host/GPU must match +- [ ] SPIR-V vs GLSL vs pipeline vs dispatch +- [ ] What GpuPack contains (binding convention) +- [ ] Why GPU jobs are launch/wait only (no mid-run sync) + +## When stuck + +1. Find the term in [14-glossary.md](./14-glossary.md). +2. Open the matching chapter from [README.md](./README.md). +3. Open the matching file in [13-map-to-our-code.md](./13-map-to-our-code.md). +4. If still unclear, ask with the chapter number + expected vs observed behavior in code. diff --git a/docs/vk_guide/16-api-reference.md b/docs/vk_guide/16-api-reference.md new file mode 100644 index 0000000..4ea3da5 --- /dev/null +++ b/docs/vk_guide/16-api-reference.md @@ -0,0 +1,1358 @@ +# 16 — Vulkan types, structs, enums, and functions (cthreads compute) + +This is a **reference chapter** for Vulkan symbols that matter to the cthreads +GPU compute path (Context, memory/transfers, and the planned launch path: +descriptors, pipelines, dispatch). It is not the full Vulkan API. + +Conventions used below: + +- **Handle** means an opaque Vulkan object id (do not dereference it as a C pointer). +- **`sType`** on create/info structs must be set to the matching `VK_STRUCTURE_TYPE_*`. +- **`pNext`** is usually `nullptr` in cthreads (extension chaining). +- **`pAllocator`** is always `nullptr` in cthreads (use the default allocator). +- **In / out** describes whether the caller fills the argument or the driver writes it. + +Related conceptual chapters: [02](./02-mental-model.md), [04](./04-instance-device-queue.md), +[06](./06-buffers-and-memory.md), [08](./08-commands-fences.md), [09](./09-descriptors-ssbo.md), +[11](./11-spirv-pipelines-dispatch.md). + +--- + +## Table of contents + +1. [Scalar and handle types](#1-scalar-and-handle-types) +2. [Results and booleans](#2-results-and-booleans) +3. [Structure type tags (`sType`)](#3-structure-type-tags-stype) +4. [Version helpers](#4-version-helpers) +5. [Instance and device bootstrap](#5-instance-and-device-bootstrap) +6. [Physical device queries](#6-physical-device-queries) +7. [Queues](#7-queues) +8. [Buffers and device memory](#8-buffers-and-device-memory) +9. [Command pools, command buffers, copies](#9-command-pools-command-buffers-copies) +10. [Fences and submit](#10-fences-and-submit) +11. [Launch path: descriptors](#11-launch-path-descriptors) +12. [Launch path: shaders, pipelines, dispatch](#12-launch-path-shaders-pipelines-dispatch) +13. [Launch path: barriers (overview)](#13-launch-path-barriers-overview) +14. [Function pointer typedefs (`PFN_*`)](#14-function-pointer-typedefs-pfn_) +15. [Quick index by name](#15-quick-index-by-name) + +--- + +## 1. Scalar and handle types + +### `VkBool32` + +- **What:** Vulkan boolean. Not C++ `bool`. +- **Values:** `VK_TRUE` (1), `VK_FALSE` (0). +- **Where used:** e.g. `vkWaitForFences(..., waitAll, ...)`. + +### `VkDeviceSize` + +- **What:** Unsigned 64-bit size/offset type for buffer sizes, copy sizes, memory offsets. +- **Why not `size_t`:** Vulkan wants a fixed width across platforms. +- **cthreads:** `GpuBuffer.size`, upload/download byte counts. + +### `VK_NULL_HANDLE` + +- **What:** Sentinel meaning "no object" for handle types. +- **Analogy:** Like `nullptr` for Vulkan handles. +- **cthreads:** Initial value of all handles on `Context` / `GpuBuffer`. + +### `VK_WHOLE_SIZE` + +- **What:** Special size meaning "the rest of the resource" (often used with map/descriptor ranges). +- **cthreads:** May appear when mapping whole allocations or descriptor buffer ranges. + +### Opaque handles (object ids) + +Handles are opaque ids created by `vkCreate*` / allocate functions and destroyed by +matching destroy/free functions. Destroy children before parents. + +Below is what each object is **for** (purpose), not only what the typedef is called. + +#### `VkInstance` + +**Purpose:** The process-wide "logged into Vulkan" object. It owns the connection to +the loader and is required before enumerating GPUs or creating a device. + +**How it works:** After `vkCreateInstance`, the loader knows this application exists +and can route instance-level calls. Almost every later bootstrap call takes the +instance (or a device created from it). Destroying the instance last tears down that +connection. cthreads keeps one instance on `Context`. + +#### `VkPhysicalDevice` + +**Purpose:** A handle representing one GPU the loader can see (hardware or software ICD). + +**How it works:** It is not "opened" yet. It is used only to **query** capabilities +(name, queue families, memory types) and as input to `vkCreateDevice`. Destroying the +instance invalidates physical device handles. cthreads picks one compute-capable +physical device (preferring discrete). + +#### `VkDevice` (logical device) + +**Purpose:** The opened session on a chosen GPU. Almost all GPU work objects (buffers, +pipelines, command pools) are created from a device. + +**How it works:** `vkCreateDevice` tells the driver which queue families this app will +use. After that, device-level entry points operate on this handle. Destroying the +device invalidates all objects created from it. cthreads uses one logical device. + +#### `VkQueue` + +**Purpose:** A submission port. Recorded command buffers are sent here to run on the GPU. + +**How it works:** Queues belong to a queue family (graphics/compute/transfer). +`vkGetDeviceQueue` returns a handle owned by the device (no separate destroy). +`vkQueueSubmit` is asynchronous: it returns when work is queued, not when finished. +cthreads uses one compute-capable queue for copies and (later) dispatches. + +#### `VkBuffer` + +**Purpose:** A GPU resource that represents a contiguous byte region with allowed uses +(copy, storage buffer, and so on). + +**How it works:** Creating a buffer only creates the *object*. Bytes live in +`VkDeviceMemory` after `vkBindBufferMemory`. Shaders never see the C++ handle directly; +descriptors bind the buffer to a binding number. cthreads uses staging buffers and +device-local SSBOs (`GpuBuffer`). + +#### `VkDeviceMemory` + +**Purpose:** An allocated slab of memory from a chosen memory type (host-visible, +device-local, …). + +**How it works:** `vkAllocateMemory` reserves memory; `vkBindBufferMemory` attaches it +to a buffer. Host-visible memory can be `vkMapMemory`'d for CPU `memcpy`. Device-local +memory is typically filled via GPU copies from staging. Free with `vkFreeMemory` +after unmapping and after the bound buffer is destroyed or no longer needs it +(cthreads destroys buffer then frees memory). + +#### `VkCommandPool` + +**Purpose:** An allocator/owner for command buffers tied to one queue family. + +**How it works:** Command buffers must come from a pool that matches the queue that +will submit them. Pools can be optimized for short-lived ("transient") buffers. +Destroying a pool frees its command buffers. cthreads currently creates a transient +pool per copy; a reused transfer engine would keep one pool alive on Context. + +#### `VkCommandBuffer` + +**Purpose:** A recorded recipe of GPU commands (copies, binds, dispatches). + +**How it works:** The CPU calls `vkBeginCommandBuffer`, then `vkCmd*` functions, then +`vkEndCommandBuffer`. Nothing runs on the GPU until `vkQueueSubmit`. After submit, the +CPU can continue; completion is tracked with a fence (or semaphores). Think of it as +building a batch job, then mailing it to the GPU. + +#### `VkFence` + +**Purpose:** A CPU-visible "this GPU submit is finished" signal. + +**How it works:** A fence starts unsignaled (unless created signaled). Passing it to +`vkQueueSubmit` asks the driver to signal it when that submit completes. The CPU calls +`vkWaitForFences` to block until signaled, then may safely read staging memory, +destroy temporary buffers, or mark a job joined. `vkResetFences` returns it to +unsignaled for reuse. Unlike semaphores (GPU-to-GPU), fences are the usual tool for +**CPU waiting on GPU** in cthreads. + +#### `VkSemaphore` + +**Purpose:** GPU-side synchronization between queue submits (and present in graphics). + +**How it works:** One submit can signal a semaphore; a later submit can wait on it at a +pipeline stage. The CPU does not typically `wait` on semaphores the way it waits on +fences. cthreads compute/copy path can ignore semaphores while everything uses one +queue and fence waits. + +#### `VkDescriptorSetLayout` + +**Purpose:** The schema for a descriptor set: which binding numbers exist and what +types they are (e.g. binding 0 = storage buffer). + +**How it works:** Created once for a shader interface. Pipeline layouts reference it. +It does not point at real buffers yet; it only describes the shape. + +#### `VkDescriptorPool` + +**Purpose:** A pool from which concrete descriptor sets are allocated. + +**How it works:** Sized by how many sets and how many descriptors of each type may be +allocated. Destroying the pool frees its sets (unless using free-individual flags). + +#### `VkDescriptorSet` + +**Purpose:** One instance of a layout filled with real resources (which `VkBuffer` is +binding 1, and so on). + +**How it works:** Allocate from a pool, then `vkUpdateDescriptorSets` to point bindings +at buffers. Before dispatch, `vkCmdBindDescriptorSets` attaches the set to the command +buffer so the shader's `layout(binding=N)` reads the right memory. + +#### `VkShaderModule` + +**Purpose:** Holds SPIR-V code for a shader stage. + +**How it works:** Created from SPIR-V bytes. Referenced when creating a pipeline. The +module itself is not "run"; the pipeline binds the compiled stage. Safe to destroy the +module after the pipeline is created (the pipeline keeps what it needs), depending on +driver rules; many apps keep modules until shutdown. + +#### `VkPipelineLayout` + +**Purpose:** Declares the interface a pipeline expects: descriptor set layouts and +optional push-constant ranges. + +**How it works:** Must match what the shader bindings and push constants use. +`vkCmdBindDescriptorSets` and `vkCmdPushConstants` take this layout. + +#### `VkPipeline` (compute) + +**Purpose:** A ready-to-bind compute program: shader stage + layout (+ state). + +**How it works:** `vkCreateComputePipelines` compiles/links the compute stage for the +device. Recording `vkCmdBindPipeline` then `vkCmdDispatch` runs it. Creating pipelines +is relatively expensive, so cthreads should cache them by shader hash. + +#### `VkPipelineCache` + +**Purpose:** Optional cache so pipeline creation can reuse previous driver compilations. + +**How it works:** Pass a cache into `vkCreateComputePipelines`. Can be saved to disk +across runs. Not required for correctness. + +--- + +## 2. Results and booleans + +### `VkResult` (enum) + +Most Vulkan functions return `VkResult`. + +| Value | Meaning for cthreads | +|-------|----------------------| +| `VK_SUCCESS` | Call succeeded | +| `VK_NOT_READY` | Not done yet (fences/queries) | +| `VK_TIMEOUT` | Wait timed out | +| `VK_EVENT_SET` / `VK_EVENT_RESET` | Event status (unused here) | +| `VK_INCOMPLETE` | Enumeration truncated (rare if count pattern is used correctly) | +| `VK_ERROR_OUT_OF_HOST_MEMORY` | CPU OOM | +| `VK_ERROR_OUT_OF_DEVICE_MEMORY` | GPU OOM | +| `VK_ERROR_INITIALIZATION_FAILED` | Init failed | +| `VK_ERROR_DEVICE_LOST` | Device lost (serious) | +| `VK_ERROR_MEMORY_MAP_FAILED` | `vkMapMemory` failed | +| `VK_ERROR_LAYER_NOT_PRESENT` | Requested layer missing | +| `VK_ERROR_EXTENSION_NOT_PRESENT` | Requested extension missing | +| `VK_ERROR_FEATURE_NOT_PRESENT` | Feature not available | +| `VK_ERROR_INCOMPATIBLE_DRIVER` | Driver too old / incompatible | + +**cthreads pattern:** treat anything other than `VK_SUCCESS` as failure and throw `cthreads.gpu.VulkanInitFailed: …` (or a more specific mapped error). + +--- + +## 3. Structure type tags (`sType`) + +### `VkStructureType` (enum, partial) + +Every create/info struct starts with: + +```text +sType: VkStructureType +pNext: const void* // usually nullptr +``` + +`sType` tells the driver which struct layout follows. Wrong `sType` is undefined behavior / validation errors. + +Values cthreads uses (or will use): + +| Enum | Struct | +|------|--------| +| `VK_STRUCTURE_TYPE_APPLICATION_INFO` | `VkApplicationInfo` | +| `VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO` | `VkInstanceCreateInfo` | +| `VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO` | `VkDeviceQueueCreateInfo` | +| `VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO` | `VkDeviceCreateInfo` | +| `VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO` | `VkBufferCreateInfo` | +| `VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO` | `VkMemoryAllocateInfo` | +| `VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO` | `VkCommandPoolCreateInfo` | +| `VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO` | `VkCommandBufferAllocateInfo` | +| `VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO` | `VkCommandBufferBeginInfo` | +| `VK_STRUCTURE_TYPE_FENCE_CREATE_INFO` | `VkFenceCreateInfo` | +| `VK_STRUCTURE_TYPE_SUBMIT_INFO` | `VkSubmitInfo` | +| `VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO` | `VkShaderModuleCreateInfo` | +| `VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO` | `VkPipelineLayoutCreateInfo` | +| `VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO` | `VkComputePipelineCreateInfo` | +| `VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO` | `VkPipelineShaderStageCreateInfo` | +| `VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO` | `VkDescriptorSetLayoutCreateInfo` | +| `VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO` | `VkDescriptorPoolCreateInfo` | +| `VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO` | `VkDescriptorSetAllocateInfo` | +| `VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET` | `VkWriteDescriptorSet` | +| `VK_STRUCTURE_TYPE_MEMORY_BARRIER` / `BUFFER_MEMORY_BARRIER` / … | Barrier structs | + +**Rule:** always zero the struct (`{}`), then set `sType`, then set needed fields. + +--- + +## 4. Version helpers + +### `VK_MAKE_VERSION(major, minor, patch)` + +Packs a version into a `uint32_t` for `VkApplicationInfo`. + +### `VK_API_VERSION_1_0` / `VK_API_VERSION_1_1` / … + +API version requested in `VkApplicationInfo.apiVersion`. + +**cthreads:** requests `VK_API_VERSION_1_1` in Context init. + +--- + +## 5. Instance and device bootstrap + +**Purpose of this layer:** turn "Vulkan exists on this machine" into a usable +`VkDevice` + `VkQueue`. Without it, no buffers or submits are possible. + +Flow: create `VkInstance` -> enumerate `VkPhysicalDevice`s -> create `VkDevice` with +a compute queue family -> `vkGetDeviceQueue`. See handle purposes in section 1. + +### `VkApplicationInfo` + +Metadata about the application (mostly for tools/drivers). + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | Must be `APPLICATION_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `pApplicationName` | `const char*` | App name string | +| `applicationVersion` | `uint32_t` | App version (`VK_MAKE_VERSION`) | +| `pEngineName` | `const char*` | Engine name (`"cthreads"`) | +| `engineVersion` | `uint32_t` | Engine version | +| `apiVersion` | `uint32_t` | Highest Vulkan API version the app uses | + +### `VkInstanceCreateInfo` + +Arguments to `vkCreateInstance`. + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `INSTANCE_CREATE_INFO` | +| `pNext` | `const void*` | Extensions via pNext (unused in basic cthreads) | +| `flags` | `VkInstanceCreateFlags` | Usually 0 | +| `pApplicationInfo` | `const VkApplicationInfo*` | Pointer to app info | +| `enabledLayerCount` | `uint32_t` | Validation layers count (0 unless debugging) | +| `ppEnabledLayerNames` | `const char* const*` | Layer name list | +| `enabledExtensionCount` | `uint32_t` | Instance extensions (0 for basic compute) | +| `ppEnabledExtensionNames` | `const char* const*` | Extension name list | + +### `vkCreateInstance` + +```text +VkResult vkCreateInstance( + const VkInstanceCreateInfo* pCreateInfo, // in: create parameters + const VkAllocationCallbacks* pAllocator, // in: nullptr in cthreads + VkInstance* pInstance // out: created instance handle +) +``` + +Creates the Vulkan instance. Resolve this entry point with a **null** instance via `vkGetInstanceProcAddr`. + +### `vkDestroyInstance` + +```text +void vkDestroyInstance( + VkInstance instance, // in: instance to destroy + const VkAllocationCallbacks* pAllocator // in: nullptr +) +``` + +Destroys the instance. Resolve with the **real** instance handle (not null). + +### `vkEnumeratePhysicalDevices` + +```text +VkResult vkEnumeratePhysicalDevices( + VkInstance instance, // in + uint32_t* pPhysicalDeviceCount, // in/out: count + VkPhysicalDevice* pPhysicalDevices // out: array, or nullptr to query count only +) +``` + +**Two-call idiom:** first call with `pPhysicalDevices == nullptr` to get count; allocate; second call to fill. + +### `VkDeviceQueueCreateInfo` + +Requests queues when creating a logical device. + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `DEVICE_QUEUE_CREATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `flags` | `VkDeviceQueueCreateFlags` | Usually 0 | +| `queueFamilyIndex` | `uint32_t` | Which family (cthreads compute family) | +| `queueCount` | `uint32_t` | How many queues (cthreads: 1) | +| `pQueuePriorities` | `const float*` | Array of priorities in `[0,1]` (cthreads: `{1.0f}`) | + +### `VkDeviceCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `DEVICE_CREATE_INFO` | +| `pNext` | `const void*` | Features/extensions via pNext if needed | +| `flags` | `VkDeviceCreateFlags` | Usually 0 | +| `queueCreateInfoCount` | `uint32_t` | Number of queue infos | +| `pQueueCreateInfos` | `const VkDeviceQueueCreateInfo*` | Queue requests | +| `enabledLayerCount` | `uint32_t` | Deprecated for device; usually 0 | +| `ppEnabledLayerNames` | `const char* const*` | Usually unused | +| `enabledExtensionCount` | `uint32_t` | Device extensions (0 for basic path) | +| `ppEnabledExtensionNames` | `const char* const*` | Extension names | +| `pEnabledFeatures` | `const VkPhysicalDeviceFeatures*` | Optional features; can be `nullptr` | + +### `vkCreateDevice` + +```text +VkResult vkCreateDevice( + VkPhysicalDevice physicalDevice, // in: chosen GPU + const VkDeviceCreateInfo* pCreateInfo, // in + const VkAllocationCallbacks* pAllocator, // in: nullptr + VkDevice* pDevice // out +) +``` + +### `vkDestroyDevice` + +```text +void vkDestroyDevice( + VkDevice device, + const VkAllocationCallbacks* pAllocator +) +``` + +Destroys the logical device. All device-child objects must already be destroyed. + +### `vkGetDeviceQueue` + +```text +void vkGetDeviceQueue( + VkDevice device, // in + uint32_t queueFamilyIndex, // in + uint32_t queueIndex, // in: 0 for first queue of that family + VkQueue* pQueue // out +) +``` + +Does not create a queue; retrieves a handle owned by the device. + +--- + +## 6. Physical device queries + +**Purpose of this layer:** ask a `VkPhysicalDevice` what it can do before opening it. +cthreads uses these queries to (1) prefer a discrete GPU with a compute queue, and +(2) pick legal memory types for staging vs device-local buffers. + +### `VkPhysicalDeviceType` (enum) + +| Value | Meaning | +|-------|---------| +| `VK_PHYSICAL_DEVICE_TYPE_OTHER` | Other / unknown | +| `VK_PHYSICAL_DEVICE_TYPE_INTEGRATED_GPU` | iGPU (CPU die) | +| `VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU` | Discrete card (preferred by cthreads scoring) | +| `VK_PHYSICAL_DEVICE_TYPE_VIRTUAL_GPU` | Virtualized | +| `VK_PHYSICAL_DEVICE_TYPE_CPU` | CPU fallback implementation | + +### `VkPhysicalDeviceProperties` + +Large struct. Fields cthreads cares about: + +| Field | Type | Meaning | +|-------|------|---------| +| `apiVersion` | `uint32_t` | Max API version supported | +| `driverVersion` | `uint32_t` | Vendor driver version | +| `vendorID` | `uint32_t` | PCI vendor id | +| `deviceID` | `uint32_t` | PCI device id | +| `deviceType` | `VkPhysicalDeviceType` | Discrete vs integrated, etc. | +| `deviceName` | `char[VK_MAX_PHYSICAL_DEVICE_NAME_SIZE]` | Human-readable GPU name | +| `pipelineCacheUUID` | `uint8_t[VK_UUID_SIZE]` | Cache identity | +| `limits` | `VkPhysicalDeviceLimits` | Many limits (max bindings, etc.) | + +### `vkGetPhysicalDeviceProperties` + +```text +void vkGetPhysicalDeviceProperties( + VkPhysicalDevice physicalDevice, // in + VkPhysicalDeviceProperties* pProperties // out +) +``` + +### `VkQueueFlagBits` / `VkQueueFlags` + +Bit flags describing what a queue family can do: + +| Flag | Meaning | +|------|---------| +| `VK_QUEUE_GRAPHICS_BIT` | Graphics commands | +| `VK_QUEUE_COMPUTE_BIT` | Compute dispatches (required by cthreads) | +| `VK_QUEUE_TRANSFER_BIT` | Dedicated transfer (often also implied by graphics/compute) | +| `VK_QUEUE_SPARSE_BINDING_BIT` | Sparse memory (unused) | + +### `VkQueueFamilyProperties` + +| Field | Type | Meaning | +|-------|------|---------| +| `queueFlags` | `VkQueueFlags` | Capability bits | +| `queueCount` | `uint32_t` | How many queues in this family | +| `timestampValidBits` | `uint32_t` | Timestamp support | +| `minImageTransferGranularity` | `VkExtent3D` | Image copy granularity | + +### `vkGetPhysicalDeviceQueueFamilyProperties` + +Two-call idiom (count then array), same as device enumeration. + +### `VkMemoryPropertyFlagBits` / `VkMemoryPropertyFlags` + +| Flag | Meaning | +|------|---------| +| `VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT` | GPU-fast memory (VRAM-like). cthreads device-local SSBOs | +| `VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT` | CPU can map. cthreads staging | +| `VK_MEMORY_PROPERTY_HOST_COHERENT_BIT` | No manual flush for CPU/GPU visibility. cthreads staging | +| `VK_MEMORY_PROPERTY_HOST_CACHED_BIT` | CPU cached (optional) | +| `VK_MEMORY_PROPERTY_LAZILY_ALLOCATED_BIT` | Transient attachments (graphics; unused) | + +### `VkMemoryType` + +| Field | Type | Meaning | +|-------|------|---------| +| `propertyFlags` | `VkMemoryPropertyFlags` | What this type can do | +| `heapIndex` | `uint32_t` | Which heap it comes from | + +### `VkMemoryHeap` + +| Field | Type | Meaning | +|-------|------|---------| +| `size` | `VkDeviceSize` | Heap size in bytes | +| `flags` | `VkMemoryHeapFlags` | e.g. device-local heap | + +### `VkPhysicalDeviceMemoryProperties` + +| Field | Type | Meaning | +|-------|------|---------| +| `memoryTypeCount` | `uint32_t` | Number of types | +| `memoryTypes` | `VkMemoryType[]` | Types array | +| `memoryHeapCount` | `uint32_t` | Number of heaps | +| `memoryHeaps` | `VkMemoryHeap[]` | Heaps array | + +### `vkGetPhysicalDeviceMemoryProperties` + +```text +void vkGetPhysicalDeviceMemoryProperties( + VkPhysicalDevice physicalDevice, + VkPhysicalDeviceMemoryProperties* pMemoryProperties +) +``` + +Used by `find_memory_type` together with `memoryTypeBits` from buffer requirements. + +--- + +## 7. Queues + +**Purpose:** a `VkQueue` is where work enters the GPU. The CPU records command +buffers, then `vkQueueSubmit` hands them to a queue. Execution is asynchronous +relative to the CPU. + +Queues are obtained with `vkGetDeviceQueue` (section 5). There is no +`vkDestroyQueue`; destroying the device invalidates queues. + +cthreads uses a single compute-capable queue for buffer copies and later for +compute dispatches. + +--- + +## 8. Buffers and device memory + +**Purpose of this layer:** give the GPU (and optionally the CPU) a place to store +bytes. A `VkBuffer` is the typed resource; `VkDeviceMemory` is the backing store. +Binding connects them. Staging memory is host-visible for `memcpy`; device-local +memory is what shaders should use for hot data. + +See also the handle write-ups for `VkBuffer` and `VkDeviceMemory` in section 1. + +### `VkBufferUsageFlagBits` / `VkBufferUsageFlags` + +| Flag | Meaning | +|------|---------| +| `VK_BUFFER_USAGE_TRANSFER_SRC_BIT` | May be source of `vkCmdCopyBuffer` | +| `VK_BUFFER_USAGE_TRANSFER_DST_BIT` | May be destination of a copy | +| `VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT` | UBO (cthreads prefers SSBO for mutable pack) | +| `VK_BUFFER_USAGE_STORAGE_BUFFER_BIT` | SSBO for compute read/write | +| `VK_BUFFER_USAGE_INDEX_BUFFER_BIT` | Graphics index buffer (unused) | +| `VK_BUFFER_USAGE_VERTEX_BUFFER_BIT` | Graphics vertex buffer (unused) | +| `VK_BUFFER_USAGE_INDIRECT_BUFFER_BIT` | Indirect dispatch/draw (optional later) | + +**cthreads Staging:** `TRANSFER_SRC | TRANSFER_DST` +**cthreads DeviceLocal:** `STORAGE_BUFFER | TRANSFER_SRC | TRANSFER_DST` + +### `VkSharingMode` (enum) + +| Value | Meaning | +|-------|---------| +| `VK_SHARING_MODE_EXCLUSIVE` | One queue family owns it (cthreads default) | +| `VK_SHARING_MODE_CONCURRENT` | Multiple families; must list indices | + +### `VkBufferCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `BUFFER_CREATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `flags` | `VkBufferCreateFlags` | Sparse flags etc.; usually 0 | +| `size` | `VkDeviceSize` | Size in bytes (> 0 in cthreads) | +| `usage` | `VkBufferUsageFlags` | Allowed uses | +| `sharingMode` | `VkSharingMode` | Exclusive vs concurrent | +| `queueFamilyIndexCount` | `uint32_t` | Needed if concurrent | +| `pQueueFamilyIndices` | `const uint32_t*` | Families if concurrent | + +### `vkCreateBuffer` + +```text +VkResult vkCreateBuffer( + VkDevice device, + const VkBufferCreateInfo* pCreateInfo, + const VkAllocationCallbacks* pAllocator, + VkBuffer* pBuffer +) +``` + +Creates the buffer object only. Memory is separate. + +### `vkDestroyBuffer` + +```text +void vkDestroyBuffer( + VkDevice device, + VkBuffer buffer, + const VkAllocationCallbacks* pAllocator +) +``` + +### `VkMemoryRequirements` + +| Field | Type | Meaning | +|-------|------|---------| +| `size` | `VkDeviceSize` | Bytes to allocate (may exceed create size due to alignment) | +| `alignment` | `VkDeviceSize` | Required alignment of the bind offset | +| `memoryTypeBits` | `uint32_t` | Bit i set => memory type i is legal | + +### `vkGetBufferMemoryRequirements` + +```text +void vkGetBufferMemoryRequirements( + VkDevice device, + VkBuffer buffer, + VkMemoryRequirements* pMemoryRequirements +) +``` + +### `VkMemoryAllocateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `MEMORY_ALLOCATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `allocationSize` | `VkDeviceSize` | Usually `mem_reqs.size` | +| `memoryTypeIndex` | `uint32_t` | From `find_memory_type` | + +### `vkAllocateMemory` + +```text +VkResult vkAllocateMemory( + VkDevice device, + const VkMemoryAllocateInfo* pAllocateInfo, + const VkAllocationCallbacks* pAllocator, + VkDeviceMemory* pMemory +) +``` + +### `vkFreeMemory` + +```text +void vkFreeMemory( + VkDevice device, + VkDeviceMemory memory, + const VkAllocationCallbacks* pAllocator +) +``` + +### `vkBindBufferMemory` + +```text +VkResult vkBindBufferMemory( + VkDevice device, + VkBuffer buffer, + VkDeviceMemory memory, + VkDeviceSize memoryOffset // must respect alignment; cthreads uses 0 with dedicated alloc +) +``` + +Associates memory with the buffer. A buffer is bound at most once (without sparse extensions). + +### `VkMemoryMapFlags` + +Usually `0` for `vkMapMemory`. + +### `vkMapMemory` + +```text +VkResult vkMapMemory( + VkDevice device, + VkDeviceMemory memory, + VkDeviceSize offset, // start offset in the allocation + VkDeviceSize size, // bytes to map, or VK_WHOLE_SIZE + VkMemoryMapFlags flags, // usually 0 + void** ppData // out: CPU pointer +) +``` + +Only valid for **host-visible** memory. cthreads maps staging buffers. + +### `vkUnmapMemory` + +```text +void vkUnmapMemory( + VkDevice device, + VkDeviceMemory memory +) +``` + +--- + +## 9. Command pools, command buffers, copies + +**Purpose of this layer:** build and own GPU "to-do lists." A command buffer is the +list; a command pool allocates those lists for a specific queue family; +`vkCmdCopyBuffer` is one kind of item on the list. Recording is CPU-side; running +happens only after submit. + +See handle write-ups for `VkCommandPool` and `VkCommandBuffer` in section 1. + +### `VkCommandPoolCreateFlagBits` + +| Flag | Meaning | +|------|---------| +| `VK_COMMAND_POOL_CREATE_TRANSIENT_BIT` | Short-lived CBs (cthreads one-shot copies) | +| `VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT` | Allow resetting individual CBs | + +### `VkCommandPoolCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `COMMAND_POOL_CREATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `flags` | `VkCommandPoolCreateFlags` | Transient / reset bits | +| `queueFamilyIndex` | `uint32_t` | Must match the queue that will submit | + +### `vkCreateCommandPool` / `vkDestroyCommandPool` + +Standard create/destroy pair on a `VkDevice`. + +### `VkCommandBufferLevel` (enum) + +| Value | Meaning | +|-------|---------| +| `VK_COMMAND_BUFFER_LEVEL_PRIMARY` | Can be submitted to a queue (cthreads) | +| `VK_COMMAND_BUFFER_LEVEL_SECONDARY` | Can be called from primary (advanced) | + +### `VkCommandBufferAllocateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `COMMAND_BUFFER_ALLOCATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `commandPool` | `VkCommandPool` | Pool to allocate from | +| `level` | `VkCommandBufferLevel` | Primary in cthreads | +| `commandBufferCount` | `uint32_t` | How many to allocate | + +### `vkAllocateCommandBuffers` / `vkFreeCommandBuffers` + +```text +VkResult vkAllocateCommandBuffers( + VkDevice device, + const VkCommandBufferAllocateInfo* pAllocateInfo, + VkCommandBuffer* pCommandBuffers // out array +) + +void vkFreeCommandBuffers( + VkDevice device, + VkCommandPool commandPool, + uint32_t commandBufferCount, + const VkCommandBuffer* pCommandBuffers +) +``` + +### `VkCommandBufferUsageFlagBits` + +| Flag | Meaning | +|------|---------| +| `VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT` | Record, submit once (copies) | +| `VK_COMMAND_BUFFER_USAGE_RENDER_PASS_CONTINUE_BIT` | Graphics secondary (unused) | +| `VK_COMMAND_BUFFER_USAGE_SIMULTANEOUS_USE_BIT` | Allow concurrent resubmit (careful) | + +### `VkCommandBufferBeginInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `COMMAND_BUFFER_BEGIN_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `flags` | `VkCommandBufferUsageFlags` | e.g. one-time | +| `pInheritanceInfo` | `const VkCommandBufferInheritanceInfo*` | For secondary; nullptr for primary | + +### `vkBeginCommandBuffer` / `vkEndCommandBuffer` + +```text +VkResult vkBeginCommandBuffer( + VkCommandBuffer commandBuffer, + const VkCommandBufferBeginInfo* pBeginInfo +) + +VkResult vkEndCommandBuffer( + VkCommandBuffer commandBuffer +) +``` + +Recording happens between begin and end. `vkCmd*` calls are only valid while recording. + +### `vkResetCommandBuffer` + +```text +VkResult vkResetCommandBuffer( + VkCommandBuffer commandBuffer, + VkCommandBufferResetFlags flags // usually 0 +) +``` + +Clears recorded commands so the buffer can be recorded again (pool must allow reset, or reset the whole pool). + +### `VkBufferCopy` + +| Field | Type | Meaning | +|-------|------|---------| +| `srcOffset` | `VkDeviceSize` | Byte offset in source | +| `dstOffset` | `VkDeviceSize` | Byte offset in destination | +| `size` | `VkDeviceSize` | Bytes to copy | + +### `vkCmdCopyBuffer` + +```text +void vkCmdCopyBuffer( + VkCommandBuffer commandBuffer, // recording CB + VkBuffer srcBuffer, + VkBuffer dstBuffer, + uint32_t regionCount, + const VkBufferCopy* pRegions +) +``` + +Records a GPU-side copy. Does not run until the CB is submitted and the GPU executes it. + +Both buffers need appropriate `TRANSFER_SRC` / `TRANSFER_DST` usage bits. + +--- + +## 10. Fences and submit + +**Purpose of this layer:** (1) send recorded work to the GPU (`vkQueueSubmit`), and +(2) let the CPU know when that work is done (`VkFence`). + +### How fences work (detail) + +The GPU and CPU run on different timelines. After `vkQueueSubmit`, the CPU might +immediately try to `memcpy` from a staging buffer that a download copy has not +finished writing. That is a race. + +A **fence** closes that gap: + +1. Create a fence (usually **unsignaled**). +2. Pass it as the last argument to `vkQueueSubmit`. +3. When the GPU finishes that submit, the driver **signals** the fence. +4. `vkWaitForFences` blocks the CPU until the fence is signaled (or times out). +5. After the wait returns successfully, it is safe to read staging memory, free + temporary buffers used by that submit, or complete a Python `join()`. +6. `vkResetFences` sets it back to unsignaled before the next submit if reusing it. + +Mental model: the fence is a doorbell the GPU rings when a batch of work is done; +the CPU sleeps on `vkWaitForFences` until it hears the ring. + +**Fence vs semaphore:** fences are for **CPU waits**. Semaphores are for **GPU waits** +between submits. cthreads upload/download and `GpuJob.join` are CPU-wait problems, +so fences are the right tool. + +### `VkFenceCreateFlagBits` + +| Flag | Meaning | +|------|---------| +| `0` | Created unsignaled (typical) | +| `VK_FENCE_CREATE_SIGNALED_BIT` | Created already signaled | + +### `VkFenceCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `FENCE_CREATE_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `flags` | `VkFenceCreateFlags` | Usually 0 | + +### `vkCreateFence` / `vkDestroyFence` + +Standard create/destroy on device. + +### `vkResetFences` + +```text +VkResult vkResetFences( + VkDevice device, + uint32_t fenceCount, + const VkFence* pFences +) +``` + +Returns fences to unsignaled so they can be reused on the next submit. + +### `vkWaitForFences` + +```text +VkResult vkWaitForFences( + VkDevice device, + uint32_t fenceCount, + const VkFence* pFences, + VkBool32 waitAll, // VK_TRUE: wait until all signal + uint64_t timeout // nanoseconds; UINT64_MAX = forever +) +``` + +CPU blocks until the fence(s) signal (or timeout). + +### `VkSubmitInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `sType` | `VkStructureType` | `SUBMIT_INFO` | +| `pNext` | `const void*` | Usually `nullptr` | +| `waitSemaphoreCount` | `uint32_t` | GPU waits (0 in basic cthreads copies) | +| `pWaitSemaphores` | `const VkSemaphore*` | Semaphores to wait on | +| `pWaitDstStageMask` | `const VkPipelineStageFlags*` | Stages to wait at | +| `commandBufferCount` | `uint32_t` | How many CBs | +| `pCommandBuffers` | `const VkCommandBuffer*` | CBs to execute | +| `signalSemaphoreCount` | `uint32_t` | Semaphores to signal (often 0) | +| `pSignalSemaphores` | `const VkSemaphore*` | Signal list | + +### `vkQueueSubmit` + +```text +VkResult vkQueueSubmit( + VkQueue queue, + uint32_t submitCount, + const VkSubmitInfo* pSubmits, + VkFence fence // optional; signals when this submit finishes +) +``` + +Returns when the submit is **queued**, not when the GPU finished. Use the fence to wait on the CPU. + +--- + +## 11. Launch path: descriptors + +**Purpose of this layer:** connect shader binding numbers to real `VkBuffer`s. +Without descriptors, a compute shader has no way to know which device memory is +`binding = 1`. Layouts describe the shape; sets hold the actual pointers/handles; +updates and binds install them for a dispatch. + +See handle write-ups for descriptor layout/pool/set in section 1. + +These are required once compute shaders bind GpuPack buffers. + +### `VkDescriptorType` (enum, partial) + +| Value | Meaning | +|-------|---------| +| `VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER` | UBO | +| `VK_DESCRIPTOR_TYPE_STORAGE_BUFFER` | SSBO (cthreads scalars + lists) | +| `VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER_DYNAMIC` | Dynamic-offset UBO | +| `VK_DESCRIPTOR_TYPE_STORAGE_BUFFER_DYNAMIC` | Dynamic-offset SSBO | + +### `VkDescriptorSetLayoutBinding` + +| Field | Type | Meaning | +|-------|------|---------| +| `binding` | `uint32_t` | Binding number (0 = scalars, 1..N = lists) | +| `descriptorType` | `VkDescriptorType` | `STORAGE_BUFFER` for cthreads | +| `descriptorCount` | `uint32_t` | Usually 1 | +| `stageFlags` | `VkShaderStageFlags` | `VK_SHADER_STAGE_COMPUTE_BIT` | +| `pImmutableSamplers` | `const VkSampler*` | nullptr for buffers | + +### `VkShaderStageFlagBits` + +| Flag | Meaning | +|------|---------| +| `VK_SHADER_STAGE_COMPUTE_BIT` | Compute shader | +| `VK_SHADER_STAGE_VERTEX_BIT` | Graphics (unused) | +| `VK_SHADER_STAGE_ALL` | All stages | + +### `VkDescriptorSetLayoutCreateInfo` + +Holds an array of `VkDescriptorSetLayoutBinding`. + +### `vkCreateDescriptorSetLayout` / `vkDestroyDescriptorSetLayout` + +Create/destroy the layout object on a device. + +### `VkDescriptorPoolSize` + +| Field | Type | Meaning | +|-------|------|---------| +| `type` | `VkDescriptorType` | e.g. storage buffer | +| `descriptorCount` | `uint32_t` | How many of that type the pool can allocate | + +### `VkDescriptorPoolCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `flags` | `VkDescriptorPoolCreateFlags` | e.g. free-set bit if freeing individual sets | +| `maxSets` | `uint32_t` | Max sets allocatable | +| `poolSizeCount` | `uint32_t` | Length of pool sizes | +| `pPoolSizes` | `const VkDescriptorPoolSize*` | Capacities per type | + +### `vkCreateDescriptorPool` / `vkDestroyDescriptorPool` + +### `VkDescriptorSetAllocateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `descriptorPool` | `VkDescriptorPool` | Pool | +| `descriptorSetCount` | `uint32_t` | How many sets | +| `pSetLayouts` | `const VkDescriptorSetLayout*` | Layout per set | + +### `vkAllocateDescriptorSets` / `vkFreeDescriptorSets` + +### `VkDescriptorBufferInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `buffer` | `VkBuffer` | Buffer to bind | +| `offset` | `VkDeviceSize` | Start offset | +| `range` | `VkDeviceSize` | Bytes, or `VK_WHOLE_SIZE` | + +### `VkWriteDescriptorSet` + +| Field | Type | Meaning | +|-------|------|---------| +| `dstSet` | `VkDescriptorSet` | Set to update | +| `dstBinding` | `uint32_t` | Binding index | +| `dstArrayElement` | `uint32_t` | Usually 0 | +| `descriptorCount` | `uint32_t` | Usually 1 | +| `descriptorType` | `VkDescriptorType` | Must match layout | +| `pImageInfo` | `const VkDescriptorImageInfo*` | For images; nullptr for buffers | +| `pBufferInfo` | `const VkDescriptorBufferInfo*` | Buffer info | +| `pTexelBufferView` | `const VkBufferView*` | Texel buffers; nullptr | + +### `vkUpdateDescriptorSets` + +```text +void vkUpdateDescriptorSets( + VkDevice device, + uint32_t descriptorWriteCount, + const VkWriteDescriptorSet* pDescriptorWrites, + uint32_t descriptorCopyCount, + const VkCopyDescriptorSet* pDescriptorCopies +) +``` + +CPU-side update; no command buffer needed. + +### `vkCmdBindDescriptorSets` + +```text +void vkCmdBindDescriptorSets( + VkCommandBuffer commandBuffer, + VkPipelineBindPoint pipelineBindPoint, // COMPUTE + VkPipelineLayout layout, + uint32_t firstSet, + uint32_t descriptorSetCount, + const VkDescriptorSet* pDescriptorSets, + uint32_t dynamicOffsetCount, + const uint32_t* pDynamicOffsets +) +``` + +--- + +## 12. Launch path: shaders, pipelines, dispatch + +**Purpose of this layer:** turn SPIR-V into something the GPU can run, then launch it. + +- **Shader module:** raw SPIR-V wrapped as a Vulkan object. +- **Pipeline layout:** declares descriptor/push-constant interface. +- **Compute pipeline:** prepared program ready to bind. +- **Dispatch:** "run N workgroups of this pipeline." + +Recording order in a command buffer is typically: bind pipeline -> bind descriptor +sets -> (optional push constants) -> `vkCmdDispatch`. Then submit + fence wait as in +section 10. + +### `VkShaderModuleCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `flags` | `VkShaderModuleCreateFlags` | Usually 0 | +| `codeSize` | `size_t` | SPIR-V size in **bytes** | +| `pCode` | `const uint32_t*` | SPIR-V words | + +### `vkCreateShaderModule` / `vkDestroyShaderModule` + +### `VkPipelineBindPoint` (enum) + +| Value | Meaning | +|-------|---------| +| `VK_PIPELINE_BIND_POINT_COMPUTE` | Compute pipeline | +| `VK_PIPELINE_BIND_POINT_GRAPHICS` | Graphics (unused) | + +### `VkPipelineShaderStageCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `stage` | `VkShaderStageFlagBits` | `COMPUTE_BIT` | +| `module` | `VkShaderModule` | Shader module | +| `pName` | `const char*` | Entry point, usually `"main"` | +| `pSpecializationInfo` | `const VkSpecializationInfo*` | Optional constants | + +### `VkPipelineLayoutCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `setLayoutCount` | `uint32_t` | Number of descriptor set layouts | +| `pSetLayouts` | `const VkDescriptorSetLayout*` | Layouts | +| `pushConstantRangeCount` | `uint32_t` | 0 if unused | +| `pPushConstantRanges` | `const VkPushConstantRange*` | Optional | + +### `vkCreatePipelineLayout` / `vkDestroyPipelineLayout` + +### `VkComputePipelineCreateInfo` + +| Field | Type | Meaning | +|-------|------|---------| +| `flags` | `VkPipelineCreateFlags` | Usually 0 | +| `stage` | `VkPipelineShaderStageCreateInfo` | Compute stage | +| `layout` | `VkPipelineLayout` | Pipeline layout | +| `basePipelineHandle` | `VkPipeline` | For derivatives; often null | +| `basePipelineIndex` | `int32_t` | Or -1 | + +### `vkCreateComputePipelines` + +```text +VkResult vkCreateComputePipelines( + VkDevice device, + VkPipelineCache pipelineCache, // optional VK_NULL_HANDLE + uint32_t createInfoCount, + const VkComputePipelineCreateInfo* pCreateInfos, + const VkAllocationCallbacks* pAllocator, + VkPipeline* pPipelines +) +``` + +### `vkDestroyPipeline` + +### `vkCmdBindPipeline` + +```text +void vkCmdBindPipeline( + VkCommandBuffer commandBuffer, + VkPipelineBindPoint pipelineBindPoint, + VkPipeline pipeline +) +``` + +### `vkCmdDispatch` + +```text +void vkCmdDispatch( + VkCommandBuffer commandBuffer, + uint32_t groupCountX, + uint32_t groupCountY, + uint32_t groupCountZ +) +``` + +Launches `groupCountX * groupCountY * groupCountZ` workgroups. +Each workgroup size comes from the shader `layout(local_size_x = …)`. + +Rough element coverage for 1D: + +```text +groupsX = ceil(elementCount / local_size_x) +``` + +### `vkCmdPushConstants` (optional) + +```text +void vkCmdPushConstants( + VkCommandBuffer commandBuffer, + VkPipelineLayout layout, + VkShaderStageFlags stageFlags, + uint32_t offset, + uint32_t size, + const void* pValues +) +``` + +Small fast uniforms. In cthreads, scalar SSBO remains writeback source of truth if both are used. + +--- + +## 13. Launch path: barriers (overview) + +**Purpose:** force ordering and memory visibility between GPU commands. The GPU may +overlap or reorder work. A barrier says: "finish these earlier writes, then allow +these later reads/writes." + +**When fences are enough:** if the CPU submits a copy-only command buffer and +`vkWaitForFences` before touching staging memory, the CPU wait provides the needed +ordering for that CPU read. No barrier required for that pattern. + +**When barriers are required:** if the **same** command buffer (or back-to-back GPU +work without a CPU wait) does `copy into buffer Y` then `dispatch that reads Y`, +insert a pipeline barrier between them so the shader cannot read stale data. + +### `VkPipelineStageFlagBits` (partial) + +| Flag | Meaning | +|------|---------| +| `VK_PIPELINE_STAGE_TRANSFER_BIT` | Copy engine | +| `VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT` | Compute shader | +| `VK_PIPELINE_STAGE_HOST_BIT` | Host access | +| `VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT` | Earliest | +| `VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT` | Latest | + +### `VkAccessFlagBits` (partial) + +| Flag | Meaning | +|------|---------| +| `VK_ACCESS_TRANSFER_WRITE_BIT` | Copy write | +| `VK_ACCESS_TRANSFER_READ_BIT` | Copy read | +| `VK_ACCESS_SHADER_READ_BIT` | Shader read | +| `VK_ACCESS_SHADER_WRITE_BIT` | Shader write | +| `VK_ACCESS_HOST_READ_BIT` | CPU read | +| `VK_ACCESS_HOST_WRITE_BIT` | CPU write | + +### `vkCmdPipelineBarrier` + +Records a barrier between earlier and later commands in the same CB. Exact struct +fields (`VkMemoryBarrier`, `VkBufferMemoryBarrier`) are filled when implementing +the launch path; until then, CPU fence waits after submit are enough for +upload/download helpers that do not dispatch in the same CB. + +--- + +## 14. Function pointer typedefs (`PFN_*`) + +Vulkan headers define: + +```text +typedef (VKAPI_PTR *PFN_vkXxx)(); +``` + +Examples: + +| Typedef | Points to | +|---------|-----------| +| `PFN_vkGetInstanceProcAddr` | `vkGetInstanceProcAddr` | +| `PFN_vkCreateInstance` | `vkCreateInstance` | +| `PFN_vkCreateBuffer` | `vkCreateBuffer` | +| `PFN_vkCmdCopyBuffer` | `vkCmdCopyBuffer` | +| … | every entry point | + +cthreads stores these on `Context` and loads them through `vkGetInstanceProcAddr` +(see [05-dynamic-loading.md](./05-dynamic-loading.md)). + +### `vkGetInstanceProcAddr` + +```text +PFN_vkVoidFunction vkGetInstanceProcAddr( + VkInstance instance, // NULL for a few globals; real instance otherwise + const char* pName // e.g. "vkCreateBuffer" +) +``` + +Returns a generic function pointer or `nullptr` if unavailable. + +--- + +## 15. Quick index by name + +### Handles / scalars + +`VkBool32`, `VkDeviceSize`, `VK_NULL_HANDLE`, `VK_WHOLE_SIZE`, +`VkInstance`, `VkPhysicalDevice`, `VkDevice`, `VkQueue`, +`VkBuffer`, `VkDeviceMemory`, `VkCommandPool`, `VkCommandBuffer`, `VkFence`, +`VkDescriptorSetLayout`, `VkDescriptorPool`, `VkDescriptorSet`, +`VkShaderModule`, `VkPipelineLayout`, `VkPipeline`, `VkPipelineCache` + +### Enums / flags (high traffic) + +`VkResult`, `VkStructureType`, `VkPhysicalDeviceType`, +`VkQueueFlagBits`, `VkMemoryPropertyFlagBits`, `VkBufferUsageFlagBits`, +`VkSharingMode`, `VkCommandBufferLevel`, `VkCommandPoolCreateFlagBits`, +`VkCommandBufferUsageFlagBits`, `VkDescriptorType`, `VkShaderStageFlagBits`, +`VkPipelineBindPoint`, `VkPipelineStageFlagBits`, `VkAccessFlagBits` + +### Structs (high traffic) + +`VkApplicationInfo`, `VkInstanceCreateInfo`, +`VkDeviceQueueCreateInfo`, `VkDeviceCreateInfo`, +`VkPhysicalDeviceProperties`, `VkQueueFamilyProperties`, +`VkPhysicalDeviceMemoryProperties`, `VkMemoryType`, `VkMemoryHeap`, +`VkBufferCreateInfo`, `VkMemoryRequirements`, `VkMemoryAllocateInfo`, +`VkCommandPoolCreateInfo`, `VkCommandBufferAllocateInfo`, `VkCommandBufferBeginInfo`, +`VkBufferCopy`, `VkFenceCreateInfo`, `VkSubmitInfo`, +`VkDescriptorSetLayoutBinding`, `VkDescriptorBufferInfo`, `VkWriteDescriptorSet`, +`VkShaderModuleCreateInfo`, `VkPipelineShaderStageCreateInfo`, +`VkPipelineLayoutCreateInfo`, `VkComputePipelineCreateInfo` + +### Functions already used in cthreads GPU code + +`vkGetInstanceProcAddr`, +`vkCreateInstance`, `vkDestroyInstance`, +`vkEnumeratePhysicalDevices`, +`vkGetPhysicalDeviceProperties`, `vkGetPhysicalDeviceQueueFamilyProperties`, +`vkGetPhysicalDeviceMemoryProperties`, +`vkCreateDevice`, `vkDestroyDevice`, `vkGetDeviceQueue`, +`vkCreateBuffer`, `vkDestroyBuffer`, `vkGetBufferMemoryRequirements`, +`vkAllocateMemory`, `vkFreeMemory`, `vkBindBufferMemory`, +`vkMapMemory`, `vkUnmapMemory`, +`vkCreateCommandPool`, `vkDestroyCommandPool`, +`vkAllocateCommandBuffers`, `vkFreeCommandBuffers`, `vkResetCommandBuffer`, +`vkBeginCommandBuffer`, `vkEndCommandBuffer`, `vkCmdCopyBuffer`, +`vkCreateFence`, `vkDestroyFence`, `vkQueueSubmit`, `vkWaitForFences`, `vkResetFences` + +### Functions needed for the launch path (not all may be loaded yet) + +`vkCreateDescriptorSetLayout`, `vkDestroyDescriptorSetLayout`, +`vkCreateDescriptorPool`, `vkDestroyDescriptorPool`, +`vkAllocateDescriptorSets`, `vkFreeDescriptorSets`, `vkUpdateDescriptorSets`, +`vkCreateShaderModule`, `vkDestroyShaderModule`, +`vkCreatePipelineLayout`, `vkDestroyPipelineLayout`, +`vkCreateComputePipelines`, `vkDestroyPipeline`, +`vkCmdBindPipeline`, `vkCmdBindDescriptorSets`, `vkCmdDispatch`, +`vkCmdPipelineBarrier`, `vkCmdPushConstants` (optional) + +When adding any of these in C++, also add the matching `PFN_*` field on `Context`, +resolve it in init, and clear it in shutdown. + +--- + +## How to use this file + +1. When reading `context.cpp` / `memory.cpp`, look up each `Vk*` / `vk*` here. +2. When implementing the launch path, start from sections 11-13 and extend Context PFNs. +3. For intuition (why fences exist, why staging exists), return to chapters 02-08. + +This reference intentionally over-explains. Prefer it over skimming random internet snippets that mix graphics-only APIs into compute work. diff --git a/docs/vk_guide/README.md b/docs/vk_guide/README.md new file mode 100644 index 0000000..5a06ea1 --- /dev/null +++ b/docs/vk_guide/README.md @@ -0,0 +1,55 @@ +# Vulkan guide for cthreads (compute path) + +**Writing `@Gpu` kernels as an application author?** Start with the user guides: +[docs/guide/gpu/README.md](../guide/gpu/README.md). This folder is the deeper +Vulkan substrate tutorial for people changing the C++ GPU backend. + +This folder is a **project-specific** Vulkan tutorial for **cthreads contributors**. +It covers the compute path used by the GPU backend: buffers, copies, descriptors, +shaders, and launch/wait -- **not** the full graphics stack (swapchains, render +passes, images, and so on). + +Architecture choices here match the cthreads GPU design (same Python types as the +CPU backend, GpuPack binding convention, device-local + staging). Implementation status in +the tree may move faster or slower than any particular roadmap document; treat +this guide as the conceptual reference, and the source under `src/cthreads/cpp/gpu/` +as ground truth for what is already landed. + +## Audience + +Contributors who know C++ and the cthreads CPU model (`@Thread`, pack, marshal, +writeback), and who may know little or no Vulkan. That is a normal starting point. + +## How to read + +Read in order the first time. Later use the glossary and the "map to the codebase" +chapter as a reference. + +| # | File | Topics | +|---|------|--------| +| 00 | [00-read-me-first.md](./00-read-me-first.md) | Goals, out of scope, Vulkan vs CUDA/OpenGL | +| 01 | [01-cthreads-gpu-big-picture.md](./01-cthreads-gpu-big-picture.md) | How GPU fits next to CPU pack/marshal | +| 02 | [02-mental-model.md](./02-mental-model.md) | Objects, handles, explicit control | +| 03 | [03-sdk-runtime-drivers.md](./03-sdk-runtime-drivers.md) | SDK vs drivers vs `vulkan-1.dll` | +| 04 | [04-instance-device-queue.md](./04-instance-device-queue.md) | Connecting to a GPU (`Context`) | +| 05 | [05-dynamic-loading.md](./05-dynamic-loading.md) | `LoadLibrary` / `PFN_*` resolution | +| 06 | [06-buffers-and-memory.md](./06-buffers-and-memory.md) | `VkBuffer`, memory types, bind | +| 07 | [07-staging-upload-download.md](./07-staging-upload-download.md) | Device-local + staging copies | +| 08 | [08-commands-fences.md](./08-commands-fences.md) | Command buffers, submit, fences | +| 09 | [09-descriptors-ssbo.md](./09-descriptors-ssbo.md) | How shaders see buffers | +| 10 | [10-std430-layouts.md](./10-std430-layouts.md) | Scalar struct packing | +| 11 | [11-spirv-pipelines-dispatch.md](./11-spirv-pipelines-dispatch.md) | Shaders, pipelines, `dispatch` | +| 12 | [12-gpupack-marshal.md](./12-gpupack-marshal.md) | GpuPack end-to-end | +| 13 | [13-map-to-our-code.md](./13-map-to-our-code.md) | Files in `src/cthreads/cpp/gpu/` | +| 14 | [14-glossary.md](./14-glossary.md) | Terms in one place | +| 15 | [15-checklist.md](./15-checklist.md) | Concepts a contributor should be able to explain | +| 16 | [16-api-reference.md](./16-api-reference.md) | Vulkan types, structs, enums, and functions used by cthreads | + +## Official docs (optional later) + +- [Vulkan Guide](https://vkguide.dev/) — general tutorial (more graphics-heavy) +- [Vulkan Tutorial](https://vulkan-tutorial.com/) — classic; skip swapchain chapters for this project +- [Khronos Vulkan Spec](https://registry.khronos.org/vulkan/) — reference, not a textbook + +This guide is intentionally longer and more hand-holding than those, and tied to +**cthreads** architecture decisions rather than a generic hello-triangle path. diff --git a/pyproject.toml b/pyproject.toml index 2c5c5a2..7862f70 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,10 +11,12 @@ name = "cthreads" version = "0.1.1" description = "Compile @Threadable / @Thread Python into native C++ kernels and run them off the GIL." requires-python = ">=3.10" +license = { file = "LICENSE" } keywords = ["threading", "multithreading", "codegen", "python compiler", "cpp", "pybind11"] classifiers = [ "Development Status :: 3 - Alpha", "Intended Audience :: Developers", + "License :: OSI Approved :: MIT License", "Programming Language :: C++", "Programming Language :: Python :: 3", "Programming Language :: Python :: 3.10", @@ -37,6 +39,9 @@ Documentation = "https://github.com/K-T0BIAS/CThreads/tree/main/docs" [project.optional-dependencies] test = ["pytest>=8"] dev = ["pytest>=8", "build", "scikit-build-core>=0.10"] +# GPU-enabled wheels are a separate PyPI project (same import: cthreads). +# Keep the pin equal to project.version when cutting a release. +gpu = ["cthreads-gpu==0.1.1"] [tool.scikit-build] minimum-version = "0.10" @@ -53,6 +58,7 @@ wheel.exclude = [ "**/*.pyc", ] sdist.include = [ + "LICENSE", "src/cthreads/cpp/**", "src/cthreads/python/cthreads/**", "src/cthreads/python/api/**", diff --git a/scripts/retarget_gpu_wheel.py b/scripts/retarget_gpu_wheel.py new file mode 100644 index 0000000..87e0bfe --- /dev/null +++ b/scripts/retarget_gpu_wheel.py @@ -0,0 +1,52 @@ +#!/usr/bin/env python3 +"""Retarget this tree to build/publish the cthreads-gpu PyPI distribution. + +Pip extras cannot select a different binary for the same project name. GPU-enabled +wheels are therefore published as ``cthreads-gpu`` (same import path ``cthreads``), +and ``pip install cthreads[gpu]`` depends on that package. + +Run from the repo root before cibuildwheel / ``python -m build`` for the GPU job. +""" + +from __future__ import annotations + +from pathlib import Path + + +def main() -> None: + root = Path(__file__).resolve().parents[1] + path = root / "pyproject.toml" + text = path.read_text(encoding="utf-8") + if 'name = "cthreads-gpu"' in text: + print("pyproject.toml already retargeted to cthreads-gpu") + return + if 'name = "cthreads"' not in text: + raise SystemExit("expected name = \"cthreads\" in pyproject.toml") + text = text.replace('name = "cthreads"', 'name = "cthreads-gpu"', 1) + # Avoid a self-referential optional extra on the GPU distribution. + old_extra = 'gpu = ["cthreads-gpu==' + if old_extra in text: + # Replace the gpu extra line with an empty marker list. + lines: list[str] = [] + for line in text.splitlines(keepends=True): + if line.startswith("gpu = ["): + lines.append( + "gpu = [] " + "# GPU wheels are this distribution; extra is a no-op here\n" + ) + else: + lines.append(line) + text = "".join(lines) + desc = 'description = "Compile @Threadable / @Thread Python into native C++ kernels and run them off the GIL."' + gpu_desc = ( + 'description = "cthreads with Vulkan GPU (@Gpu) support built into _ext. ' + 'Install via pip install cthreads[gpu] or pip install cthreads-gpu."' + ) + if desc in text: + text = text.replace(desc, gpu_desc, 1) + path.write_text(text, encoding="utf-8") + print("retargeted pyproject.toml -> name = \"cthreads-gpu\"") + + +if __name__ == "__main__": + main() diff --git a/src/cthreads/cpp/CMakeLists.txt b/src/cthreads/cpp/CMakeLists.txt index 84f84ce..0d71548 100644 --- a/src/cthreads/cpp/CMakeLists.txt +++ b/src/cthreads/cpp/CMakeLists.txt @@ -16,6 +16,13 @@ set(CMAKE_CXX_STANDARD_REQUIRED ON) set(CMAKE_CXX_EXTENSIONS OFF) set(CMAKE_POSITION_INDEPENDENT_CODE ON) +# Enable locally (pick one shell): +# cmd: set CMAKE_ARGS=-DCTHREADS_GPU=ON +# PowerShell: $env:CMAKE_ARGS="-DCTHREADS_GPU=ON" +# either: pip install -e . --config-settings=cmake.define.CTHREADS_GPU=ON +# Wipe build/ if toggling ON/OFF so CMake reconfigures (cached OFF sticks otherwise). +option(CTHREADS_GPU "Build Vulkan GPU support into _ext" OFF) + # --- Python + pybind11 ------------------------------------------------------- find_package(Python COMPONENTS Interpreter Development.Module REQUIRED) @@ -56,7 +63,7 @@ if(UNIX AND NOT APPLE) target_link_libraries(_ext PRIVATE ${CMAKE_DL_LIBS}) endif() -# --- Standalone linalg micro-bench (dot + matmul, light/medium/heavy) -------- +# --- Standalone linalg micro-bench (dot + matmul, light/medium/heavy) (contact @K-T0BIAS for src) -------- set(_CTHREADS_BENCH_SRC "${CMAKE_CURRENT_SOURCE_DIR}/../../../demo/bench/cpp/linalg_bench.cpp" ) @@ -82,7 +89,7 @@ if(EXISTS "${_CTHREADS_BENCH_SRC}") message(STATUS "linalg_bench: ${_CTHREADS_BENCH_SRC}") endif() -# Python package dir (sibling of this cpp/ tree). Never use cpp/ as output/build. +# Python package dir (sibling of this cpp/ tree). set(_cthreads_py_out "${CMAKE_CURRENT_SOURCE_DIR}/../python/cthreads") # Bare cmake (no pip): build straight into the package tree. @@ -132,3 +139,108 @@ message(STATUS "Python: ${Python_EXECUTABLE} (${Python_VERSION})") if(DEFINED SKBUILD_STATE) message(STATUS "SKBUILD_STATE: ${SKBUILD_STATE}") endif() + +# --- GPU ---- + +if(CTHREADS_GPU) + find_package(Vulkan REQUIRED) # vulkan sdk for headers/includes + message(STATUS "cthreads GPU: ON (Vulkan)") + + # Vendor Khronos glslang (GLSL -> SPIR-V). Same compiler engine shaderc uses. + # Linked statically into _ext so end users need no glslc / extra SDK tools. + include(FetchContent) + set(ENABLE_GLSLANG_BINARIES OFF CACHE BOOL "" FORCE) + set(ENABLE_HLSL OFF CACHE BOOL "" FORCE) + set(ENABLE_OPT OFF CACHE BOOL "" FORCE) + set(ENABLE_SPVREMAPPER OFF CACHE BOOL "" FORCE) + set(ENABLE_CTEST OFF CACHE BOOL "" FORCE) + set(SKIP_GLSLANG_INSTALL ON CACHE BOOL "" FORCE) + set(BUILD_EXTERNAL OFF CACHE BOOL "" FORCE) + set(GLSLANG_TESTS OFF CACHE BOOL "" FORCE) + set(ENABLE_GLSLANG_JS OFF CACHE BOOL "" FORCE) + FetchContent_Declare( + glslang + GIT_REPOSITORY https://github.com/KhronosGroup/glslang.git + GIT_TAG 15.1.0 + GIT_SHALLOW TRUE + ) + FetchContent_MakeAvailable(glslang) + message(STATUS "cthreads GPU: vendored glslang for compile_glsl") + + target_sources(_ext PRIVATE + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/context.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/memory.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/pack.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/shader_cache.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/state.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/shader.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/descriptors.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/module.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/impl/compile_glsl.cpp" + "${CMAKE_CURRENT_SOURCE_DIR}/bindings/gpu_module.cpp" + ) + + # Local-only substrate smokes under gpu/testing/ (gitignored via testing/). + # Product GPU builds (CI / wheels) do not need them. Python coverage of the + # public path lives in repo tests/; + set(_cthreads_gpu_testing_pack + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/testing/pack_roundtrip.cpp") + set(_cthreads_gpu_testing_smoke + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/testing/shader_smoke.cpp") + if(EXISTS "${_cthreads_gpu_testing_pack}" AND EXISTS "${_cthreads_gpu_testing_smoke}") + message(STATUS "cthreads GPU: local gpu/testing helpers ON") + target_sources(_ext PRIVATE + "${_cthreads_gpu_testing_pack}" + "${_cthreads_gpu_testing_smoke}" + "${CMAKE_CURRENT_SOURCE_DIR}/bindings/gpu_testing_module.cpp" + ) + target_compile_definitions(_ext PRIVATE CTHREADS_GPU_TESTING=1) + else() + message(STATUS "cthreads GPU: local gpu/testing helpers OFF (not in tree)") + endif() + + target_include_directories(_ext PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/gpu/headers + ${CMAKE_CURRENT_SOURCE_DIR}/gpu + ${Vulkan_INCLUDE_DIRS} + ) + target_compile_definitions(_ext PRIVATE CTHREADS_WITH_GPU=1) + target_link_libraries(_ext PRIVATE + glslang + SPIRV + glslang-default-resource-limits + ) + + # Ship upstream license texts next to the Python package (Apache/BSD notices). + set(_cthreads_gpu_notices_out + "${_cthreads_py_out}/gpu/third_party_notices") + set(_cthreads_gpu_notices_src + "${CMAKE_CURRENT_SOURCE_DIR}/gpu/third_party_notices") + file(MAKE_DIRECTORY "${_cthreads_gpu_notices_out}") + if(EXISTS "${_cthreads_gpu_notices_src}/README.md") + configure_file( + "${_cthreads_gpu_notices_src}/README.md" + "${_cthreads_gpu_notices_out}/README.md" + COPYONLY + ) + endif() + if(DEFINED glslang_SOURCE_DIR) + foreach(_lic IN ITEMS LICENSE.txt LICENSE.TXT LICENSE) + if(EXISTS "${glslang_SOURCE_DIR}/${_lic}") + configure_file( + "${glslang_SOURCE_DIR}/${_lic}" + "${_cthreads_gpu_notices_out}/glslang-${_lic}" + COPYONLY + ) + break() + endif() + endforeach() + endif() + + if(DEFINED SKBUILD AND NOT (DEFINED SKBUILD_STATE AND SKBUILD_STATE STREQUAL "editable")) + install(DIRECTORY "${_cthreads_gpu_notices_out}/" + DESTINATION cthreads/gpu/third_party_notices + OPTIONAL + ) + endif() +endif() diff --git a/src/cthreads/cpp/bindings/gpu_module.cpp b/src/cthreads/cpp/bindings/gpu_module.cpp new file mode 100644 index 0000000..26daa0b --- /dev/null +++ b/src/cthreads/cpp/bindings/gpu_module.cpp @@ -0,0 +1,277 @@ +// Copyright (c) 2026 Tobias Karusseit +// This source code is licensed under the MIT license found in the +// LICENSE file in the root directory of this source tree. + +#include "gpu_module.hpp" +#ifdef CTHREADS_GPU_TESTING +#include "gpu_testing_module.hpp" +#endif + +#include "../gpu/headers/context.hpp" +#include "../gpu/headers/compile_glsl.hpp" +#include "../gpu/headers/memory.hpp" +#include "../gpu/headers/module.hpp" +#include "../gpu/headers/shader_cache.hpp" +#include "../gpu/headers/state.hpp" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace py = pybind11; + +void bind_gpu(py::module_& parent) { + py::module_ g = parent.def_submodule( + "gpu", + "Vulkan GPU runtime (loader dynamically loaded at init)" + ); + + g.def( + "available", + &cthreads::gpu::available, + "True if Vulkan loader + compute device initialized successfully." + ); + + g.def( + "device_name", + &cthreads::gpu::device_name, + "GPU deviceName from Vulkan; calls init() (may raise)." + ); + + g.def( + "init", + &cthreads::gpu::init, + "Explicitly initialize Vulkan context (optional; available/device_name also init)." + ); + + g.def( + "shutdown", + &cthreads::gpu::shutdown, + "Destroy device/instance and unload the Vulkan loader." + ); + + // Product launch handle (mirror of CPU SpawnedKernel / Job surface). + // join() always uses the process Context so Python never holds a Context&. + py::class_< + cthreads::gpu::SpawnedGpuKernel, + std::shared_ptr>(g, "GpuJob") + .def( + "start", + &cthreads::gpu::SpawnedGpuKernel::start, + "No-op: work is submitted at launch_gpu_kernel time." + ) + .def( + "join", + [](cthreads::gpu::SpawnedGpuKernel& self, bool download) { + self.join(cthreads::gpu::context(), download); + }, + py::arg("download") = true, + "Wait for the GPU fence; if download=True, write ref lists back, then " + "release inflight state." + ) + .def( + "done", + &cthreads::gpu::SpawnedGpuKernel::done, + "True after join (or failure) has marked the job complete." + ) + .def( + "wait", + &cthreads::gpu::SpawnedGpuKernel::wait, + py::call_guard(), + "Block until done_flag; does not download. Prefer join()." + ); + + g.def( + "compile_glsl", + [](const std::string& source) { + // glslang can take noticeable time; release the GIL while compiling. + std::vector words; + { + py::gil_scoped_release release; + words = cthreads::gpu::compile_glsl_to_spirv(source); + } + const char* raw = reinterpret_cast(words.data()); + const py::ssize_t nbytes = + static_cast(words.size() * sizeof(std::uint32_t)); + return py::bytes(raw, nbytes); + }, + py::arg("source"), + "Compile a GLSL compute shader string to SPIR-V bytes (vendored glslang). " + "No external glslc / Vulkan SDK tools required." + ); + + g.def( + "register_shader", + [](const std::string& symbol, const py::bytes& spirv, uint32_t binding_count) { + cthreads::gpu::init(); + const std::string raw = spirv; + if (raw.empty() || (raw.size() % 4) != 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: register_shader spirv must " + "be a non-empty multiple of 4 bytes"); + } + if (binding_count == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: register_shader " + "binding_count must be >= 1"); + } + // SPIR-V is a stream of little-endian uint32 words. + std::vector words(raw.size() / 4); + std::memcpy(words.data(), raw.data(), raw.size()); + (void)cthreads::gpu::shader::ShaderRegistry::register_spirv( + cthreads::gpu::context(), + symbol, + words.data(), + words.size(), + binding_count); + }, + py::arg("symbol"), + py::arg("spirv"), + py::arg("binding_count"), + "Create a compute pipeline from SPIR-V and insert it into ShaderCache. " + "Sole product writer path (ShaderRegistry). Duplicate symbol throws." + ); + + g.def( + "launch_gpu_kernel", + &cthreads::gpu::launch_gpu_kernel, + py::arg("meta"), + py::arg("ordered_values"), + "Marshal args from meta + ordered_values, submit compute, return GpuJob. " + "Does not wait; call job.join() for fence wait and list writeback. " + "Requires the kernel symbol to already be in ShaderCache." + ); + + // Named device-local buffer registry (singleton). Not a public DeviceBuffer: + // Python only sees names + sizes; VkBuffer stays inside the registry. + py::class_< + cthreads::gpu::memory::GpuState, + std::unique_ptr< + cthreads::gpu::memory::GpuState, + py::nodelete>>(g, "GpuState") + .def_static( + "instance", + []() -> cthreads::gpu::memory::GpuState& { + return cthreads::gpu::memory::GpuState::getInstance(); + }, + py::return_value_policy::reference, + "Process-wide GpuState singleton." + ) + .def( + "contains", + &cthreads::gpu::memory::GpuState::contains, + py::arg("name"), + "True if name is already registered." + ) + .def( + "size", + &cthreads::gpu::memory::GpuState::size, + "Number of registered buffers." + ) + .def( + "names", + &cthreads::gpu::memory::GpuState::names, + "Snapshot of registered names (order is not meaningful)." + ) + .def( + "add", + [](cthreads::gpu::memory::GpuState& self, + const std::string& name, + std::uint64_t nbytes) { + cthreads::gpu::init(); + auto buffer = cthreads::gpu::memory::create_buffer( + cthreads::gpu::context(), + static_cast(nbytes), + cthreads::gpu::memory::BufferKind::DeviceLocal + ); + self.add(name, std::move(buffer)); + }, + py::arg("name"), + py::arg("nbytes"), + "Allocate a device-local buffer of nbytes and register it under name. " + "Duplicate or empty names raise." + ) + .def( + "remove", + [](cthreads::gpu::memory::GpuState& self, const std::string& name) { + self.remove(cthreads::gpu::context(), name); + }, + py::arg("name"), + "Destroy and unregister name. Raises if unknown or in_use." + ) + .def( + "buffer_size", + [](cthreads::gpu::memory::GpuState& self, const std::string& name) { + return static_cast(self.get(name).size); + }, + py::arg("name"), + "Byte size of the buffer registered under name." + ) + .def( + "is_in_use", + &cthreads::gpu::memory::GpuState::is_in_use, + py::arg("name"), + "True if name is checked out for a launch." + ) + .def( + "mark_in_use", + &cthreads::gpu::memory::GpuState::mark_in_use, + py::arg("name"), + "Check out name for a launch. Raises if unknown or already in_use." + ) + .def( + "release_in_use", + &cthreads::gpu::memory::GpuState::release_in_use, + py::arg("name"), + "Clear in_use after a launch finishes. Raises if unknown or not in_use." + ) + .def( + "upload", + [](cthreads::gpu::memory::GpuState& self, + const std::string& name, + const py::bytes& data) { + cthreads::gpu::init(); + const std::string raw = data; + self.upload( + cthreads::gpu::context(), + name, + raw.empty() ? nullptr : raw.data(), + static_cast(raw.size()) + ); + }, + py::arg("name"), + py::arg("data"), + "Upload host bytes into a registered device-local buffer (H2D)." + ) + .def( + "download", + [](cthreads::gpu::memory::GpuState& self, const std::string& name) { + cthreads::gpu::init(); + const std::uint64_t nbytes = static_cast( + self.get(name).size); + std::string raw(static_cast(nbytes), '\0'); + self.download( + cthreads::gpu::context(), + name, + raw.empty() ? nullptr : raw.data(), + static_cast(nbytes) + ); + return py::bytes(raw); + }, + py::arg("name"), + "Download a registered device-local buffer into bytes (D2H)." + ); + + // Test-only pack/shader smokes: only when local gpu/testing/ is present + // (gitignored). Product CI/wheels build without _ext.gpu.testing. +#ifdef CTHREADS_GPU_TESTING + bind_gpu_testing(g); +#endif +} diff --git a/src/cthreads/cpp/bindings/gpu_module.hpp b/src/cthreads/cpp/bindings/gpu_module.hpp new file mode 100644 index 0000000..a2494af --- /dev/null +++ b/src/cthreads/cpp/bindings/gpu_module.hpp @@ -0,0 +1,12 @@ +#pragma once + +#include + +namespace py = pybind11; + +/** + * Register ``cthreads._ext.gpu`` (probe API, launch_gpu_kernel / GpuJob, + * GpuState singleton). Optional ``testing`` submodule only when + * CTHREADS_GPU_TESTING is set (local gpu/testing/ sources present). + */ +void bind_gpu(py::module_& parent); diff --git a/src/cthreads/cpp/bindings/gpu_testing_module.cpp b/src/cthreads/cpp/bindings/gpu_testing_module.cpp new file mode 100644 index 0000000..e71af90 --- /dev/null +++ b/src/cthreads/cpp/bindings/gpu_testing_module.cpp @@ -0,0 +1,119 @@ +// Copyright (c) 2026 Tobias Karusseit +// This source code is licensed under the MIT license found in the +// LICENSE file in the root directory of this source tree. + +#include "gpu_testing_module.hpp" + +#include "../gpu/testing/pack_roundtrip.hpp" +#include "../gpu/testing/shader_smoke.hpp" + +#include +#include + +#include +#include +#include +#include + +namespace py = pybind11; + +namespace { + +py::bytes vector_to_bytes(const std::vector& data) { + return py::bytes(reinterpret_cast(data.data()), static_cast(data.size())); +} + +} // namespace + +void bind_gpu_testing(py::module_& gpu_parent) { + py::module_ t = gpu_parent.def_submodule( + "testing", + "TEST-ONLY GpuPack helpers. Not a product API; do not use from library code." + ); + + t.def( + "roundtrip_float_pack", + [](const py::bytes& scalar, const std::vector>& lists) { + const std::string raw = scalar; + auto result = cthreads::gpu::testing::roundtrip_float_pack( + raw.empty() ? nullptr : raw.data(), + raw.size(), + lists + ); + return py::make_tuple(vector_to_bytes(result.scalars), py::cast(result.float_lists)); + }, + py::arg("scalar"), + py::arg("float_lists"), + "Upload/download scalar bytes + list[list[float]] through GpuPack (test-only)." + ); + + t.def( + "roundtrip_int_pack", + [](const py::bytes& scalar, const std::vector>& lists) { + const std::string raw = scalar; + auto result = cthreads::gpu::testing::roundtrip_int_pack( + raw.empty() ? nullptr : raw.data(), + raw.size(), + lists + ); + return py::make_tuple(vector_to_bytes(result.scalars), py::cast(result.int_lists)); + }, + py::arg("scalar"), + py::arg("int_lists"), + "Upload/download scalar bytes + list[list[int]] through GpuPack (test-only)." + ); + + t.def( + "probe_invalid_elem_bytes", + &cthreads::gpu::testing::probe_invalid_elem_bytes, + "Raises GpuInvalidArgument (zero elem_bytes). Test-only." + ); + + t.def( + "probe_use_after_destroy_scalars", + &cthreads::gpu::testing::probe_use_after_destroy_scalars, + "Raises GpuUseAfterDestroy (upload into pack with no scalar buffer). Test-only." + ); + + t.def( + "smoke_create_entry", + &cthreads::gpu::testing::smoke_create_entry, + "create_entry + destroy for committed smoke SPIR-V. Test-only." + ); + t.def( + "smoke_update_descriptors", + &cthreads::gpu::testing::smoke_update_descriptors, + "create_entry + pack + update_descriptors smoke. Test-only." + ); + t.def( + "smoke_cache_register_and_get", + &cthreads::gpu::testing::smoke_cache_register_and_get, + "ShaderCache add/get/clear smoke. Test-only." + ); + t.def( + "probe_cache_duplicate_add", + &cthreads::gpu::testing::probe_cache_duplicate_add, + "Raises on duplicate ShaderCache add. Test-only." + ); + t.def( + "probe_update_empty_list_slot", + &cthreads::gpu::testing::probe_update_empty_list_slot, + "Raises GpuInvalidArgument for empty list descriptor. Test-only." + ); + t.def( + "probe_create_entry_zero_bindings", + &cthreads::gpu::testing::probe_create_entry_zero_bindings, + "Raises GpuInvalidArgument for binding_count 0. Test-only." + ); + t.def( + "clear_shader_cache", + &cthreads::gpu::testing::clear_shader_cache, + "Clear process ShaderCache. Test-only teardown." + ); + t.def( + "register_smoke_saxpy", + &cthreads::gpu::testing::register_smoke_saxpy, + "Register smoke saxpy SPIR-V in ShaderCache; returns symbol key. " + "Does not launch - use _ext.gpu.launch_gpu_kernel. Test-only." + ); +} diff --git a/src/cthreads/cpp/bindings/gpu_testing_module.hpp b/src/cthreads/cpp/bindings/gpu_testing_module.hpp new file mode 100644 index 0000000..3f385ed --- /dev/null +++ b/src/cthreads/cpp/bindings/gpu_testing_module.hpp @@ -0,0 +1,11 @@ +#pragma once + +#include + +namespace py = pybind11; + +/** + * Register `cthreads._ext.gpu.testing` (GpuPack round-trip smoke API). + * Test-only. Not part of the public `cthreads.gpu` package. + */ +void bind_gpu_testing(py::module_& gpu_parent); diff --git a/src/cthreads/cpp/bindings/module.cpp b/src/cthreads/cpp/bindings/module.cpp index 1a96d48..25f5e41 100644 --- a/src/cthreads/cpp/bindings/module.cpp +++ b/src/cthreads/cpp/bindings/module.cpp @@ -21,6 +21,10 @@ #include "../headers/pool/threadPool.hpp" #include "../headers/shared_host.hpp" +#ifdef CTHREADS_WITH_GPU +#include "gpu_module.hpp" +#endif + #include #include #include @@ -1255,4 +1259,9 @@ PYBIND11_MODULE(_ext, m) { bind_linalg(m); bind_pool(m); + +// runs when the user has the gpu capable version +#ifdef CTHREADS_WITH_GPU + bind_gpu(m); +#endif } diff --git a/src/cthreads/cpp/gpu/headers/compile_glsl.hpp b/src/cthreads/cpp/gpu/headers/compile_glsl.hpp new file mode 100644 index 0000000..60961c9 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/compile_glsl.hpp @@ -0,0 +1,26 @@ +#pragma once + +#include +#include +#include + +namespace cthreads::gpu { + +/** + * Compile a GLSL compute shader string to SPIR-V words. + * + * Uses the vendored Khronos glslang library (same compiler engine shaderc + * wraps). No external glslc / Vulkan SDK tools required at runtime. + * + * #### Args: + * - source: std::string = full compute GLSL (`#version` + buffers + main) + * + * #### Returns + * - std::vector = SPIR-V words + * + * #### Raises + * - std::runtime_error = parse/link failure (message includes glslang log) + */ +std::vector compile_glsl_to_spirv(const std::string& source); + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/headers/context.hpp b/src/cthreads/cpp/gpu/headers/context.hpp new file mode 100644 index 0000000..a3e29e0 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/context.hpp @@ -0,0 +1,172 @@ +#pragma once +#include +#include +#include +#include +#include + +#include "memory.hpp" + +namespace cthreads::gpu { + +/** + * Reused copy machinery owned by Context for the process lifetime. + * + * Pool + fence are created in init. Staging scratch is optional and may stay + * empty until the first upload/download grows it (see memory::). + * + * Destroy via shutdown_transfer_engine before destroying the logical device. + */ +struct TransferEngine { + VkCommandPool command_pool = VK_NULL_HANDLE; + VkFence fence = VK_NULL_HANDLE; + // Host-visible scratch for H2D/D2H; empty until allocated. Grow-on-demand. + memory::GpuBuffer staging; +}; + +/** + * Process-lifetime launch command pool with per-job CB + fence checkout. + * + * Overlapping jobs each hold one command_buffer and one fence until join. + * The pool itself is never created/destroyed per launch. Free lists grow on + * demand; checkout resets a returned CB/fence before reuse. + * + * Guard with Context::launch_engine_mutex for checkout, return, and queue submit. + */ +struct LaunchEngine { + VkCommandPool command_pool = VK_NULL_HANDLE; + std::vector free_command_buffers; + std::vector free_fences; +}; + +/** + * Per-job handles checked out from LaunchEngine (not owned by the job forever). + * Return via return_launch_resources after the fence has been waited. + */ +struct LaunchResources { + VkCommandBuffer command_buffer = VK_NULL_HANDLE; + VkFence fence = VK_NULL_HANDLE; +}; + +struct Context { + // OS handle to the loader shared library (HMODULE on Windows, void* on Linux). + // Purpose: keep the DLL mapped and FreeLibrary/dlclose on shutdown. + void* loader_module = nullptr; + + // Bootstrap entry from the loader. Type: pointer-to-function matching + // VkResult-less GetInstanceProcAddr signature from the headers. + // Usage: resolve almost every other Vulkan function by name string. + // Naming Note: GetPRocAddress => GetFunctionAddress (Proc -> procedure -> function) + PFN_vkGetInstanceProcAddr vkGetInstanceProcAddr = nullptr; + // Global (pre-instance) / instance-level entry points. + // Each is a typed function pointer; assigned in init() via GetInstanceProcAddr. + PFN_vkCreateInstance vkCreateInstance = nullptr; // creates the Vulkan instance (app <-> loader connection) + PFN_vkDestroyInstance vkDestroyInstance = nullptr; // destroys the instance and its child resources owned at instance level + PFN_vkEnumeratePhysicalDevices vkEnumeratePhysicalDevices = nullptr; // lists GPUs the loader can see + PFN_vkGetPhysicalDeviceProperties vkGetPhysicalDeviceProperties = nullptr; // reads name, type, limits for one GPU + PFN_vkGetPhysicalDeviceQueueFamilyProperties vkGetPhysicalDeviceQueueFamilyProperties = nullptr; // lists queue families (graphics/compute/transfer) on one GPU + PFN_vkCreateDevice vkCreateDevice = nullptr; // opens a logical device on a chosen physical GPU + PFN_vkDestroyDevice vkDestroyDevice = nullptr; // destroys the logical device + PFN_vkGetDeviceQueue vkGetDeviceQueue = nullptr; // gets a queue handle used to submit work + + // Buffer and memory functions + PFN_vkCreateBuffer vkCreateBuffer = nullptr; // creates a buffer object (byte region descriptor; needs memory bound) + PFN_vkDestroyBuffer vkDestroyBuffer = nullptr; // destroys a buffer object + PFN_vkGetBufferMemoryRequirements vkGetBufferMemoryRequirements = nullptr; // reports size, alignment, and allowed memory types for a buffer + PFN_vkAllocateMemory vkAllocateMemory = nullptr; // allocates a block of device (or host-visible) memory + PFN_vkFreeMemory vkFreeMemory = nullptr; // frees a memory block from AllocateMemory + PFN_vkBindBufferMemory vkBindBufferMemory = nullptr; // attaches a memory block to a buffer at an offset + PFN_vkMapMemory vkMapMemory = nullptr; // exposes host-visible memory as a CPU pointer for read/write + PFN_vkUnmapMemory vkUnmapMemory = nullptr; // releases a CPU mapping from MapMemory + PFN_vkGetPhysicalDeviceMemoryProperties vkGetPhysicalDeviceMemoryProperties = nullptr; // lists memory heaps/types and their flags (e.g. host-visible, device-local) + + // Command pool, command buffer, copy, and sync functions + PFN_vkCreateCommandPool vkCreateCommandPool = nullptr; // creates a pool that owns command buffers for one queue family + PFN_vkDestroyCommandPool vkDestroyCommandPool = nullptr; // destroys a command pool and its buffers + PFN_vkAllocateCommandBuffers vkAllocateCommandBuffers = nullptr; // allocates one or more command buffers from a pool + PFN_vkFreeCommandBuffers vkFreeCommandBuffers = nullptr; // returns command buffers to the pool / frees them + PFN_vkResetCommandBuffer vkResetCommandBuffer = nullptr; // clears a command buffer so it can be recorded again + PFN_vkBeginCommandBuffer vkBeginCommandBuffer = nullptr; // starts recording commands into a command buffer + PFN_vkEndCommandBuffer vkEndCommandBuffer = nullptr; // finishes recording; buffer is ready to submit + PFN_vkCmdCopyBuffer vkCmdCopyBuffer = nullptr; // records a GPU copy from one buffer to another + PFN_vkCreateFence vkCreateFence = nullptr; // creates a fence (CPU waits until GPU work finishes) + PFN_vkDestroyFence vkDestroyFence = nullptr; // destroys a fence + PFN_vkQueueSubmit vkQueueSubmit = nullptr; // submits recorded command buffers to a queue + PFN_vkWaitForFences vkWaitForFences = nullptr; // blocks the CPU until the given fences signal + PFN_vkResetFences vkResetFences = nullptr; // resets fences back to unsignaled for reuse + + // Shader / pipeline create + destroy (entry build and ShaderCache::clear). + PFN_vkCreateShaderModule vkCreateShaderModule = nullptr; + PFN_vkDestroyShaderModule vkDestroyShaderModule = nullptr; + PFN_vkCreateDescriptorSetLayout vkCreateDescriptorSetLayout = nullptr; + PFN_vkDestroyDescriptorSetLayout vkDestroyDescriptorSetLayout = nullptr; + PFN_vkCreatePipelineLayout vkCreatePipelineLayout = nullptr; + PFN_vkDestroyPipelineLayout vkDestroyPipelineLayout = nullptr; + PFN_vkCreateComputePipelines vkCreateComputePipelines = nullptr; + PFN_vkDestroyPipeline vkDestroyPipeline = nullptr; + + // Descriptor pool / set / update (per-launch wiring of GpuPack buffers). + PFN_vkCreateDescriptorPool vkCreateDescriptorPool = nullptr; + PFN_vkDestroyDescriptorPool vkDestroyDescriptorPool = nullptr; + PFN_vkAllocateDescriptorSets vkAllocateDescriptorSets = nullptr; + PFN_vkFreeDescriptorSets vkFreeDescriptorSets = nullptr; + PFN_vkUpdateDescriptorSets vkUpdateDescriptorSets = nullptr; + + // Compute dispatch recording (launch_gpu_kernel command buffers). + PFN_vkCmdBindPipeline vkCmdBindPipeline = nullptr; + PFN_vkCmdBindDescriptorSets vkCmdBindDescriptorSets = nullptr; + PFN_vkCmdDispatch vkCmdDispatch = nullptr; + PFN_vkCmdPipelineBarrier vkCmdPipelineBarrier = nullptr; + + // Opaque Vulkan handles. + VkInstance instance = VK_NULL_HANDLE; // connection to the loader/app + VkPhysicalDevice physical_device = VK_NULL_HANDLE; // chosen GPU + VkDevice device = VK_NULL_HANDLE; // logical device (opened GPU) + VkQueue queue = VK_NULL_HANDLE; // compute submission port + // Which queue family index we passed to vkCreateDevice (needed for pools). + uint32_t queue_family = 0; + // Human-readable GPU name from VkPhysicalDeviceProperties::deviceName. + std::string device_name; + // True only after init() fully succeeded. + bool ready = false; + + TransferEngine transfer_engine; + std::mutex transfer_engine_mutex; + + LaunchEngine launch_engine; + std::mutex launch_engine_mutex; +}; +// Process-wide singleton accessor. +Context& context(); +// Create loader + instance + device + queue. Throws on failure. +void init(); +// Destroy device/instance; unload loader; clear pointers. Safe to call if not ready. +void shutdown(); +// If not ready, try init once; return ready without throwing (for available()). +bool available(); +// Requires ready context; returns device_name. +const std::string& device_name(); + +/** + * Checkout a command buffer + fence from Context::launch_engine. + * Caller records the CB, then submit_launch, then join waits the fence, then + * return_launch_resources. Holds launch_engine_mutex only for the checkout. + */ +LaunchResources checkout_launch_resources(Context& context); + +/** + * Return CB + fence to the free lists after the fence has been waited. + * Clears the handles in resources. Holds launch_engine_mutex. + */ +void return_launch_resources(Context& context, LaunchResources& resources); + +/** + * vkQueueSubmit for a launch CB under launch_engine_mutex (same lock domain as checkout). + */ +void submit_launch( + Context& context, + VkCommandBuffer command_buffer, + VkFence fence +); + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/headers/descriptors.hpp b/src/cthreads/cpp/gpu/headers/descriptors.hpp new file mode 100644 index 0000000..f64dd37 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/descriptors.hpp @@ -0,0 +1,168 @@ +#pragma once + +#include +#include + +#include "pack.hpp" +#include "shader_cache.hpp" + +namespace cthreads::gpu { +struct Context; +} + +/** + * Per-launch descriptor helpers for GpuPack wiring (namespace pack). + * + * The set layout and pipeline live on ShaderCacheEntry (created once per + * symbol). Each job allocates a set, calls update_descriptors with that job's + * GpuPack, then binds the set at dispatch time. + * + * #### Technical terms: + * - descriptor set layout: schema of bindings (from create_entry / cache). + * - descriptor pool: allocator from which concrete descriptor sets are taken. + * - descriptor set: one instance of the layout; holds VkBuffer pointers for a launch. + * - update_descriptors: writes binding i -> GpuPack buffer i (scalars then lists). + */ +namespace cthreads::gpu::pack { + +/** + * Pool that can allocate descriptor sets for a fixed STORAGE_BUFFER binding count. + * + * Created with FREE_DESCRIPTOR_SET_BIT so sets can be returned with free_set + * when a job finishes. Destroy the pool only after all sets from it are freed + * (or the device is shutting down and no jobs remain). + * + * #### Fields: + * - pool: VkDescriptorPool = Vulkan pool handle; VK_NULL_HANDLE if empty. + * - max_sets: uint32_t = how many sets this pool can still hold at create time. + * - binding_count: uint32_t = STORAGE_BUFFER descriptors per set (binding convention). + */ +struct DescriptorPool { + VkDescriptorPool pool = VK_NULL_HANDLE; + uint32_t max_sets = 0; + uint32_t binding_count = 0; +}; + +/** + * Creates a descriptor pool for compute sets that follow the binding convention. + * + * Each set needs binding_count STORAGE_BUFFER descriptors. The pool can + * allocate up to max_sets such sets. Sets may be freed individually. + * + * #### Parameters: + * - context: Context& = initialized GPU context with descriptor pool entry points. + * - binding_count: uint32_t = bindings per set (must match ShaderCacheEntry). + * - max_sets: uint32_t = maximum sets this pool may allocate (>= 1). + * + * #### Returns: + * - DescriptorPool = owned pool; caller must destroy_pool when done. + * + * #### Throws: + * - runtime_error if context is not ready, counts are 0, PFNs are missing, or + * vkCreateDescriptorPool fails. + */ +DescriptorPool create_pool( + cthreads::gpu::Context& context, + uint32_t binding_count, + uint32_t max_sets +); + +/** + * Destroys a descriptor pool and resets the DescriptorPool fields. + * + * All sets allocated from this pool become invalid. Prefer free_set on each + * live set first when jobs may still hold sets. + * + * #### Parameters: + * - context: Context& = same device that created the pool. + * - pool: DescriptorPool& = pool to destroy; left empty on return. + */ +void destroy_pool( + cthreads::gpu::Context& context, + DescriptorPool& pool +); + +/** + * Allocates one descriptor set from the pool using the given set layout. + * + * The layout must match the pool's binding_count (same layout used in + * ShaderCacheEntry::set_layout). The set is empty until update_descriptors. + * + * #### Parameters: + * - context: Context& = initialized GPU context. + * - pool: DescriptorPool& = pool with remaining capacity. + * - set_layout: VkDescriptorSetLayout = layout from ShaderCacheEntry. + * + * #### Returns: + * - VkDescriptorSet = allocated set (not a owned C++ type; free with free_set). + * + * #### Throws: + * - runtime_error if the pool is empty/invalid, layout is null, or allocate fails. + */ +VkDescriptorSet allocate_set( + cthreads::gpu::Context& context, + DescriptorPool& pool, + VkDescriptorSetLayout set_layout +); + +/** + * Returns a set to its pool. Safe no-op if set is VK_NULL_HANDLE. + * + * #### Parameters: + * - context: Context& = same device as the pool. + * - pool: DescriptorPool& = pool that allocated the set. + * - set: VkDescriptorSet& = set to free; set to VK_NULL_HANDLE on return. + * + * #### Throws: + * - runtime_error if free fails (pool must have been created with free bit). + */ +void free_set( + cthreads::gpu::Context& context, + DescriptorPool& pool, + VkDescriptorSet& set +); + +/** + * Writes buffer bindings from a GpuPack into a descriptor set. + * + * If pack has a scalar buffer: binding 0 = scalars, 1..N = lists. + * If not: binding 0..N-1 = lists (no scalar descriptor). + * binding_count must equal (has_scalars ? 1 : 0) + container_slots.size(). + * Every written binding must have a non-null VkBuffer. + * + * Called by the launch path after allocate_set and before recording bind/dispatch. + * + * #### Parameters: + * - context: Context& = initialized GPU context with vkUpdateDescriptorSets. + * - set: VkDescriptorSet = destination set from allocate_set. + * - binding_count: uint32_t = number of STORAGE_BUFFER bindings to write. + * - pack: const GpuPack& = source buffers (optional scalars, then lists). + * + * #### Throws: + * - runtime_error if set is null, binding_count mismatches the pack, any + * required buffer handle is null, or update entry points are missing. + */ +void update_descriptors( + cthreads::gpu::Context& context, + VkDescriptorSet set, + uint32_t binding_count, + const GpuPack& pack +); + +/** + * Convenience: update_descriptors using entry.binding_count. + * + * #### Parameters: + * - context: Context& = initialized GPU context. + * - set: VkDescriptorSet = destination set. + * - entry: const ShaderCacheEntry& = cached layout metadata (binding_count). + * - pack: const GpuPack& = source buffers. + */ +void update_descriptors( + cthreads::gpu::Context& context, + VkDescriptorSet set, + const cthreads::gpu::shader::ShaderCacheEntry& entry, + const GpuPack& pack +); + +} // namespace cthreads::gpu::pack diff --git a/src/cthreads/cpp/gpu/headers/inflight_store.hpp b/src/cthreads/cpp/gpu/headers/inflight_store.hpp new file mode 100644 index 0000000..6f70f09 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/inflight_store.hpp @@ -0,0 +1 @@ +#pragma once diff --git a/src/cthreads/cpp/gpu/headers/memory.hpp b/src/cthreads/cpp/gpu/headers/memory.hpp new file mode 100644 index 0000000..88b4702 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/memory.hpp @@ -0,0 +1,194 @@ +#pragma once + +#include +#include + +namespace cthreads::gpu { +struct Context; +} + +/** + * Stateless helpers for one GPU byte region at a time: allocate, free, and + * copy between host memory and device-local storage buffers. + * + * Shader-facing buffers are device-local. Host traffic goes through the + * Context TransferEngine: reused command pool + fence, and a grow-on-demand + * host-visible staging scratch. Pass a ready Context each time. GpuPack builds + * on top. + * + * #### Technical terms: + * - Context: process-wide Vulkan connection (device, queue, loaded entry points). + * - host: CPU / process memory used by C++ and Python. + * - device-local: GPU memory optimized for shader access; not persistently mapped. + * - staging: host-visible buffer used only as a temporary for uploads/downloads. + * - SSBO: storage buffer - a buffer shaders can read and write. + */ +namespace cthreads::gpu::memory { + +/** + * Which memory path create_buffer should use. + * + * One create API, two property sets - not two parallel designs. + */ +enum class BufferKind : uint8_t { + // Host-visible + coherent; used only as the memcpy side of transfers. + Staging = 0, + // Device-local storage buffer for shaders (scalar SSBO or one list SSBO). + DeviceLocal = 1, +}; + +/** + * One contiguous byte region on the GPU (staging scratch, scalar SSBO, or one list). + * + * Owns the Vulkan buffer object and the device memory bound to it. GpuPacks + * use several device-local GpuBuffers: one for all scalars, then one per list. + * + * #### Fields: + * - buffer: VkBuffer = Vulkan handle for the buffer object. VK_NULL_HANDLE if empty. + * - memory: VkDeviceMemory = allocated memory slab bound to buffer. VK_NULL_HANDLE if empty. + * - size: VkDeviceSize = caller-facing byte count requested at create (0 if empty). + * - mapped: void* = CPU pointer when this is a mapped staging buffer; nullptr for + * device-local buffers (they are never persistently mapped). + * - kind: BufferKind = Staging or DeviceLocal; drives destroy and upload/download. + * + * #### Technical terms: + * - VkBuffer: opaque id for a buffer resource (not a raw C pointer to bytes). + * - VkDeviceMemory: opaque id for an allocated memory block from the driver. + * - VK_NULL_HANDLE: sentinel meaning no object. + */ +struct GpuBuffer { + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + VkDeviceSize size = 0; + void* mapped = nullptr; + BufferKind kind = BufferKind::DeviceLocal; +}; + +/** + * Picks which memory type index on this GPU can back a new buffer. + * + * Vulkan gives a bitmask of allowed types for a buffer (type_bits). You also + * request properties (for example host-visible, or device-local). This walks the + * device memory types and returns the first index that is allowed and has every + * requested property flag. + * + * #### Parameters: + * - context: Context& = live GPU context (needs physical device + memory query entry point). + * - type_bits: uint32_t = bit i set means memory type i is allowed (from memory requirements). + * - properties: VkMemoryPropertyFlags = required flags for that type. + * + * #### Returns: + * - uint32_t = memory type index for vkAllocateMemory. + * + * #### Throws: + * - runtime_error if context is not ready or no type matches. + * + * #### Technical terms: + * - memory type: one of the heaps/types the driver exposes (device-local, host-visible, etc.). + * - type_bits: bitmask; bit N means this buffer may use memory type N. + */ +uint32_t find_memory_type( + cthreads::gpu::Context& context, + uint32_t type_bits, + VkMemoryPropertyFlags properties +); + +/** + * Creates a GpuBuffer of the given kind and size. + * + * Staging: host-visible + coherent, transfer usage, left mapped for memcpy. + * DeviceLocal: device-local memory, storage + transfer usage, not mapped. + * Size must be greater than 0 (empty lists are handled by GpuPack, not a zero-size buffer). + * + * #### Parameters: + * - context: Context& = initialized device and buffer/memory entry points. + * - size: VkDeviceSize = number of bytes to store (must be > 0). + * - kind: BufferKind = Staging or DeviceLocal. + * + * #### Returns: + * - GpuBuffer = owned handles; caller must destroy_buffer when done. + * + * #### Throws: + * - runtime_error on zero size, missing entry points, or Vulkan create/allocate/bind/map failure. + */ +GpuBuffer create_buffer( + cthreads::gpu::Context& context, + VkDeviceSize size, + BufferKind kind +); + +/** + * Releases a GpuBuffer and clears its fields. + * + * Unmaps if mapped, destroys the VkBuffer, frees VkDeviceMemory, then zeroes the + * handle. Safe no-op if the buffer is already empty. + * + * #### Parameters: + * - context: Context& = same device that created the buffer. + * - buffer: GpuBuffer& = allocation to free; left empty after return. + * + * #### Returns: + * - void + */ +void destroy_buffer(cthreads::gpu::Context& context, GpuBuffer& buffer); + +/** + * Copies host bytes into a device-local GpuBuffer (host to device). + * + * Final path: memcpy into Context TransferEngine staging, then GPU-copy into + * the device-local buffer (engine pool + fence). + * + * #### Parameters: + * - context: Context& = device + transfer engine entry points. + * - buffer: GpuBuffer& = device-local destination; must be large enough. + * - data: const void* = host source bytes. + * - size: VkDeviceSize = bytes to copy; must be <= buffer.size. + * + * #### Returns: + * - void + * + * #### Throws: + * - runtime_error if buffer is not device-local, data is null, size is invalid, + * transfer engine is missing, or the transfer fails. + * + * #### Technical terms: + * - upload: copy from host (CPU) toward device (GPU) memory. + */ +void upload_buffer( + cthreads::gpu::Context& context, + GpuBuffer& buffer, + const void* data, + VkDeviceSize size +); + +/** + * Copies bytes from a device-local GpuBuffer into host memory (device to host). + * + * Final path: GPU-copy device-local into TransferEngine staging, then memcpy + * staging to data (engine pool + fence). + * + * #### Parameters: + * - context: Context& = device + transfer engine entry points. + * - buffer: GpuBuffer& = device-local source; must be large enough. + * - data: void* = host destination bytes. + * - size: VkDeviceSize = bytes to copy; must be <= buffer.size. + * + * #### Returns: + * - void + * + * #### Throws: + * - runtime_error if buffer is not device-local, data is null, size is invalid, + * transfer engine is missing, or the transfer fails. + * + * #### Technical terms: + * - download: copy from device (GPU) memory to host (CPU). + * - writeback: copying native results into the same host/Python objects the caller passed in. + */ +void download_buffer( + cthreads::gpu::Context& context, + GpuBuffer& buffer, + void* data, + VkDeviceSize size +); + +} // namespace cthreads::gpu::memory diff --git a/src/cthreads/cpp/gpu/headers/module.hpp b/src/cthreads/cpp/gpu/headers/module.hpp new file mode 100644 index 0000000..dba01bd --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/module.hpp @@ -0,0 +1,221 @@ +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "pack.hpp" +#include "descriptors.hpp" + +namespace py = pybind11; + +namespace cthreads::gpu { + +struct Context; + +/** + * Per-launch GPU job handle (mirror of CPU SpawnedKernel, without an OS thread). + * + * CPU kernels run on a CThread. GPU kernels run on the device after + * vkQueueSubmit; this struct owns the host-side inflight state and waits on a + * fence in join(). There is no CGpuThread. + * + * Typical lifetime: + * 1. Launch path fills pack, descriptor set, records/submits with fence. + * 2. Caller may overlap other host work. + * 3. join() waits on the fence, downloads, writeback, then releases GPU objects. + * + * Methods are declared here; record/submit/join bodies land with the launch path. + * + * #### Fields: + * - pack: GpuPack = device-local scalar + list SSBOs for this launch. + * - descriptor_pool: DescriptorPool = pool that allocated descriptor_set (for free_set). + * - descriptor_set: VkDescriptorSet = bindings wired to pack buffers. + * - command_buffer: VkCommandBuffer = checked out from Context LaunchEngine. + * - command_pool: VkCommandPool = Context launch pool (borrowed; not destroyed on join). + * - fence: VkFence = checked out per job; returned to LaunchEngine after wait. + * - symbol: string = shader cache key for this kernel. + * - group_count_x/y/z: uint32_t = vkCmdDispatch workgroup counts. + * - values_keep: shared_ptr to py::list = Python args kept alive for list writeback. + * - writeback_lists: plan of ref list slots to download into values_keep on join. + * - finished: bool = true after join completed writeback / teardown. + * - eptr: exception_ptr = error captured during launch or join (rethrown on join). + * + * #### Technical terms: + * - Fence: GPU timeline object the CPU waits on until submitted work completes. + * - Descriptor set: per-launch table binding i -> pack buffer i. + * - GpuPack: binding-convention buffer bag (scalars at 0, lists at 1..N). + */ +struct SpawnedGpuKernel { + /** + * One list argument to download into the kept Python list on join. + * Built at launch from meta (pass_as ref) + live numel. + */ + struct WritebackListSlot { + size_t value_index = 0; // index into values_keep + size_t container_index = 0; // index into pack.container_slots + size_t numel = 0; + std::string elem_kind; // "float" / "int" / "double" / "bool" + + }; + + pack::GpuPack pack{}; + pack::DescriptorPool descriptor_pool{}; + VkDescriptorSet descriptor_set = VK_NULL_HANDLE; + VkCommandBuffer command_buffer = VK_NULL_HANDLE; + VkCommandPool command_pool = VK_NULL_HANDLE; + VkFence fence = VK_NULL_HANDLE; + + std::string symbol; + uint32_t group_count_x = 1; + uint32_t group_count_y = 1; + uint32_t group_count_z = 1; + + // Same Python list objects the caller passed (CPU SpawnedKernel values_keep). + std::shared_ptr values_keep; + // Ref list slots only; value scalars are not written back. + std::vector writeback_lists; + + // Parallel to pack.container_slots: true => destroy_buffer on release. + // False => borrowed from GpuState; handles cleared without destroy. + std::vector container_owned; + // GpuState names marked in_use for this launch; released in release_inflight. + std::vector resident_names; + + bool finished = false; + std::mutex done_mu; + std::condition_variable done_cv; + bool done_flag = false; + std::exception_ptr eptr; + + SpawnedGpuKernel() = default; + SpawnedGpuKernel(const SpawnedGpuKernel&) = delete; + SpawnedGpuKernel& operator=(const SpawnedGpuKernel&) = delete; + SpawnedGpuKernel(SpawnedGpuKernel&&) = delete; + SpawnedGpuKernel& operator=(SpawnedGpuKernel&&) = delete; + + /** + * No-op for the default path (work is submitted at launch time). + * Kept so Python Job-shaped wrappers can share a start/join/done surface + * with CPU SpawnedKernel. + */ + void start(); + + /** + * Wait until the GPU fence signals, optionally download/writeback, then + * release inflight GPU objects. Rethrows eptr if set. Idempotent after finished. + * + * #### Parameters: + * - context: Context& = same device that created pack / submitted work. + * - download: bool = if true (default), download ref lists into values_keep. + * If false, skip writeback (resident buffers stay device-authoritative). + */ + void join(Context& context, bool download = true); + + /** + * Block until done_flag is set (join or failure path). Does not download. + * Prefer join() for the full writeback and teardown path. + */ + void wait(); + + /** + * True after the GPU work is complete (done_flag). Does not perform writeback. + */ + bool done(); + + /** + * Destroy remaining Vulkan objects if join was never called. + * Safe if already finished / empty. + */ + ~SpawnedGpuKernel(); +}; + +/** + * GPU analog of CPU spawn_from_meta: marshal args, submit compute, return a job. + * + * Ensures the process Context is ready, resolves the kernel from meta (shader + * cache symbol / SPIR-V), builds and uploads a GpuPack from ordered_values, + * allocates and updates a descriptor set, records bind+dispatch, submits with + * a fence, and returns an owned SpawnedGpuKernel. Does not wait; call + * job->join(context) for fence wait, download, and writeback. + * + * Unlike spawn_from_meta there is no CThread and no pool argument. Work runs on + * the GPU after vkQueueSubmit on the calling host thread. + * + * #### Parameters: + * - meta: py::dict = kernel metadata (symbol, binding layout / scalar size, + * container specs, dispatch size, and later types/schemas for writeback). + * Shape will align with @Gpu __kernel_meta__ when emit exists; smoke tests may + * pass a minimal dict. + * - ordered_values: py::list = Python args in parameter order matching meta + * (scalars flattened into the scalar SSBO; lists map to bindings 1..N). + * + * #### Returns: + * - shared_ptr = inflight job (pack, descriptors, fence). + * Caller owns the pointer until join/destroy. + * + * #### Throws: + * - runtime_error / type_error style errors if Context is not ready, meta is + * incomplete, arity mismatches, cache/SPIR-V is missing, or Vulkan submit fails. + * + * #### Example meta shape (saxpy-style @Gpu): + * ``py + * { + * "symbol": "saxpy", # ShaderCache key / kernel name + * "binding_count": 3, # STORAGE_BUFFER bindings (0 scalars + 2 lists) + * "scalar_bytes": 8, # std430 scalar SSBO size (e.g. int n + float a) + * "local_size_x": 64, # compute workgroup size (shader layout) + * "group_count_x": None, # optional override; else ceil(n / local_size_x) + * "group_count_y": 1, + * "group_count_z": 1, + * "params": [ + * { + * "name": "n", + * "kind": "int", # packed into scalar SSBO (binding 0) + * "pass_as": "value", + * }, + * { + * "name": "a", + * "kind": "float", + * "pass_as": "value", + * }, + * { + * "name": "x", + * "kind": "list", # binding 1 + * "pass_as": "ref", + * "elem_kind": "float", + * "elem_bytes": 4, + * }, + * { + * "name": "y", + * "kind": "list", # binding 2 + * "pass_as": "ref", + * "elem_kind": "float", + * "elem_bytes": 4, + * }, + * ], + * "types": {}, # optional: Python classes for writeback + * "schemas": {}, # optional: field layouts for Threadables + * } + * `` + * ordered_values for that meta would be like ``[n, a, x_list, y_list]``. + * Scalars are flattened into one SSBO in param order; each list gets its own + * binding 1..N. Smoke tests may omit types/schemas and pass a smaller dict. + * + * #### Technical terms: + * - spawn_from_meta: CPU launcher that builds SpawnedKernel + CThread/pool task. + * - Shader cache: process map of symbol -> reusable pipeline and set layout. + * - Descriptor set: per-launch binding table from GpuPack buffers to the shader. + */ +std::shared_ptr launch_gpu_kernel( + py::dict meta, + py::list ordered_values +); + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/headers/pack.hpp b/src/cthreads/cpp/gpu/headers/pack.hpp new file mode 100644 index 0000000..d989a82 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/pack.hpp @@ -0,0 +1,227 @@ +#pragma once + +#include +#include +#include + +#include "memory.hpp" + +namespace cthreads::gpu { +struct Context; +} + +/** + * GpuPack: one device-local scalar SSBO plus one device-local SSBO per + * list/container. Host traffic goes through memory:: upload/download helpers. + * Descriptor helpers wire a pack into a per-launch descriptor set for dispatch. + * + * This is a generic runtime bag of buffers. Per-kernel std430 layout and which + * Python arg maps to which slot are marshal/codegen concerns, not this type. + * + * #### Technical terms: + * - scalar SSBO: single buffer holding all packed scalar bytes for one launch. + * - container slot: one list (or similar) argument; empty numel means no VkBuffer. + * - upload / download: host <-> device-local copy via staging (see memory::). + */ +namespace cthreads::gpu::pack { + +/** + * How large one container slot should be at create time. + * + * #### Fields: + * - elem_bytes: size_t = bytes per element (e.g. 4 for float or int32). + * - numel: size_t = element count. 0 means no buffer is allocated for that slot. + */ +struct ContainerSpec { + size_t elem_bytes = 0; + size_t numel = 0; +}; + +/** + * One list/container argument inside a GpuPack. + * + * #### Fields: + * - buffer: GpuBuffer = device-local SSBO when numel > 0; empty handles when numel == 0. + * - spec: ContainerSpec = elem size and count used at create (and for bounds checks). + */ +struct ContainerSlot { + cthreads::gpu::memory::GpuBuffer buffer; + ContainerSpec spec; +}; + +/** + * Per-launch GPU argument pack (binding convention: scalars then lists). + * + * Owns Vulkan allocations until destroy_gpu_pack. Returning GpuPack by value + * moves handles only; device bytes are not copied. + * + * #### Fields: + * - scalar_buffer: GpuBuffer = device-local blob for all scalars (empty if scalar_bytes was 0). + * - container_slots: vector of ContainerSlot = binding order 1..N matching create specs. + */ +struct GpuPack { + cthreads::gpu::memory::GpuBuffer scalar_buffer; + std::vector container_slots; +}; + +/** + * Allocates a GpuPack: device-local scalar buffer (if scalar_bytes > 0) and one + * device-local buffer per container with numel > 0. + * + * Empty containers (numel == 0) keep a slot with no VkBuffer. Zero-size Vulkan + * buffers are never created. + * + * #### Parameters: + * - context: Context& = initialized GPU context with buffer/memory entry points. + * - scalar_bytes: size_t = byte size of the scalar blob (0 = no scalar buffer). + * - container_specs: vector = per-list elem_bytes and numel, in binding order. + * + * #### Returns: + * - GpuPack = owned buffers; caller must destroy_gpu_pack when done. + * + * #### Throws: + * - runtime_error on bad specs (e.g. numel > 0 but elem_bytes == 0) or Vulkan alloc failure. + */ +GpuPack create_gpu_pack( + cthreads::gpu::Context& context, + size_t scalar_bytes, + std::vector container_specs +); + +/** + * Copies host scalar bytes into pack.scalar_buffer (host to device). + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = destination pack (must have a scalar buffer large enough). + * - data: const void* = host source bytes. + * - size: size_t = bytes to copy; must be <= scalar_buffer.size. + * + * #### Throws: + * - runtime_error if there is no scalar buffer, data is null, size is invalid, or transfer fails. + */ +void upload_scalars( + cthreads::gpu::Context& context, + GpuPack& pack, + const void* data, + size_t size +); + +/** + * Copies host bytes into one container slot (host to device). + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = destination pack. + * - index: size_t = container_slots index. + * - data: const void* = host source bytes. + * - size: size_t = bytes to copy; must equal elem_bytes * numel for that slot. + * + * #### Throws: + * - runtime_error if index is out of range, the slot is empty (numel == 0), data is null, + * size mismatches, or transfer fails. + */ +void upload_container( + cthreads::gpu::Context& context, + GpuPack& pack, + size_t index, + const void* data, + size_t size +); + +/** + * Uploads every non-empty container slot from parallel host pointers. + * + * data[i] / sizes[i] correspond to container_slots[i]. Slots with numel == 0 + * are skipped (pointer/size for those entries are ignored). + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = destination pack. + * - data: vector of host pointers, one per slot. + * - sizes: vector of byte counts, one per slot (same length as data and slots). + * + * #### Throws: + * - runtime_error if lengths mismatch pack.container_slots or any per-slot upload fails. + */ +void upload_containers( + cthreads::gpu::Context& context, + GpuPack& pack, + const std::vector& data, + const std::vector& sizes +); + +/** + * Copies pack.scalar_buffer into host memory (device to host). + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = source pack. + * - data: void* = host destination bytes. + * - size: size_t = bytes to copy; must be non-zero and <= scalar_buffer.size. + * + * #### Throws: + * - runtime_error if there is no scalar buffer, data is null, size is invalid, or transfer fails. + */ +void download_scalars( + cthreads::gpu::Context& context, + GpuPack& pack, + void* data, + size_t size +); + +/** + * Copies one container slot into host memory (device to host). + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = source pack. + * - index: size_t = container_slots index. + * - data: void* = host destination bytes. + * - size: size_t = bytes to copy; must be <= container_slots[index].buffer.size. + * + * #### Throws: + * - runtime_error if index is out of range, the slot is empty (numel == 0), data is null, + * size mismatches, or transfer fails. + */ +void download_container( + cthreads::gpu::Context& context, + GpuPack& pack, + size_t index, + void* data, + size_t size +); + +/** + * Downloads every non-empty container slot into parallel host pointers. + * + * Slots with numel == 0 are skipped. data/sizes length must match container_slots. + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = source pack. + * - data: vector of host destinations, one per slot. + * - sizes: vector of byte counts, one per slot (same length as data and slots). + * + * #### Throws: + * - runtime_error if lengths mismatch pack.container_slots or any per-slot download fails. + */ +void download_containers( + cthreads::gpu::Context& context, + GpuPack& pack, + std::vector& data, + const std::vector& sizes +); + +/** + * Destroys all buffers owned by the pack and clears its fields. + * + * Safe to call on an already-empty pack. Does not destroy the Context. + * + * #### Parameters: + * - context: Context& = same device that created the pack. + * - pack: GpuPack& = pack to free; left empty after return. + */ +void destroy_gpu_pack(cthreads::gpu::Context& context, GpuPack& pack); + +} // namespace cthreads::gpu::pack diff --git a/src/cthreads/cpp/gpu/headers/shader.hpp b/src/cthreads/cpp/gpu/headers/shader.hpp new file mode 100644 index 0000000..8f53466 --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/shader.hpp @@ -0,0 +1,85 @@ +#pragma once + +#include +#include + +#include "shader_cache.hpp" + +namespace cthreads::gpu { +struct Context; +} + +/** + * Stateless helpers that build and tear down ShaderCacheEntry Vulkan objects. + * + * create_entry turns SPIR-V + a binding count into the reusable pipeline row + * stored in ShaderCache. It does not allocate per-launch descriptor sets or + * bind a GpuPack; that is the launch / inflight path. + * + * Pass a ready Context with shader/pipeline create (and destroy) entry points. + * + * #### Technical terms: + * - SPIR-V: binary compute shader (uint32 words). Built for tests as committed + * bytes or via shaderc; @Gpu emit feeds the same helper later. + * - binding_count: STORAGE_BUFFER bindings for the binding convention + * ((scalars ? 1 : 0) + number of list SSBOs; at least 1 total). + * - ShaderCacheEntry: module + set layout + pipeline layout + compute pipeline. + */ +namespace cthreads::gpu::shader { + +/** + * Creates a ShaderCacheEntry from SPIR-V and a binding-convention binding count. + * + * Builds, in order: shader module, descriptor set layout (bindings 0 .. + * binding_count-1 as STORAGE_BUFFER), pipeline layout, compute pipeline. + * On failure, destroys any objects already created and throws. + * + * Does not insert into ShaderCache; ShaderRegistry::register_spirv calls + * create_entry then ShaderCache::add. + * + * #### Parameters: + * - context: Context& = initialized GPU context with device and create entry points. + * - spirv: const uint32_t* = SPIR-V code words (must be valid compute SPIR-V). + * - spirv_word_count: size_t = number of uint32 words in spirv (byte size / 4). + * - binding_count: uint32_t = number of SSBO bindings (must be >= 1). + * + * #### Returns: + * - ShaderCacheEntry = owned Vulkan handles; move into ShaderCache::add or + * destroy_entry when abandoning the entry. + * + * #### Throws: + * - runtime_error if context is not ready, spirv is null/empty, binding_count + * is 0, create entry points are missing, or any Vulkan create fails. + * + * #### Technical terms: + * - VkShaderModule: Vulkan wrapper around SPIR-V bytes. + * - VkDescriptorSetLayout: schema only; concrete descriptor sets are per launch. + * - VkPipeline: compiled compute program ready for vkCmdBindPipeline. + */ +ShaderCacheEntry create_entry( + cthreads::gpu::Context& context, + const uint32_t* spirv, + size_t spirv_word_count, + uint32_t binding_count +); + +/** + * Destroys Vulkan objects owned by a ShaderCacheEntry and resets handles to null. + * + * Safe to call on an already-empty entry. Used by ShaderCache::clear and by + * callers that abandon an entry before add. + * + * #### Parameters: + * - context: Context& = same device that created the entry (needs destroy PFNs). + * - entry: ShaderCacheEntry& = entry to destroy; left empty on return. + * + * #### Throws: + * - Does not throw on destroy failure paths that only null handles when the + * device is already gone (shutdown edge cases). + */ +void destroy_entry( + cthreads::gpu::Context& context, + ShaderCacheEntry& entry +); + +} // namespace cthreads::gpu::shader diff --git a/src/cthreads/cpp/gpu/headers/shader_cache.hpp b/src/cthreads/cpp/gpu/headers/shader_cache.hpp new file mode 100644 index 0000000..7e99eeb --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/shader_cache.hpp @@ -0,0 +1,121 @@ +#pragma once + +#include +#include +#include +#include +#include + +namespace cthreads::gpu { +struct Context; +} + +/** + * Shader cache: reusable per-symbol Vulkan pipeline objects. + * + * Binding schema and compute pipelines are fixed for a kernel symbol. Per-launch + * buffers and descriptor sets live on the job / inflight path, not here. + * + * #### Technical terms: + * - Context: process-wide Vulkan connection (device, queue, loaded entry points). + * - SPIR-V: binary shader IR the driver consumes (see create_entry in shader.hpp). + * - descriptor set layout: schema of SSBO bindings (0 scalars, 1..N lists). + * - compute pipeline: compiled shader + layout ready to bind and dispatch. + */ +namespace cthreads::gpu::shader { + +/** + * Per-kernel Vulkan objects reused across launches of the same symbol. + * + * Move-only: Vulkan handles must not be copied (would double-destroy). + * + * #### Fields: + * - shader_module: VkShaderModule = SPIR-V module (may be null after pipeline create). + * - set_layout: VkDescriptorSetLayout = bindings 0..N-1 (scalars at 0 if present, then lists). + * - pipeline_layout: VkPipelineLayout = layout used to create the compute pipeline. + * - pipeline: VkPipeline = compute pipeline ready to bind. + * - binding_count: uint32_t = STORAGE_BUFFER bindings ((scalars?1:0) + list count). + */ +struct ShaderCacheEntry { + VkShaderModule shader_module = VK_NULL_HANDLE; + VkDescriptorSetLayout set_layout = VK_NULL_HANDLE; + VkPipelineLayout pipeline_layout = VK_NULL_HANDLE; + VkPipeline pipeline = VK_NULL_HANDLE; + uint32_t binding_count = 0; + + ShaderCacheEntry() = default; + ShaderCacheEntry(const ShaderCacheEntry&) = delete; + ShaderCacheEntry& operator=(const ShaderCacheEntry&) = delete; + ShaderCacheEntry(ShaderCacheEntry&& other) noexcept; + ShaderCacheEntry& operator=(ShaderCacheEntry&& other) noexcept; +}; + +/** + * Sole writer into ShaderCache (create_entry + add). + * + * Python reaches this via `_ext.gpu.register_shader`. Tests and future native + * callers use the same type - do not friend other writers or call add() directly. + */ +struct ShaderRegistry { + /** + * Build a pipeline entry from SPIR-V and insert it under `symbol`. + * + * #### Parameters: + * - context: Context& = ready GPU context + * - symbol: string = ShaderCache key (kernel name) + * - spirv: const uint32_t* = SPIR-V words + * - spirv_word_count: size_t = word count + * - binding_count: uint32_t = SSBO binding count (>= 1) + * + * #### Returns: + * - const ShaderCacheEntry& = entry stored in the cache + * + * #### Throws: + * - runtime_error = create_entry failure or duplicate symbol + */ + static const ShaderCacheEntry& register_spirv( + cthreads::gpu::Context& context, + const std::string& symbol, + const uint32_t* spirv, + size_t spirv_word_count, + uint32_t binding_count + ); +}; + +/** + * Process-wide map of kernel symbol -> reusable pipeline objects. + * + * Writers: ShaderRegistry only. Everyone else: get / clear. + * Entries are immutable after insert; clear runs on Context shutdown. + */ +class ShaderCache { +private: + std::unordered_map _cache; + std::mutex _cache_mutex; + + ShaderCache() = default; + ~ShaderCache(); + + // ShaderRegistry-only. Duplicate key throws. + const ShaderCacheEntry& add(const std::string& key, ShaderCacheEntry&& entry); + +public: + static ShaderCache& getInstance(); + + ShaderCache(const ShaderCache&) = delete; + ShaderCache& operator=(const ShaderCache&) = delete; + ShaderCache(ShaderCache&&) = delete; + ShaderCache& operator=(ShaderCache&&) = delete; + + // Throws if the symbol is not registered. + const ShaderCacheEntry& get(const std::string& key); + + // Destroy all Vulkan objects on the entry, then empty the map. + // Call from Context shutdown before destroying the logical device. + void clear(cthreads::gpu::Context& context); + + friend struct cthreads::gpu::Context; // shutdown / future access + friend struct ShaderRegistry; +}; + +} // namespace cthreads::gpu::shader diff --git a/src/cthreads/cpp/gpu/headers/state.hpp b/src/cthreads/cpp/gpu/headers/state.hpp new file mode 100644 index 0000000..b35135f --- /dev/null +++ b/src/cthreads/cpp/gpu/headers/state.hpp @@ -0,0 +1,245 @@ +#pragma once + +#include +#include +#include +#include +#include + +#include "memory.hpp" + +namespace cthreads::gpu { +struct Context; +} + +/** + * Process-wide registry of named device-local GpuBuffers. + * + * Owns GPU memory outside of a single launch so kernels can reuse the same + * buffers across dispatches. Names are unique: adding a duplicate name throws. + * Intended for a later Python binding of this singleton so host code can + * allocate, upload, launch, download, and free without leaking or double-owning + * Vulkan handles. + * + * This is not a public DeviceBuffer type. Python should see a controlled state + * / arena API that calls into this registry; VkBuffer stays inside _ext. + * + * #### Technical terms: + * - GpuBuffer: one contiguous device (or staging) byte region (see memory.hpp). + * - Context: process-wide Vulkan connection (device, queue, loaded entry points). + * - in_use: true while a launch has checked out this name; blocks remove and a + * second mark_in_use until release_in_use. + * - singleton: one process-wide instance via getInstance(); not copyable. + */ +namespace cthreads::gpu::memory { + +/** + * One named row in GpuState: owned buffer plus launch checkout flag. + * + * #### Fields: + * - buffer: GpuBuffer = owned Vulkan allocation (moved in on add). + * - in_use: bool = true while a kernel job holds this name for dispatch. + */ +struct GpuStateEntry { + GpuBuffer buffer{}; + bool in_use = false; +}; + +/** + * Singleton map of unique string names to owned GpuBuffers. + * + * Thread-safe: every public method locks an internal mutex. Call clear(context) + * from Context shutdown before destroying the logical device so VkBuffer / + * VkDeviceMemory handles are freed while the device is still alive. + * + * Writers / readers: any native or pybind caller. Duplicate names are rejected + * on add. Removing or clearing while in_use is rejected on remove; clear on + * shutdown still destroys (process teardown). + */ +class GpuState { +private: + std::unordered_map _entries; + mutable std::mutex _mutex; + + GpuState() = default; + ~GpuState(); + +public: + static GpuState& getInstance(); + + GpuState(const GpuState&) = delete; + GpuState& operator=(const GpuState&) = delete; + GpuState(GpuState&&) = delete; + GpuState& operator=(GpuState&&) = delete; + + /** + * Destroy every registered buffer and empty the map. + * + * Call from Context shutdown before destroying the logical device. + * + * #### Parameters: + * - context: Context& = device used to destroy buffers + */ + void clear(cthreads::gpu::Context& context); + + /** + * True if `name` is already registered. + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Returns: + * - bool = true when an entry exists for name + */ + bool contains(const std::string& name) const; + + /** + * Number of registered buffers. + * + * #### Returns: + * - size_t = entry count + */ + size_t size() const; + + /** + * Snapshot of all registered names (order is not meaningful). + * + * #### Returns: + * - vector = copy of keys for introspection / pybind + */ + std::vector names() const; + + /** + * Take ownership of a buffer under a unique name. + * + * Moves `buffer` into the registry. The caller must not use or destroy the + * moved-from GpuBuffer afterward (handles are null after a successful add). + * + * #### Parameters: + * - name: const string& = unique key (must be non-empty and not already used) + * - buffer: GpuBuffer&& = owned allocation to store (typically DeviceLocal) + * + * #### Throws: + * - runtime_error = empty name, duplicate name, or empty buffer handles + */ + void add(const std::string& name, GpuBuffer&& buffer); + + /** + * Destroy and unregister one buffer by name. + * + * #### Parameters: + * - context: Context& = same device that created the buffer + * - name: const string& = registry key + * + * #### Throws: + * - runtime_error = unknown name, or entry is in_use (release first) + */ + void remove(cthreads::gpu::Context& context, const std::string& name); + + /** + * Mutable reference to the buffer stored under `name`. + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Returns: + * - GpuBuffer& = owned buffer (valid until remove/clear) + * + * #### Throws: + * - runtime_error = unknown name + */ + GpuBuffer& get(const std::string& name); + + /** + * Const reference to the buffer stored under `name`. + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Returns: + * - const GpuBuffer& = owned buffer (valid until remove/clear) + * + * #### Throws: + * - runtime_error = unknown name + */ + const GpuBuffer& get(const std::string& name) const; + + /** + * True if the named entry is checked out for a launch. + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Returns: + * - bool = entry.in_use + * + * #### Throws: + * - runtime_error = unknown name + */ + bool is_in_use(const std::string& name) const; + + /** + * Mark a buffer as checked out so remove and a second mark fail. + * + * Call from the launch path before recording work that binds this buffer. + * Pair with release_in_use after the fence wait (or on launch failure). + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Throws: + * - runtime_error = unknown name, or already in_use + */ + void mark_in_use(const std::string& name); + + /** + * Clear the in_use flag after a launch finishes (or aborts). + * + * #### Parameters: + * - name: const string& = registry key + * + * #### Throws: + * - runtime_error = unknown name, or entry was not in_use + */ + void release_in_use(const std::string& name); + + /** + * Upload host bytes into a registered device-local buffer (H2D). + * + * #### Parameters: + * - context: Context& = same device that created the buffer + * - name: const string& = registry key + * - data: const void* = host source bytes + * - size: VkDeviceSize = bytes to copy; must be > 0 and <= buffer.size + * + * #### Throws: + * - runtime_error = unknown name, in_use, bad size, or transfer failure + */ + void upload( + cthreads::gpu::Context& context, + const std::string& name, + const void* data, + VkDeviceSize size + ); + + /** + * Download a registered device-local buffer into host bytes (D2H). + * + * #### Parameters: + * - context: Context& = same device that created the buffer + * - name: const string& = registry key + * - data: void* = host destination bytes + * - size: VkDeviceSize = bytes to copy; must be > 0 and <= buffer.size + * + * #### Throws: + * - runtime_error = unknown name, in_use, bad size, or transfer failure + */ + void download( + cthreads::gpu::Context& context, + const std::string& name, + void* data, + VkDeviceSize size + ); +}; + +} // namespace cthreads::gpu::memory diff --git a/src/cthreads/cpp/gpu/impl/compile_glsl.cpp b/src/cthreads/cpp/gpu/impl/compile_glsl.cpp new file mode 100644 index 0000000..cc5c7a1 --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/compile_glsl.cpp @@ -0,0 +1,91 @@ +// Copyright (c) 2026 Tobias Karusseit +// This source code is licensed under the MIT license found in the +// LICENSE file in the root directory of this source tree. + +#include "../headers/compile_glsl.hpp" + +#include +#include +#include + +#include +#include +#include + +namespace cthreads::gpu { +namespace { + +std::once_flag g_glslang_once; + +void ensure_glslang_initialized() { + std::call_once(g_glslang_once, []() { + if (!glslang::InitializeProcess()) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: glslang InitializeProcess failed"); + } + }); +} + +} // namespace + +std::vector compile_glsl_to_spirv(const std::string& source) { + if (source.empty()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: compile_glsl source is empty"); + } + + ensure_glslang_initialized(); + + const EShLanguage stage = EShLangCompute; + glslang::TShader shader(stage); + const char* strings[] = {source.c_str()}; + shader.setStrings(strings, 1); + + // Vulkan 1.0 / SPIR-V 1.0 is enough for our compute SSBOs + builtins. + shader.setEnvInput( + glslang::EShSourceGlsl, stage, glslang::EShClientVulkan, 100); + shader.setEnvClient(glslang::EShClientVulkan, glslang::EShTargetVulkan_1_0); + shader.setEnvTarget(glslang::EShTargetSpv, glslang::EShTargetSpv_1_0); + + const EShMessages messages = + static_cast(EShMsgSpvRules | EShMsgVulkanRules); + const TBuiltInResource* resources = GetDefaultResources(); + + std::string log; + if (!shader.parse(resources, 100, false, messages)) { + log = shader.getInfoLog(); + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GLSL parse failed:\n" + log); + } + + glslang::TProgram program; + program.addShader(&shader); + if (!program.link(messages)) { + log = program.getInfoLog(); + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GLSL link failed:\n" + log); + } + + const glslang::TIntermediate* intermediate = program.getIntermediate(stage); + if (intermediate == nullptr) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GLSL link produced no intermediate"); + } + + std::vector spirv; + spv::SpvBuildLogger logger; + glslang::SpvOptions options; + options.generateDebugInfo = false; + options.disableOptimizer = true; + options.optimizeSize = false; + glslang::GlslangToSpv(*intermediate, spirv, &logger, &options); + + if (spirv.empty()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GlslangToSpv produced empty SPIR-V: " + + logger.getAllMessages()); + } + return spirv; +} + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/impl/context.cpp b/src/cthreads/cpp/gpu/impl/context.cpp new file mode 100644 index 0000000..70d4b57 --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/context.cpp @@ -0,0 +1,687 @@ +#include "../headers/context.hpp" +#include +#include +#include +#include +#include + +#include "../headers/memory.hpp" +#include "../headers/shader_cache.hpp" +#include "../headers/state.hpp" + +#if defined(_WIN32) + // Windows (32-bit or 64-bit) + #include +#elif defined(__linux__) + // Linux + #include +#else + // No MacOs support yet !!!! + // Unknown -> throw error + #error "cthreads: Unsupported OS" +#endif + +namespace cthreads::gpu { + +namespace { + + // ------ Hidden TransferEngineHelpers ------ + + // Caller must hold c.transfer_engine_mutex (also used from init while locked). + void shutdown_transfer_engine_unlocked(Context& c) { + // Safe no-op if the engine was never created or already cleared. + TransferEngine& te = c.transfer_engine; + if (te.command_pool == VK_NULL_HANDLE && + te.fence == VK_NULL_HANDLE && + te.staging.buffer == VK_NULL_HANDLE) { + return; + } + + // Finish any in-flight copy before freeing GPU objects. + if (te.fence != VK_NULL_HANDLE && c.device != VK_NULL_HANDLE && + c.vkWaitForFences) { + c.vkWaitForFences(c.device, 1, &te.fence, VK_TRUE, UINT64_MAX); + } + + // Staging first (uses device + buffer PFNs), then fence, then pool. + if (te.staging.buffer != VK_NULL_HANDLE && c.device != VK_NULL_HANDLE) { + memory::destroy_buffer(c, te.staging); + } + if (te.fence != VK_NULL_HANDLE && c.device != VK_NULL_HANDLE && + c.vkDestroyFence) { + c.vkDestroyFence(c.device, te.fence, nullptr); + te.fence = VK_NULL_HANDLE; + } + if (te.command_pool != VK_NULL_HANDLE && c.device != VK_NULL_HANDLE && + c.vkDestroyCommandPool) { + c.vkDestroyCommandPool(c.device, te.command_pool, nullptr); + te.command_pool = VK_NULL_HANDLE; + } + te = TransferEngine{}; + } + + void shutdown_transfer_engine(Context& c) { + std::lock_guard lock(c.transfer_engine_mutex); + shutdown_transfer_engine_unlocked(c); + } + + void init_transfer_engine(Context& c) { + std::lock_guard lock(c.transfer_engine_mutex); + // Pool + fence only. Staging is allocated later on demand so idle + // contexts do not hold a fixed 1 MiB host-visible buffer. + if (c.transfer_engine.command_pool != VK_NULL_HANDLE && + c.transfer_engine.fence != VK_NULL_HANDLE) { + return; + } + + if (c.device == VK_NULL_HANDLE || !c.vkCreateCommandPool || + !c.vkDestroyCommandPool || !c.vkCreateFence || !c.vkDestroyFence) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: init_transfer_engine missing " + "device or command/fence entry points"); + } + + // If a previous attempt left a half-built engine, clear it first. + if (c.transfer_engine.command_pool != VK_NULL_HANDLE || + c.transfer_engine.fence != VK_NULL_HANDLE || + c.transfer_engine.staging.buffer != VK_NULL_HANDLE) { + shutdown_transfer_engine_unlocked(c); + } + + VkCommandPoolCreateInfo pool_info{}; + pool_info.sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO; + pool_info.queueFamilyIndex = c.queue_family; + // TRANSIENT: short-lived recordings. RESET: allow vkResetCommandBuffer reuse. + pool_info.flags = VK_COMMAND_POOL_CREATE_TRANSIENT_BIT | + VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; + if (c.vkCreateCommandPool( + c.device, &pool_info, nullptr, &c.transfer_engine.command_pool) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateCommandPool failed"); + } + + VkFenceCreateInfo fence_info{}; + fence_info.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO; + // Signaled so the first wait/reset path can treat it as "idle". + fence_info.flags = VK_FENCE_CREATE_SIGNALED_BIT; + if (c.vkCreateFence( + c.device, &fence_info, nullptr, &c.transfer_engine.fence) != + VK_SUCCESS) { + c.vkDestroyCommandPool( + c.device, c.transfer_engine.command_pool, nullptr); + c.transfer_engine.command_pool = VK_NULL_HANDLE; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateFence failed"); + } + } + + // ------ Hidden LaunchEngine helpers ------ + + void shutdown_launch_engine_unlocked(Context& c) { + LaunchEngine& le = c.launch_engine; + if (le.command_pool == VK_NULL_HANDLE && + le.free_command_buffers.empty() && + le.free_fences.empty()) { + return; + } + + if (c.device != VK_NULL_HANDLE && c.vkDestroyFence) { + for (VkFence fence : le.free_fences) { + if (fence != VK_NULL_HANDLE) { + c.vkDestroyFence(c.device, fence, nullptr); + } + } + } + le.free_fences.clear(); + // Destroying the pool frees every CB allocated from it (idle and any + // still checked out if shutdown races a live job — process teardown). + le.free_command_buffers.clear(); + if (le.command_pool != VK_NULL_HANDLE && c.device != VK_NULL_HANDLE && + c.vkDestroyCommandPool) { + c.vkDestroyCommandPool(c.device, le.command_pool, nullptr); + } + le.command_pool = VK_NULL_HANDLE; + } + + void shutdown_launch_engine(Context& c) { + std::lock_guard lock(c.launch_engine_mutex); + shutdown_launch_engine_unlocked(c); + } + + void init_launch_engine(Context& c) { + std::lock_guard lock(c.launch_engine_mutex); + if (c.launch_engine.command_pool != VK_NULL_HANDLE) { + return; + } + + if (c.device == VK_NULL_HANDLE || !c.vkCreateCommandPool || + !c.vkDestroyCommandPool) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: init_launch_engine missing " + "device or command pool entry points"); + } + + if (!c.launch_engine.free_command_buffers.empty() || + !c.launch_engine.free_fences.empty()) { + shutdown_launch_engine_unlocked(c); + } + + VkCommandPoolCreateInfo pool_info{}; + pool_info.sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO; + pool_info.queueFamilyIndex = c.queue_family; + // RESET: checkout path calls vkResetCommandBuffer before re-record. + pool_info.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; + if (c.vkCreateCommandPool( + c.device, &pool_info, nullptr, &c.launch_engine.command_pool) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateCommandPool failed for " + "LaunchEngine"); + } + } + + // ------ Hidden Context Helpers ------ + + // Look up one export inside the already-loaded loader module. + // module: void* (HMODULE on Windows). name: C string export name. + // returns: raw code address, or nullptr if missing. + static void* load_fn(void* module, const char* name) { +#if defined(_WIN32) + return reinterpret_cast( + // gets the function ptr address by name from the module lookup table + GetProcAddress(static_cast(module), name)); +#else + return dlsym(module, name); +#endif + } + + // Resolve a Vulkan entry point by name and cast to the typed PFN_*. + // c: Context with vkGetInstanceProcAddr already set. + // instance: VK_NULL_HANDLE for global functions; real instance after create. + // name: e.g. "vkCreateInstance". + template + static PFN get_fn(Context& c, VkInstance instance, const char* name) { + // PFN_vkVoidFunction = generic "pointer to some Vulkan fn". + PFN_vkVoidFunction raw = c.vkGetInstanceProcAddr(instance, name); + if (!raw) { + throw std::runtime_error( + std::string("cthreads.gpu.VulkanInitFailed: missing ") + name); + } + return reinterpret_cast(raw); // PFN = e.g. PFN_vkCreateInstance + } + + // Open the Vulkan DLL / shared library and put the file pointer on the Context.loader_module + // Throws on failure. + // also assign the vkGetInstanceProcAddr. + // Naming Note: GetPRocAddress 0> GetFunctionAddress + static void open_loader(Context& c) { +#if defined(_WIN32) + // Map the Vulkan loader DLL into this process (driver-installed). + HMODULE mod = LoadLibraryA("vulkan-1.dll"); + if (!mod) { + throw std::runtime_error( + "cthreads.gpu.VulkanLoaderNotFound: vulkan-1.dll not found"); + } + c.loader_module = static_cast(mod); +#else + void* mod = dlopen("libvulkan.so.1", RTLD_NOW); + if (!mod) { + throw std::runtime_error( + "cthreads.gpu.VulkanLoaderNotFound: libvulkan.so.1 not found"); + } + c.loader_module = mod; +#endif + // Only this first symbol comes from GetProcAddress/dlsym. + // Everything else goes through vkGetInstanceProcAddr. + c.vkGetInstanceProcAddr = + reinterpret_cast( + load_fn(c.loader_module, "vkGetInstanceProcAddr")); + if (!c.vkGetInstanceProcAddr) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkGetInstanceProcAddr missing"); + } + } + + static void create_instance_and_device(Context& c) { + // Only globals may be resolved with VK_NULL_HANDLE (Vulkan loader rules). + // Instance-level procs (DestroyInstance, EnumeratePhysicalDevices, …) + // must be resolved after vkCreateInstance with the real instance. + c.vkCreateInstance = get_fn( + c, VK_NULL_HANDLE, "vkCreateInstance"); + // VkApplicationInfo: tells the loader who we are (required sType pattern). + VkApplicationInfo app{}; // vulkan app metadata + app.sType = VK_STRUCTURE_TYPE_APPLICATION_INFO; // set type for vulkan to interprete this struct + app.pApplicationName = "cthreads"; + app.applicationVersion = VK_MAKE_VERSION(0, 1, 0); + app.pEngineName = "cthreads"; + app.engineVersion = VK_MAKE_VERSION(0, 1, 0); + app.apiVersion = VK_API_VERSION_1_1; // request 1.1 + // VkInstanceCreateInfo: parameters for vkCreateInstance. + VkInstanceCreateInfo ici{}; + ici.sType = VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO; // set type for vulkan to interprete this struct + ici.pApplicationInfo = &app; + // no layers/extensions for Issue 1 + // Out-param: writes the new VkInstance into c.instance. + if (c.vkCreateInstance(&ici, nullptr, &c.instance) != VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateInstance failed"); + } + // Instance-level procs (pass c.instance). + c.vkDestroyInstance = get_fn( + c, c.instance, "vkDestroyInstance"); + c.vkEnumeratePhysicalDevices = get_fn( + c, c.instance, "vkEnumeratePhysicalDevices"); + c.vkGetPhysicalDeviceProperties = + get_fn( // get fn from vulkan (gets the properties of the physical device) + c, c.instance, "vkGetPhysicalDeviceProperties"); + c.vkGetPhysicalDeviceQueueFamilyProperties = + get_fn( + c, c.instance, "vkGetPhysicalDeviceQueueFamilyProperties"); + c.vkGetPhysicalDeviceMemoryProperties = + get_fn( + c, c.instance, "vkGetPhysicalDeviceMemoryProperties"); + c.vkCreateDevice = get_fn( + c, c.instance, "vkCreateDevice"); + c.vkDestroyDevice = get_fn( + c, c.instance, "vkDestroyDevice"); + c.vkGetDeviceQueue = get_fn( + c, c.instance, "vkGetDeviceQueue"); + // --- list GPUs (two-call idiom: count, then data) --- + uint32_t dev_count = 0; + c.vkEnumeratePhysicalDevices(c.instance, &dev_count, nullptr); + if (dev_count == 0) { + throw std::runtime_error( + "cthreads.gpu.VulkanNoDevice: no physical devices"); + } + std::vector devices(dev_count); + c.vkEnumeratePhysicalDevices(c.instance, &dev_count, devices.data()); + // Pick: need COMPUTE queue; prefer discrete GPU. + int best_score = -1; + for (VkPhysicalDevice pd : devices) { + VkPhysicalDeviceProperties props{}; + c.vkGetPhysicalDeviceProperties(pd, &props); + uint32_t qcount = 0; + c.vkGetPhysicalDeviceQueueFamilyProperties(pd, &qcount, nullptr); + std::vector qprops(qcount); + c.vkGetPhysicalDeviceQueueFamilyProperties(pd, &qcount, qprops.data()); + for (uint32_t fi = 0; fi < qcount; ++fi) { + if (!(qprops[fi].queueFlags & VK_QUEUE_COMPUTE_BIT)) { + continue; // graphics-only family: skip + } + int score = (props.deviceType == VK_PHYSICAL_DEVICE_TYPE_DISCRETE_GPU) + ? 1000 + : 100; + if (score > best_score) { + best_score = score; + c.physical_device = pd; + c.queue_family = fi; + c.device_name = props.deviceName; // C string -> std::string + } + } + } + if (c.physical_device == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanNoDevice: no compute queue family"); + } + // --- logical device = "open" that GPU for our process --- + float priority = 1.0f; // single queue, max priority in [0,1] + VkDeviceQueueCreateInfo qci{}; + qci.sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; + qci.queueFamilyIndex = c.queue_family; + qci.queueCount = 1; + qci.pQueuePriorities = &priority; + VkDeviceCreateInfo dci{}; + dci.sType = VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO; + dci.queueCreateInfoCount = 1; + dci.pQueueCreateInfos = &qci; + if (c.vkCreateDevice(c.physical_device, &dci, nullptr, &c.device) != VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateDevice failed"); + } + // Queue handle is owned by the device; index 0 of that family. + c.vkGetDeviceQueue(c.device, c.queue_family, 0, &c.queue); + + // Device-level buffer / memory / transfer entry points (Issue 2). + // Resolved after the logical device exists; GIPA still returns loader trampolines. + c.vkCreateBuffer = get_fn( + c, c.instance, "vkCreateBuffer"); + c.vkDestroyBuffer = get_fn( + c, c.instance, "vkDestroyBuffer"); + c.vkGetBufferMemoryRequirements = get_fn( + c, c.instance, "vkGetBufferMemoryRequirements"); + c.vkAllocateMemory = get_fn( + c, c.instance, "vkAllocateMemory"); + c.vkFreeMemory = get_fn( + c, c.instance, "vkFreeMemory"); + c.vkBindBufferMemory = get_fn( + c, c.instance, "vkBindBufferMemory"); + c.vkMapMemory = get_fn( + c, c.instance, "vkMapMemory"); + c.vkUnmapMemory = get_fn( + c, c.instance, "vkUnmapMemory"); + c.vkCreateCommandPool = get_fn( + c, c.instance, "vkCreateCommandPool"); + c.vkDestroyCommandPool = get_fn( + c, c.instance, "vkDestroyCommandPool"); + c.vkAllocateCommandBuffers = get_fn( + c, c.instance, "vkAllocateCommandBuffers"); + c.vkFreeCommandBuffers = get_fn( + c, c.instance, "vkFreeCommandBuffers"); + c.vkResetCommandBuffer = get_fn( + c, c.instance, "vkResetCommandBuffer"); + c.vkBeginCommandBuffer = get_fn( + c, c.instance, "vkBeginCommandBuffer"); + c.vkEndCommandBuffer = get_fn( + c, c.instance, "vkEndCommandBuffer"); + c.vkCmdCopyBuffer = get_fn( + c, c.instance, "vkCmdCopyBuffer"); + c.vkCreateFence = get_fn( + c, c.instance, "vkCreateFence"); + c.vkDestroyFence = get_fn( + c, c.instance, "vkDestroyFence"); + c.vkQueueSubmit = get_fn( + c, c.instance, "vkQueueSubmit"); + c.vkWaitForFences = get_fn( + c, c.instance, "vkWaitForFences"); + c.vkResetFences = get_fn( + c, c.instance, "vkResetFences"); + c.vkCreateShaderModule = get_fn( + c, c.instance, "vkCreateShaderModule"); + c.vkDestroyShaderModule = get_fn( + c, c.instance, "vkDestroyShaderModule"); + c.vkCreateDescriptorSetLayout = get_fn( + c, c.instance, "vkCreateDescriptorSetLayout"); + c.vkDestroyDescriptorSetLayout = + get_fn( + c, c.instance, "vkDestroyDescriptorSetLayout"); + c.vkCreatePipelineLayout = get_fn( + c, c.instance, "vkCreatePipelineLayout"); + c.vkDestroyPipelineLayout = get_fn( + c, c.instance, "vkDestroyPipelineLayout"); + c.vkCreateComputePipelines = get_fn( + c, c.instance, "vkCreateComputePipelines"); + c.vkDestroyPipeline = get_fn( + c, c.instance, "vkDestroyPipeline"); + c.vkCreateDescriptorPool = get_fn( + c, c.instance, "vkCreateDescriptorPool"); + c.vkDestroyDescriptorPool = get_fn( + c, c.instance, "vkDestroyDescriptorPool"); + c.vkAllocateDescriptorSets = get_fn( + c, c.instance, "vkAllocateDescriptorSets"); + c.vkFreeDescriptorSets = get_fn( + c, c.instance, "vkFreeDescriptorSets"); + c.vkUpdateDescriptorSets = get_fn( + c, c.instance, "vkUpdateDescriptorSets"); + c.vkCmdBindPipeline = get_fn( + c, c.instance, "vkCmdBindPipeline"); + c.vkCmdBindDescriptorSets = get_fn( + c, c.instance, "vkCmdBindDescriptorSets"); + c.vkCmdDispatch = get_fn( + c, c.instance, "vkCmdDispatch"); + c.vkCmdPipelineBarrier = get_fn( + c, c.instance, "vkCmdPipelineBarrier"); + + c.ready = true; + // After device + PFNs + ready: reusable copy + launch pools. + init_transfer_engine(c); + init_launch_engine(c); + } + + void shutdown_unlocked(Context& c) { + // Children before parents: launch engine, transfer engine, named + // GpuState buffers, shader cache, then device. + shutdown_launch_engine(c); + shutdown_transfer_engine(c); + memory::GpuState::getInstance().clear(c); + shader::ShaderCache::getInstance().clear(c); + + // 1) release logical device + if (c.device != VK_NULL_HANDLE && c.vkDestroyDevice) { // check if device is set and if theres a destroy fn for it + c.vkDestroyDevice(c.device, nullptr); // set nullptr + c.device = VK_NULL_HANDLE; // must be nulled out + c.queue = VK_NULL_HANDLE; // must be nulled out + } + + // 2) release instance + if (c.instance != VK_NULL_HANDLE && c.vkDestroyInstance) { // ensure instance is set and if theres a destroy fn for it + c.vkDestroyInstance(c.instance, nullptr); // set to nullptr + c.instance = VK_NULL_HANDLE; // null the handle to avoid dangling pointers + } + c.physical_device = VK_NULL_HANDLE; // can be nulled now that the instance and logical device are destroyed + + // 3) Unmap loader DLL so OS can unload it. + if (c.loader_module) { + // closes the files and nulls the pointer to the module + #if defined(_WIN32) + FreeLibrary(static_cast(c.loader_module)); + #else + dlclose(c.loader_module); + #endif + c.loader_module = nullptr; + } + + // 4) Clear function pointers so a buggy late call can't jump into freed DLL! + // instance functions + c.vkGetInstanceProcAddr = nullptr; + c.vkCreateInstance = nullptr; + c.vkDestroyInstance = nullptr; + c.vkEnumeratePhysicalDevices = nullptr; + c.vkGetPhysicalDeviceProperties = nullptr; + c.vkGetPhysicalDeviceQueueFamilyProperties = nullptr; + c.vkCreateDevice = nullptr; + c.vkDestroyDevice = nullptr; + c.vkGetDeviceQueue = nullptr; + + // buffer and memory functions + c.vkCreateBuffer = nullptr; + c.vkDestroyBuffer = nullptr; + c.vkGetBufferMemoryRequirements = nullptr; + c.vkAllocateMemory = nullptr; + c.vkFreeMemory = nullptr; + c.vkBindBufferMemory = nullptr; + c.vkMapMemory = nullptr; + c.vkUnmapMemory = nullptr; + c.vkGetPhysicalDeviceMemoryProperties = nullptr; + + // Command pool, command buffer, copy, and sync functions + c.vkCreateCommandPool = nullptr; + c.vkDestroyCommandPool = nullptr; + c.vkAllocateCommandBuffers = nullptr; + c.vkFreeCommandBuffers = nullptr; + c.vkResetCommandBuffer = nullptr; + c.vkBeginCommandBuffer = nullptr; + c.vkEndCommandBuffer = nullptr; + c.vkCmdCopyBuffer = nullptr; + c.vkCreateFence = nullptr; + c.vkDestroyFence = nullptr; + c.vkQueueSubmit = nullptr; + c.vkWaitForFences = nullptr; + c.vkResetFences = nullptr; + c.vkCreateShaderModule = nullptr; + c.vkDestroyShaderModule = nullptr; + c.vkCreateDescriptorSetLayout = nullptr; + c.vkDestroyDescriptorSetLayout = nullptr; + c.vkCreatePipelineLayout = nullptr; + c.vkDestroyPipelineLayout = nullptr; + c.vkCreateComputePipelines = nullptr; + c.vkDestroyPipeline = nullptr; + c.vkCreateDescriptorPool = nullptr; + c.vkDestroyDescriptorPool = nullptr; + c.vkAllocateDescriptorSets = nullptr; + c.vkFreeDescriptorSets = nullptr; + c.vkUpdateDescriptorSets = nullptr; + c.vkCmdBindPipeline = nullptr; + c.vkCmdBindDescriptorSets = nullptr; + c.vkCmdDispatch = nullptr; + c.vkCmdPipelineBarrier = nullptr; + + c.queue_family = 0; + c.device_name.clear(); + c.ready = false; + } + +} // namespace anonymous + + // to lock the init and shutdown functions aswell as any thread unsafe gpu functions + static std::mutex& gpu_mutex() { + static std::mutex m; + return m; + } + + Context& context() { + static Context ctx; + return ctx; + } + + const std::string& device_name() { + try { + init(); // try to initialize (noops if already initialized) + return context().device_name; + } catch (const std::exception& e) { + throw std::runtime_error("cthreads: " + std::string(e.what())); // this could be cleaner but isnt relevant for now + } + } + + bool available() { + try { + init(); // try to initialize (noops if already initialized) + return context().ready; // return success/failure + } catch (...) { + return false; // initialization failed + } + } + + void init() { + Context& c = context(); + if (c.ready) return; + + std::lock_guard lock(gpu_mutex()); + if (c.ready) return; + + try { + open_loader(c); + create_instance_and_device(c); + } catch (...) { + shutdown_unlocked(c); + throw; // original exception, nothing stored + } + } + + + void shutdown() { + std::lock_guard lock(gpu_mutex()); + shutdown_unlocked(context()); + } + + +LaunchResources checkout_launch_resources(Context& context) { + std::lock_guard lock(context.launch_engine_mutex); + LaunchEngine& le = context.launch_engine; + if (!context.ready || le.command_pool == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: checkout_launch_resources needs a " + "ready LaunchEngine"); + } + if (!context.vkAllocateCommandBuffers || !context.vkResetCommandBuffer || + !context.vkCreateFence || !context.vkResetFences) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: checkout_launch_resources missing " + "command/fence entry points"); + } + + LaunchResources out{}; + + if (!le.free_command_buffers.empty()) { + out.command_buffer = le.free_command_buffers.back(); + le.free_command_buffers.pop_back(); + if (context.vkResetCommandBuffer(out.command_buffer, 0) != VK_SUCCESS) { + le.free_command_buffers.push_back(out.command_buffer); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkResetCommandBuffer failed in " + "checkout_launch_resources"); + } + } else { + VkCommandBufferAllocateInfo alloc_info{}; + alloc_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO; + alloc_info.commandPool = le.command_pool; + alloc_info.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + alloc_info.commandBufferCount = 1; + if (context.vkAllocateCommandBuffers( + context.device, &alloc_info, &out.command_buffer) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkAllocateCommandBuffers failed " + "in checkout_launch_resources"); + } + } + + if (!le.free_fences.empty()) { + out.fence = le.free_fences.back(); + le.free_fences.pop_back(); + if (context.vkResetFences(context.device, 1, &out.fence) != VK_SUCCESS) { + le.free_fences.push_back(out.fence); + le.free_command_buffers.push_back(out.command_buffer); + out = LaunchResources{}; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkResetFences failed in " + "checkout_launch_resources"); + } + } else { + VkFenceCreateInfo fence_info{}; + fence_info.sType = VK_STRUCTURE_TYPE_FENCE_CREATE_INFO; + if (context.vkCreateFence( + context.device, &fence_info, nullptr, &out.fence) != + VK_SUCCESS) { + le.free_command_buffers.push_back(out.command_buffer); + out.command_buffer = VK_NULL_HANDLE; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateFence failed in " + "checkout_launch_resources"); + } + } + + return out; +} + +void return_launch_resources(Context& context, LaunchResources& resources) { + std::lock_guard lock(context.launch_engine_mutex); + LaunchEngine& le = context.launch_engine; + if (resources.command_buffer != VK_NULL_HANDLE) { + le.free_command_buffers.push_back(resources.command_buffer); + resources.command_buffer = VK_NULL_HANDLE; + } + if (resources.fence != VK_NULL_HANDLE) { + le.free_fences.push_back(resources.fence); + resources.fence = VK_NULL_HANDLE; + } +} + +void submit_launch( + Context& context, + VkCommandBuffer command_buffer, + VkFence fence +) { + std::lock_guard lock(context.launch_engine_mutex); + if (!context.ready || !context.queue || !context.vkQueueSubmit) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: submit_launch needs a ready queue"); + } + if (command_buffer == VK_NULL_HANDLE || fence == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: submit_launch null command buffer " + "or fence"); + } + + VkSubmitInfo submit{}; + submit.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO; + submit.commandBufferCount = 1; + submit.pCommandBuffers = &command_buffer; + if (context.vkQueueSubmit(context.queue, 1, &submit, fence) != VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkQueueSubmit failed in " + "submit_launch"); + } +} + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/impl/descriptors.cpp b/src/cthreads/cpp/gpu/impl/descriptors.cpp new file mode 100644 index 0000000..39f50cb --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/descriptors.cpp @@ -0,0 +1,216 @@ +#include "../headers/descriptors.hpp" +#include "../headers/context.hpp" +#include "../headers/shader_cache.hpp" + +#include +#include +#include + +namespace cthreads::gpu::pack { + +DescriptorPool create_pool( + Context& context, + uint32_t binding_count, + uint32_t max_sets +) { + if (!context.ready || context.device == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: create_pool needs an initialized " + "device"); + } + if (binding_count == 0 || max_sets == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: create_pool binding_count and " + "max_sets must be >= 1"); + } + if (!context.vkCreateDescriptorPool) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: create_pool missing " + "vkCreateDescriptorPool"); + } + + VkDescriptorPoolSize pool_size{}; + pool_size.type = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + pool_size.descriptorCount = binding_count * max_sets; + + VkDescriptorPoolCreateInfo pool_info{}; + pool_info.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO; + // Allow free_set per job when the launch completes. + pool_info.flags = VK_DESCRIPTOR_POOL_CREATE_FREE_DESCRIPTOR_SET_BIT; + pool_info.maxSets = max_sets; + pool_info.poolSizeCount = 1; + pool_info.pPoolSizes = &pool_size; + + DescriptorPool out{}; + out.binding_count = binding_count; + out.max_sets = max_sets; + if (context.vkCreateDescriptorPool( + context.device, &pool_info, nullptr, &out.pool) != VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateDescriptorPool failed"); + } + return out; +} + +void destroy_pool(Context& context, DescriptorPool& pool) { + if (pool.pool == VK_NULL_HANDLE) { + pool = DescriptorPool{}; + return; + } + if (context.device != VK_NULL_HANDLE && context.vkDestroyDescriptorPool) { + context.vkDestroyDescriptorPool(context.device, pool.pool, nullptr); + } + pool = DescriptorPool{}; +} + +VkDescriptorSet allocate_set( + Context& context, + DescriptorPool& pool, + VkDescriptorSetLayout set_layout +) { + if (!context.ready || context.device == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: allocate_set needs an initialized " + "device"); + } + if (pool.pool == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: allocate_set pool is empty"); + } + if (set_layout == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: allocate_set set_layout is null"); + } + if (!context.vkAllocateDescriptorSets) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: allocate_set missing " + "vkAllocateDescriptorSets"); + } + + VkDescriptorSetAllocateInfo alloc_info{}; + alloc_info.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO; + alloc_info.descriptorPool = pool.pool; + alloc_info.descriptorSetCount = 1; + alloc_info.pSetLayouts = &set_layout; + + VkDescriptorSet set = VK_NULL_HANDLE; + if (context.vkAllocateDescriptorSets(context.device, &alloc_info, &set) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkAllocateDescriptorSets failed"); + } + return set; +} + +void free_set(Context& context, DescriptorPool& pool, VkDescriptorSet& set) { + if (set == VK_NULL_HANDLE) { + return; + } + if (pool.pool == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: free_set pool is empty"); + } + if (!context.vkFreeDescriptorSets || context.device == VK_NULL_HANDLE) { + set = VK_NULL_HANDLE; + return; + } + if (context.vkFreeDescriptorSets(context.device, pool.pool, 1, &set) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkFreeDescriptorSets failed"); + } + set = VK_NULL_HANDLE; +} + +void update_descriptors( + Context& context, + VkDescriptorSet set, + uint32_t binding_count, + const GpuPack& pack +) { + if (!context.ready || context.device == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: update_descriptors needs an " + "initialized device"); + } + if (set == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: update_descriptors set is null"); + } + if (binding_count == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: update_descriptors binding_count " + "must be >= 1"); + } + const bool has_scalars = (pack.scalar_buffer.buffer != VK_NULL_HANDLE); + const uint32_t expected = + (has_scalars ? 1u : 0u) + + static_cast(pack.container_slots.size()); + if (binding_count != expected) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: update_descriptors binding_count " + "must equal (scalars?1:0) + container_slots.size()"); + } + if (!context.vkUpdateDescriptorSets) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: update_descriptors missing " + "vkUpdateDescriptorSets"); + } + + // One buffer info + write per binding; infos must stay alive for the call. + std::vector buffer_infos(binding_count); + std::vector writes(binding_count); + + for (uint32_t i = 0; i < binding_count; ++i) { + VkBuffer buffer = VK_NULL_HANDLE; + VkDeviceSize size = 0; + if (has_scalars && i == 0) { + buffer = pack.scalar_buffer.buffer; + size = pack.scalar_buffer.size; + } else { + const uint32_t list_i = has_scalars ? (i - 1u) : i; + const ContainerSlot& slot = pack.container_slots[list_i]; + buffer = slot.buffer.buffer; + size = slot.buffer.size; + } + if (buffer == VK_NULL_HANDLE || size == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: update_descriptors binding " + + std::to_string(i) + + " needs a non-empty GpuBuffer (empty pack slots not supported " + "yet)"); + } + + buffer_infos[i] = {}; + buffer_infos[i].buffer = buffer; + buffer_infos[i].offset = 0; + buffer_infos[i].range = size; + + writes[i] = {}; + writes[i].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + writes[i].dstSet = set; + writes[i].dstBinding = i; + writes[i].dstArrayElement = 0; + writes[i].descriptorCount = 1; + writes[i].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + writes[i].pBufferInfo = &buffer_infos[i]; + } + + context.vkUpdateDescriptorSets( + context.device, + binding_count, + writes.data(), + 0, + nullptr); +} + +void update_descriptors( + Context& context, + VkDescriptorSet set, + const shader::ShaderCacheEntry& entry, + const GpuPack& pack +) { + update_descriptors(context, set, entry.binding_count, pack); +} + +} // namespace cthreads::gpu::pack diff --git a/src/cthreads/cpp/gpu/impl/memory.cpp b/src/cthreads/cpp/gpu/impl/memory.cpp new file mode 100644 index 0000000..1d4c2cb --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/memory.cpp @@ -0,0 +1,372 @@ +#include "../headers/memory.hpp" +#include "../headers/context.hpp" + +#include +#include +#include +#include + +namespace cthreads::gpu::memory { + +namespace { + +/** +* Ensure that the global context singleton has been initialized and is ready to use. +* +* #### Parameters: +* - context: The global context singleton (holds all dynamically linked vulkan function ptrs aswell as the device information) +* - where: The name of the function that is calling this helper. This is used to construct the error message. +* +* #### Throws: +* - std::runtime_error: If the context is not ready. +*/ +void require_ready(const Context& context, const char* where) { + if (!context.ready || context.device == VK_NULL_HANDLE) { + throw std::runtime_error( + std::string("cthreads.gpu.VulkanInitFailed: ") + where + + " needs an initialized device"); + } +} + +void require_transfer_engine(const Context& context, const char* where) { + // assumes the engine mutex is locked + if (context.transfer_engine.command_pool == VK_NULL_HANDLE || + context.transfer_engine.fence == VK_NULL_HANDLE) { + throw std::runtime_error( + std::string("cthreads.gpu.VulkanInitFailed: ") + where + + " needs an initialized TransferEngine (pool + fence)"); + } +} + +// GPU copy via Context TransferEngine pool + fence, then CPU wait. +// Caller must hold context.transfer_engine_mutex for the whole call. +void copy_buffer_and_wait( + Context& context, + VkBuffer src, + VkBuffer dst, + VkDeviceSize size +) { + require_transfer_engine(context, "copy_buffer_and_wait"); + if (!context.vkAllocateCommandBuffers || !context.vkFreeCommandBuffers || + !context.vkBeginCommandBuffer || !context.vkEndCommandBuffer || + !context.vkCmdCopyBuffer || !context.vkQueueSubmit || + !context.vkWaitForFences || !context.vkResetFences || !context.queue) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: copy_buffer_and_wait missing " + "command/fence entry points or queue"); + } + + TransferEngine& te = context.transfer_engine; + VkCommandPool pool = te.command_pool; + VkFence fence = te.fence; + + VkCommandBufferAllocateInfo alloc_info{}; + alloc_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO; + alloc_info.commandPool = pool; + alloc_info.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + alloc_info.commandBufferCount = 1; + VkCommandBuffer cmd = VK_NULL_HANDLE; + if (context.vkAllocateCommandBuffers(context.device, &alloc_info, &cmd) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkAllocateCommandBuffers failed"); + } + + VkCommandBufferBeginInfo begin_info{}; + begin_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO; + begin_info.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + if (context.vkBeginCommandBuffer(cmd, &begin_info) != VK_SUCCESS) { + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkBeginCommandBuffer failed"); + } + + VkBufferCopy region{}; + region.srcOffset = 0; + region.dstOffset = 0; + region.size = size; + context.vkCmdCopyBuffer(cmd, src, dst, 1, ®ion); + + if (context.vkEndCommandBuffer(cmd) != VK_SUCCESS) { + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkEndCommandBuffer failed"); + } + + // Fence is created signaled and left signaled after each wait; reset for reuse. + if (context.vkResetFences(context.device, 1, &fence) != VK_SUCCESS) { + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkResetFences failed"); + } + + VkSubmitInfo submit{}; + submit.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO; + submit.commandBufferCount = 1; + submit.pCommandBuffers = &cmd; + if (context.vkQueueSubmit(context.queue, 1, &submit, fence) != VK_SUCCESS) { + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkQueueSubmit failed"); + } + + if (context.vkWaitForFences( + context.device, 1, &fence, VK_TRUE, UINT64_MAX) != VK_SUCCESS) { + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkWaitForFences failed"); + } + + context.vkFreeCommandBuffers(context.device, pool, 1, &cmd); +} + +} // namespace + +uint32_t find_memory_type( + Context& context, + uint32_t type_bits, + VkMemoryPropertyFlags properties +) { + // ensure the context is setup correctly + if (context.physical_device == VK_NULL_HANDLE || + !context.vkGetPhysicalDeviceMemoryProperties) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: find_memory_type needs a ready " + "physical device and vkGetPhysicalDeviceMemoryProperties" + ); + } + + VkPhysicalDeviceMemoryProperties memory_props{}; // get mem properties from the ctx device + context.vkGetPhysicalDeviceMemoryProperties( + context.physical_device, &memory_props); + + for (uint32_t i = 0; i < memory_props.memoryTypeCount; ++i) { + const bool allowed_by_buffer = (type_bits & (1u << i)) != 0; // check if the ith bit is 1 + if (!allowed_by_buffer) { + continue; + } + const VkMemoryPropertyFlags flags = + memory_props.memoryTypes[i].propertyFlags; + if ((flags & properties) == properties) { // checck if all properties bits are set in flags + return i; + } + } + + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: no memory type matches type_bits and " + "requested properties"); +} + +GpuBuffer create_buffer( + Context& context, + VkDeviceSize size, + BufferKind kind +) { + require_ready(context, "create_buffer"); // ensure the ctx is ready (the device must be initialized) + if (size == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: create_buffer size must be greater than 0"); + } + if (!context.vkCreateBuffer || !context.vkGetBufferMemoryRequirements || + !context.vkAllocateMemory || !context.vkBindBufferMemory || + !context.vkDestroyBuffer || !context.vkFreeMemory) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: create_buffer missing Vulkan entry points"); + } + + // setup flags based of the buffers kind + VkBufferUsageFlags usage = 0; + VkMemoryPropertyFlags mem_props = 0; + if (kind == BufferKind::Staging) { + // Host can memcpy here; GPU copies to/from device-local buffers. + usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT; + mem_props = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + } else { + // Shader SSBO + copies for marshal upload/download. + usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | + VK_BUFFER_USAGE_TRANSFER_SRC_BIT | + VK_BUFFER_USAGE_TRANSFER_DST_BIT; + mem_props = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT; + } + + GpuBuffer out{}; // create gpu buffer struct (internally owns vulkan buffer and memory handles) + out.kind = kind; + + // create the vk buffer + VkBufferCreateInfo bci{}; + bci.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO; + bci.size = size; + bci.usage = usage; + bci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + if (context.vkCreateBuffer(context.device, &bci, nullptr, &out.buffer) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateBuffer failed"); + } + + VkMemoryRequirements mem_reqs{}; // get the memory requirements for the buffer + context.vkGetBufferMemoryRequirements(context.device, out.buffer, &mem_reqs); + + const uint32_t memory_type = // find the memory type that matches the requirements + find_memory_type(context, mem_reqs.memoryTypeBits, mem_props); + + // allocate the memory (allocated to the gpuBuffers memory handle) + VkMemoryAllocateInfo mai{}; + mai.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO; + mai.allocationSize = mem_reqs.size; + mai.memoryTypeIndex = memory_type; + if (context.vkAllocateMemory(context.device, &mai, nullptr, &out.memory) != + VK_SUCCESS) { + context.vkDestroyBuffer(context.device, out.buffer, nullptr); + out.buffer = VK_NULL_HANDLE; + throw std::runtime_error( + "cthreads.gpu.VulkanOutOfMemory: vkAllocateMemory failed"); + } + + if (context.vkBindBufferMemory(context.device, out.buffer, out.memory, 0) != + VK_SUCCESS) { + context.vkFreeMemory(context.device, out.memory, nullptr); + context.vkDestroyBuffer(context.device, out.buffer, nullptr); + out.memory = VK_NULL_HANDLE; + out.buffer = VK_NULL_HANDLE; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkBindBufferMemory failed"); + } + + // if the buffer is for staging then also set the cpu pointer to the memory in buffer.mapped + if (kind == BufferKind::Staging) { + if (!context.vkMapMemory) { // if the mem isnt mapped sth went wrong -> free and throw + context.vkFreeMemory(context.device, out.memory, nullptr); + context.vkDestroyBuffer(context.device, out.buffer, nullptr); + out.memory = VK_NULL_HANDLE; + out.buffer = VK_NULL_HANDLE; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkMapMemory missing"); + } + if (context.vkMapMemory( // map the memory to the cpu pointer + context.device, out.memory, 0, mem_reqs.size, 0, &out.mapped) != + VK_SUCCESS) { + context.vkFreeMemory(context.device, out.memory, nullptr); + context.vkDestroyBuffer(context.device, out.buffer, nullptr); + out.memory = VK_NULL_HANDLE; + out.buffer = VK_NULL_HANDLE; + out.mapped = nullptr; + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkMapMemory failed"); + } + } + + out.size = size; + return out; +} + +void destroy_buffer(Context& context, GpuBuffer& buffer) { + // check if the buffer is already destroyed + if (buffer.buffer == VK_NULL_HANDLE && buffer.memory == VK_NULL_HANDLE) { + buffer = GpuBuffer{}; + return; + } + require_ready(context, "destroy_buffer"); // ensure the ctx is ready (the device must be initialized) + + // if the buffer is mapped then unmap it + if (buffer.mapped != nullptr && context.vkUnmapMemory && + buffer.memory != VK_NULL_HANDLE) { + context.vkUnmapMemory(context.device, buffer.memory); + buffer.mapped = nullptr; + } + // destroy the vk buffer + if (buffer.buffer != VK_NULL_HANDLE && context.vkDestroyBuffer) { + context.vkDestroyBuffer(context.device, buffer.buffer, nullptr); + buffer.buffer = VK_NULL_HANDLE; + } + // free the vk memory + if (buffer.memory != VK_NULL_HANDLE && context.vkFreeMemory) { + context.vkFreeMemory(context.device, buffer.memory, nullptr); + buffer.memory = VK_NULL_HANDLE; + } + buffer.size = 0; + buffer.kind = BufferKind::DeviceLocal; +} + +namespace { + +// Grow-only host-visible scratch on the TransferEngine. Never shrinks until +// Context shutdown. Caller must hold context.transfer_engine_mutex. +void ensure_staging(Context& context, VkDeviceSize size) { + TransferEngine& te = context.transfer_engine; + if (te.staging.buffer != VK_NULL_HANDLE && te.staging.size >= size && + te.staging.mapped != nullptr) { + return; + } + if (te.staging.buffer != VK_NULL_HANDLE) { + destroy_buffer(context, te.staging); + } + te.staging = create_buffer(context, size, BufferKind::Staging); +} + +} // namespace + +void upload_buffer( + Context& context, + GpuBuffer& buffer, + const void* data, + VkDeviceSize size +) { + require_ready(context, "upload_buffer"); + if (buffer.kind != BufferKind::DeviceLocal || + buffer.buffer == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: upload_buffer requires a device-local " + "GpuBuffer"); + } + if (data == nullptr) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: upload_buffer data is null"); + } + if (size == 0 || size > buffer.size) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: upload_buffer size invalid"); + } + + // One lock for engine check, staging grow, host memcpy, and GPU copy. + std::lock_guard lock(context.transfer_engine_mutex); + require_transfer_engine(context, "upload_buffer"); + ensure_staging(context, size); + GpuBuffer& staging = context.transfer_engine.staging; + std::memcpy(staging.mapped, data, static_cast(size)); + copy_buffer_and_wait(context, staging.buffer, buffer.buffer, size); +} + +void download_buffer( + Context& context, + GpuBuffer& buffer, + void* data, + VkDeviceSize size +) { + require_ready(context, "download_buffer"); + if (buffer.kind != BufferKind::DeviceLocal || + buffer.buffer == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: download_buffer requires a " + "device-local GpuBuffer"); + } + if (data == nullptr) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: download_buffer data is null"); + } + if (size == 0 || size > buffer.size) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: download_buffer size invalid"); + } + + // One lock for engine check, staging grow, GPU copy, and host memcpy. + std::lock_guard lock(context.transfer_engine_mutex); + require_transfer_engine(context, "download_buffer"); + ensure_staging(context, size); + GpuBuffer& staging = context.transfer_engine.staging; + copy_buffer_and_wait(context, buffer.buffer, staging.buffer, size); + std::memcpy(data, staging.mapped, static_cast(size)); +} + +} // namespace cthreads::gpu::memory diff --git a/src/cthreads/cpp/gpu/impl/module.cpp b/src/cthreads/cpp/gpu/impl/module.cpp new file mode 100644 index 0000000..d1f7102 --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/module.cpp @@ -0,0 +1,821 @@ +#include "../headers/module.hpp" +#include "../headers/context.hpp" +#include "../headers/shader.hpp" +#include "../headers/shader_cache.hpp" +#include "../headers/pack.hpp" +#include "../headers/descriptors.hpp" +#include "../headers/state.hpp" +#include "../headers/memory.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace cthreads::gpu { +namespace { + +// Host byte sizes for the scalar SSBO (std430 / SPIR-V). Not C++ sizeof for +// bool/int: GLSL bool is stored as 32-bit; use int32_t so Win/Linux match. +// string is not a scalar-SSBO type on the GPU path. +static const std::unordered_map py_size_of = { + {"bool", 4}, // int32 0/1 + {"int", 4}, // int32_t + {"float", 4}, // float + {"double", 8}, // float64; align 8 when laying out +}; + +size_t align_up(size_t value, size_t alignment) { + return (value + alignment - 1) & ~(alignment - 1); +} + +size_t std430_align_of(const std::string& kind) { + // std430: scalar alignment equals its size for these types. + return py_size_of.at(kind); +} + +void release_inflight(Context& context, SpawnedGpuKernel& job) { + // Return checked-out CB + fence to LaunchEngine (do not destroy the pool). + if (job.command_buffer != VK_NULL_HANDLE || job.fence != VK_NULL_HANDLE) { + LaunchResources resources{}; + resources.command_buffer = job.command_buffer; + resources.fence = job.fence; + job.command_buffer = VK_NULL_HANDLE; + job.fence = VK_NULL_HANDLE; + job.command_pool = VK_NULL_HANDLE; + return_launch_resources(context, resources); + } else { + job.command_pool = VK_NULL_HANDLE; + } + + if (job.descriptor_set != VK_NULL_HANDLE) { // free the descriptors (if not freed yet) + pack::free_set(context, job.descriptor_pool, job.descriptor_set); + } + pack::destroy_pool(context, job.descriptor_pool); + + // Borrowed GpuState buffers: drop handles without destroy_buffer. + for (size_t i = 0; i < job.pack.container_slots.size(); ++i) { + const bool owned = + (i < job.container_owned.size()) ? (job.container_owned[i] != 0) : true; + if (!owned) { + job.pack.container_slots[i].buffer = memory::GpuBuffer{}; + } + } + for (const std::string& name : job.resident_names) { + try { + memory::GpuState::getInstance().release_in_use(name); + } catch (...) { + // Best-effort on teardown / double-release paths. + } + } + job.resident_names.clear(); + job.container_owned.clear(); + + pack::destroy_gpu_pack(context, job.pack); + job.symbol.clear(); + job.writeback_lists.clear(); + job.values_keep.reset(); +} + +void mark_done(SpawnedGpuKernel& job) { + { // lock the done_mu mutex and set the done flag + std::lock_guard lock(job.done_mu); + job.done_flag = true; + } + job.done_cv.notify_all(); // notify all waiting threads (main thread currently, in later version also cthreads) +} + +// After dispatch fence: make SHADER_WRITE visible to TRANSFER_READ for downloads. +// Uses TransferEngine pool/fence under transfer_engine_mutex (same as uploads). +void compute_to_transfer_barrier(Context& context) { + std::lock_guard lock(context.transfer_engine_mutex); + if (context.transfer_engine.command_pool == VK_NULL_HANDLE || + context.transfer_engine.fence == VK_NULL_HANDLE || + !context.vkCmdPipelineBarrier || !context.vkAllocateCommandBuffers || + !context.vkBeginCommandBuffer || !context.vkEndCommandBuffer || + !context.vkQueueSubmit || !context.vkResetFences || + !context.vkWaitForFences || !context.vkFreeCommandBuffers || + !context.queue) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier missing " + "TransferEngine or entry points"); + } + + VkCommandBufferAllocateInfo alloc_info{}; + alloc_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO; + alloc_info.commandPool = context.transfer_engine.command_pool; + alloc_info.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + alloc_info.commandBufferCount = 1; + VkCommandBuffer cmd = VK_NULL_HANDLE; + if (context.vkAllocateCommandBuffers(context.device, &alloc_info, &cmd) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkAllocateCommandBuffers failed"); + } + + VkCommandBufferBeginInfo begin_info{}; + begin_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO; + begin_info.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + if (context.vkBeginCommandBuffer(cmd, &begin_info) != VK_SUCCESS) { + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkBeginCommandBuffer failed"); + } + + VkMemoryBarrier mem_barrier{}; + mem_barrier.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER; + mem_barrier.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT; + mem_barrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + context.vkCmdPipelineBarrier( + cmd, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, + 1, + &mem_barrier, + 0, + nullptr, + 0, + nullptr); + + if (context.vkEndCommandBuffer(cmd) != VK_SUCCESS) { + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkEndCommandBuffer failed"); + } + + VkFence te_fence = context.transfer_engine.fence; + if (context.vkResetFences(context.device, 1, &te_fence) != VK_SUCCESS) { + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkResetFences failed"); + } + + VkSubmitInfo submit{}; + submit.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO; + submit.commandBufferCount = 1; + submit.pCommandBuffers = &cmd; + if (context.vkQueueSubmit(context.queue, 1, &submit, te_fence) != + VK_SUCCESS) { + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkQueueSubmit failed"); + } + if (context.vkWaitForFences( + context.device, 1, &te_fence, VK_TRUE, UINT64_MAX) != VK_SUCCESS) { + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: compute_to_transfer_barrier " + "vkWaitForFences failed"); + } + context.vkFreeCommandBuffers( + context.device, context.transfer_engine.command_pool, 1, &cmd); +} + +// Download each ref list SSBO into the kept Python list (in place). +void writeback_ref_lists(Context& context, SpawnedGpuKernel& job) { + if (job.writeback_lists.empty()) { + return; + } + if (!job.values_keep) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: join writeback missing values_keep"); + } + + py::gil_scoped_acquire gil; + py::list& values = *job.values_keep; + + for (const SpawnedGpuKernel::WritebackListSlot& slot : job.writeback_lists) { + if (slot.numel == 0) { + continue; + } + if (slot.value_index >= static_cast(values.size())) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: writeback value_index out of " + "range"); + } + py::list list_val = values[slot.value_index].cast(); + if (static_cast(list_val.size()) != slot.numel) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: writeback list length changed " + "during job"); + } + + if (slot.elem_kind == "float") { + std::vector host(slot.numel); + pack::download_container( + context, + job.pack, + slot.container_index, + host.data(), + host.size() * sizeof(float)); + for (size_t j = 0; j < slot.numel; ++j) { + list_val[j] = host[j]; + } + } else if (slot.elem_kind == "int") { + std::vector host(slot.numel); + pack::download_container( + context, + job.pack, + slot.container_index, + host.data(), + host.size() * sizeof(std::int32_t)); + for (size_t j = 0; j < slot.numel; ++j) { + list_val[j] = host[j]; + } + } else if (slot.elem_kind == "bool") { + // GLSL bool is std430 32-bit 0/1 (same packing as scalar bool). + std::vector host(slot.numel); + pack::download_container( + context, + job.pack, + slot.container_index, + host.data(), + host.size() * sizeof(std::int32_t)); + for (size_t j = 0; j < slot.numel; ++j) { + list_val[j] = host[j] != 0; + } + } else if (slot.elem_kind == "double") { + std::vector host(slot.numel); + pack::download_container( + context, + job.pack, + slot.container_index, + host.data(), + host.size() * sizeof(double)); + for (size_t j = 0; j < slot.numel; ++j) { + list_val[j] = host[j]; + } + } else { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unsupported writeback " + "elem_kind: " + + slot.elem_kind); + } + } +} + +// Write one Python scalar into the host scalar blob at offset (std430 layout). +void write_scalar_bytes( + std::vector& blob, + size_t offset, + const std::string& kind, + const py::object& value +) { + if (kind == "int") { + const std::int32_t v = value.cast(); + std::memcpy(blob.data() + offset, &v, sizeof(v)); + } else if (kind == "float") { + const float v = value.cast(); + std::memcpy(blob.data() + offset, &v, sizeof(v)); + } else if (kind == "double") { + const double v = value.cast(); + std::memcpy(blob.data() + offset, &v, sizeof(v)); + } else if (kind == "bool") { + // GLSL bool in std430 is 32-bit; store 0/1 as int32. + const std::int32_t v = value.cast() ? 1 : 0; + std::memcpy(blob.data() + offset, &v, sizeof(v)); + } else { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: cannot pack scalar kind: " + kind); + } +} + +} // namespace + +void SpawnedGpuKernel::start() { + // Default path submits at launch time; start is a shared API no-op. +} + +void SpawnedGpuKernel::wait() { + std::unique_lock lock(done_mu); + done_cv.wait(lock, [this] { return done_flag; }); +} + +bool SpawnedGpuKernel::done() { + std::lock_guard lock(done_mu); + return done_flag; +} + +void SpawnedGpuKernel::join(Context& context, bool download) { + if (finished) { + if (eptr) { + std::rethrow_exception(eptr); + } + return; + } + + try { + if (fence != VK_NULL_HANDLE) { + if (!context.ready || context.device == VK_NULL_HANDLE || + !context.vkWaitForFences) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: SpawnedGpuKernel::join " + "needs a ready device and vkWaitForFences"); + } + // wait for the fence to be signaled + if (context.vkWaitForFences( + context.device, 1, &fence, VK_TRUE, UINT64_MAX) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkWaitForFences failed in " + "SpawnedGpuKernel::join"); + } + } + + // Permanent list writeback path (Threadable/schema marshal is later). + // download=false: fence only; resident GpuState buffers stay authoritative. + if (download && !writeback_lists.empty()) { + compute_to_transfer_barrier(context); + writeback_ref_lists(context, *this); + } + + mark_done(*this); // mark the job as done + release_inflight(context, *this); // release the inflight GPU state (clears all buffers, fences and cmd structures) + finished = true; + } catch (...) { + eptr = std::current_exception(); + mark_done(*this); + try { + release_inflight(context, *this); + } catch (...) { + // Prefer the original join error. + } + finished = true; + std::rethrow_exception(eptr); + } + + if (eptr) { + std::rethrow_exception(eptr); + } +} + +SpawnedGpuKernel::~SpawnedGpuKernel() { + if (finished) { + return; + } + try { + // try to await the fence and clear the inflight GPU state before destroying the object + Context& ctx = context(); + if (ctx.ready && ctx.device != VK_NULL_HANDLE) { + if (fence != VK_NULL_HANDLE && ctx.vkWaitForFences) { + ctx.vkWaitForFences( + ctx.device, 1, &fence, VK_TRUE, UINT64_MAX); + } + release_inflight(ctx, *this); + } + } catch (...) { + // Destructor must not throw. + } + mark_done(*this); // mark the job as done (doesnt mean the job finished successfully, just means it was terminated) + finished = true; +} + +std::shared_ptr launch_gpu_kernel( + py::dict meta, + py::list ordered_values +) { + // Launch path: + // Python args + gpu meta + // -> create/fill GpuPack (upload via staging) + // -> ShaderCache.get (create_entry/add is registry-only; missing => throw) + // -> allocate descriptor set, update_descriptors(pack) + // -> record CB: barrier, bind pipeline, bind set, dispatch + // -> create fence, vkQueueSubmit(..., fence) [no wait here] + // -> stash handles on SpawnedGpuKernel (pack, set, pool, CB, fence, symbol, groups) + // -> return SpawnedGpuKernel (no CThread) + // join(context): wait fence -> download/writeback -> release_inflight + + // Ensure Vulkan Context is ready (loader + device + TransferEngine). + init(); // init the context (noop if already initialized) + Context& context = cthreads::gpu::context(); + if (!context.ready) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: launch_gpu_kernel needs a ready " + "Context"); + } + + if (!meta.contains("symbol") || !meta.contains("params")) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: launch_gpu_kernel meta needs " + "'symbol' and 'params'"); + } + + const std::string symbol = meta["symbol"].cast(); // kernel / fn name + py::list params = meta["params"]; // list of input params (see docs string for example) + // ensure this fn call mathces the number of params in the kernels meta + if (static_cast(ordered_values.size()) != static_cast(params.size())) { + throw py::type_error( + "cthreads.gpu: expected " + std::to_string(params.size()) + + " args for '" + symbol + "', got " + + std::to_string(ordered_values.size())); + } + + // Walk params: scalars -> binding 0 blob; lists -> ContainerSpec (bindings 1..N). + // build the container specs + scalar layout (std430 align while summing) + std::vector container_specs; + std::vector scalar_host; // filled after we know scalar_bytes + struct ScalarSlot { // private helper struct to record metadata of scalar params in the kernel call + size_t value_index = 0; + size_t offset = 0; + std::string kind; + }; + std::vector scalar_slots; + struct ContainerSlotPlan { // private helper struct to record metadata of list params in the kernel call + size_t value_index = 0; + size_t elem_bytes = 0; + size_t numel = 0; + std::string elem_kind; + bool writeback = true; // pass_as ref (default for lists) + }; + std::vector container_plans; + + size_t scalar_bytes = 0; + // iter all the params and build the container specs + scalar layout (std430 align while summing) + for (size_t i = 0; i < static_cast(params.size()); ++i) { + // each param is a dict from meta (see launch_gpu_kernel docstring example) + py::dict param = params[i].cast(); // this param is a dict with meta like name, kind, pass_as, numel, elem_kind, elem_bytes, ... + std::string name = param["name"].cast(); // the var name / identifier (set by usr code) + std::string kind = param["kind"].cast(); // the type + const bool is_list = (kind == "list"); + // Scalars default value; lists default ref (join writeback). + std::string pass_as = param.contains("pass_as") + ? param["pass_as"].cast() + : (is_list ? std::string("ref") : std::string("value")); + (void)name; + + if (is_list) { // lists get theri own ssbo (single buffer) + // List SSBO: elem size from meta; numel from the Python list length. + std::string elem_kind = param.contains("elem_kind") //type (default float) + ? param["elem_kind"].cast() + : std::string("float"); + size_t elem_bytes = 0; + // check if the bytes are set in the meta, otherwise try to infer from the type or throw an err + if (param.contains("elem_bytes")) { + elem_bytes = param["elem_bytes"].cast(); + } else if (py_size_of.find(elem_kind) != py_size_of.end()) { + elem_bytes = py_size_of.at(elem_kind); + } else { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unknown list elem type: " + + elem_kind); + } + // Lists default to pass_as ref (in-place writeback on join). + if (pass_as != "ref" && pass_as != "value") { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unsupported pass_as for " + "list '" + + name + "': " + pass_as); + } + const bool do_writeback = (pass_as != "value"); + // gte the actuall value from the kernel call (py side) + py::list list_val = ordered_values[i].cast(); + const size_t numel = static_cast(list_val.size()); + // optional meta numel must match the live list if both present + if (param.contains("numel") && param["numel"].cast() != numel) { // check that sizes match the expectations form the meta + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: meta numel does not match " + "list length for '" + + name + "'"); + } + container_specs.push_back(pack::ContainerSpec{elem_bytes, numel}); + container_plans.push_back(ContainerSlotPlan{ + i, elem_bytes, numel, std::move(elem_kind), do_writeback}); + continue; + } + // --- handle scalar args --- + + // Scalar field into the binding-0 SSBO (not a separate descriptor). + if (py_size_of.find(kind) == py_size_of.end()) { // check if we can interpret this type + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unknown variable type: " + + kind); + } + if (pass_as != "value" && pass_as != "ref") { + // Scalars are always packed by value into the SSBO; pass_as is + // reserved for future semantics. Reject unknown tags early. + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unsupported pass_as for " + "scalar '" + + name + "': " + pass_as); + } + const size_t size = py_size_of.at(kind); + const size_t alignment = std430_align_of(kind); + scalar_bytes = align_up(scalar_bytes, alignment); + scalar_slots.push_back(ScalarSlot{i, scalar_bytes, kind}); + scalar_bytes += size; + } + + // Prefer meta scalar_bytes when present (codegen truth); must match walk. + if (meta.contains("scalar_bytes")) { + const size_t meta_bytes = meta["scalar_bytes"].cast(); + if (meta_bytes != scalar_bytes) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: meta scalar_bytes (" + + std::to_string(meta_bytes) + ") != layout sum (" + + std::to_string(scalar_bytes) + ")"); + } + } + + // Optional residency: value_index -> GpuState name (Python GpuArena). + // Those list SSBOs are borrowed; skip create/upload for them. + std::unordered_map resident_by_value_index; + if (meta.contains("resident") && !meta["resident"].is_none()) { + py::dict res = meta["resident"].cast(); + for (auto item : res) { + const size_t value_index = + py::reinterpret_borrow(item.first).cast(); + const std::string state_name = + py::reinterpret_borrow(item.second) + .cast(); + resident_by_value_index.emplace(value_index, state_name); + } + } + + // Job owns GPU objects from here on so failures can release_inflight. + auto job = std::make_shared(); + job->symbol = symbol; + // Keep the same Python arg objects for join writeback (list identity). + job->values_keep = std::make_shared(ordered_values); + for (size_t c = 0; c < container_plans.size(); ++c) { + const ContainerSlotPlan& plan = container_plans[c]; + if (!plan.writeback || plan.numel == 0) { + continue; + } + job->writeback_lists.push_back(SpawnedGpuKernel::WritebackListSlot{ + plan.value_index, + c, + plan.numel, + plan.elem_kind, + }); + } + // collect launch group data (required to ensure the correct num threads a re launched and the correct thread block shape is used) + if (meta.contains("group_count_x") && !meta["group_count_x"].is_none()) { + job->group_count_x = meta["group_count_x"].cast(); + } + if (meta.contains("group_count_y") && !meta["group_count_y"].is_none()) { + job->group_count_y = meta["group_count_y"].cast(); + } + if (meta.contains("group_count_z") && !meta["group_count_z"].is_none()) { + job->group_count_z = meta["group_count_z"].cast(); + } + + try { + // Build pack: owned scalars + per-list either create or borrow from GpuState. + job->container_owned.assign(container_plans.size(), 1); + if (scalar_bytes > 0) { + job->pack.scalar_buffer = memory::create_buffer( + context, + static_cast(scalar_bytes), + memory::BufferKind::DeviceLocal + ); + } + job->pack.container_slots.resize(container_plans.size()); + memory::GpuState& state = memory::GpuState::getInstance(); + + for (size_t c = 0; c < container_plans.size(); ++c) { + const ContainerSlotPlan& plan = container_plans[c]; + job->pack.container_slots[c].spec = + pack::ContainerSpec{plan.elem_bytes, plan.numel}; + if (plan.numel == 0) { + continue; + } + + auto res_it = resident_by_value_index.find(plan.value_index); + if (res_it != resident_by_value_index.end()) { + const std::string& state_name = res_it->second; + // Checkout before reading handles so remove cannot race. + state.mark_in_use(state_name); + job->resident_names.push_back(state_name); + memory::GpuBuffer& registered = state.get(state_name); + const VkDeviceSize need = + static_cast(plan.elem_bytes * plan.numel); + if (registered.size < need) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: resident buffer '" + + state_name + "' is too small for list arg"); + } + // Borrow handles; GpuState remains the owner. + job->pack.container_slots[c].buffer = registered; + job->container_owned[c] = 0; + continue; + } + + job->pack.container_slots[c].buffer = memory::create_buffer( + context, + static_cast(plan.elem_bytes * plan.numel), + memory::BufferKind::DeviceLocal + ); + job->container_owned[c] = 1; + } + + // Pack Python scalars into a host byte blob, then upload through staging. + if (scalar_bytes > 0) { + scalar_host.assign(scalar_bytes, 0); + for (const ScalarSlot& slot : scalar_slots) { + write_scalar_bytes( + scalar_host, + slot.offset, + slot.kind, + ordered_values[slot.value_index].cast()); + } + // upload the scalars + pack::upload_scalars( + context, job->pack, scalar_host.data(), scalar_bytes); + } + + // Upload each non-resident list container from ordered_values. + for (size_t c = 0; c < container_plans.size(); ++c) { + const ContainerSlotPlan& plan = container_plans[c]; + if (plan.numel == 0 || job->container_owned[c] == 0) { + continue; // empty or resident (already on device) + } + py::list list_val = ordered_values[plan.value_index].cast(); + if (plan.elem_kind == "float") { + std::vector host(plan.numel); + for (size_t j = 0; j < plan.numel; ++j) { + host[j] = list_val[j].cast(); + } + pack::upload_container( + context, + job->pack, + c, + host.data(), + host.size() * sizeof(float)); + } else if (plan.elem_kind == "int") { + std::vector host(plan.numel); + for (size_t j = 0; j < plan.numel; ++j) { + host[j] = list_val[j].cast(); + } + pack::upload_container( + context, + job->pack, + c, + host.data(), + host.size() * sizeof(std::int32_t)); + } else if (plan.elem_kind == "bool") { + // std430 bool = 4 bytes; coerce Python bool to 0/1 int32. + std::vector host(plan.numel); + for (size_t j = 0; j < plan.numel; ++j) { + host[j] = list_val[j].cast() ? 1 : 0; + } + pack::upload_container( + context, + job->pack, + c, + host.data(), + host.size() * sizeof(std::int32_t)); + } else if (plan.elem_kind == "double") { + std::vector host(plan.numel); + for (size_t j = 0; j < plan.numel; ++j) { + host[j] = list_val[j].cast(); + } + pack::upload_container( + context, + job->pack, + c, + host.data(), + host.size() * sizeof(double)); + } else { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: unsupported list elem_kind: " + + plan.elem_kind); + } + } + + // get the shader cache entry (must already be registered) + const shader::ShaderCacheEntry& entry = + shader::ShaderCache::getInstance().get(symbol); + + // binding_count = (scalars ? 1 : 0) + list count (matches Python Signature) + const uint32_t expected_bindings = + (scalar_bytes > 0 ? 1u : 0u) + + static_cast(container_specs.size()); + if (entry.binding_count != expected_bindings) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: ShaderCacheEntry binding_count (" + + std::to_string(entry.binding_count) + + ") != (scalars?1:0) + list count (" + + std::to_string(expected_bindings) + ")"); + } + if (entry.pipeline == VK_NULL_HANDLE || + entry.pipeline_layout == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: ShaderCacheEntry missing " + "pipeline for '" + + symbol + "'"); + } + + // get the descriptor pool to allocate the descriptor set next + job->descriptor_pool = + pack::create_pool(context, entry.binding_count, 1); + // allocate the descriptor set + job->descriptor_set = + pack::allocate_set(context, job->descriptor_pool, entry.set_layout); + // wire binding i -> pack buffer i (schema from entry, buffers from this pack) + pack::update_descriptors( + context, job->descriptor_set, entry, job->pack); + + // Need bind/dispatch/barrier PFNs (CB/fence come from LaunchEngine). + if (!context.vkBeginCommandBuffer || !context.vkEndCommandBuffer || + !context.vkCmdPipelineBarrier || !context.vkCmdBindPipeline || + !context.vkCmdBindDescriptorSets || !context.vkCmdDispatch) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: launch_gpu_kernel missing " + "dispatch/command entry points"); + } + + // Checkout CB + fence from Context LaunchEngine (pool is process-lifetime). + LaunchResources launch = checkout_launch_resources(context); + job->command_buffer = launch.command_buffer; + job->fence = launch.fence; + job->command_pool = context.launch_engine.command_pool; + + VkCommandBufferBeginInfo begin_info{}; + begin_info.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO; + begin_info.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + if (context.vkBeginCommandBuffer(job->command_buffer, &begin_info) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkBeginCommandBuffer failed in " + "launch_gpu_kernel"); + } + + // Uploads already waited on the TransferEngine fence, but Vulkan still + // needs a barrier so compute sees TRANSFER_WRITE results. + VkMemoryBarrier mem_barrier{}; + mem_barrier.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER; + mem_barrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + mem_barrier.dstAccessMask = + VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT; + context.vkCmdPipelineBarrier( + job->command_buffer, + VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + 0, + 1, + &mem_barrier, + 0, + nullptr, + 0, + nullptr); + + // Bind compute pipeline + this launch's descriptor set, then dispatch. + context.vkCmdBindPipeline( + job->command_buffer, + VK_PIPELINE_BIND_POINT_COMPUTE, + entry.pipeline); + context.vkCmdBindDescriptorSets( + job->command_buffer, + VK_PIPELINE_BIND_POINT_COMPUTE, + entry.pipeline_layout, + 0, + 1, + &job->descriptor_set, + 0, + nullptr); + context.vkCmdDispatch( + job->command_buffer, + job->group_count_x, + job->group_count_y, + job->group_count_z); + + if (context.vkEndCommandBuffer(job->command_buffer) != VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkEndCommandBuffer failed in " + "launch_gpu_kernel"); + } + + // Fence was checked out unsignaled; submit under launch_engine_mutex. + submit_launch(context, job->command_buffer, job->fence); + // Do not wait here — join() waits on job->fence. + } catch (...) { + // Tear down any handles already stashed; then rethrow to Python. + try { + release_inflight(context, *job); + } catch (...) { + // Prefer the original launch error. + } + throw; + } + + return job; +} + +} // namespace cthreads::gpu diff --git a/src/cthreads/cpp/gpu/impl/pack.cpp b/src/cthreads/cpp/gpu/impl/pack.cpp new file mode 100644 index 0000000..1da3cff --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/pack.cpp @@ -0,0 +1,227 @@ +#include "../headers/pack.hpp" +#include "../headers/memory.hpp" +#include "../headers/context.hpp" + +#include +#include + +namespace cthreads::gpu::pack { +namespace { + +[[noreturn]] void invalid_arg(const char* detail) { + throw std::invalid_argument( + std::string("cthreads.gpu.GpuInvalidArgument: ") + detail + ); +} + +[[noreturn]] void use_after_destroy(const char* detail) { + throw std::runtime_error( + std::string("cthreads.gpu.GpuUseAfterDestroy: ") + detail + ); +} + +} // namespace + +GpuPack create_gpu_pack( + cthreads::gpu::Context& context, + size_t scalar_bytes, + std::vector container_specs +) { + GpuPack pack; + pack.container_slots.reserve(container_specs.size()); + + if (scalar_bytes > 0) { + pack.scalar_buffer = memory::create_buffer( + context, + scalar_bytes, + memory::BufferKind::DeviceLocal + ); + } + + for (const auto& spec : container_specs) { + if (spec.elem_bytes == 0) { + invalid_arg("ContainerSpec must have positive elem_bytes"); + } + + ContainerSlot slot; + slot.spec = spec; + if (spec.numel > 0) { + slot.buffer = memory::create_buffer( + context, + static_cast(spec.numel * spec.elem_bytes), + memory::BufferKind::DeviceLocal + ); + } + pack.container_slots.push_back(std::move(slot)); + } + return pack; +} + +void upload_scalars( + cthreads::gpu::Context& context, + GpuPack& pack, + const void* data, + size_t size +) { + if (!data) { + invalid_arg("data is null"); + } + if (size == 0) { + invalid_arg("size must be non-zero"); + } + if (pack.scalar_buffer.buffer == VK_NULL_HANDLE) { + use_after_destroy("scalar buffer is not initialized"); + } + if (size > pack.scalar_buffer.size) { + invalid_arg("size is greater than the scalar buffer size"); + } + + memory::upload_buffer(context, pack.scalar_buffer, data, size); +} + +void upload_container( + cthreads::gpu::Context& context, + GpuPack& pack, + size_t index, + const void* data, + size_t size +) { + if (index >= pack.container_slots.size()) { + invalid_arg("container index is out of range"); + } + if (!data) { + invalid_arg("data is null"); + } + if (size == 0) { + invalid_arg("size must be non-zero"); + } + + ContainerSlot& slot = pack.container_slots[index]; + if (slot.spec.numel == 0 || slot.buffer.buffer == VK_NULL_HANDLE) { + use_after_destroy("container buffer is not initialized"); + } + const size_t expected = slot.spec.elem_bytes * slot.spec.numel; + if (size != expected) { + invalid_arg("size does not match the container size"); + } + + memory::upload_buffer(context, slot.buffer, data, size); +} + +void upload_containers( + cthreads::gpu::Context& context, + GpuPack& pack, + const std::vector& data, + const std::vector& sizes +) { + if (data.size() != sizes.size()) { + invalid_arg("data and sizes must have the same length"); + } + if (data.size() != pack.container_slots.size()) { + invalid_arg("data length must match container_slots size"); + } + + for (size_t i = 0; i < data.size(); ++i) { + if (pack.container_slots[i].spec.numel == 0) { + continue; + } + try { + upload_container(context, pack, i, data[i], sizes[i]); + } catch (const std::exception& e) { + throw std::runtime_error( + std::string(e.what()) + " [container " + std::to_string(i) + "]" + ); + } + } +} + +void download_scalars( + cthreads::gpu::Context& context, + GpuPack& pack, + void* data, + size_t size +) { + if (!data) { + invalid_arg("data is null"); + } + if (size == 0) { + invalid_arg("size must be non-zero"); + } + if (pack.scalar_buffer.buffer == VK_NULL_HANDLE) { + use_after_destroy("scalar buffer is not initialized"); + } + if (size > pack.scalar_buffer.size) { + invalid_arg("size is greater than the scalar buffer size"); + } + + memory::download_buffer(context, pack.scalar_buffer, data, size); +} + +void download_container( + cthreads::gpu::Context& context, + GpuPack& pack, + size_t index, + void* data, + size_t size +) { + if (index >= pack.container_slots.size()) { + invalid_arg("container index is out of range"); + } + if (!data) { + invalid_arg("data is null"); + } + if (size == 0) { + invalid_arg("size must be non-zero"); + } + + ContainerSlot& slot = pack.container_slots[index]; + if (slot.spec.numel == 0 || slot.buffer.buffer == VK_NULL_HANDLE) { + use_after_destroy("container buffer is not initialized"); + } + if (size > slot.buffer.size) { + invalid_arg("size is greater than the container buffer size"); + } + + memory::download_buffer(context, slot.buffer, data, size); +} + +void download_containers( + cthreads::gpu::Context& context, + GpuPack& pack, + std::vector& data, + const std::vector& sizes +) { + if (data.size() != sizes.size()) { + invalid_arg("data and sizes must have the same length"); + } + if (data.size() != pack.container_slots.size()) { + invalid_arg("data length must match container_slots size"); + } + + for (size_t i = 0; i < data.size(); ++i) { + if (pack.container_slots[i].spec.numel == 0) { + continue; + } + try { + download_container(context, pack, i, data[i], sizes[i]); + } catch (const std::exception& e) { + throw std::runtime_error( + std::string(e.what()) + " [container " + std::to_string(i) + "]" + ); + } + } +} + +void destroy_gpu_pack(cthreads::gpu::Context& context, GpuPack& pack) { + if (pack.scalar_buffer.buffer != VK_NULL_HANDLE) { + memory::destroy_buffer(context, pack.scalar_buffer); + } + for (auto& slot : pack.container_slots) { + if (slot.buffer.buffer != VK_NULL_HANDLE) { + memory::destroy_buffer(context, slot.buffer); + } + } + pack = GpuPack{}; +} + +} // namespace cthreads::gpu::pack diff --git a/src/cthreads/cpp/gpu/impl/shader.cpp b/src/cthreads/cpp/gpu/impl/shader.cpp new file mode 100644 index 0000000..ad5011f --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/shader.cpp @@ -0,0 +1,119 @@ +#include "../headers/shader.hpp" +#include "../headers/shader_cache.hpp" +#include "../headers/context.hpp" + +#include +#include + +namespace cthreads::gpu::shader { + +ShaderCacheEntry create_entry( + Context& context, + const uint32_t* spirv, + size_t spirv_word_count, + uint32_t binding_count +) { + if (!context.ready || context.device == VK_NULL_HANDLE) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: create_entry needs an initialized " + "device"); + } + if (spirv == nullptr || spirv_word_count == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: create_entry spirv is empty"); + } + if (binding_count == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: create_entry binding_count must " + "be >= 1"); + } + if (!context.vkCreateShaderModule || !context.vkCreateDescriptorSetLayout || + !context.vkCreatePipelineLayout || !context.vkCreateComputePipelines) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: create_entry missing shader/" + "pipeline create entry points"); + } + + ShaderCacheEntry entry{}; + entry.binding_count = binding_count; + + // 1) SPIR-V -> shader module + VkShaderModuleCreateInfo module_info{}; + module_info.sType = VK_STRUCTURE_TYPE_SHADER_MODULE_CREATE_INFO; + module_info.codeSize = spirv_word_count * sizeof(uint32_t); + module_info.pCode = spirv; + if (context.vkCreateShaderModule( + context.device, &module_info, nullptr, &entry.shader_module) != + VK_SUCCESS) { + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateShaderModule failed"); + } + + // 2) set layout: binding i is one STORAGE_BUFFER (compute). + std::vector bindings(binding_count); + for (uint32_t i = 0; i < binding_count; ++i) { + bindings[i] = {}; + bindings[i].binding = i; + bindings[i].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + bindings[i].descriptorCount = 1; + bindings[i].stageFlags = VK_SHADER_STAGE_COMPUTE_BIT; + bindings[i].pImmutableSamplers = nullptr; + } + + VkDescriptorSetLayoutCreateInfo layout_info{}; + layout_info.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO; + layout_info.bindingCount = binding_count; + layout_info.pBindings = bindings.data(); + if (context.vkCreateDescriptorSetLayout( + context.device, &layout_info, nullptr, &entry.set_layout) != + VK_SUCCESS) { + destroy_entry(context, entry); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateDescriptorSetLayout failed"); + } + + // 3) Pipeline layout (one set, no push constants. IF OPTIMIZATION REQUIRES THEM ADD PUSH CONSTS HERE). + VkPipelineLayoutCreateInfo pipe_layout_info{}; + pipe_layout_info.sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO; + pipe_layout_info.setLayoutCount = 1; + pipe_layout_info.pSetLayouts = &entry.set_layout; + pipe_layout_info.pushConstantRangeCount = 0; + pipe_layout_info.pPushConstantRanges = nullptr; + if (context.vkCreatePipelineLayout( + context.device, &pipe_layout_info, nullptr, &entry.pipeline_layout) != + VK_SUCCESS) { + destroy_entry(context, entry); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreatePipelineLayout failed"); + } + + // 4) Compute pipeline from module + layout (entry point "main"). + VkPipelineShaderStageCreateInfo stage{}; + stage.sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO; + stage.stage = VK_SHADER_STAGE_COMPUTE_BIT; + stage.module = entry.shader_module; + stage.pName = "main"; + + VkComputePipelineCreateInfo pipe_info{}; + pipe_info.sType = VK_STRUCTURE_TYPE_COMPUTE_PIPELINE_CREATE_INFO; + pipe_info.stage = stage; + pipe_info.layout = entry.pipeline_layout; + pipe_info.basePipelineHandle = VK_NULL_HANDLE; + pipe_info.basePipelineIndex = -1; + + if (context.vkCreateComputePipelines( + context.device, + VK_NULL_HANDLE, + 1, + &pipe_info, + nullptr, + &entry.pipeline) != VK_SUCCESS) { + destroy_entry(context, entry); + throw std::runtime_error( + "cthreads.gpu.VulkanInitFailed: vkCreateComputePipelines failed"); + } + + return entry; +} + +} // namespace cthreads::gpu::shader diff --git a/src/cthreads/cpp/gpu/impl/shader_cache.cpp b/src/cthreads/cpp/gpu/impl/shader_cache.cpp new file mode 100644 index 0000000..a953b61 --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/shader_cache.cpp @@ -0,0 +1,135 @@ +#include "../headers/shader_cache.hpp" +#include "../headers/context.hpp" +#include "../headers/shader.hpp" + +#include +#include + +namespace cthreads::gpu::shader { + +void destroy_entry(Context& context, ShaderCacheEntry& entry) { + if (context.device == VK_NULL_HANDLE) { + // Cannot destroy without a device; drop handle values only. + entry.shader_module = VK_NULL_HANDLE; + entry.set_layout = VK_NULL_HANDLE; + entry.pipeline_layout = VK_NULL_HANDLE; + entry.pipeline = VK_NULL_HANDLE; + entry.binding_count = 0; + return; + } + // Pipeline before layouts/module (children before parents). + if (entry.pipeline != VK_NULL_HANDLE && context.vkDestroyPipeline) { + context.vkDestroyPipeline(context.device, entry.pipeline, nullptr); + entry.pipeline = VK_NULL_HANDLE; + } + if (entry.pipeline_layout != VK_NULL_HANDLE && + context.vkDestroyPipelineLayout) { + context.vkDestroyPipelineLayout( + context.device, entry.pipeline_layout, nullptr); + entry.pipeline_layout = VK_NULL_HANDLE; + } + if (entry.set_layout != VK_NULL_HANDLE && + context.vkDestroyDescriptorSetLayout) { + context.vkDestroyDescriptorSetLayout( + context.device, entry.set_layout, nullptr); + entry.set_layout = VK_NULL_HANDLE; + } + if (entry.shader_module != VK_NULL_HANDLE && + context.vkDestroyShaderModule) { + context.vkDestroyShaderModule( + context.device, entry.shader_module, nullptr); + entry.shader_module = VK_NULL_HANDLE; + } + entry.binding_count = 0; +} + +ShaderCacheEntry::ShaderCacheEntry(ShaderCacheEntry&& other) noexcept + : shader_module(other.shader_module), + set_layout(other.set_layout), + pipeline_layout(other.pipeline_layout), + pipeline(other.pipeline), + binding_count(other.binding_count) { + other.shader_module = VK_NULL_HANDLE; + other.set_layout = VK_NULL_HANDLE; + other.pipeline_layout = VK_NULL_HANDLE; + other.pipeline = VK_NULL_HANDLE; + other.binding_count = 0; +} + +ShaderCacheEntry& ShaderCacheEntry::operator=(ShaderCacheEntry&& other) noexcept { + if (this == &other) { + return *this; + } + // Assumes this entry's handles are already null or ownership was transferred. + // clear() destroys before erase; move-assign is only for empty or stolen rows. + shader_module = other.shader_module; + set_layout = other.set_layout; + pipeline_layout = other.pipeline_layout; + pipeline = other.pipeline; + binding_count = other.binding_count; + other.shader_module = VK_NULL_HANDLE; + other.set_layout = VK_NULL_HANDLE; + other.pipeline_layout = VK_NULL_HANDLE; + other.pipeline = VK_NULL_HANDLE; + other.binding_count = 0; + return *this; +} + +ShaderCache& ShaderCache::getInstance() { + static ShaderCache instance; + return instance; +} + +ShaderCache::~ShaderCache() { + // Static teardown order vs Context is undefined. Shutdown must clear first + // so handles are already null here; only drop the map. + std::lock_guard lock(_cache_mutex); + _cache.clear(); +} + +const ShaderCacheEntry& ShaderCache::add( + const std::string& key, ShaderCacheEntry&& entry) { + std::lock_guard lock(_cache_mutex); + auto [it, inserted] = + _cache.emplace(key, std::move(entry)); + if (!inserted) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: shader cache entry already " + "exists: " + + key); + } + return it->second; +} + +const ShaderCacheEntry& ShaderCache::get(const std::string& key) { + std::lock_guard lock(_cache_mutex); + const auto it = _cache.find(key); + if (it == _cache.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: shader not found in cache: " + + key); + } + return it->second; +} + +void ShaderCache::clear(Context& context) { + std::lock_guard lock(_cache_mutex); + for (auto& [key, entry] : _cache) { + (void)key; + destroy_entry(context, entry); + } + _cache.clear(); +} + +const ShaderCacheEntry& ShaderRegistry::register_spirv( + Context& context, + const std::string& symbol, + const uint32_t* spirv, + size_t spirv_word_count, + uint32_t binding_count +) { + ShaderCacheEntry entry = create_entry(context, spirv, spirv_word_count, binding_count); + return ShaderCache::getInstance().add(symbol, std::move(entry)); +} + +} // namespace cthreads::gpu::shader diff --git a/src/cthreads/cpp/gpu/impl/state.cpp b/src/cthreads/cpp/gpu/impl/state.cpp new file mode 100644 index 0000000..749dd8c --- /dev/null +++ b/src/cthreads/cpp/gpu/impl/state.cpp @@ -0,0 +1,200 @@ +#include "../headers/state.hpp" +#include "../headers/context.hpp" + +#include +#include + +namespace cthreads::gpu::memory { + +GpuState& GpuState::getInstance() { + static GpuState instance; + return instance; +} + +GpuState::~GpuState() { + // Static teardown order vs Context is undefined. Shutdown must clear first + // so Vulkan handles are already destroyed; only drop the map here. + std::lock_guard lock(_mutex); + _entries.clear(); +} + +bool GpuState::contains(const std::string& name) const { + std::lock_guard lock(_mutex); + return _entries.find(name) != _entries.end(); +} + +size_t GpuState::size() const { + std::lock_guard lock(_mutex); + return _entries.size(); +} + +std::vector GpuState::names() const { + std::lock_guard lock(_mutex); + std::vector out; + out.reserve(_entries.size()); + for (const auto& pair : _entries) { + out.push_back(pair.first); + } + return out; +} + +void GpuState::add(const std::string& name, GpuBuffer&& buffer) { + if (name.empty()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState.add requires a non-empty " + "name"); + } + if (buffer.buffer == VK_NULL_HANDLE || buffer.memory == VK_NULL_HANDLE || + buffer.size == 0) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState.add requires a non-empty " + "GpuBuffer"); + } + + std::lock_guard lock(_mutex); + if (_entries.find(name) != _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState name already registered: " + + name); + } + + GpuStateEntry entry{}; + // GpuBuffer has no custom move that clears handles, so copy then null the + // caller only after the map insert succeeds (otherwise we would leak). + entry.buffer = buffer; + entry.in_use = false; + _entries.emplace(name, std::move(entry)); + buffer = GpuBuffer{}; +} + +void GpuState::remove(Context& context, const std::string& name) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + if (it->second.in_use) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState cannot remove in-use " + "buffer: " + + name); + } + destroy_buffer(context, it->second.buffer); + _entries.erase(it); +} + +GpuBuffer& GpuState::get(const std::string& name) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + return it->second.buffer; +} + +const GpuBuffer& GpuState::get(const std::string& name) const { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + return it->second.buffer; +} + +bool GpuState::is_in_use(const std::string& name) const { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + return it->second.in_use; +} + +void GpuState::mark_in_use(const std::string& name) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + if (it->second.in_use) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState buffer already in use: " + + name); + } + it->second.in_use = true; +} + +void GpuState::release_in_use(const std::string& name) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + if (!it->second.in_use) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState buffer is not in use: " + + name); + } + it->second.in_use = false; +} + +void GpuState::upload( + Context& context, + const std::string& name, + const void* data, + VkDeviceSize size +) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + if (it->second.in_use) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState cannot upload in-use " + "buffer: " + + name); + } + // upload_buffer validates size / kind; keep the map lock so remove cannot race. + upload_buffer(context, it->second.buffer, data, size); +} + +void GpuState::download( + Context& context, + const std::string& name, + void* data, + VkDeviceSize size +) { + std::lock_guard lock(_mutex); + auto it = _entries.find(name); + if (it == _entries.end()) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState unknown name: " + name); + } + if (it->second.in_use) { + throw std::runtime_error( + "cthreads.gpu.GpuInvalidArgument: GpuState cannot download in-use " + "buffer: " + + name); + } + download_buffer(context, it->second.buffer, data, size); +} + +void GpuState::clear(Context& context) { + std::lock_guard lock(_mutex); + for (auto& pair : _entries) { + // Process teardown: destroy even if a launch left in_use set. + destroy_buffer(context, pair.second.buffer); + pair.second.in_use = false; + } + _entries.clear(); +} + +} // namespace cthreads::gpu::memory diff --git a/src/cthreads/cpp/gpu/third_party_notices/README.md b/src/cthreads/cpp/gpu/third_party_notices/README.md new file mode 100644 index 0000000..8dd0642 --- /dev/null +++ b/src/cthreads/cpp/gpu/third_party_notices/README.md @@ -0,0 +1,12 @@ +# Third-party notices for native GLSL -> SPIR-V + +When `CTHREADS_GPU=ON`, cthreads links **Khronos glslang** into `_ext` so +`compile_glsl` works without installing `glslc` or other Vulkan SDK tools. + +glslang is the same compiler engine Google **shaderc** wraps. License texts +copied here at build time (see also the root `LICENSE` third-party section): + +- `glslang-LICENSE*` — Khronos glslang (Apache-2.0 / BSD-style components) + +End-user wheels that include the GPU extension must redistribute these notices +alongside the binary. diff --git a/src/cthreads/python/.gitignore b/src/cthreads/python/.gitignore new file mode 100644 index 0000000..0bf994e --- /dev/null +++ b/src/cthreads/python/.gitignore @@ -0,0 +1,11 @@ +# >>> cthreads (auto) +__Thread__/ +__Threadable__/ +__Gpu__/ +.cthreads_cache.json +cthreads_kernels.dll +cthreads_kernels.so +cthreads_kernels.lib +libcthreads_kernels.so +libcthreads_kernels.dylib +# <<< cthreads (auto) diff --git a/src/cthreads/python/cthreads/cache.py b/src/cthreads/python/cthreads/cache.py index fce73d4..55d6324 100644 --- a/src/cthreads/python/cthreads/cache.py +++ b/src/cthreads/python/cthreads/cache.py @@ -40,6 +40,7 @@ def source_fingerprint(*objs: Any) -> str: _GITIGNORE_PATTERNS = ( "__Thread__/", "__Threadable__/", + "__Gpu__/", ".cthreads_cache.json", "cthreads_kernels.dll", "cthreads_kernels.so", @@ -66,18 +67,30 @@ def cache_path_for_root(root: Path) -> Path: return root / CACHE_FILENAME +def _empty_cache(version: str) -> dict[str, Any]: + """Fresh cache document: CPU units + GPU units share one file, separate maps.""" + return { + "version": version, + "units": {}, + "gpu_units": {}, + "link_hash": None, + "binary": None, + } + + def load_cache(root: Path) -> dict[str, Any]: path = cache_path_for_root(root) version = REGISTRY.VERSION if not path.is_file(): - return {"version": version, "units": {}, "link_hash": None, "binary": None} + return _empty_cache(version) try: data = json.loads(path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError): - return {"version": version, "units": {}, "link_hash": None, "binary": None} + return _empty_cache(version) if data.get("version") != version: - return {"version": version, "units": {}, "link_hash": None, "binary": None} + return _empty_cache(version) data.setdefault("units", {}) + data.setdefault("gpu_units", {}) return data diff --git a/src/cthreads/python/cthreads/frontend/Gpu/Wrapper.py b/src/cthreads/python/cthreads/frontend/Gpu/Wrapper.py new file mode 100644 index 0000000..1d0d064 --- /dev/null +++ b/src/cthreads/python/cthreads/frontend/Gpu/Wrapper.py @@ -0,0 +1,3 @@ +from ...gpu.frontend.wrapper import Gpu + +__all__ = ["Gpu"] \ No newline at end of file diff --git a/src/cthreads/python/cthreads/frontend/Gpu/__init__.py b/src/cthreads/python/cthreads/frontend/Gpu/__init__.py new file mode 100644 index 0000000..1df7f83 --- /dev/null +++ b/src/cthreads/python/cthreads/frontend/Gpu/__init__.py @@ -0,0 +1,3 @@ +from .Wrapper import Gpu + +__all__ = ["Gpu"] \ No newline at end of file diff --git a/src/cthreads/python/cthreads/frontend/Registry/registry.py b/src/cthreads/python/cthreads/frontend/Registry/registry.py index 62f26c3..37c5183 100644 --- a/src/cthreads/python/cthreads/frontend/Registry/registry.py +++ b/src/cthreads/python/cthreads/frontend/Registry/registry.py @@ -10,7 +10,7 @@ if TYPE_CHECKING: from ...compiler.orchestrator.units import ThreadableUnit, ThreadUnit - + from ...gpu.compiler.orchestrator.gpu_unit import GpuUnit class Registry: """ @@ -29,6 +29,10 @@ def __init__(self) -> None: self.threadable_units: dict[str, ThreadableUnit] = {} self.thread_units: dict[str, ThreadUnit] = {} + # gpu stuff + self.gpu_functions: dict[str, object] = {} # function qualname -> function (@Gpu) + self.gpu_function_units: dict[str, GpuUnit] = {} # function qualname -> GpuUnit + def register_threadable(self, cls: type) -> None: """Registers a threadable class for code generation""" self.threadables[cls.__name__] = cls @@ -37,15 +41,23 @@ def register_thread(self, fn: object) -> None: """Registers a thread function for code generation""" self.threads[fn.__qualname__] = fn + def register_gpu_function(self, fn: object) -> None: + """Registers a gpu function for code generation""" + self.gpu_functions[fn.__qualname__] = fn + def clear(self) -> None: """Clears the registry""" self.threadables.clear() self.threads.clear() self.threadable_units.clear() self.thread_units.clear() + self.gpu_functions.clear() + self.gpu_function_units.clear() from ...kernel_meta import KERNELS + from ...gpu.gpu_kernel_meta import GPU_KERNELS KERNELS.clear() + GPU_KERNELS.clear() REGISTRY = Registry() diff --git a/src/cthreads/python/cthreads/gpu/__init__.py b/src/cthreads/python/cthreads/gpu/__init__.py new file mode 100644 index 0000000..2d508f0 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/__init__.py @@ -0,0 +1,72 @@ +""" +Vulkan GPU runtime probe API. + +Native access goes through `_ext_gpu_api` (`cthreads._ext.gpu`). +Public helpers and error types are re-exported from `frontend`. + +Launch helpers (`prepare` / `gpu` / `compile`) live in `runtime` so the +callable name `prepare` does not shadow a submodule. +""" + +from . import _ext_gpu_api +from .frontend import ( + BlockDim, + BlockIdx, + CThreadsGPUError, + GPUNotAvailable, + GlobalIdx, + GridDim, + Gpu, + GpuInvalidArgument, + GpuUseAfterDestroy, + ThreadIdx, + VulkanInitFailed, + VulkanLoaderNotFound, + VulkanNoDevice, + VulkanNotBuiltError, + VulkanOutOfMemory, + _map_error, + available, + device_name, + init, + shutdown, +) +from .arena import GpuArena +from .runtime import GpuJob, compile, gpu, prepare + + +def __getattr__(name: str): + if name == "_gpu": + return _ext_gpu_api._gpu + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + + +__all__ = [ + "Gpu", + "GpuArena", + "GpuJob", + "BlockDim", + "BlockIdx", + "GlobalIdx", + "GridDim", + "ThreadIdx", + "CThreadsGPUError", + "GPUNotAvailable", + "GpuInvalidArgument", + "GpuUseAfterDestroy", + "VulkanInitFailed", + "VulkanLoaderNotFound", + "VulkanNoDevice", + "VulkanNotBuiltError", + "VulkanOutOfMemory", + "_map_error", + "_gpu", + "_ext_gpu_api", + "available", + "compile", + "device_name", + "gpu", + "init", + "prepare", + "shutdown", +] diff --git a/src/cthreads/python/cthreads/gpu/_ext_gpu_api.py b/src/cthreads/python/cthreads/gpu/_ext_gpu_api.py new file mode 100644 index 0000000..befb337 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/_ext_gpu_api.py @@ -0,0 +1,189 @@ +""" +Lazy binding to `cthreads._ext.gpu` (native Vulkan submodule). + +Central entry for all Python code that talks to the C++ GPU package. +Soft-imports so CPU-only builds still import `cthreads.gpu` cleanly. +""" + +from __future__ import annotations + +from typing import Any + +try: + from cthreads._ext import gpu as _gpu +except ImportError: + _gpu = None # type: ignore[assignment] + + +def _require_ext_gpu(): + """ + Return the native `_ext.gpu` module. + + #### Returns + - module = pybind `cthreads._ext.gpu` submodule + + #### Raises + - RuntimeError = extension was built without CTHREADS_GPU + """ + if _gpu is None: + raise RuntimeError( + "cthreads._ext.gpu is not available - rebuild with -DCTHREADS_GPU=ON" + ) + return _gpu + + +def available() -> bool: + """ + Report whether the Vulkan loader and a compute device can initialize. + + Never raises. Returns False when the GPU extension is missing or init + would fail. + + #### Returns + - bool = True when a compute-capable GPU context can be created + """ + if _gpu is None: + return False + return bool(_gpu.available()) + + +def device_name() -> str: + """ + Return the active GPU device name (may call init first). + + #### Returns + - str = Vulkan device name string + + #### Raises + - RuntimeError = `_ext.gpu` is not built + - Exception = native init / device query failures (unmapped) + """ + return str(_require_ext_gpu().device_name()) + + +def init() -> None: + """ + Explicitly initialize the process-wide Vulkan context. + + #### Returns + - None + + #### Raises + - RuntimeError = `_ext.gpu` is not built + - Exception = native init failures (unmapped) + """ + _require_ext_gpu().init() + + +def shutdown() -> None: + """ + Destroy the Vulkan device/instance and unload the loader. + + No-op when the GPU extension is not built. + + #### Returns + - None + """ + if _gpu is None: + return + _gpu.shutdown() + + +def testing() -> Any | None: + """ + Return the test-only `_ext.gpu.testing` submodule when present. + + #### Returns + - module | None = testing helpers, or None if missing / not built + """ + if _gpu is None: + return None + return getattr(_gpu, "testing", None) + + +def launch_gpu_kernel(meta: dict[str, Any], ordered_values: list[Any]) -> Any: + """ + Submit one GPU kernel from metadata and ordered Python arguments. + + Returns a native GpuJob handle. Does not wait; the caller joins that + handle for fence wait and list writeback. The kernel `symbol` must already + be registered in the ShaderCache. + + #### Args: + - meta: dict[str, Any] = kernel metadata (`symbol`, `params`, layout fields) + - ordered_values: list[Any] = arguments in parameter order matching `meta` + + #### Returns + - Any = native `_ext.gpu.GpuJob` (SpawnedGpuKernel) + + #### Raises + - RuntimeError = `_ext.gpu` is not built + - Exception = native launch failures (unmapped) + + #### Technical terms: + - GpuJob: per-launch GPU job handle (fence, pack, writeback plan) + - writeback: download of ref list buffers into the same Python list objects + """ + return _require_ext_gpu().launch_gpu_kernel(meta, ordered_values) + + +def compile_glsl(source: str) -> bytes: + """ + Compile GLSL compute source to SPIR-V via vendored glslang in `_ext`. + + #### Args: + - source: str = full compute shader text + + #### Returns + - bytes = SPIR-V binary + + #### Raises + - RuntimeError = `_ext.gpu` is not built or lacks compile_glsl + - Exception = native compile failures (unmapped) + """ + gpu = _require_ext_gpu() + compile_native = getattr(gpu, "compile_glsl", None) + if not callable(compile_native): + raise RuntimeError( + "cthreads._ext.gpu.compile_glsl is missing - rebuild with " + "CTHREADS_GPU=ON (glslang vendored into _ext)" + ) + return bytes(compile_native(source)) + + +def register_shader(symbol: str, spirv: bytes, binding_count: int) -> None: + """ + Insert a compute pipeline into the process ShaderCache from SPIR-V bytes. + + Calls native `ShaderRegistry::register_spirv` (sole writer). Duplicate + symbols raise. + + #### Args: + - symbol: str = ShaderCache key / kernel name + - spirv: bytes = SPIR-V binary (multiple of 4 bytes) + - binding_count: int = number of STORAGE_BUFFER bindings (>= 1) + + #### Returns + - None + + #### Raises + - RuntimeError = `_ext.gpu` is not built + - Exception = native create/insert failures (unmapped) + """ + _require_ext_gpu().register_shader(symbol, spirv, binding_count) + + +def gpu_state() -> Any: + """ + Return the process-wide native GpuState singleton. + + Named device-local buffers live here outside of a single launch. Does not + expose Vulkan handles; use `add(name, nbytes)` / `remove(name)` / etc. + + #### Returns + - Any = native `_ext.gpu.GpuState` + + #### Raises + - RuntimeError = `_ext.gpu` is not built + """ + return _require_ext_gpu().GpuState.instance() diff --git a/src/cthreads/python/cthreads/gpu/arena.py b/src/cthreads/python/cthreads/gpu/arena.py new file mode 100644 index 0000000..f5ec6b4 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/arena.py @@ -0,0 +1,309 @@ +""" +GpuArena: bind Python lists into process GpuState for launch reuse. + +Option B launch style: pass the same list objects to `gpu()` after bind. +Launch checks id + length; resident buffers skip alloc/upload. Host edits +between syncs are the caller's responsibility (no proxy yet). +""" + +from __future__ import annotations + +import struct +import threading +import uuid +from dataclasses import dataclass +from typing import Any, Iterator + +from . import _ext_gpu_api +from .frontend.errors import GPUNotAvailable, GpuInvalidArgument + +_ELEM_BYTES: dict[str, int] = { + "bool": 4, + "int": 4, + "float": 4, + "double": 8, +} + +# id(list) -> BoundSlot for every live arena bind (process-wide Option B lookup). +_ID_TO_SLOT: dict[int, "BoundSlot"] = {} +_ID_LOCK = threading.Lock() + + +@dataclass +class BoundSlot: + """One arena-bound Python list and its GpuState name.""" + + arena_id: str + name: str + state_name: str + host: list[Any] + numel: int + elem_kind: str + elem_bytes: int + + +def lookup_resident(value: Any) -> BoundSlot | None: + """ + Return the BoundSlot for a Python list if it is currently arena-bound. + + #### Args: + - value: Any = launch argument (typically a list) + + #### Returns + - BoundSlot | None = slot when id(value) is registered + """ + if not isinstance(value, list): + return None + with _ID_LOCK: + return _ID_TO_SLOT.get(id(value)) + + +def infer_elem_kind(values: list[Any]) -> str: + """ + Infer GPU list elem_kind from the first element (bool before int). + + #### Args: + - values: list[Any] = non-empty host list + + #### Returns + - str = "bool" | "int" | "float" | "double" + + #### Raises + - GpuInvalidArgument = empty list or unsupported element type + """ + if not values: + raise GpuInvalidArgument( + "GpuArena.bind: cannot infer elem type from an empty list " + "(pass a non-empty list)" + ) + sample: Any = values[0] + if isinstance(sample, bool): + return "bool" + if isinstance(sample, int): + return "int" + if isinstance(sample, float): + return "float" + raise GpuInvalidArgument( + f"GpuArena.bind: unsupported list element type {type(sample)!r}" + ) + + +def list_to_bytes(values: list[Any], elem_kind: str) -> bytes: + """Pack a Python list into std430-friendly host bytes.""" + if elem_kind == "float": + return struct.pack(f"{len(values)}f", *[float(v) for v in values]) + if elem_kind == "double": + return struct.pack(f"{len(values)}d", *[float(v) for v in values]) + if elem_kind == "int": + return struct.pack(f"{len(values)}i", *[int(v) for v in values]) + if elem_kind == "bool": + return struct.pack( + f"{len(values)}i", *[1 if bool(v) else 0 for v in values] + ) + raise GpuInvalidArgument(f"unsupported elem_kind: {elem_kind!r}") + + +def bytes_into_list(data: bytes, values: list[Any], elem_kind: str) -> None: + """Write downloaded bytes back into the same Python list object.""" + n: int = len(values) + if elem_kind == "float": + unpacked = struct.unpack(f"{n}f", data) + for i, v in enumerate(unpacked): + values[i] = float(v) + return + if elem_kind == "double": + unpacked = struct.unpack(f"{n}d", data) + for i, v in enumerate(unpacked): + values[i] = float(v) + return + if elem_kind == "int": + unpacked = struct.unpack(f"{n}i", data) + for i, v in enumerate(unpacked): + values[i] = int(v) + return + if elem_kind == "bool": + unpacked = struct.unpack(f"{n}i", data) + for i, v in enumerate(unpacked): + values[i] = bool(v) + return + raise GpuInvalidArgument(f"unsupported elem_kind: {elem_kind!r}") + + +class GpuArena: + """ + Session that keeps named list buffers resident in GpuState. + + #### Example: + ``py + with GpuArena() as arena: + arena.bind(x=x, y=y) + for _ in range(100): + gpu(saxpy, n, 2.0, x, y).join(download=False) + arena.sync() + `` + """ + + def __init__(self) -> None: + self._id: str = uuid.uuid4().hex + self._slots: dict[str, BoundSlot] = {} + self._released: bool = False + + def __enter__(self) -> "GpuArena": + return self + + def __exit__(self, *args: Any) -> None: + self.release() + + def bind(self, **named_lists: list[Any]) -> "GpuArena": + """ + Allocate/upload device buffers for the given lists (by kwarg name). + + Reuses GpuState entries when the same slot name is rebound with the + same length and elem kind; otherwise removes and recreates. + + #### Args: + - **named_lists: list[Any] = keyword slot name -> Python list object + + #### Returns + - GpuArena = this arena (for chaining) + + #### Raises + - GPUNotAvailable = GPU extension missing + - GpuInvalidArgument = bad args, empty lists, duplicate list ids + """ + if self._released: + raise GpuInvalidArgument("GpuArena.bind: arena already released") + if not named_lists: + raise GpuInvalidArgument("GpuArena.bind: expected at least one list") + if not _ext_gpu_api.available(): + raise GPUNotAvailable( + "GPU is not available (build with CTHREADS_GPU=ON and a Vulkan device)" + ) + + state = _ext_gpu_api.gpu_state() + _ext_gpu_api.init() + + for slot_name, host in named_lists.items(): + if not isinstance(host, list): + raise GpuInvalidArgument( + f"GpuArena.bind: {slot_name!r} must be a list, got {type(host)!r}" + ) + elem_kind: str = infer_elem_kind(host) + elem_bytes: int = _ELEM_BYTES[elem_kind] + numel: int = len(host) + nbytes: int = numel * elem_bytes + state_name: str = f"{self._id}/{slot_name}" + host_id: int = id(host) + + with _ID_LOCK: + existing = _ID_TO_SLOT.get(host_id) + if existing is not None and ( + existing.arena_id != self._id or existing.name != slot_name + ): + raise GpuInvalidArgument( + f"GpuArena.bind: list for {slot_name!r} is already bound " + f"as {existing.arena_id}/{existing.name}" + ) + + # Replace prior slot with the same name if shape changed. + prior = self._slots.get(slot_name) + if prior is not None: + if ( + prior.host is host + and prior.numel == numel + and prior.elem_kind == elem_kind + ): + state.upload(state_name, list_to_bytes(host, elem_kind)) + continue + self._unbind_slot(prior, state) + + if state.contains(state_name): + state.remove(state_name) + state.add(state_name, nbytes) + state.upload(state_name, list_to_bytes(host, elem_kind)) + + slot = BoundSlot( + arena_id=self._id, + name=slot_name, + state_name=state_name, + host=host, + numel=numel, + elem_kind=elem_kind, + elem_bytes=elem_bytes, + ) + self._slots[slot_name] = slot + with _ID_LOCK: + _ID_TO_SLOT[host_id] = slot + return self + + def sync(self, *names: str) -> None: + """ + Download resident buffers into the bound Python lists. + + #### Args: + - *names: str = optional slot names; default = all bound slots + + #### Raises + - GpuInvalidArgument = unknown name or arena released + """ + if self._released: + raise GpuInvalidArgument("GpuArena.sync: arena already released") + state = _ext_gpu_api.gpu_state() + targets: list[BoundSlot] + if names: + targets = [] + for name in names: + slot = self._slots.get(name) + if slot is None: + raise GpuInvalidArgument( + f"GpuArena.sync: unknown slot {name!r}" + ) + targets.append(slot) + else: + targets = list(self._slots.values()) + + for slot in targets: + if len(slot.host) != slot.numel: + raise GpuInvalidArgument( + f"GpuArena.sync: list for {slot.name!r} changed length " + f"(was {slot.numel}, now {len(slot.host)}); rebind" + ) + data: bytes = bytes(state.download(slot.state_name)) + bytes_into_list(data, slot.host, slot.elem_kind) + + def release(self) -> None: + """Destroy all GpuState buffers owned by this arena and drop id map entries.""" + if self._released: + return + state = None + try: + if _ext_gpu_api.available(): + state = _ext_gpu_api.gpu_state() + except Exception: + state = None + for slot in list(self._slots.values()): + self._unbind_slot(slot, state) + self._slots.clear() + self._released = True + + def _unbind_slot(self, slot: BoundSlot, state: Any) -> None: + with _ID_LOCK: + cur = _ID_TO_SLOT.get(id(slot.host)) + if cur is slot: + del _ID_TO_SLOT[id(slot.host)] + if state is not None and state.contains(slot.state_name): + try: + state.remove(slot.state_name) + except Exception: + pass + self._slots.pop(slot.name, None) + + def __contains__(self, name: str) -> bool: + return name in self._slots + + def names(self) -> list[str]: + """Registered slot names in this arena.""" + return list(self._slots.keys()) + + def __iter__(self) -> Iterator[str]: + return iter(self._slots) diff --git a/src/cthreads/python/cthreads/gpu/compiler/orchestrator/__init__.py b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/__init__.py new file mode 100644 index 0000000..53a1657 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/__init__.py @@ -0,0 +1,7 @@ +from .gpu_unit import GpuUnit +from .gpu_compile_session import GpuCompileSession + +__all__ = [ + "GpuUnit", + "GpuCompileSession", +] diff --git a/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_compile_session.py b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_compile_session.py new file mode 100644 index 0000000..e0ba5f0 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_compile_session.py @@ -0,0 +1,93 @@ +""" +Drain `REGISTRY.gpu_functions` into `GpuUnit`s and run emit. + +Does not keep its own unit maps; units live on REGISTRY. +""" + +from __future__ import annotations + +import inspect +from pathlib import Path +from typing import Any, get_type_hints + +from ....cache import ensure_gitignore, load_cache, save_cache +from ....compiler.orchestrator.units.handle import Handle +from ....frontend.Registry import REGISTRY +from ....gpu.gpu_kernel_meta import GPU_KERNELS, build_gpu_kernel_meta +from ....types import PyType, hint_to_pytype +from .gpu_unit import GpuUnit + + +class GpuCompileSession: + """ + Stateless GPU compile pass: registry functions -> units -> emit -> cache. + """ + + @staticmethod + def compile(force: bool = False) -> dict[str, Any]: + """ + Build GpuUnits for every registered `@Gpu` function and emit. + + #### Args: + - force: bool = passed through to `GpuUnit.emit` (default False) + + #### Returns + - dict[str, Any] = `root`, `cache`, `rewritten` (unit names that rewrote) + + #### Raises + - RuntimeError = nothing registered + - TypeError = bad annotations / non-None return + """ + REGISTRY.gpu_function_units.clear() + GPU_KERNELS.clear() + + if not REGISTRY.gpu_functions: + raise RuntimeError("Nothing registered to compile") + + sample: Any = next(iter(REGISTRY.gpu_functions.values())) + root = Path(inspect.getfile(sample)).resolve().parent + ensure_gitignore(root) + cache = load_cache(root) + rewritten: list[str] = [] + + for qualname, fn in list(REGISTRY.gpu_functions.items()): + src_file = Path(inspect.getfile(fn)).resolve() + hints = get_type_hints(fn) + params: list[tuple[str, PyType]] = [] + for pname in inspect.signature(fn).parameters: + if pname not in hints: + raise TypeError( + f"GPU function {qualname}: " + f"parameter {pname!r} needs a type annotation" + ) + params.append((pname, hint_to_pytype(hints[pname]))) + + ret_hint = hints.get("return") + if ret_hint not in (None, type(None)): + raise TypeError( + f"GPU function {qualname}: " + f"return type {ret_hint!r} is not allowed; use -> None " + f"with in-place list updates" + ) + return_type = None + + # Rebuild meta after GPU_KERNELS.clear (decorator may have built it earlier). + build_gpu_kernel_meta(fn) + + gpu_unit = GpuUnit( + handle=Handle( + name=qualname, + path=str(src_file), + target=fn, + ), + params=params, + return_type=return_type, + ) + REGISTRY.gpu_function_units[qualname] = gpu_unit + + for unit in REGISTRY.gpu_function_units.values(): + if unit.emit(force=force, cache=cache): + rewritten.append(unit.handle.name) + + save_cache(root, cache) + return {"root": root, "cache": cache, "rewritten": rewritten} diff --git a/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_unit.py b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_unit.py new file mode 100644 index 0000000..198f96f --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/orchestrator/gpu_unit.py @@ -0,0 +1,119 @@ +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from ....cache import source_fingerprint, write_if_changed +from ....compiler.orchestrator.units.baseUnit import BaseUnit +from ....types.pyType import PyType +from ... import _ext_gpu_api +from ...gpu_kernel_meta import build_gpu_kernel_meta +from ..translation.translate import translate_function_for_gpu + + +@dataclass +class GpuUnit(BaseUnit): + """ + One `@Gpu` free function: translate GLSL, compile SPIR-V, register shader. + """ + + params: list[tuple[str, PyType]] + return_type: PyType | None + + def validate(self) -> None: + """ + Ensure the handle target is a `@Gpu` function. + + #### Raises + - TypeError = target is not marked `@Gpu` + """ + fn = self.handle.target + if not getattr(fn, "__gpu__", False): + raise TypeError(f"{self.handle.name} is not a @Gpu function") + + def emit(self, *, force: bool = False, cache: dict[str, Any] | None = None) -> bool: + """ + Translate, compile with shaderc, write artifacts, register SPIR-V. + + Always registers into the process ShaderCache (Vulkan state is not on + disk). The source fingerprint only skips rewriting `__Gpu__` files. + + #### Args: + - force: bool = rewrite `__Gpu__` artifacts even if the hash matches + - cache: dict[str, Any] | None = shared `.cthreads_cache.json` document + + #### Returns + - bool = True if `__Gpu__` files were rewritten + + #### Raises + - RuntimeError = GLSL/SPIR-V compile failed or GPU ext missing + """ + self.validate() + fn = self.handle.target + meta = getattr(fn, "__gpu_kernel_meta__", None) + if not isinstance(meta, dict): + meta = build_gpu_kernel_meta(fn).to_dict() + + src_hash: str = source_fingerprint(fn) + local_size_x: int = int(meta.get("local_size_x", 64)) + result = translate_function_for_gpu( + fn, local_size_x=local_size_x, compile_spirv=True + ) + if result.spirv is None: + raise RuntimeError( + f"GPU unit {self.handle.name}: SPIR-V compile produced no bytes" + ) + + src_file: Path = Path(self.handle.path).resolve() + out_dir: Path = src_file.parent / "__Gpu__" + out_dir.mkdir(parents=True, exist_ok=True) + comp_path: Path = out_dir / f"{result.func_name}.comp" + spv_path: Path = out_dir / f"{result.func_name}.spv" + + rewritten: bool = False + gpu_units: dict[str, Any] = {} + if cache is not None: + gpu_units = cache.setdefault("gpu_units", {}) + prev = gpu_units.get(self.handle.name) + hash_ok = ( + isinstance(prev, dict) + and prev.get("hash") == src_hash + and not force + ) + else: + hash_ok = False + + if not hash_ok: + rewritten = write_if_changed(comp_path, result.source) + prev_spv: bytes | None = None + if spv_path.is_file(): + try: + prev_spv = spv_path.read_bytes() + except OSError: + prev_spv = None + if prev_spv != result.spirv: + spv_path.write_bytes(result.spirv) + rewritten = True + + symbol: str = str(meta.get("symbol", result.func_name)) + binding_count: int = int(meta.get("binding_count", result.binding_count)) + # Always populate process ShaderCache (also after shutdown released it). + # Disk fingerprint only gates `__Gpu__` file rewrites above. + try: + _ext_gpu_api.register_shader(symbol, result.spirv, binding_count) + except Exception as exc: + msg: str = str(exc) + if "already exists" not in msg: + raise + + if cache is not None: + gpu_units[self.handle.name] = { + "hash": src_hash, + "symbol": symbol, + "binding_count": binding_count, + "registered": True, + "comp": str(comp_path), + "spv": str(spv_path), + } + return rewritten diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/Glsl.py b/src/cthreads/python/cthreads/gpu/compiler/translation/Glsl.py new file mode 100644 index 0000000..3e3d8cd --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/Glsl.py @@ -0,0 +1,86 @@ +""" +Map cthreads PyTypes to GLSL type names and std430 sizes for GPU Signature. + +Python float -> GLSL float (f32). CPU @Thread uses double; GPU v1 does not. +Lists are not a GLSL type string — use type_name on the element type. +""" + +from __future__ import annotations + +from ....types import ( + PyBool, + PyDict, + PyFloat, + PyInt, + PyList, + PyString, + PyThreadable, + PyType, + is_shared_pytype, + is_sync_pytype, + is_tbuffer_pytype, +) + +# GLSL / host pack sizes (must match gpu_kernel_meta._GPU_SCALAR_BYTES). +_SCALAR: dict[type, tuple[str, int]] = { + PyInt: ("int", 4), + PyFloat: ("float", 4), + PyBool: ("bool", 4), +} + + +def _scalar_entry(py_type: PyType) -> tuple[str, int]: + for cls, entry in _SCALAR.items(): + if isinstance(py_type, cls): + return entry + raise TypeError( + f"GPU GLSL: unsupported scalar type {py_type.name!r} " + f"(allowed: int, float, bool)" + ) + + +def type_name(py_type: PyType) -> str: + """ + GLSL type name for a scalar PyType (`int` / `float` / `bool`). + + For lists, pass `py_type.inner_type` (or use `elem_type_name`). + """ + if isinstance(py_type, PyList): + raise TypeError( + "GPU GLSL: list is not a scalar type name; " + "use elem_type_name(py_type) or type_name(py_type.inner_type)" + ) + if ( + is_sync_pytype(py_type) + or is_shared_pytype(py_type) + or is_tbuffer_pytype(py_type) + or isinstance(py_type, (PyDict, PyString, PyThreadable)) + ): + raise TypeError( + f"GPU GLSL: {py_type.name!r} is not supported on the GPU path" + ) + return _scalar_entry(py_type)[0] + + +def elem_type_name(py_type: PyList) -> str: + """GLSL element type for a `list[...]` PyType (e.g. `float` for list[float]).""" + if not isinstance(py_type, PyList): + raise TypeError( + f"GPU GLSL: elem_type_name expects PyList, got {type(py_type)!r}" + ) + return type_name(py_type.inner_type) + + +def size_bytes(py_type: PyType) -> int: + """std430 size in bytes for a scalar (lists: use size_bytes on the element).""" + if isinstance(py_type, PyList): + raise TypeError( + "GPU GLSL: list has no single scalar size; " + "use size_bytes(py_type.inner_type) for the element" + ) + return _scalar_entry(py_type)[1] + + +def align_bytes(py_type: PyType) -> int: + """std430 alignment for a v1 scalar (same as size for int/float/bool).""" + return size_bytes(py_type) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/Signature.py b/src/cthreads/python/cthreads/gpu/compiler/translation/Signature.py new file mode 100644 index 0000000..5ce7e65 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/Signature.py @@ -0,0 +1,101 @@ +from __future__ import annotations + +import ast +from typing import get_type_hints + +from ....types import PyList, hint_to_pytype +from .Glsl import align_bytes, elem_type_name, size_bytes, type_name +from .context import GpuTranslationContext +from .result import GpuSignatureResult, ListField, ScalarField + + +class GpuSignature: + """Emit GLSL buffer preamble; fill ctx.symbols and binding_names.""" + + @staticmethod + def translate( + func_def: ast.FunctionDef, ctx: GpuTranslationContext + ) -> GpuSignatureResult: + # No Threadable owners yet — no localns. + hints = get_type_hints(ctx.fn) + args = list(func_def.args.args) + + if func_def.args.vararg or func_def.args.kwarg or func_def.args.kwonlyargs: + raise TypeError( + f"GPU function {ctx.func_name}: " + "*args/**kwargs/kw-only args are not supported" + ) + + ret_hint = hints.get("return") + if ret_hint is not None and ret_hint is not type(None): + raise TypeError( + f"GPU function {ctx.func_name}: return must be None " + f"(in-place list writeback only), got {ret_hint!r}" + ) + + scalar_fields: list[ScalarField] = [] + # (name, glsl_elem) — binding numbers assigned after we know if scalars exist. + pending_lists: list[tuple[str, str]] = [] + scalar_bytes = 0 + + for arg in args: + if arg.arg not in hints: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"parameter {arg.arg!r} needs a type annotation" + ) + py_type = hint_to_pytype(hints[arg.arg]) + ctx.symbols[arg.arg] = py_type + + if isinstance(py_type, PyList): + glsl_elem = elem_type_name(py_type) + pending_lists.append((arg.arg, glsl_elem)) + ctx.list_params.add(arg.arg) + else: + glsl_ty = type_name(py_type) + align = align_bytes(py_type) + scalar_bytes = (scalar_bytes + align - 1) & ~(align - 1) + scalar_bytes += size_bytes(py_type) + scalar_fields.append((arg.arg, glsl_ty)) + ctx.scalar_params.add(arg.arg) + + # Scalars at binding 0 when present; lists start at 1 or 0 accordingly. + list_base = 1 if scalar_fields else 0 + list_fields: list[ListField] = [] + for i, (name, glsl_elem) in enumerate(pending_lists): + binding = list_base + i + list_fields.append((binding, name, glsl_elem)) + ctx.binding_names.append((binding, name, "list")) + + if scalar_fields: + ctx.binding_names.insert(0, (0, "scalars", "scalar")) + + binding_count = (1 if scalar_fields else 0) + len(list_fields) + + lines: list[str] = [f"layout(local_size_x = {ctx.local_size_x}) in;", ""] + if scalar_fields: + lines.append("layout(set = 0, binding = 0, std430) buffer Scalars {") + for name, glsl_ty in scalar_fields: + lines.append(f" {glsl_ty} {name};") + lines.append("} scalars;") + lines.append("") + for binding, name, glsl_elem in list_fields: + block = name[:1].upper() + name[1:] + lines.append( + f"layout(set = 0, binding = {binding}, std430) buffer {block} {{" + ) + lines.append(f" {glsl_elem} data[];") + lines.append(f"}} {name};") + lines.append("") + + preamble = "\n".join(lines).rstrip() + "\n" + + return GpuSignatureResult( + func_name=func_def.name, + preamble=preamble, + binding_count=binding_count, + scalar_bytes=scalar_bytes, + local_size_x=ctx.local_size_x, + scalar_fields=scalar_fields, + list_fields=list_fields, + ) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/Typeof.py b/src/cthreads/python/cthreads/gpu/compiler/translation/Typeof.py new file mode 100644 index 0000000..e69de29 diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/assemble.py b/src/cthreads/python/cthreads/gpu/compiler/translation/assemble.py new file mode 100644 index 0000000..6e6dc63 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/assemble.py @@ -0,0 +1,40 @@ +""" +Assemble a full GLSL compute shader (`.comp`) from preamble + body. +""" + +from __future__ import annotations + +_DEFAULT_GLSL_VERSION: int = 450 + + +def assemble_comp( + preamble: str, + body: str, + *, + version: int = _DEFAULT_GLSL_VERSION, +) -> str: + """ + Build a complete compute shader string: `#version`, buffers, `void main()`. + + Body lines from GpuSyntax are already indented; they become the main body. + + #### Args: + - preamble: str = Signature output (local_size + buffer layouts) + - body: str = lowered statement lines (already indented) + - version: int = GLSL version directive (default 450) + + #### Returns + - str = full `.comp` source text + """ + pre: str = preamble.rstrip() + bod: str = body.rstrip() + lines: list[str] = [f"#version {version}", ""] + if pre: + lines.append(pre) + lines.append("") + lines.append("void main() {") + if bod: + lines.append(bod) + lines.append("}") + lines.append("") + return "\n".join(lines) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/context.py b/src/cthreads/python/cthreads/gpu/compiler/translation/context.py new file mode 100644 index 0000000..94a9663 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/context.py @@ -0,0 +1,27 @@ +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Callable + +from ....types import PyType + +# (binding index, param name, kind) — kind is "scalar" or "list" +BindingName = tuple[int, str, str] + + +@dataclass +class GpuTranslationContext: + """Mutable codegen state for one @Gpu function.""" + + fn: Callable + local_size_x: int = 64 + symbols: dict[str, PyType] = field(default_factory=dict) + binding_names: list[BindingName] = field(default_factory=list) + # Param names that live in the binding-0 Scalars SSBO (not bare GLSL ids). + scalar_params: set[str] = field(default_factory=set) + # Param names that are list SSBO instances (x -> x.data[i] via Index). + list_params: set[str] = field(default_factory=set) + + @property + def func_name(self) -> str: + return self.fn.__name__ diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/__init__.py b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/__init__.py new file mode 100644 index 0000000..093ea55 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/__init__.py @@ -0,0 +1,109 @@ +""" +Ordered GPU plugin lists. GpuSyntax.expr tries these for Call / Attribute. +""" + +from __future__ import annotations + +import ast + +from ..context import GpuTranslationContext +from .base import AttrPlugin, CallPlugin, TranslateExpr + +CALL_PLUGINS: list[CallPlugin] = [] +ATTR_PLUGINS: list[AttrPlugin] = [] + + +def register_call(plugin: CallPlugin) -> CallPlugin: + """ + Append a CallPlugin to the GPU call registry. + + #### Args: + - plugin: CallPlugin = plugin instance to register + + #### Returns + - CallPlugin = the same plugin (for chaining) + """ + CALL_PLUGINS.append(plugin) + return plugin + + +def register_attr(plugin: AttrPlugin) -> AttrPlugin: + """ + Append an AttrPlugin to the GPU attribute registry. + + #### Args: + - plugin: AttrPlugin = plugin instance to register + + #### Returns + - AttrPlugin = the same plugin (for chaining) + """ + ATTR_PLUGINS.append(plugin) + return plugin + + +def lower_call( + node: ast.Call, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, +) -> str | None: + """ + Try each CallPlugin until one returns GLSL text. + + #### Args: + - node: ast.Call = call expression + - ctx: GpuTranslationContext = current GPU translation state + - translate_expr: TranslateExpr = nested expression lowerer + + #### Returns + - str | None = GLSL text, or None if no plugin matched + """ + for plugin in CALL_PLUGINS: + out = plugin.try_lower(node, ctx, translate_expr) + if out is not None: + return out + return None + + +def lower_attr( + node: ast.Attribute, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, +) -> str | None: + """ + Try each AttrPlugin until one returns GLSL text. + + #### Args: + - node: ast.Attribute = attribute expression + - ctx: GpuTranslationContext = current GPU translation state + - translate_expr: TranslateExpr = nested expression lowerer + + #### Returns + - str | None = GLSL text, or None if no plugin matched + """ + for plugin in ATTR_PLUGINS: + out = plugin.try_lower(node, ctx, translate_expr) + if out is not None: + return out + return None + + +__all__ = [ + "CALL_PLUGINS", + "ATTR_PLUGINS", + "AttrPlugin", + "CallPlugin", + "TranslateExpr", + "register_call", + "register_attr", + "lower_call", + "lower_attr", +] + +# Side-effect: register concrete plugins. +from .indexes import IndexAttrPlugin # noqa: E402 +from .math_calls import MathCallPlugin # noqa: E402 +from .sync_threads import SyncThreadsPlugin # noqa: E402 + +register_attr(IndexAttrPlugin()) +register_call(MathCallPlugin()) +register_call(SyncThreadsPlugin()) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/base.py b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/base.py new file mode 100644 index 0000000..745bc9b --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/base.py @@ -0,0 +1,64 @@ +""" +Plugin bases for GPU Call / Attribute lowering. +""" + +from __future__ import annotations + +import ast +from abc import ABC, abstractmethod +from collections.abc import Callable + +from ..context import GpuTranslationContext + +# Already-translated subexpressions; typically GpuSyntax.expr +TranslateExpr = Callable[[ast.expr, GpuTranslationContext], str] + + +class AttrPlugin(ABC): + """ + Handles `ast.Attribute` (index builtins, props - not calls). + """ + + @abstractmethod + def try_lower( + self, + node: ast.Attribute, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, + ) -> str | None: + """ + Return GLSL text if this plugin handles `node`, else None. + + #### Args: + - node: ast.Attribute = attribute expression + - ctx: GpuTranslationContext = current GPU translation state + - translate_expr: TranslateExpr = nested expression lowerer + + #### Returns + - str | None = GLSL expression, or None to try the next plugin + """ + + +class CallPlugin(ABC): + """ + Handles `ast.Call` (math builtins later). + """ + + @abstractmethod + def try_lower( + self, + node: ast.Call, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, + ) -> str | None: + """ + Return GLSL text if this plugin handles `node`, else None. + + #### Args: + - node: ast.Call = call expression + - ctx: GpuTranslationContext = current GPU translation state + - translate_expr: TranslateExpr = nested expression lowerer + + #### Returns + - str | None = GLSL expression, or None to try the next plugin + """ diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/indexes.py b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/indexes.py new file mode 100644 index 0000000..9c3c87b --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/indexes.py @@ -0,0 +1,61 @@ +""" +AttrPlugin: ThreadIdx / BlockIdx / GlobalIdx / ... .x|.y|.z -> GLSL builtins. +""" +import ast +from typing import Any + +from ....frontend.indexes import GpuIndexBuiltin +from ..context import GpuTranslationContext +from .base import AttrPlugin, TranslateExpr + +_AXES: frozenset[str] = frozenset({"x", "y", "z"}) + + +def _globals(ctx: GpuTranslationContext) -> dict: + g = getattr(ctx.fn, "__globals__", None) + return g if isinstance(g, dict) else {} + + +def _resolve_value_obj(node: ast.expr, globals_ns: dict) -> Any: + """ + Resolve a simple Name or module.Name expression to a Python object. + """ + if isinstance(node, ast.Name): + return globals_ns.get(node.id) + if isinstance(node, ast.Attribute) and isinstance(node.value, ast.Name): + mod = globals_ns.get(node.value.id) + if mod is None: + return None + return getattr(mod, node.attr, None) + return None + + +def _glsl_base_for(obj: Any) -> str | None: + if obj is None: + return None + cls = obj if isinstance(obj, type) else type(obj) + if not isinstance(cls, type) or not issubclass(cls, GpuIndexBuiltin): + return None + base: str = getattr(cls, "_glsl_base", "") + return base or None + + +class IndexAttrPlugin(AttrPlugin): + """ + Lower `GlobalIdx.x` / `gpu.ThreadIdx.y` to `gl_*Invocation*.*`. + """ + + def try_lower( + self, + node: ast.Attribute, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, + ) -> str | None: + if node.attr not in _AXES: + return None + obj = _resolve_value_obj(node.value, _globals(ctx)) + base = _glsl_base_for(obj) + if base is None: + return None + # GLSL invocation IDs are uint; cast to int so `i: int = GlobalIdx.x` typechecks. + return f"int({base}.{node.attr})" diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/math_calls.py b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/math_calls.py new file mode 100644 index 0000000..3362059 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/math_calls.py @@ -0,0 +1,56 @@ +""" +Minimal GLSL math CallPlugins for @Gpu (sqrt / floor / int — SPH grid + forces). +""" + +from __future__ import annotations + +import ast + +from ..context import GpuTranslationContext +from .base import CallPlugin, TranslateExpr + + +class MathCallPlugin(CallPlugin): + """ + Lower a small set of math / cast calls to GLSL. + """ + + def try_lower( + self, + node: ast.Call, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, + ) -> str | None: + if node.keywords: + return None + if len(node.args) != 1: + return None + fn = node.func + arg = translate_expr(node.args[0], ctx) + + if isinstance(fn, ast.Name) and fn.id == "sqrt": + return f"sqrt({arg})" + if ( + isinstance(fn, ast.Attribute) + and fn.attr == "sqrt" + and isinstance(fn.value, ast.Name) + and fn.value.id == "math" + ): + return f"sqrt({arg})" + + # Truncate toward -inf (GLSL floor); used for cell indices. + if isinstance(fn, ast.Name) and fn.id == "floor": + return f"floor({arg})" + if ( + isinstance(fn, ast.Attribute) + and fn.attr == "floor" + and isinstance(fn.value, ast.Name) + and fn.value.id == "math" + ): + return f"floor({arg})" + + # Python int(x) on floats -> GLSL int(x) (trunc toward zero). + if isinstance(fn, ast.Name) and fn.id == "int": + return f"int({arg})" + + return None diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/sync_threads.py b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/sync_threads.py new file mode 100644 index 0000000..8d33404 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/plugins/sync_threads.py @@ -0,0 +1,68 @@ +""" +Workgroup barrier lowering for @Gpu. + +Maps CUDA-style `__sync_threads()` and beginner-friendly +`Barrier.arrive_and_wait()` (no instance) to the same GLSL workgroup barrier. +Does not implement host lock semantics. +""" + +from __future__ import annotations + +import ast + +from ..context import GpuTranslationContext +from .base import CallPlugin, TranslateExpr + +# Single GLSL fragment used as an expression-statement body (trailing ; added +# by GpuFlow.expr_stmt). Workgroup scope only — not grid-wide. +_GPU_BARRIER_GLSL = "barrier(); memoryBarrierShared()" + + +def _require_no_args(node: ast.Call, ctx: GpuTranslationContext, label: str) -> None: + if node.keywords or node.args: + raise TypeError( + f"GPU function {ctx.func_name}: {label} takes no arguments" + ) + + +class SyncThreadsPlugin(CallPlugin): + """ + `__sync_threads()` / `Barrier.arrive_and_wait()` -> GLSL workgroup barrier. + + Rejects `Barrier(...)` construction inside @Gpu (use arrive_and_wait). + """ + + def try_lower( + self, + node: ast.Call, + ctx: GpuTranslationContext, + translate_expr: TranslateExpr, + ) -> str | None: + del translate_expr # barrier forms take no nested exprs + fn = node.func + + # CUDA-style free call. + if isinstance(fn, ast.Name) and fn.id == "__sync_threads": + _require_no_args(node, ctx, "__sync_threads()") + return _GPU_BARRIER_GLSL + + # Beginner form: Barrier.arrive_and_wait() — not Barrier(...). + if ( + isinstance(fn, ast.Attribute) + and fn.attr == "arrive_and_wait" + and isinstance(fn.value, ast.Name) + and fn.value.id == "Barrier" + ): + _require_no_args(node, ctx, "Barrier.arrive_and_wait()") + return _GPU_BARRIER_GLSL + + # Clear error if someone writes Barrier(n) or Barrier() in a @Gpu body. + if isinstance(fn, ast.Name) and fn.id == "Barrier": + raise TypeError( + f"GPU function {ctx.func_name}: " + "Barrier(...) construction is not valid inside @Gpu; " + "use Barrier.arrive_and_wait() or __sync_threads() " + "(workgroup barrier)" + ) + + return None diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/result.py b/src/cthreads/python/cthreads/gpu/compiler/translation/result.py new file mode 100644 index 0000000..5a2dda1 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/result.py @@ -0,0 +1,42 @@ +from __future__ import annotations + +from dataclasses import dataclass + +# (param_name, glsl_type) e.g. ("n", "int") +ScalarField = tuple[str, str] +# (binding, param_name, glsl_elem_type) e.g. (1, "x", "float") +ListField = tuple[int, str, str] + + +@dataclass +class GpuSignatureResult: + """GLSL buffer preamble + layout facts for one @Gpu Signature.""" + + func_name: str + preamble: str + binding_count: int + scalar_bytes: int + local_size_x: int + scalar_fields: list[ScalarField] + list_fields: list[ListField] + + +@dataclass +class GpuTranslationResult: + """ + Emit contract for one translated `@Gpu` function. + + `source` is the full `.comp` text (`#version` + preamble + `main`). + `spirv` is set when compile_spirv was requested and compilation succeeded. + """ + + func_name: str + preamble: str + body: str + source: str + binding_count: int + scalar_bytes: int + local_size_x: int + scalar_fields: list[ScalarField] + list_fields: list[ListField] + spirv: bytes | None = None diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/spirv.py b/src/cthreads/python/cthreads/gpu/compiler/translation/spirv.py new file mode 100644 index 0000000..9f9d32d --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/spirv.py @@ -0,0 +1,133 @@ +""" +Compile GLSL compute source to SPIR-V. + +Prefers in-process `_ext.gpu.compile_glsl` (vendored Khronos glslang linked +into the GPU extension — no extra user tools). Falls back to the Vulkan SDK +`glslc` CLI when the native binding is unavailable (CPU-only builds / dev). +""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import tempfile +from pathlib import Path + + +def _find_glslc() -> str | None: + """ + Locate the shaderc `glslc` executable (dev fallback only). + + #### Returns + - str | None = path to glslc, or None if not found + """ + found: str | None = shutil.which("glslc") + if found: + return found + sdk: str | None = os.environ.get("VULKAN_SDK") + if not sdk: + return None + for rel in ("Bin/glslc.exe", "Bin/glslc", "bin/glslc"): + candidate: Path = Path(sdk) / rel + if candidate.is_file(): + return str(candidate) + return None + + +def compile_glsl_to_spirv(source: str) -> bytes: + """ + Compile a GLSL compute shader string to SPIR-V bytes. + + Uses native `cthreads._ext.gpu.compile_glsl` when the GPU extension was + built with glslang. Otherwise invokes `glslc` if present. + + #### Args: + - source: str = full `.comp` text (`#version` + buffers + `main`) + + #### Returns + - bytes = SPIR-V binary (multiple of 4 bytes, magic 0x07230203) + + #### Raises + - RuntimeError = no compiler available, or compile failed + + #### Technical terms: + - SPIR-V: intermediate binary Vulkan drivers consume + - glslang: Khronos GLSL compiler linked into `_ext` for GPU builds + """ + try: + from cthreads._ext import gpu as _gpu # type: ignore + + compile_native = getattr(_gpu, "compile_glsl", None) + if callable(compile_native): + try: + out = compile_native(source) + except Exception as exc: + raise RuntimeError(_format_glsl_compile_error(str(exc))) from exc + if not isinstance(out, (bytes, bytearray)): + raise RuntimeError( + "cthreads.gpu: _ext.gpu.compile_glsl did not return bytes" + ) + data: bytes = bytes(out) + if not data or (len(data) % 4) != 0: + raise RuntimeError( + "cthreads.gpu: native compile_glsl returned invalid SPIR-V" + ) + if data[:4] != b"\x03\x02\x23\x07": + raise RuntimeError( + "cthreads.gpu: native compile_glsl missing SPIR-V magic" + ) + return data + except ImportError: + pass + + glslc: str | None = _find_glslc() + if glslc is None: + raise RuntimeError( + "cthreads.gpu: no GLSL compiler found. Rebuild with " + "-DCTHREADS_GPU=ON (vendors glslang into _ext), or install the " + "Vulkan SDK glslc for a temporary CLI fallback" + ) + + with tempfile.TemporaryDirectory(prefix="cthreads_glsl_") as tmp: + tmp_path: Path = Path(tmp) + comp_path: Path = tmp_path / "kernel.comp" + spv_path: Path = tmp_path / "kernel.spv" + comp_path.write_text(source, encoding="utf-8") + cmd: list[str] = [ + glslc, + "-fshader-stage=compute", + str(comp_path), + "-o", + str(spv_path), + ] + proc = subprocess.run(cmd, capture_output=True, text=True) + if proc.returncode != 0: + err: str = (proc.stderr or proc.stdout or "").strip() + raise RuntimeError( + _format_glsl_compile_error( + "glslc (shaderc CLI fallback) failed:\n" + err + ) + ) + data = spv_path.read_bytes() + + if not data or (len(data) % 4) != 0: + raise RuntimeError("cthreads.gpu: glslc produced invalid SPIR-V") + if data[:4] != b"\x03\x02\x23\x07": + raise RuntimeError("cthreads.gpu: glslc output missing SPIR-V magic") + return data + + +def _format_glsl_compile_error(detail: str) -> str: + """ + Wrap a glslang/glslc failure with a short identifier hint. + + Reserved GLSL words used as buffer instance names (e.g. `out`, `in`) + fail at compile time; we surface that instead of maintaining a keyword list. + """ + return ( + "cthreads.gpu: GLSL compile failed:\n" + f"{detail.strip()}\n" + "Hint: kernel / parameter names become GLSL identifiers. Avoid " + "reserved words (e.g. out, in, buffer, shared, uniform, flat)." + ) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Assign.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Assign.py new file mode 100644 index 0000000..4d2b3c7 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Assign.py @@ -0,0 +1,144 @@ +""" +Assignment lowering for @Gpu bodies. + +Same shape as CPU Assign, without C++ includes / to_cpp / std::pow. +Annotated locals use Glsl.type_name. Power (`**`) is deferred to a future +math builtin mirror (like CPU stdlib -> C++). +""" +import ast + +from .....compiler.translation.Source import Source +from .....types import PyList, hint_to_pytype +from ..Glsl import type_name +from ..context import GpuTranslationContext + + +class GpuAssign: + """ + Lower ast.AnnAssign / Assign / AugAssign for GPU shader bodies. + """ + + @staticmethod + def ann_assign(node: ast.AnnAssign, ctx: GpuTranslationContext) -> list[str]: + """ + Declare a typed local and optionally initialize it. + + #### Args: + - node: ast.AnnAssign = annotated assignment statement + - ctx: GpuTranslationContext = symbols table for later Name lowering + + #### Returns + - list[str] = one indented GLSL declaration line + """ + from .Syntax import GpuSyntax + + if not isinstance(node.target, ast.Name): + raise TypeError( + f"GPU function {ctx.func_name}: " + "AnnAssign target must be a plain name" + ) + var_name: str = node.target.id + if var_name in ctx.symbols: + raise TypeError( + f"GPU function {ctx.func_name}: redeclaration of {var_name!r}" + ) + + hint = Source.resolve_annotation(node.annotation, ctx.fn.__globals__) + py_type = hint_to_pytype(hint) + if isinstance(py_type, PyList): + raise TypeError( + f"GPU function {ctx.func_name}: " + "local list declarations are not supported" + ) + + glsl_ty: str = type_name(py_type) + # Locals are bare ids; do not add to scalar_params (those are SSBO fields). + ctx.symbols[var_name] = py_type + + if node.value is None: + return [f" {glsl_ty} {var_name};"] + rhs: str = GpuSyntax.expr(node.value, ctx) + return [f" {glsl_ty} {var_name} = {rhs};"] + + @staticmethod + def assign(node: ast.Assign, ctx: GpuTranslationContext) -> list[str]: + """ + Lower a single-target assignment (`lhs = rhs`). + + Name and Index already rewrite scalar/list SSBO accessors. + + #### Args: + - node: ast.Assign = assignment statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = one indented GLSL assignment line + """ + from .Syntax import GpuSyntax + + if len(node.targets) != 1: + raise TypeError( + f"GPU function {ctx.func_name}: " + "only single-target assignment is supported" + ) + target = node.targets[0] + if isinstance(target, ast.Name): + if target.id not in ctx.symbols: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"assign to unknown name {target.id!r} " + "(declare it with an annotated assignment first)" + ) + if target.id in ctx.list_params: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"cannot assign to list parameter {target.id!r} " + "(assign elements via indexing)" + ) + elif isinstance(target, ast.Subscript): + if isinstance(target.slice, ast.Slice): + raise TypeError( + f"GPU function {ctx.func_name}: " + "slice assignment is not supported" + ) + else: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported assign target {type(target).__name__}" + ) + + lhs: str = GpuSyntax.expr(target, ctx) + rhs: str = GpuSyntax.expr(node.value, ctx) + return [f" {lhs} = {rhs};"] + + @staticmethod + def aug_assign(node: ast.AugAssign, ctx: GpuTranslationContext) -> list[str]: + """ + Lower augmented assignment (`lhs op= rhs`). Power is not supported yet. + + #### Args: + - node: ast.AugAssign = augmented assignment statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = one indented GLSL aug-assign line + """ + from .Op import GpuOp + from .Syntax import GpuSyntax + + if isinstance(node.op, ast.Pow): + raise TypeError( + f"GPU function {ctx.func_name}: " + "** / pow is not supported yet " + "(planned via a math builtin mirror)" + ) + + target: str = GpuSyntax.expr(node.target, ctx) + value: str = GpuSyntax.expr(node.value, ctx) + op = GpuOp.BINOPS.get(type(node.op)) + if not op: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported aug-assign operator {type(node.op).__name__}" + ) + return [f" {target} {op}= {value};"] diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Flow.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Flow.py new file mode 100644 index 0000000..3f4f733 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Flow.py @@ -0,0 +1,185 @@ +""" +Control-flow lowering for @Gpu bodies. + +Subclasses CPU Flow to reuse nest / pass / break / continue. +Overrides handlers that import Syntax so they use GpuSyntax. +for-in over lists (C++ auto&) is rejected; range-for is kept. +""" +import ast + +from .....compiler.translation.syntax.Flow import Flow +from .....types import PyInt +from ..context import GpuTranslationContext +from .Op import GpuOp + + +class GpuFlow(Flow): + """ + Lower if / while / for / return for GPU shader bodies. Reuses CPU Flow for shared helpers. + """ + + @staticmethod + def if_stmt(node: ast.If, ctx: GpuTranslationContext) -> list[str]: + """ + Lower an if/else statement to GLSL. + + #### Args: + - node: ast.If = if statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = indented GLSL if/else lines + """ + from .Syntax import GpuSyntax + + test: str = GpuSyntax.expr(node.test, ctx) + lines: list[str] = [f" if ({test}) {{"] + for stmt in node.body: + lines.extend(GpuFlow.nest(GpuSyntax.stmt(stmt, ctx))) + lines.append(" }") + if node.orelse: + lines.append(" else {") + for stmt in node.orelse: + lines.extend(GpuFlow.nest(GpuSyntax.stmt(stmt, ctx))) + lines.append(" }") + return lines + + @staticmethod + def while_stmt(node: ast.While, ctx: GpuTranslationContext) -> list[str]: + """ + Lower a while loop to GLSL (no while/else). + + #### Args: + - node: ast.While = while statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = indented GLSL while lines + """ + from .Syntax import GpuSyntax + + if node.orelse: + raise TypeError( + f"GPU function {ctx.func_name}: while/else is not supported" + ) + test: str = GpuSyntax.expr(node.test, ctx) + lines: list[str] = [f" while ({test}) {{"] + for stmt in node.body: + lines.extend(GpuFlow.nest(GpuSyntax.stmt(stmt, ctx))) + lines.append(" }") + return lines + + @staticmethod + def for_stmt(node: ast.For, ctx: GpuTranslationContext) -> list[str]: + """ + Lower `for i in range(...)` to a C-style GLSL for loop. + + Iteration over list parameters is not supported (no C++ range-for). + + #### Args: + - node: ast.For = for statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = indented GLSL for-loop lines + """ + from .Syntax import GpuSyntax + + if node.orelse: + raise TypeError( + f"GPU function {ctx.func_name}: for/else is not supported" + ) + if not isinstance(node.target, ast.Name): + raise TypeError( + f"GPU function {ctx.func_name}: " + "for-loop target must be a plain name" + ) + loop_var: str = node.target.id + if loop_var in ctx.symbols: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"for-loop rebinds existing name {loop_var!r}" + ) + + it = node.iter + if not GpuOp.is_builtin_call(it, "range"): + raise TypeError( + f"GPU function {ctx.func_name}: " + "for-iter must be range(...) " + "(list iteration is not supported on the GPU path)" + ) + assert isinstance(it, ast.Call) + if it.keywords: + raise TypeError( + f"GPU function {ctx.func_name}: " + "range() keyword args are not supported" + ) + n: int = len(it.args) + if n == 1: + start, stop, step = "0", GpuSyntax.expr(it.args[0], ctx), "1" + elif n == 2: + start = GpuSyntax.expr(it.args[0], ctx) + stop = GpuSyntax.expr(it.args[1], ctx) + step = "1" + elif n == 3: + start = GpuSyntax.expr(it.args[0], ctx) + stop = GpuSyntax.expr(it.args[1], ctx) + step = GpuSyntax.expr(it.args[2], ctx) + else: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"range() expects 1..3 args, got {n}" + ) + + ctx.symbols[loop_var] = PyInt() + lines: list[str] = [ + f" for (int {loop_var} = {start}; " + f"{loop_var} < {stop}; " + f"{loop_var} += {step}) {{" + ] + for stmt in node.body: + lines.extend(GpuFlow.nest(GpuSyntax.stmt(stmt, ctx))) + lines.append(" }") + del ctx.symbols[loop_var] + return lines + + @staticmethod + def return_stmt(node: ast.Return, ctx: GpuTranslationContext) -> list[str]: + """ + Lower return to a void early exit. + + Valued returns are not lowered yet; always emit bare `return;`. + + #### Args: + - node: ast.Return = return statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = one indented `return;` line + """ + return [" return;"] + + @staticmethod + def expr_stmt(node: ast.Expr, ctx: GpuTranslationContext) -> list[str]: + """ + Lower expression statements; string doc-exprs are ignored. + + Call expressions go through GpuSyntax.expr (CallPlugins), e.g. + `__sync_threads()` -> GLSL barrier. + + #### Args: + - node: ast.Expr = expression statement + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = GLSL statement lines, empty for docstrings, or a comment + """ + from .Syntax import GpuSyntax + + if isinstance(node.value, ast.Constant) and isinstance(node.value.value, str): + return [] + if isinstance(node.value, ast.Call): + return [f" {GpuSyntax.expr(node.value, ctx)};"] + return [ + f" // unsupported statement: Expr ({type(node.value).__name__})" + ] diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Index.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Index.py new file mode 100644 index 0000000..fc2e9ca --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Index.py @@ -0,0 +1,51 @@ +""" +ast.Subscript lowering for @Gpu bodies. + +List indexing becomes SSBO unsized-array access: x[i] -> x.data[i]. +""" + +import ast + +from ..context import GpuTranslationContext + + +class GpuIndex: + """ + Lower ast.Subscript for GPU list SSBOs (not slices). + """ + + @staticmethod + def subscript(node: ast.Subscript, ctx: GpuTranslationContext) -> str: + """ + Lower list indexing to GLSL `base.data[index]`. + + #### Args: + - node: ast.Subscript = Python subscript expression + - ctx: GpuTranslationContext = symbols and list param set + + #### Returns + - str = GLSL text such as `(x.data[i])` + + #### Raises + - TypeError = slice syntax, or subscript of a non-list param + + #### Technical terms: + - SSBO: shader storage buffer object; list params use `T data[]` + """ + from .Syntax import GpuSyntax + + if isinstance(node.slice, ast.Slice): + raise TypeError( + f"GPU function {ctx.func_name}: slice syntax is not supported" + ) + + # Only list kernel params expose a `data[]` member in the preamble. + if not isinstance(node.value, ast.Name) or node.value.id not in ctx.list_params: + raise TypeError( + f"GPU function {ctx.func_name}: " + "subscript is only supported on list parameters" + ) + + base: str = GpuSyntax.expr(node.value, ctx) + index: str = GpuSyntax.expr(node.slice, ctx) + return f"({base}.data[{index}])" diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Name.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Name.py new file mode 100644 index 0000000..2ba81fa --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Name.py @@ -0,0 +1,67 @@ +""" +ast.Name lowering for @Gpu bodies. + +Scalar params become fields on the binding-0 Scalars SSBO. +List params stay bare buffer instance names (Index adds .data[i]). +Locals stay bare identifiers. +self is stubbed until method kernels exist. +""" +import ast + +from ..context import GpuTranslationContext + + +class GpuName: + """ + Lower ast.Name nodes for GPU shader bodies with SSBO-aware rewriting. + """ + + @staticmethod + def name(node: ast.Name, ctx: GpuTranslationContext) -> str: + """ + Lower a name to a GLSL identifier or Scalars SSBO field access. + + #### Args: + - node: ast.Name = Python name expression + - ctx: GpuTranslationContext = symbols and scalar/list param sets + + #### Returns + - str = GLSL text (`scalars.n`, bare local/list id, or self hook) + + #### Technical terms: + - SSBO: shader storage buffer object holding kernel params on device + """ + if node.id == "self": + # Separate hook so method kernels can land here without rewriting name(). + return GpuName.lower_self(ctx) + + if node.id not in ctx.symbols: + raise TypeError( + f"GPU function {ctx.func_name}: unknown name {node.id!r}" + ) + + # Scalar kernel params live in the binding-0 block instance `scalars`. + if node.id in ctx.scalar_params: + return f"scalars.{node.id}" + + # List params are SSBO instance names; locals are ordinary GLSL ids. + return node.id + + @staticmethod + def lower_self(ctx: GpuTranslationContext) -> str: + """ + Lower `self` for method kernels (not implemented yet). + + #### Args: + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - str = GLSL receiver expression (when supported) + + #### Raises + - TypeError = self / method kernels are not supported yet + """ + raise TypeError( + f"GPU function {ctx.func_name}: " + "self / method kernels are not supported yet" + ) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Op.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Op.py new file mode 100644 index 0000000..1a29634 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Op.py @@ -0,0 +1,102 @@ +""" +GLSL operator lowering. + +Reuses CppOp operator tables (BINOPS / UNARYOPS / CMPOPS / BOOLOPS). +Handlers are overridden so they call GPU Syntax and emit GLSL (pow, no includes). +""" +import ast + +from .....compiler.translation.syntax.Op import Op as CppOp +from ..context import GpuTranslationContext + + +class GpuOp(CppOp): + """ast.BinOp / UnaryOp / Compare / BoolOp for @Gpu bodies.""" + + # No CPU sync builtin on the GPU path. + BUILTINS: frozenset[str] = frozenset({"range", "len"}) + + @staticmethod + def is_builtin_call(node: ast.AST, name: str) -> bool: + return ( + isinstance(node, ast.Call) + and isinstance(node.func, ast.Name) + and node.func.id == name + and name in GpuOp.BUILTINS + ) + + @staticmethod + def bin_op(node: ast.BinOp, ctx: GpuTranslationContext) -> str: + from .Syntax import GpuSyntax + + left = GpuSyntax.expr(node.left, ctx) + right = GpuSyntax.expr(node.right, ctx) + # Power stays out until math builtins mirror Python -> GLSL (like CPU). + if isinstance(node.op, ast.Pow): + raise TypeError( + f"GPU function {ctx.func_name}: " + "** / pow is not supported yet " + "(planned via a math builtin mirror)" + ) + op = GpuOp.BINOPS.get(type(node.op)) + if not op: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported binary operator {type(node.op).__name__}" + ) + return f"({left} {op} {right})" + + @staticmethod + def unary_op(node: ast.UnaryOp, ctx: GpuTranslationContext) -> str: + from .Syntax import GpuSyntax + + op = GpuOp.UNARYOPS.get(type(node.op)) + if not op: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported unary operator {type(node.op).__name__}" + ) + operand = GpuSyntax.expr(node.operand, ctx) + return f"({op}{operand})" + + @staticmethod + def compare(node: ast.Compare, ctx: GpuTranslationContext) -> str: + from .Syntax import GpuSyntax + + if len(node.ops) != len(node.comparators): + raise TypeError( + f"GPU function {ctx.func_name}: malformed Compare node" + ) + left = GpuSyntax.expr(node.left, ctx) + parts: list[str] = [] + prev = left + for op_node, comparator in zip(node.ops, node.comparators): + op = GpuOp.CMPOPS.get(type(op_node)) + if not op: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported compare operator {type(op_node).__name__}" + ) + right = GpuSyntax.expr(comparator, ctx) + parts.append(f"({prev} {op} {right})") + prev = right + if len(parts) == 1: + return parts[0] + return "(" + " && ".join(parts) + ")" + + @staticmethod + def bool_op(node: ast.BoolOp, ctx: GpuTranslationContext) -> str: + from .Syntax import GpuSyntax + + op = GpuOp.BOOLOPS.get(type(node.op)) + if not op: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported bool operator {type(node.op).__name__}" + ) + if len(node.values) < 2: + raise TypeError( + f"GPU function {ctx.func_name}: BoolOp needs at least two values" + ) + parts = [GpuSyntax.expr(v, ctx) for v in node.values] + return "(" + f" {op} ".join(parts) + ")" diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Syntax.py b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Syntax.py new file mode 100644 index 0000000..032e67b --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/syntax/Syntax.py @@ -0,0 +1,106 @@ +""" +Syntax dispatcher for @Gpu bodies. + +Wires GpuOp / GpuName / GpuIndex / GpuAssign / GpuFlow / Literal.constant. +Attribute and Call go through the GPU plugin registry. +""" + +from __future__ import annotations + +import ast +from typing import Callable + +from .....compiler.translation.syntax.Literal import Literal +from ..context import GpuTranslationContext +from ..plugins import lower_attr, lower_call +from .Assign import GpuAssign +from .Flow import GpuFlow +from .Index import GpuIndex +from .Name import GpuName +from .Op import GpuOp + +ExprHandler = Callable[[ast.AST, GpuTranslationContext], str] +StmtHandler = Callable[[ast.AST, GpuTranslationContext], list[str]] + + +class GpuSyntax: + """ + Dispatcher: GpuSyntax.expr / GpuSyntax.stmt to area static methods. + """ + + _EXPR: dict[type, ExprHandler] = { + ast.Constant: Literal.constant, + ast.Name: GpuName.name, + ast.Subscript: GpuIndex.subscript, + ast.BinOp: GpuOp.bin_op, + ast.UnaryOp: GpuOp.unary_op, + ast.Compare: GpuOp.compare, + ast.BoolOp: GpuOp.bool_op, + } + _STMT: dict[type, StmtHandler] = { + ast.AnnAssign: GpuAssign.ann_assign, + ast.Assign: GpuAssign.assign, + ast.AugAssign: GpuAssign.aug_assign, + ast.Pass: GpuFlow.pass_stmt, + ast.Break: GpuFlow.break_stmt, + ast.Continue: GpuFlow.continue_stmt, + ast.Return: GpuFlow.return_stmt, + ast.Expr: GpuFlow.expr_stmt, + ast.If: GpuFlow.if_stmt, + ast.For: GpuFlow.for_stmt, + ast.While: GpuFlow.while_stmt, + } + + @staticmethod + def expr(node: ast.expr, ctx: GpuTranslationContext) -> str: + """ + Lower one expression AST node to GLSL text. + + #### Args: + - node: ast.expr = expression node + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - str = GLSL expression text + """ + if isinstance(node, ast.Call): + out = lower_call(node, ctx, GpuSyntax.expr) + if out is not None: + return out + raise TypeError( + f"GPU function {ctx.func_name}: " + "unsupported call (no CallPlugin matched)" + ) + if isinstance(node, ast.Attribute): + out = lower_attr(node, ctx, GpuSyntax.expr) + if out is not None: + return out + raise TypeError( + f"GPU function {ctx.func_name}: " + "unsupported attribute (no AttrPlugin matched)" + ) + + handler = GpuSyntax._EXPR.get(type(node)) + if handler is None: + raise TypeError( + f"GPU function {ctx.func_name}: " + f"unsupported expression {type(node).__name__}" + ) + return handler(node, ctx) + + @staticmethod + def stmt(node: ast.stmt, ctx: GpuTranslationContext) -> list[str]: + """ + Lower one statement AST node to indented GLSL lines. + + #### Args: + - node: ast.stmt = statement node + - ctx: GpuTranslationContext = current GPU translation state + + #### Returns + - list[str] = GLSL lines (comment stub if unsupported) + """ + handler = GpuSyntax._STMT.get(type(node)) + if handler is None: + return [f" // unsupported statement: {type(node).__name__}"] + return handler(node, ctx) diff --git a/src/cthreads/python/cthreads/gpu/compiler/translation/translate.py b/src/cthreads/python/cthreads/gpu/compiler/translation/translate.py new file mode 100644 index 0000000..2ec8068 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/compiler/translation/translate.py @@ -0,0 +1,68 @@ +""" +One-shot GPU translate: parse -> Signature -> Syntax -> assemble -> optional SPIR-V. +""" + +from __future__ import annotations + +import ast +from typing import Callable + +from ....compiler.translation.Source import Source +from .Signature import GpuSignature +from .assemble import assemble_comp +from .context import GpuTranslationContext +from .result import GpuTranslationResult +from .spirv import compile_glsl_to_spirv +from .syntax.Syntax import GpuSyntax + + +def translate_function_for_gpu( + fn: Callable, + *, + local_size_x: int = 64, + compile_spirv: bool = False, +) -> GpuTranslationResult: + """ + Translate one `@Gpu` function to GLSL, optionally compile to SPIR-V. + + #### Args: + - fn: Callable = annotated GPU kernel function + - local_size_x: int = workgroup size x (default 64) + - compile_spirv: bool = if True, run shaderc (`glslc` / native) on `source` + + #### Returns + - GpuTranslationResult = preamble, body, `source`, optional `spirv`, layout + + #### Raises + - RuntimeError = SPIR-V compile requested but compiler missing / failed + """ + ctx: GpuTranslationContext = GpuTranslationContext( + fn=fn, local_size_x=local_size_x + ) + func_def: ast.FunctionDef = Source.parse_function(fn) + sig = GpuSignature.translate(func_def, ctx) + + body_lines: list[str] = [] + for stmt in func_def.body: + body_lines.extend(GpuSyntax.stmt(stmt, ctx)) + body: str = "\n".join(body_lines) + if body and not body.endswith("\n"): + body += "\n" + + source: str = assemble_comp(sig.preamble, body) + spirv: bytes | None = None + if compile_spirv: + spirv = compile_glsl_to_spirv(source) + + return GpuTranslationResult( + func_name=sig.func_name, + preamble=sig.preamble, + body=body, + source=source, + binding_count=sig.binding_count, + scalar_bytes=sig.scalar_bytes, + local_size_x=sig.local_size_x, + scalar_fields=sig.scalar_fields, + list_fields=sig.list_fields, + spirv=spirv, + ) diff --git a/src/cthreads/python/cthreads/gpu/frontend/__init__.py b/src/cthreads/python/cthreads/gpu/frontend/__init__.py new file mode 100644 index 0000000..81c3f87 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/frontend/__init__.py @@ -0,0 +1,49 @@ +from .wrapper import Gpu +from .errors import ( + CThreadsGPUError, + GPUNotAvailable, + GpuInvalidArgument, + GpuUseAfterDestroy, + VulkanInitFailed, + VulkanLoaderNotFound, + VulkanNoDevice, + VulkanNotBuiltError, + VulkanOutOfMemory, + _map_error, +) +from .indexes import ( + BlockDim, + BlockIdx, + GlobalIdx, + GridDim, + ThreadIdx, +) +from .lib import ( + available, + device_name, + init, + shutdown, +) + +__all__ = [ + "Gpu", + "BlockDim", + "BlockIdx", + "GlobalIdx", + "GridDim", + "ThreadIdx", + "CThreadsGPUError", + "GPUNotAvailable", + "GpuInvalidArgument", + "GpuUseAfterDestroy", + "VulkanInitFailed", + "VulkanLoaderNotFound", + "VulkanNoDevice", + "VulkanNotBuiltError", + "VulkanOutOfMemory", + "_map_error", + "available", + "device_name", + "init", + "shutdown", +] diff --git a/src/cthreads/python/cthreads/gpu/frontend/errors.py b/src/cthreads/python/cthreads/gpu/frontend/errors.py new file mode 100644 index 0000000..f016a79 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/frontend/errors.py @@ -0,0 +1,85 @@ +"""ctypes-style error types for cthreads.gpu (mapped from C++ message prefixes).""" + +class CThreadsGPUError(Exception): + def __init__(self, detail: str = "Unknown Error") -> None: + self.detail = detail + self.message = f"\033[91mcthreads gpu error\033[0m: {detail}" + super().__init__(self.message) + + def __str__(self) -> str: + return self.message + + +class VulkanNotBuiltError(CThreadsGPUError): + """_ext was compiled without CTHREADS_GPU.""" + + def __init__(self, detail: str = "cthreads built without CTHREADS_GPU") -> None: + super().__init__(detail) + + +class VulkanLoaderNotFound(CThreadsGPUError): + """vulkan-1.dll / libvulkan.so.1 could not be loaded.""" + + def __init__(self, detail: str = "Vulkan loader not found") -> None: + super().__init__(detail) + + +class VulkanNoDevice(CThreadsGPUError): + """Loader ok but no compute-capable device.""" + + def __init__(self, detail: str = "No Vulkan compute device") -> None: + super().__init__(detail) + + +class VulkanInitFailed(CThreadsGPUError): + """Instance/device creation or missing Vulkan entry point.""" + + def __init__(self, detail: str = "Vulkan init failed") -> None: + super().__init__(detail) + + +class VulkanOutOfMemory(CThreadsGPUError): + """vkAllocateMemory (or related) failed — device/host GPU memory exhausted.""" + + def __init__(self, detail: str = "Vulkan out of memory") -> None: + super().__init__(detail) + + +class GpuInvalidArgument(CThreadsGPUError): + """Bad size, dtype, null pointer, index, or other pack/memory argument error.""" + + def __init__(self, detail: str = "Invalid GPU argument") -> None: + super().__init__(detail) + + +class GpuUseAfterDestroy(CThreadsGPUError): + """Buffer/pack used after destroy or never initialized for the requested op.""" + + def __init__(self, detail: str = "GPU resource used after destroy") -> None: + super().__init__(detail) + + +class GPUNotAvailable(CThreadsGPUError): + """Generic: GPU path not usable (not built, no loader, or no device).""" + + def __init__(self, detail: str = "GPU not available") -> None: + super().__init__(detail) + + +def _map_error(exc: BaseException) -> CThreadsGPUError: + msg = str(exc) + if "VulkanLoaderNotFound" in msg: + return VulkanLoaderNotFound(msg) + if "VulkanNoDevice" in msg: + return VulkanNoDevice(msg) + if "VulkanOutOfMemory" in msg: + return VulkanOutOfMemory(msg) + if "GpuUseAfterDestroy" in msg: + return GpuUseAfterDestroy(msg) + if "GpuInvalidArgument" in msg: + return GpuInvalidArgument(msg) + if "VulkanNotBuilt" in msg: + return VulkanNotBuiltError(msg) + if "VulkanInitFailed" in msg: + return VulkanInitFailed(msg) + return VulkanInitFailed(msg) \ No newline at end of file diff --git a/src/cthreads/python/cthreads/gpu/frontend/indexes.py b/src/cthreads/python/cthreads/gpu/frontend/indexes.py new file mode 100644 index 0000000..9f93a12 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/frontend/indexes.py @@ -0,0 +1,110 @@ +""" +CUDA-style index builtins for `@Gpu` kernels. + +Frontend markers only. Codegen lowers `GlobalIdx.x` (and friends) via the +GPU AttrPlugin registry to GLSL invocation IDs. + +#### Mapping: +- ThreadIdx -> gl_LocalInvocationID +- BlockIdx -> gl_WorkGroupID +- BlockDim -> gl_WorkGroupSize +- GridDim -> gl_NumWorkGroups +- GlobalIdx -> gl_GlobalInvocationID +""" + +from __future__ import annotations + + +class _Axis: + """ + Placeholder for one axis (`.x` / `.y` / `.z`) of an index builtin. + """ + + __slots__ = ("_label",) + + def __init__(self, label: str) -> None: + self._label: str = label + + def __repr__(self) -> str: + return f"" + + +class GpuIndexBuiltin: + """ + Base for ThreadIdx / BlockIdx / GlobalIdx marker classes. + + #### Technical terms: + - GLSL: shading language used for Vulkan compute shaders + """ + + # Subclasses set the GLSL built-in vector name (without .x/.y/.z). + _glsl_base: str = "" + + +class ThreadIdx(GpuIndexBuiltin): + """ + Local invocation index within a workgroup (`gl_LocalInvocationID`). + """ + + _glsl_base: str = "gl_LocalInvocationID" + x: _Axis = _Axis("ThreadIdx.x") + y: _Axis = _Axis("ThreadIdx.y") + z: _Axis = _Axis("ThreadIdx.z") + + +class BlockIdx(GpuIndexBuiltin): + """ + Workgroup index (`gl_WorkGroupID`). + """ + + _glsl_base: str = "gl_WorkGroupID" + x: _Axis = _Axis("BlockIdx.x") + y: _Axis = _Axis("BlockIdx.y") + z: _Axis = _Axis("BlockIdx.z") + + +class BlockDim(GpuIndexBuiltin): + """ + Workgroup size (`gl_WorkGroupSize`). + """ + + _glsl_base: str = "gl_WorkGroupSize" + x: _Axis = _Axis("BlockDim.x") + y: _Axis = _Axis("BlockDim.y") + z: _Axis = _Axis("BlockDim.z") + + +class GridDim(GpuIndexBuiltin): + """ + Number of workgroups (`gl_NumWorkGroups`). + """ + + _glsl_base: str = "gl_NumWorkGroups" + x: _Axis = _Axis("GridDim.x") + y: _Axis = _Axis("GridDim.y") + z: _Axis = _Axis("GridDim.z") + + +class GlobalIdx(GpuIndexBuiltin): + """ + Global invocation index (`gl_GlobalInvocationID`). + + Prefer this for 1D element-wise kernels (for example saxpy). + + #### Example: + ``py + from cthreads.gpu import GlobalIdx, Gpu + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + `` + """ + + _glsl_base: str = "gl_GlobalInvocationID" + x: _Axis = _Axis("GlobalIdx.x") + y: _Axis = _Axis("GlobalIdx.y") + z: _Axis = _Axis("GlobalIdx.z") diff --git a/src/cthreads/python/cthreads/gpu/frontend/lib.py b/src/cthreads/python/cthreads/gpu/frontend/lib.py new file mode 100644 index 0000000..2dd1df0 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/frontend/lib.py @@ -0,0 +1,106 @@ +""" +Public GPU probe wrappers. + +Maps native C++ error prefixes to `frontend.errors` types and calls through +`_ext_gpu_api`. +""" + +from .. import _ext_gpu_api +from .errors import ( + VulkanNotBuiltError, + _map_error, +) + + +def available() -> bool: + """ + Return True if the Vulkan loader and a compute device can be initialized. + + #### Returns + - bool = True when GPU init would succeed + + #### Example: + ``py + from cthreads import gpu + if gpu.available(): + print(gpu.device_name()) + `` + """ + return _ext_gpu_api.available() + + +def device_name() -> str: + """ + Return the active GPU name (calls init). Raises on failure or not built. + + #### Returns + - str = Vulkan device name + + #### Raises + - VulkanNotBuiltError = extension compiled without CTHREADS_GPU + - CThreadsGPUError = mapped native init / device errors + + #### Example: + ``py + from cthreads import gpu + name = gpu.device_name() + `` + """ + if _ext_gpu_api._gpu is None: + raise VulkanNotBuiltError( + "cthreads built without CTHREADS_GPU; rebuild with -DCTHREADS_GPU=ON" + ) + try: + return _ext_gpu_api.device_name() + except Exception as exc: + raise _map_error(exc) from exc + + +def init() -> None: + """ + Explicitly initialize the Vulkan context. + + #### Returns + - None + + #### Raises + - VulkanNotBuiltError = extension compiled without CTHREADS_GPU + - CThreadsGPUError = mapped native init errors + + #### Example: + ``py + from cthreads import gpu + gpu.init() + `` + """ + if _ext_gpu_api._gpu is None: + raise VulkanNotBuiltError( + "cthreads built without CTHREADS_GPU; rebuild with -DCTHREADS_GPU=ON" + ) + try: + _ext_gpu_api.init() + except Exception as exc: + raise _map_error(exc) from exc + + +def shutdown() -> None: + """ + Destroy the device/instance and unload the Vulkan loader. + + Native ShaderCache is released with the device (explicit memory management). + Marks `prepare` so the next `prepare()` / `gpu()` rewalks the registry and + re-registers SPIR-V. No-op when the GPU extension is not built. + + #### Returns + - None + + #### Example: + ``py + from cthreads import gpu + gpu.shutdown() + `` + """ + _ext_gpu_api.shutdown() + from .. import runtime as runtime_mod + + runtime_mod._gpu_prepared = False diff --git a/src/cthreads/python/cthreads/gpu/frontend/wrapper.py b/src/cthreads/python/cthreads/gpu/frontend/wrapper.py new file mode 100644 index 0000000..8955af7 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/frontend/wrapper.py @@ -0,0 +1,78 @@ +""" +`@Gpu` decorator: mark, validate, and register a GPU kernel function. + +Does not emit SPIR-V or launch. That happens later via GpuCompileSession / `gpu()`. +""" + +from __future__ import annotations + +from collections.abc import Callable +from typing import Any, TypeVar + +from ...frontend.Registry import REGISTRY +from ..gpu_kernel_meta import build_gpu_kernel_meta +from .errors import GPUNotAvailable +from .lib import available, device_name + +F = TypeVar("F", bound=Callable[..., Any]) + + +def Gpu(fn: F | None = None, *, log: bool = False): + """ + Mark a function as a `@Gpu` kernel and register it for later compile/launch. + + Supports `@Gpu` and `@Gpu(log=True)`. Validates annotations against the GPU + allowlist (`-> None`, scalars and `list` of scalars), attaches + `fn.__gpu_kernel_meta__`, and returns the same function (still runs as + normal Python when called directly). + + #### Args: + - fn: Callable | None = function when used as `@Gpu`; omit for `@Gpu(log=...)` + - log: bool = if True, print the assigned device name (default False) + + #### Returns + - Callable = the marked function, or a decorator when `fn` is omitted + + #### Raises + - GPUNotAvailable = Vulkan GPU path is not usable in this process + - TypeError = missing or unsupported annotations + + #### Example: + ``py + from cthreads.gpu import Gpu + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + @Gpu(log=True) + def saxpy_logged(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + `` + """ + + def apply(f: F) -> F: + if not available(): + raise GPUNotAvailable( + "GPU is not available (build with CTHREADS_GPU=ON and a Vulkan " + "device)" + ) + + f.__gpu__ = True # type: ignore[attr-defined] + f.__gpu_version__ = REGISTRY.VERSION # type: ignore[attr-defined] + + # Validate annotations and attach launch meta (no SPIR-V yet). + build_gpu_kernel_meta(f) + + REGISTRY.register_gpu_function(f) + + if log: + print( + f"\033[92mGPU LOG:\033[0m function {f.__name__} is assigned to " + f"GPU {device_name()}" + ) + return f + + if fn is not None: + return apply(fn) + return apply diff --git a/src/cthreads/python/cthreads/gpu/gpu_kernel_meta.py b/src/cthreads/python/cthreads/gpu/gpu_kernel_meta.py new file mode 100644 index 0000000..e7f975f --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/gpu_kernel_meta.py @@ -0,0 +1,379 @@ +""" +Compile-time metadata for one `@Gpu` kernel. + +Shape matches `launch_gpu_kernel` in `module.hpp`: `symbol`, binding layout, +`params` with flat `kind` / `pass_as` / `elem_kind` / `elem_bytes`, and optional +dispatch overrides. + +v1 is in-place list writeback only (`pass_as="ref"` on lists). Kernels are +`-> None`; scalars are not written back on join. +""" + +from __future__ import annotations + +import inspect +from dataclasses import dataclass, field +from typing import Any, Callable, get_type_hints + +from ..types import ( + PyBool, + PyDict, + PyFloat, + PyInt, + PyList, + PyString, + PyThreadable, + PyType, + hint_to_pytype, + is_shared_pytype, + is_sync_pytype, + is_tbuffer_pytype, +) + +# Populated by build_gpu_kernel_meta(); keyed by symbol. +GPU_KERNELS: dict[str, "GpuKernelMeta"] = {} + +# Host / std430 sizes used by module.cpp py_size_of (GLSL bool is 32-bit). +_GPU_SCALAR_BYTES: dict[str, int] = { + "bool": 4, + "int": 4, + "float": 4, + "double": 8, +} + +_DEFAULT_LOCAL_SIZE_X = 64 + + +def _align_up(value: int, alignment: int) -> int: + """ + Round `value` up to the next multiple of `alignment` (std430 field packing). + + #### Args: + - value: int = current byte offset + - alignment: int = required alignment in bytes + + #### Returns + - int = aligned offset + """ + return (value + alignment - 1) & ~(alignment - 1) + + +def _gpu_kind_and_bytes(py_type: PyType) -> tuple[str, int]: + """ + Map a scalar PyType to the launch `kind` string and std430 byte size. + + #### Args: + - py_type: PyType = scalar type from hint_to_pytype + + #### Returns + - tuple[str, int] = (kind, elem_bytes) for int / float / bool + + #### Raises + - TypeError = type is not a supported GPU scalar + """ + if isinstance(py_type, PyInt): + return "int", _GPU_SCALAR_BYTES["int"] + if isinstance(py_type, PyFloat): + # Shader float (f32). CPU @Thread uses double; GPU v1 follows GLSL float. + return "float", _GPU_SCALAR_BYTES["float"] + if isinstance(py_type, PyBool): + return "bool", _GPU_SCALAR_BYTES["bool"] + raise TypeError( + f"GPU kernel: unsupported scalar type {py_type.name!r} " + f"(allowed: int, float, bool)" + ) + + +def pytype_to_gpu_schema(py_type: PyType) -> GpuTypeSchema: + """ + Convert a PyType into a GPU schema for marshal and codegen. + + Lists must be `list[int|float|bool]`. Dict, str, Threadable, sync, and + shared types are rejected. + + #### Args: + - py_type: PyType = type from hint_to_pytype + + #### Returns + - GpuTypeSchema = scalar or list schema with spirv_type / elem_bytes + + #### Raises + - TypeError = unsupported GPU type + """ + if is_sync_pytype(py_type) or is_shared_pytype(py_type) or is_tbuffer_pytype( + py_type + ): + raise TypeError( + f"GPU kernel: {py_type.name!r} is not supported on the GPU path" + ) + if isinstance(py_type, (PyDict, PyString, PyThreadable)): + raise TypeError( + f"GPU kernel: {py_type.name!r} is not supported on the GPU path" + ) + if isinstance(py_type, PyList): + kind, elem_bytes = _gpu_kind_and_bytes(py_type.inner_type) + return GpuTypeSchema( + kind="list", + spirv_type=f"{kind}[]", + inner=GpuTypeSchema( + kind=kind, + spirv_type=kind, + elem_bytes=elem_bytes, + ), + elem_bytes=elem_bytes, + ) + kind, elem_bytes = _gpu_kind_and_bytes(py_type) + return GpuTypeSchema(kind=kind, spirv_type=kind, elem_bytes=elem_bytes) + + +@dataclass +class GpuTypeSchema: + """ + Layout for one marshal or codegen slot (scalar or list of scalars). + + Serialized under each param's `schema` key. Launch also reads flat + `kind` / `elem_kind` / `elem_bytes` from GpuParamMeta.to_dict. + + #### Technical terms: + - std430: GLSL/SPIR-V storage buffer packing rules for scalar fields + """ + + kind: str # int, float, bool, list + spirv_type: str + elem_bytes: int = 0 + inner: GpuTypeSchema | None = None + + def to_dict(self) -> dict[str, Any]: + """ + Serialize this schema for `fn.__gpu_kernel_meta__`. + + #### Returns + - dict[str, Any] = JSON-friendly schema node for marshal and tests + + #### Raises + - ValueError = list schema is missing an inner element schema + """ + d: dict[str, Any] = { + "kind": self.kind, + "spirv_type": self.spirv_type, + "elem_bytes": self.elem_bytes, + } + if self.kind == "list": + if self.inner is None: + raise ValueError("list schema requires inner element schema") + d["inner"] = self.inner.to_dict() + return d + + +@dataclass +class GpuParamMeta: + """ + One kernel parameter: Python name, pass mode, and GpuTypeSchema. + + Scalars default to `pass_as="value"` (packed into binding 0). Lists default + to `pass_as="ref"` (downloaded into the same Python list on join). + + #### Technical terms: + - pass_as: value packs into the scalar SSBO; ref lists are written back on join + - binding 0: scalar storage buffer; list buffers use bindings 1..N + """ + + name: str + pass_as: str # value | ref + schema: GpuTypeSchema + + def __post_init__(self) -> None: + if self.pass_as not in ("value", "ref"): + raise TypeError( + f"GPU param {self.name!r}: pass_as must be 'value' or 'ref', " + f"got {self.pass_as!r}" + ) + if self.schema.kind == "list" and self.pass_as not in ("value", "ref"): + raise TypeError( + f"GPU list param {self.name!r}: pass_as must be 'value' or 'ref'" + ) + + @property + def kind(self) -> str: + """ + Return the top-level schema kind for this parameter. + + #### Returns + - str = `int`, `float`, `bool`, or `list` + """ + return self.schema.kind + + @property + def elem_kind(self) -> str | None: + """ + Return the list element kind, or None for scalars. + + #### Returns + - str | None = inner kind when `kind == "list"`, else None + """ + if self.schema.kind == "list" and self.schema.inner is not None: + return self.schema.inner.kind + return None + + @property + def elem_bytes(self) -> int | None: + """ + Return the list element size in bytes, or None for scalars. + + #### Returns + - int | None = element byte width when `kind == "list"`, else None + """ + if self.schema.kind == "list": + return self.schema.elem_bytes + return None + + def to_dict(self) -> dict[str, Any]: + """ + Flatten this parameter for `launch_gpu_kernel`. + + Emits top-level `kind` / `pass_as` and, for lists, `elem_kind` / + `elem_bytes` as read by module.cpp. + + #### Returns + - dict[str, Any] = parameter metadata consumed by marshal and launch + """ + d: dict[str, Any] = { + "name": self.name, + "pass_as": self.pass_as, + "kind": self.schema.kind, + "schema": self.schema.to_dict(), + } + if self.schema.kind == "list": + d["elem_kind"] = self.elem_kind + d["elem_bytes"] = self.elem_bytes + return d + + +@dataclass +class GpuKernelMeta: + """ + Compile-time record for one `@Gpu` kernel (ShaderCache key + launch meta). + + No trampolines or DLL symbols. `symbol` is the ShaderCache key. Return is + always void (`-> None`); results are ref-list writeback on join. + + #### Technical terms: + - ShaderCache: process map of symbol to reusable pipeline and set layout + - writeback: download of ref list buffers into caller-owned Python lists + """ + + symbol: str + binding_count: int + scalar_bytes: int + params: list[GpuParamMeta] + local_size_x: int = _DEFAULT_LOCAL_SIZE_X + # Optional dispatch overrides; None => launch may ceil(n / local_size_x). + group_count_x: int | None = None + group_count_y: int | None = 1 + group_count_z: int | None = 1 + types: dict[str, Any] = field(default_factory=dict) + schemas: dict[str, Any] = field(default_factory=dict) + + def to_dict(self) -> dict[str, Any]: + """ + Serialize kernel metadata for `fn.__gpu_kernel_meta__` and launch. + + #### Returns + - dict[str, Any] = dict accepted by `launch_gpu_kernel` + """ + return { + "symbol": self.symbol, + "binding_count": self.binding_count, + "scalar_bytes": self.scalar_bytes, + "local_size_x": self.local_size_x, + "group_count_x": self.group_count_x, + "group_count_y": self.group_count_y, + "group_count_z": self.group_count_z, + "params": [p.to_dict() for p in self.params], + "types": dict(self.types), + "schemas": dict(self.schemas), + } + + +def build_gpu_kernel_meta( + fn: Callable, + *, + symbol: str | None = None, + local_size_x: int = _DEFAULT_LOCAL_SIZE_X, +) -> GpuKernelMeta: + """ + Build and attach metadata for one `@Gpu` function from its annotations. + + Stores the record in `GPU_KERNELS[symbol]` and sets + `fn.__gpu_kernel_meta__` to `meta.to_dict()`. + + #### Args: + - fn: Callable = `@Gpu` function with type hints + - symbol: str | None = ShaderCache key (default: `fn.__name__`) + - local_size_x: int = compute workgroup size (default: 64) + + #### Returns + - GpuKernelMeta = full metadata record for this kernel + + #### Raises + - TypeError = missing annotations, non-None return, or unsupported types + """ + hints = get_type_hints(fn) + ret = hints.get("return", None) + if ret not in (None, type(None)): + raise TypeError( + f"GPU kernel {fn.__qualname__}: return must be None " + f"(in-place list writeback only), got {ret!r}" + ) + + sig = inspect.signature(fn) + if any( + p.kind + in ( + inspect.Parameter.VAR_POSITIONAL, + inspect.Parameter.VAR_KEYWORD, + inspect.Parameter.KEYWORD_ONLY, + ) + for p in sig.parameters.values() + ): + raise TypeError( + f"GPU kernel {fn.__qualname__}: *args / **kwargs / keyword-only " + "args are not supported" + ) + + params: list[GpuParamMeta] = [] + scalar_bytes = 0 + list_count = 0 + + for pname in sig.parameters: + if pname not in hints: + raise TypeError( + f"GPU kernel {fn.__qualname__}: parameter {pname!r} needs a " + "type annotation" + ) + py_type = hint_to_pytype(hints[pname]) + schema = pytype_to_gpu_schema(py_type) + if schema.kind == "list": + list_count += 1 + pass_as = "ref" + else: + align = schema.elem_bytes + scalar_bytes = _align_up(scalar_bytes, align) + scalar_bytes += schema.elem_bytes + pass_as = "value" + params.append(GpuParamMeta(name=pname, pass_as=pass_as, schema=schema)) + + # Binding 0 = scalar SSBO (reserved when any scalars); lists at 1..N. + binding_count = (1 if scalar_bytes > 0 else 0) + list_count + sym = symbol if symbol is not None else fn.__name__ + + meta = GpuKernelMeta( + symbol=sym, + binding_count=binding_count, + scalar_bytes=scalar_bytes, + params=params, + local_size_x=local_size_x, + ) + GPU_KERNELS[sym] = meta + fn.__gpu_kernel_meta__ = meta.to_dict() # type: ignore[attr-defined] + return meta diff --git a/src/cthreads/python/cthreads/gpu/gpu_marshal.py b/src/cthreads/python/cthreads/gpu/gpu_marshal.py new file mode 100644 index 0000000..48c2734 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/gpu_marshal.py @@ -0,0 +1,75 @@ +""" +Marshal helpers for `gpu()` launch (ordered args + dispatch sizing). +""" + +from typing import Any + + +def ordered_values_for_meta(meta: dict[str, Any], args: tuple[Any, ...]) -> list[Any]: + """ + Build the ordered argument list for `launch_gpu_kernel`. + + #### Args: + - meta: dict[str, Any] = kernel metadata (`params`, …) + - args: tuple[Any, ...] = positional args from `gpu(fn, *args)` + + #### Returns + - list[Any] = args in parameter order (same Python list objects for refs) + + #### Raises + - TypeError = arity mismatch + """ + params = meta.get("params") + if not isinstance(params, list): + raise TypeError("GPU meta is missing a params list") + if len(args) != len(params): + raise TypeError( + f"GPU kernel {meta.get('symbol', '?')!r}: expected {len(params)} " + f"args, got {len(args)}" + ) + return list(args) + + +def infer_group_count_x(meta: dict[str, Any], ordered_values: list[Any]) -> int: + """ + Choose `group_count_x` as ceil(n / local_size_x). + + Prefers a scalar parameter named `n`, else the longest list argument length. + + #### Args: + - meta: dict[str, Any] = kernel metadata + - ordered_values: list[Any] = launch args in param order + + #### Returns + - int = workgroup count in X (>= 1) + """ + local_size_x: int = int(meta.get("local_size_x") or 64) + if local_size_x < 1: + local_size_x = 64 + + params = meta.get("params") + if not isinstance(params, list): + return 1 + + n: int | None = None + for i, param in enumerate(params): + if not isinstance(param, dict): + continue + name = param.get("name") + kind = param.get("kind") + if name == "n" and kind == "int": + n = int(ordered_values[i]) + break + + if n is None: + best = 0 + for i, param in enumerate(params): + if isinstance(param, dict) and param.get("kind") == "list": + val = ordered_values[i] + if isinstance(val, list): + best = max(best, len(val)) + n = best + + if n is None or n < 1: + return 1 + return (int(n) + local_size_x - 1) // local_size_x diff --git a/src/cthreads/python/cthreads/gpu/runtime.py b/src/cthreads/python/cthreads/gpu/runtime.py new file mode 100644 index 0000000..931b076 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/runtime.py @@ -0,0 +1,227 @@ +""" +High-level GPU prepare + `gpu()` launch entry. + +Mirrors CPU `cthreads.prepare` / `thread`: compile registered `@Gpu` kernels, +then launch via `_ext.gpu.launch_gpu_kernel`. +""" + +from typing import Any, Callable + +from ..job import Job +from . import _ext_gpu_api +from .arena import lookup_resident +from .compiler.orchestrator import GpuCompileSession +from .frontend.errors import GPUNotAvailable, GpuInvalidArgument, _map_error +from .gpu_kernel_meta import build_gpu_kernel_meta +from .gpu_marshal import infer_group_count_x, ordered_values_for_meta + +# True after a successful GpuCompileSession.compile in this process. +# Cleared by frontend shutdown() when native ShaderCache is released. +_gpu_prepared: bool = False + + +class GpuJob(Job): + """ + Job wrapper for native GpuJob handles (void kernels; `result()` is None). + """ + + def join(self, download: bool = True) -> None: + """ + Wait for the GPU fence; optionally write ref lists back into Python. + + #### Args: + - download: bool = if True (default), download ref lists on join. + If False, skip writeback (use GpuArena.sync for resident lists). + + #### Returns + - None + """ + if not self._started: + self.start() + raw_join = self._raw.join + try: + raw_join(download) + except TypeError: + # Older native builds without the download argument. + if download is False: + raise GpuInvalidArgument( + "GpuJob.join(download=False) requires a rebuild with " + "CTHREADS_GPU residency support" + ) from None + raw_join() + + def result(self) -> None: + """ + GPU kernels are writeback-only; there is no scalar return value. + + #### Returns + - None + """ + return None + + +def compile(force: bool = False) -> dict[str, Any]: + """ + Drain registered `@Gpu` functions through GpuCompileSession (SPIR-V emit). + + #### Args: + - force: bool = rewrite `__Gpu__` artifacts even when fingerprints match + + #### Returns + - dict[str, Any] = session result (`root`, `cache`, `rewritten`) + """ + global _gpu_prepared + info = GpuCompileSession.compile(force=force) + _gpu_prepared = True + return info + + +def prepare(force: bool = False) -> dict[str, Any]: + """ + Compile all registered `@Gpu` kernels and register SPIR-V in ShaderCache. + + #### Args: + - force: bool = if True, re-init Vulkan and force-rebuild GPU units + + #### Returns + - dict[str, Any] = compile session info + + #### Raises + - GPUNotAvailable = no usable Vulkan device / GPU extension + - RuntimeError = nothing registered or compile failed + """ + global _gpu_prepared + if not _ext_gpu_api.available(): + raise GPUNotAvailable( + "GPU is not available (build with CTHREADS_GPU=ON and a Vulkan device)" + ) + if force: + _ext_gpu_api.shutdown() + _ext_gpu_api.init() + _gpu_prepared = False + return compile(force=force) + + +def _resident_meta_for_args( + meta: dict[str, Any], ordered: list[Any] +) -> dict[int, str]: + """ + Map value_index -> GpuState name for arena-bound list args. + + Raises if a bound list length no longer matches the registered numel. + """ + params = meta.get("params") + if not isinstance(params, list): + return {} + resident: dict[int, str] = {} + for i, param in enumerate(params): + if not isinstance(param, dict) or param.get("kind") != "list": + continue + slot = lookup_resident(ordered[i]) + if slot is None: + continue + host = ordered[i] + if not isinstance(host, list): + continue + if len(host) != slot.numel: + raise GpuInvalidArgument( + f"gpu(): bound list {slot.name!r} length changed " + f"(was {slot.numel}, now {len(host)}); call arena.bind again" + ) + meta_kind = param.get("elem_kind") + if meta_kind is not None and meta_kind != slot.elem_kind: + raise GpuInvalidArgument( + f"gpu(): bound list {slot.name!r} elem_kind {slot.elem_kind!r} " + f"does not match kernel {meta_kind!r}" + ) + resident[i] = slot.state_name + return resident + + +def gpu( + fn: Callable[..., Any], + *args: Any, + force: bool = False, + **kwargs: Any, +) -> GpuJob: + """ + Launch a `@Gpu` kernel and return a joinable job handle. + + Ensures GPU compile/emit has run, then submits via `launch_gpu_kernel`. + List arguments are written back in place on `join()` unless + `join(download=False)` is used with GpuArena-resident lists. + + #### Args: + - fn: Callable = `@Gpu`-decorated kernel + - *args: Any = positional kernel arguments (param order) + - force: bool = force recompile + Vulkan re-init before launch + - **kwargs: Any = not supported yet + + #### Returns + - GpuJob = awaitable / joinable handle (result is always None) + + #### Raises + - TypeError = missing `@Gpu`, bad arity, or unexpected kwargs + - GPUNotAvailable = Vulkan path not usable + - Exception = mapped native launch failures + + #### Example: + ``py + from cthreads.gpu import Gpu, GlobalIdx, gpu + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [10.0, 20.0, 30.0, 40.0] + gpu(saxpy, len(x), 2.0, x, y).join() + `` + """ + global _gpu_prepared + + if kwargs: + raise TypeError("gpu(): keyword arguments are not supported yet") + if not callable(fn): + raise TypeError("gpu(): fn must be callable") + if not getattr(fn, "__gpu__", False): + raise TypeError( + f"gpu(): {getattr(fn, '__qualname__', fn)!r} is not a @Gpu function" + ) + + if not _ext_gpu_api.available(): + raise GPUNotAvailable( + "GPU is not available (build with CTHREADS_GPU=ON and a Vulkan device)" + ) + + if force or not _gpu_prepared: + prepare(force=force) + + meta_obj = getattr(fn, "__gpu_kernel_meta__", None) + if not isinstance(meta_obj, dict): + meta_obj = build_gpu_kernel_meta(fn).to_dict() + meta: dict[str, Any] = dict(meta_obj) + + ordered = ordered_values_for_meta(meta, args) + if meta.get("group_count_x") is None: + meta["group_count_x"] = infer_group_count_x(meta, ordered) + if meta.get("group_count_y") is None: + meta["group_count_y"] = 1 + if meta.get("group_count_z") is None: + meta["group_count_z"] = 1 + + resident = _resident_meta_for_args(meta, ordered) + if resident: + meta["resident"] = resident + + try: + raw = _ext_gpu_api.launch_gpu_kernel(meta, ordered) + except Exception as exc: + raise _map_error(exc) from exc + + job = GpuJob(raw) + job.start() + return job diff --git a/src/cthreads/python/cthreads/gpu/third_party_notices/README.md b/src/cthreads/python/cthreads/gpu/third_party_notices/README.md new file mode 100644 index 0000000..8dd0642 --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/third_party_notices/README.md @@ -0,0 +1,12 @@ +# Third-party notices for native GLSL -> SPIR-V + +When `CTHREADS_GPU=ON`, cthreads links **Khronos glslang** into `_ext` so +`compile_glsl` works without installing `glslc` or other Vulkan SDK tools. + +glslang is the same compiler engine Google **shaderc** wraps. License texts +copied here at build time (see also the root `LICENSE` third-party section): + +- `glslang-LICENSE*` — Khronos glslang (Apache-2.0 / BSD-style components) + +End-user wheels that include the GPU extension must redistribute these notices +alongside the binary. diff --git a/src/cthreads/python/cthreads/gpu/third_party_notices/glslang-LICENSE.txt b/src/cthreads/python/cthreads/gpu/third_party_notices/glslang-LICENSE.txt new file mode 100644 index 0000000..054e68a --- /dev/null +++ b/src/cthreads/python/cthreads/gpu/third_party_notices/glslang-LICENSE.txt @@ -0,0 +1,1016 @@ +Here, glslang proper means core GLSL parsing, HLSL parsing, and SPIR-V code +generation. Glslang proper requires use of a number of licenses, one that covers +preprocessing and others that covers non-preprocessing. + +Bison was removed long ago. You can build glslang from the source grammar, +using tools of your choice, without using bison or any bison files. + +Other parts, outside of glslang proper, include: + +- gl_types.h, only needed for OpenGL-like reflection, and can be left out of + a parse and codegen project. See it for its license. + +- update_glslang_sources.py, which is not part of the project proper and does + not need to be used. + +- the SPIR-V "remapper", which is optional, but has the same license as + glslang proper + +- Google tests and SPIR-V tools, and anything in the external subdirectory + are external and optional; see them for their respective licenses. + +-------------------------------------------------------------------------------- + +The core of glslang-proper, minus the preprocessor is licenced as follows: + +-------------------------------------------------------------------------------- +3-Clause BSD License +-------------------------------------------------------------------------------- + +// +// Copyright (C) 2015-2018 Google, Inc. +// Copyright (C) +// +// All rights reserved. +// +// Redistribution and use in source and binary forms, with or without +// modification, are permitted provided that the following conditions +// are met: +// +// Redistributions of source code must retain the above copyright +// notice, this list of conditions and the following disclaimer. +// +// Redistributions in binary form must reproduce the above +// copyright notice, this list of conditions and the following +// disclaimer in the documentation and/or other materials provided +// with the distribution. +// +// Neither the name of 3Dlabs Inc. Ltd. nor the names of its +// contributors may be used to endorse or promote products derived +// from this software without specific prior written permission. +// +// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS +// FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE +// COPYRIGHT HOLDERS OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, +// INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, +// BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +// LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT +// LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN +// ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +// POSSIBILITY OF SUCH DAMAGE. +// + + +-------------------------------------------------------------------------------- +2-Clause BSD License +-------------------------------------------------------------------------------- + +Copyright 2020 The Khronos Group Inc + +Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright notice, this list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright notice, this list of conditions and the following disclaimer in the documentation and/or other materials provided with the distribution. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + +-------------------------------------------------------------------------------- +The MIT License +-------------------------------------------------------------------------------- + +Copyright 2020 The Khronos Group Inc + +Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + + +-------------------------------------------------------------------------------- +APACHE LICENSE, VERSION 2.0 +-------------------------------------------------------------------------------- + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. + + +-------------------------------------------------------------------------------- +GPL 3 with special bison exception +-------------------------------------------------------------------------------- + + GNU GENERAL PUBLIC LICENSE + Version 3, 29 June 2007 + + Copyright (C) 2007 Free Software Foundation, Inc. + Everyone is permitted to copy and distribute verbatim copies + of this license document, but changing it is not allowed. + + Preamble + + The GNU General Public License is a free, copyleft license for +software and other kinds of works. + + The licenses for most software and other practical works are designed +to take away your freedom to share and change the works. By contrast, +the GNU General Public License is intended to guarantee your freedom to +share and change all versions of a program--to make sure it remains free +software for all its users. We, the Free Software Foundation, use the +GNU General Public License for most of our software; it applies also to +any other work released this way by its authors. You can apply it to +your programs, too. + + When we speak of free software, we are referring to freedom, not +price. Our General Public Licenses are designed to make sure that you +have the freedom to distribute copies of free software (and charge for +them if you wish), that you receive source code or can get it if you +want it, that you can change the software or use pieces of it in new +free programs, and that you know you can do these things. + + To protect your rights, we need to prevent others from denying you +these rights or asking you to surrender the rights. Therefore, you have +certain responsibilities if you distribute copies of the software, or if +you modify it: responsibilities to respect the freedom of others. + + For example, if you distribute copies of such a program, whether +gratis or for a fee, you must pass on to the recipients the same +freedoms that you received. You must make sure that they, too, receive +or can get the source code. And you must show them these terms so they +know their rights. + + Developers that use the GNU GPL protect your rights with two steps: +(1) assert copyright on the software, and (2) offer you this License +giving you legal permission to copy, distribute and/or modify it. + + For the developers' and authors' protection, the GPL clearly explains +that there is no warranty for this free software. For both users' and +authors' sake, the GPL requires that modified versions be marked as +changed, so that their problems will not be attributed erroneously to +authors of previous versions. + + Some devices are designed to deny users access to install or run +modified versions of the software inside them, although the manufacturer +can do so. This is fundamentally incompatible with the aim of +protecting users' freedom to change the software. The systematic +pattern of such abuse occurs in the area of products for individuals to +use, which is precisely where it is most unacceptable. Therefore, we +have designed this version of the GPL to prohibit the practice for those +products. If such problems arise substantially in other domains, we +stand ready to extend this provision to those domains in future versions +of the GPL, as needed to protect the freedom of users. + + Finally, every program is threatened constantly by software patents. +States should not allow patents to restrict development and use of +software on general-purpose computers, but in those that do, we wish to +avoid the special danger that patents applied to a free program could +make it effectively proprietary. To prevent this, the GPL assures that +patents cannot be used to render the program non-free. + + The precise terms and conditions for copying, distribution and +modification follow. + + TERMS AND CONDITIONS + + 0. Definitions. + + "This License" refers to version 3 of the GNU General Public License. + + "Copyright" also means copyright-like laws that apply to other kinds of +works, such as semiconductor masks. + + "The Program" refers to any copyrightable work licensed under this +License. Each licensee is addressed as "you". "Licensees" and +"recipients" may be individuals or organizations. + + To "modify" a work means to copy from or adapt all or part of the work +in a fashion requiring copyright permission, other than the making of an +exact copy. The resulting work is called a "modified version" of the +earlier work or a work "based on" the earlier work. + + A "covered work" means either the unmodified Program or a work based +on the Program. + + To "propagate" a work means to do anything with it that, without +permission, would make you directly or secondarily liable for +infringement under applicable copyright law, except executing it on a +computer or modifying a private copy. Propagation includes copying, +distribution (with or without modification), making available to the +public, and in some countries other activities as well. + + To "convey" a work means any kind of propagation that enables other +parties to make or receive copies. Mere interaction with a user through +a computer network, with no transfer of a copy, is not conveying. + + An interactive user interface displays "Appropriate Legal Notices" +to the extent that it includes a convenient and prominently visible +feature that (1) displays an appropriate copyright notice, and (2) +tells the user that there is no warranty for the work (except to the +extent that warranties are provided), that licensees may convey the +work under this License, and how to view a copy of this License. If +the interface presents a list of user commands or options, such as a +menu, a prominent item in the list meets this criterion. + + 1. Source Code. + + The "source code" for a work means the preferred form of the work +for making modifications to it. "Object code" means any non-source +form of a work. + + A "Standard Interface" means an interface that either is an official +standard defined by a recognized standards body, or, in the case of +interfaces specified for a particular programming language, one that +is widely used among developers working in that language. + + The "System Libraries" of an executable work include anything, other +than the work as a whole, that (a) is included in the normal form of +packaging a Major Component, but which is not part of that Major +Component, and (b) serves only to enable use of the work with that +Major Component, or to implement a Standard Interface for which an +implementation is available to the public in source code form. A +"Major Component", in this context, means a major essential component +(kernel, window system, and so on) of the specific operating system +(if any) on which the executable work runs, or a compiler used to +produce the work, or an object code interpreter used to run it. + + The "Corresponding Source" for a work in object code form means all +the source code needed to generate, install, and (for an executable +work) run the object code and to modify the work, including scripts to +control those activities. However, it does not include the work's +System Libraries, or general-purpose tools or generally available free +programs which are used unmodified in performing those activities but +which are not part of the work. For example, Corresponding Source +includes interface definition files associated with source files for +the work, and the source code for shared libraries and dynamically +linked subprograms that the work is specifically designed to require, +such as by intimate data communication or control flow between those +subprograms and other parts of the work. + + The Corresponding Source need not include anything that users +can regenerate automatically from other parts of the Corresponding +Source. + + The Corresponding Source for a work in source code form is that +same work. + + 2. Basic Permissions. + + All rights granted under this License are granted for the term of +copyright on the Program, and are irrevocable provided the stated +conditions are met. This License explicitly affirms your unlimited +permission to run the unmodified Program. The output from running a +covered work is covered by this License only if the output, given its +content, constitutes a covered work. This License acknowledges your +rights of fair use or other equivalent, as provided by copyright law. + + You may make, run and propagate covered works that you do not +convey, without conditions so long as your license otherwise remains +in force. You may convey covered works to others for the sole purpose +of having them make modifications exclusively for you, or provide you +with facilities for running those works, provided that you comply with +the terms of this License in conveying all material for which you do +not control copyright. Those thus making or running the covered works +for you must do so exclusively on your behalf, under your direction +and control, on terms that prohibit them from making any copies of +your copyrighted material outside their relationship with you. + + Conveying under any other circumstances is permitted solely under +the conditions stated below. Sublicensing is not allowed; section 10 +makes it unnecessary. + + 3. Protecting Users' Legal Rights From Anti-Circumvention Law. + + No covered work shall be deemed part of an effective technological +measure under any applicable law fulfilling obligations under article +11 of the WIPO copyright treaty adopted on 20 December 1996, or +similar laws prohibiting or restricting circumvention of such +measures. + + When you convey a covered work, you waive any legal power to forbid +circumvention of technological measures to the extent such circumvention +is effected by exercising rights under this License with respect to +the covered work, and you disclaim any intention to limit operation or +modification of the work as a means of enforcing, against the work's +users, your or third parties' legal rights to forbid circumvention of +technological measures. + + 4. Conveying Verbatim Copies. + + You may convey verbatim copies of the Program's source code as you +receive it, in any medium, provided that you conspicuously and +appropriately publish on each copy an appropriate copyright notice; +keep intact all notices stating that this License and any +non-permissive terms added in accord with section 7 apply to the code; +keep intact all notices of the absence of any warranty; and give all +recipients a copy of this License along with the Program. + + You may charge any price or no price for each copy that you convey, +and you may offer support or warranty protection for a fee. + + 5. Conveying Modified Source Versions. + + You may convey a work based on the Program, or the modifications to +produce it from the Program, in the form of source code under the +terms of section 4, provided that you also meet all of these conditions: + + a) The work must carry prominent notices stating that you modified + it, and giving a relevant date. + + b) The work must carry prominent notices stating that it is + released under this License and any conditions added under section + 7. This requirement modifies the requirement in section 4 to + "keep intact all notices". + + c) You must license the entire work, as a whole, under this + License to anyone who comes into possession of a copy. This + License will therefore apply, along with any applicable section 7 + additional terms, to the whole of the work, and all its parts, + regardless of how they are packaged. This License gives no + permission to license the work in any other way, but it does not + invalidate such permission if you have separately received it. + + d) If the work has interactive user interfaces, each must display + Appropriate Legal Notices; however, if the Program has interactive + interfaces that do not display Appropriate Legal Notices, your + work need not make them do so. + + A compilation of a covered work with other separate and independent +works, which are not by their nature extensions of the covered work, +and which are not combined with it such as to form a larger program, +in or on a volume of a storage or distribution medium, is called an +"aggregate" if the compilation and its resulting copyright are not +used to limit the access or legal rights of the compilation's users +beyond what the individual works permit. Inclusion of a covered work +in an aggregate does not cause this License to apply to the other +parts of the aggregate. + + 6. Conveying Non-Source Forms. + + You may convey a covered work in object code form under the terms +of sections 4 and 5, provided that you also convey the +machine-readable Corresponding Source under the terms of this License, +in one of these ways: + + a) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by the + Corresponding Source fixed on a durable physical medium + customarily used for software interchange. + + b) Convey the object code in, or embodied in, a physical product + (including a physical distribution medium), accompanied by a + written offer, valid for at least three years and valid for as + long as you offer spare parts or customer support for that product + model, to give anyone who possesses the object code either (1) a + copy of the Corresponding Source for all the software in the + product that is covered by this License, on a durable physical + medium customarily used for software interchange, for a price no + more than your reasonable cost of physically performing this + conveying of source, or (2) access to copy the + Corresponding Source from a network server at no charge. + + c) Convey individual copies of the object code with a copy of the + written offer to provide the Corresponding Source. This + alternative is allowed only occasionally and noncommercially, and + only if you received the object code with such an offer, in accord + with subsection 6b. + + d) Convey the object code by offering access from a designated + place (gratis or for a charge), and offer equivalent access to the + Corresponding Source in the same way through the same place at no + further charge. You need not require recipients to copy the + Corresponding Source along with the object code. If the place to + copy the object code is a network server, the Corresponding Source + may be on a different server (operated by you or a third party) + that supports equivalent copying facilities, provided you maintain + clear directions next to the object code saying where to find the + Corresponding Source. Regardless of what server hosts the + Corresponding Source, you remain obligated to ensure that it is + available for as long as needed to satisfy these requirements. + + e) Convey the object code using peer-to-peer transmission, provided + you inform other peers where the object code and Corresponding + Source of the work are being offered to the general public at no + charge under subsection 6d. + + A separable portion of the object code, whose source code is excluded +from the Corresponding Source as a System Library, need not be +included in conveying the object code work. + + A "User Product" is either (1) a "consumer product", which means any +tangible personal property which is normally used for personal, family, +or household purposes, or (2) anything designed or sold for incorporation +into a dwelling. In determining whether a product is a consumer product, +doubtful cases shall be resolved in favor of coverage. For a particular +product received by a particular user, "normally used" refers to a +typical or common use of that class of product, regardless of the status +of the particular user or of the way in which the particular user +actually uses, or expects or is expected to use, the product. A product +is a consumer product regardless of whether the product has substantial +commercial, industrial or non-consumer uses, unless such uses represent +the only significant mode of use of the product. + + "Installation Information" for a User Product means any methods, +procedures, authorization keys, or other information required to install +and execute modified versions of a covered work in that User Product from +a modified version of its Corresponding Source. The information must +suffice to ensure that the continued functioning of the modified object +code is in no case prevented or interfered with solely because +modification has been made. + + If you convey an object code work under this section in, or with, or +specifically for use in, a User Product, and the conveying occurs as +part of a transaction in which the right of possession and use of the +User Product is transferred to the recipient in perpetuity or for a +fixed term (regardless of how the transaction is characterized), the +Corresponding Source conveyed under this section must be accompanied +by the Installation Information. But this requirement does not apply +if neither you nor any third party retains the ability to install +modified object code on the User Product (for example, the work has +been installed in ROM). + + The requirement to provide Installation Information does not include a +requirement to continue to provide support service, warranty, or updates +for a work that has been modified or installed by the recipient, or for +the User Product in which it has been modified or installed. Access to a +network may be denied when the modification itself materially and +adversely affects the operation of the network or violates the rules and +protocols for communication across the network. + + Corresponding Source conveyed, and Installation Information provided, +in accord with this section must be in a format that is publicly +documented (and with an implementation available to the public in +source code form), and must require no special password or key for +unpacking, reading or copying. + + 7. Additional Terms. + + "Additional permissions" are terms that supplement the terms of this +License by making exceptions from one or more of its conditions. +Additional permissions that are applicable to the entire Program shall +be treated as though they were included in this License, to the extent +that they are valid under applicable law. If additional permissions +apply only to part of the Program, that part may be used separately +under those permissions, but the entire Program remains governed by +this License without regard to the additional permissions. + + When you convey a copy of a covered work, you may at your option +remove any additional permissions from that copy, or from any part of +it. (Additional permissions may be written to require their own +removal in certain cases when you modify the work.) You may place +additional permissions on material, added by you to a covered work, +for which you have or can give appropriate copyright permission. + + Notwithstanding any other provision of this License, for material you +add to a covered work, you may (if authorized by the copyright holders of +that material) supplement the terms of this License with terms: + + a) Disclaiming warranty or limiting liability differently from the + terms of sections 15 and 16 of this License; or + + b) Requiring preservation of specified reasonable legal notices or + author attributions in that material or in the Appropriate Legal + Notices displayed by works containing it; or + + c) Prohibiting misrepresentation of the origin of that material, or + requiring that modified versions of such material be marked in + reasonable ways as different from the original version; or + + d) Limiting the use for publicity purposes of names of licensors or + authors of the material; or + + e) Declining to grant rights under trademark law for use of some + trade names, trademarks, or service marks; or + + f) Requiring indemnification of licensors and authors of that + material by anyone who conveys the material (or modified versions of + it) with contractual assumptions of liability to the recipient, for + any liability that these contractual assumptions directly impose on + those licensors and authors. + + All other non-permissive additional terms are considered "further +restrictions" within the meaning of section 10. If the Program as you +received it, or any part of it, contains a notice stating that it is +governed by this License along with a term that is a further +restriction, you may remove that term. If a license document contains +a further restriction but permits relicensing or conveying under this +License, you may add to a covered work material governed by the terms +of that license document, provided that the further restriction does +not survive such relicensing or conveying. + + If you add terms to a covered work in accord with this section, you +must place, in the relevant source files, a statement of the +additional terms that apply to those files, or a notice indicating +where to find the applicable terms. + + Additional terms, permissive or non-permissive, may be stated in the +form of a separately written license, or stated as exceptions; +the above requirements apply either way. + + 8. Termination. + + You may not propagate or modify a covered work except as expressly +provided under this License. Any attempt otherwise to propagate or +modify it is void, and will automatically terminate your rights under +this License (including any patent licenses granted under the third +paragraph of section 11). + + However, if you cease all violation of this License, then your +license from a particular copyright holder is reinstated (a) +provisionally, unless and until the copyright holder explicitly and +finally terminates your license, and (b) permanently, if the copyright +holder fails to notify you of the violation by some reasonable means +prior to 60 days after the cessation. + + Moreover, your license from a particular copyright holder is +reinstated permanently if the copyright holder notifies you of the +violation by some reasonable means, this is the first time you have +received notice of violation of this License (for any work) from that +copyright holder, and you cure the violation prior to 30 days after +your receipt of the notice. + + Termination of your rights under this section does not terminate the +licenses of parties who have received copies or rights from you under +this License. If your rights have been terminated and not permanently +reinstated, you do not qualify to receive new licenses for the same +material under section 10. + + 9. Acceptance Not Required for Having Copies. + + You are not required to accept this License in order to receive or +run a copy of the Program. Ancillary propagation of a covered work +occurring solely as a consequence of using peer-to-peer transmission +to receive a copy likewise does not require acceptance. However, +nothing other than this License grants you permission to propagate or +modify any covered work. These actions infringe copyright if you do +not accept this License. Therefore, by modifying or propagating a +covered work, you indicate your acceptance of this License to do so. + + 10. Automatic Licensing of Downstream Recipients. + + Each time you convey a covered work, the recipient automatically +receives a license from the original licensors, to run, modify and +propagate that work, subject to this License. You are not responsible +for enforcing compliance by third parties with this License. + + An "entity transaction" is a transaction transferring control of an +organization, or substantially all assets of one, or subdividing an +organization, or merging organizations. If propagation of a covered +work results from an entity transaction, each party to that +transaction who receives a copy of the work also receives whatever +licenses to the work the party's predecessor in interest had or could +give under the previous paragraph, plus a right to possession of the +Corresponding Source of the work from the predecessor in interest, if +the predecessor has it or can get it with reasonable efforts. + + You may not impose any further restrictions on the exercise of the +rights granted or affirmed under this License. For example, you may +not impose a license fee, royalty, or other charge for exercise of +rights granted under this License, and you may not initiate litigation +(including a cross-claim or counterclaim in a lawsuit) alleging that +any patent claim is infringed by making, using, selling, offering for +sale, or importing the Program or any portion of it. + + 11. Patents. + + A "contributor" is a copyright holder who authorizes use under this +License of the Program or a work on which the Program is based. The +work thus licensed is called the contributor's "contributor version". + + A contributor's "essential patent claims" are all patent claims +owned or controlled by the contributor, whether already acquired or +hereafter acquired, that would be infringed by some manner, permitted +by this License, of making, using, or selling its contributor version, +but do not include claims that would be infringed only as a +consequence of further modification of the contributor version. For +purposes of this definition, "control" includes the right to grant +patent sublicenses in a manner consistent with the requirements of +this License. + + Each contributor grants you a non-exclusive, worldwide, royalty-free +patent license under the contributor's essential patent claims, to +make, use, sell, offer for sale, import and otherwise run, modify and +propagate the contents of its contributor version. + + In the following three paragraphs, a "patent license" is any express +agreement or commitment, however denominated, not to enforce a patent +(such as an express permission to practice a patent or covenant not to +sue for patent infringement). To "grant" such a patent license to a +party means to make such an agreement or commitment not to enforce a +patent against the party. + + If you convey a covered work, knowingly relying on a patent license, +and the Corresponding Source of the work is not available for anyone +to copy, free of charge and under the terms of this License, through a +publicly available network server or other readily accessible means, +then you must either (1) cause the Corresponding Source to be so +available, or (2) arrange to deprive yourself of the benefit of the +patent license for this particular work, or (3) arrange, in a manner +consistent with the requirements of this License, to extend the patent +license to downstream recipients. "Knowingly relying" means you have +actual knowledge that, but for the patent license, your conveying the +covered work in a country, or your recipient's use of the covered work +in a country, would infringe one or more identifiable patents in that +country that you have reason to believe are valid. + + If, pursuant to or in connection with a single transaction or +arrangement, you convey, or propagate by procuring conveyance of, a +covered work, and grant a patent license to some of the parties +receiving the covered work authorizing them to use, propagate, modify +or convey a specific copy of the covered work, then the patent license +you grant is automatically extended to all recipients of the covered +work and works based on it. + + A patent license is "discriminatory" if it does not include within +the scope of its coverage, prohibits the exercise of, or is +conditioned on the non-exercise of one or more of the rights that are +specifically granted under this License. You may not convey a covered +work if you are a party to an arrangement with a third party that is +in the business of distributing software, under which you make payment +to the third party based on the extent of your activity of conveying +the work, and under which the third party grants, to any of the +parties who would receive the covered work from you, a discriminatory +patent license (a) in connection with copies of the covered work +conveyed by you (or copies made from those copies), or (b) primarily +for and in connection with specific products or compilations that +contain the covered work, unless you entered into that arrangement, +or that patent license was granted, prior to 28 March 2007. + + Nothing in this License shall be construed as excluding or limiting +any implied license or other defenses to infringement that may +otherwise be available to you under applicable patent law. + + 12. No Surrender of Others' Freedom. + + If conditions are imposed on you (whether by court order, agreement or +otherwise) that contradict the conditions of this License, they do not +excuse you from the conditions of this License. If you cannot convey a +covered work so as to satisfy simultaneously your obligations under this +License and any other pertinent obligations, then as a consequence you may +not convey it at all. For example, if you agree to terms that obligate you +to collect a royalty for further conveying from those to whom you convey +the Program, the only way you could satisfy both those terms and this +License would be to refrain entirely from conveying the Program. + + 13. Use with the GNU Affero General Public License. + + Notwithstanding any other provision of this License, you have +permission to link or combine any covered work with a work licensed +under version 3 of the GNU Affero General Public License into a single +combined work, and to convey the resulting work. The terms of this +License will continue to apply to the part which is the covered work, +but the special requirements of the GNU Affero General Public License, +section 13, concerning interaction through a network will apply to the +combination as such. + + 14. Revised Versions of this License. + + The Free Software Foundation may publish revised and/or new versions of +the GNU General Public License from time to time. Such new versions will +be similar in spirit to the present version, but may differ in detail to +address new problems or concerns. + + Each version is given a distinguishing version number. If the +Program specifies that a certain numbered version of the GNU General +Public License "or any later version" applies to it, you have the +option of following the terms and conditions either of that numbered +version or of any later version published by the Free Software +Foundation. If the Program does not specify a version number of the +GNU General Public License, you may choose any version ever published +by the Free Software Foundation. + + If the Program specifies that a proxy can decide which future +versions of the GNU General Public License can be used, that proxy's +public statement of acceptance of a version permanently authorizes you +to choose that version for the Program. + + Later license versions may give you additional or different +permissions. However, no additional obligations are imposed on any +author or copyright holder as a result of your choosing to follow a +later version. + + 15. Disclaimer of Warranty. + + THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY +APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT +HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY +OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, +THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR +PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM +IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF +ALL NECESSARY SERVICING, REPAIR OR CORRECTION. + + 16. Limitation of Liability. + + IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING +WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS +THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY +GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE +USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF +DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD +PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), +EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF +SUCH DAMAGES. + + 17. Interpretation of Sections 15 and 16. + + If the disclaimer of warranty and limitation of liability provided +above cannot be given local legal effect according to their terms, +reviewing courts shall apply local law that most closely approximates +an absolute waiver of all civil liability in connection with the +Program, unless a warranty or assumption of liability accompanies a +copy of the Program in return for a fee. + +Bison Exception + +As a special exception, you may create a larger work that contains part or all +of the Bison parser skeleton and distribute that work under terms of your +choice, so long as that work isn't itself a parser generator using the skeleton +or a modified version thereof as a parser skeleton. Alternatively, if you +modify or redistribute the parser skeleton itself, you may (at your option) +remove this special exception, which will cause the skeleton and the resulting +Bison output files to be licensed under the GNU General Public License without +this special exception. + +This special exception was added by the Free Software Foundation in version +2.2 of Bison. + + END OF TERMS AND CONDITIONS + +-------------------------------------------------------------------------------- +================================================================================ +-------------------------------------------------------------------------------- + +The preprocessor has the core licenses stated above, plus additional licences: + +/****************************************************************************\ +Copyright (c) 2002, NVIDIA Corporation. + +NVIDIA Corporation("NVIDIA") supplies this software to you in +consideration of your agreement to the following terms, and your use, +installation, modification or redistribution of this NVIDIA software +constitutes acceptance of these terms. If you do not agree with these +terms, please do not use, install, modify or redistribute this NVIDIA +software. + +In consideration of your agreement to abide by the following terms, and +subject to these terms, NVIDIA grants you a personal, non-exclusive +license, under NVIDIA's copyrights in this original NVIDIA software (the +"NVIDIA Software"), to use, reproduce, modify and redistribute the +NVIDIA Software, with or without modifications, in source and/or binary +forms; provided that if you redistribute the NVIDIA Software, you must +retain the copyright notice of NVIDIA, this notice and the following +text and disclaimers in all such redistributions of the NVIDIA Software. +Neither the name, trademarks, service marks nor logos of NVIDIA +Corporation may be used to endorse or promote products derived from the +NVIDIA Software without specific prior written permission from NVIDIA. +Except as expressly stated in this notice, no other rights or licenses +express or implied, are granted by NVIDIA herein, including but not +limited to any patent rights that may be infringed by your derivative +works or by other works in which the NVIDIA Software may be +incorporated. No hardware is licensed hereunder. + +THE NVIDIA SOFTWARE IS BEING PROVIDED ON AN "AS IS" BASIS, WITHOUT +WARRANTIES OR CONDITIONS OF ANY KIND, EITHER EXPRESS OR IMPLIED, +INCLUDING WITHOUT LIMITATION, WARRANTIES OR CONDITIONS OF TITLE, +NON-INFRINGEMENT, MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, OR +ITS USE AND OPERATION EITHER ALONE OR IN COMBINATION WITH OTHER +PRODUCTS. + +IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY SPECIAL, INDIRECT, +INCIDENTAL, EXEMPLARY, CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED +TO, LOST PROFITS; PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF +USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) OR ARISING IN ANY WAY +OUT OF THE USE, REPRODUCTION, MODIFICATION AND/OR DISTRIBUTION OF THE +NVIDIA SOFTWARE, HOWEVER CAUSED AND WHETHER UNDER THEORY OF CONTRACT, +TORT (INCLUDING NEGLIGENCE), STRICT LIABILITY OR OTHERWISE, EVEN IF +NVIDIA HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +\****************************************************************************/ + +/* +** Copyright (c) 2014-2016 The Khronos Group Inc. +** +** Permission is hereby granted, free of charge, to any person obtaining a copy +** of this software and/or associated documentation files (the "Materials"), +** to deal in the Materials without restriction, including without limitation +** the rights to use, copy, modify, merge, publish, distribute, sublicense, +** and/or sell copies of the Materials, and to permit persons to whom the +** Materials are furnished to do so, subject to the following conditions: +** +** The above copyright notice and this permission notice shall be included in +** all copies or substantial portions of the Materials. +** +** MODIFICATIONS TO THIS FILE MAY MEAN IT NO LONGER ACCURATELY REFLECTS KHRONOS +** STANDARDS. THE UNMODIFIED, NORMATIVE VERSIONS OF KHRONOS SPECIFICATIONS AND +** HEADER INFORMATION ARE LOCATED AT https://www.khronos.org/registry/ +** +** THE MATERIALS ARE PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +** OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +** FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL +** THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +** LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +** FROM,OUT OF OR IN CONNECTION WITH THE MATERIALS OR THE USE OR OTHER DEALINGS +** IN THE MATERIALS. +*/ diff --git a/src/cthreads/python/cthreads/marshal.py b/src/cthreads/python/cthreads/marshal.py index 9a0c219..56afda4 100644 --- a/src/cthreads/python/cthreads/marshal.py +++ b/src/cthreads/python/cthreads/marshal.py @@ -228,7 +228,7 @@ def _pack_c(pack: int | ctypes.c_void_p) -> ctypes.c_void_p: return pack if not pack: raise RuntimeError("cthreads.marshal: null pack pointer") - return ctypes.c_void_p(int(pack)) + return ctypes.c_void_p(int(pack)) # cast to void (trampolines expect a void pointer and static cast internally) def _extra(path: _Path) -> list: diff --git a/src/cthreads/python/cthreads/sync/__init__.py b/src/cthreads/python/cthreads/sync/__init__.py index e8d82d4..b7869fa 100644 --- a/src/cthreads/python/cthreads/sync/__init__.py +++ b/src/cthreads/python/cthreads/sync/__init__.py @@ -4,6 +4,8 @@ - Annotation: `TBuffer[...]` (from types) - Host alloc: `create_tbuffer` / `TBufferHandle` / … - Native locks/events: re-exported from `cthreads._ext.sync` when present +- GPU workgroup barrier stub: `__sync_threads`; inside `@Gpu` also + `Barrier.arrive_and_wait()` (same GLSL lowering, no Barrier(...) call) """ from __future__ import annotations @@ -19,6 +21,24 @@ tbuffer_read_copy_ptr, ) + +def __sync_threads() -> None: + """ + Workgroup barrier stub (CUDA-style). + + Only valid inside `@Gpu` bodies; compiled to GLSL barrier() / + memoryBarrierShared(). Same device sync as Barrier.arrive_and_wait() + on the GPU path. + + #### Raises + - RuntimeError = called from ordinary Python (not compiled @Gpu) + """ + raise RuntimeError( + "cthreads.sync.__sync_threads() is only valid inside @Gpu bodies " + "(workgroup barrier; compiled to GLSL barrier())" + ) + + try: from cthreads import _ext as _ext except ImportError: @@ -52,4 +72,5 @@ "RWLock", "Barrier", "TBufferI64", + "__sync_threads", ] diff --git a/tests/helpers_gpu.py b/tests/helpers_gpu.py new file mode 100644 index 0000000..6078775 --- /dev/null +++ b/tests/helpers_gpu.py @@ -0,0 +1,42 @@ +""" +Shared helpers for GPU unit / pipeline tests. +""" + +from __future__ import annotations + +import os +from types import ModuleType + +import cthreads.gpu.runtime as runtime_mod + + +def prepare_module() -> ModuleType: + """ + Return `cthreads.gpu.runtime` (holds `_gpu_prepared` / prepare / gpu). + """ + return runtime_mod + + +def on_github_actions() -> bool: + """True when running under GitHub Actions CI.""" + return os.environ.get("GITHUB_ACTIONS", "").lower() == "true" + + +def glsl_compiler_available() -> bool: + """ + True when in-process compile_glsl or glslc can compile a tiny compute shader. + + Always False on GitHub Actions for now (CI builds without CTHREADS_GPU / + glslang). Broad except: probe must never fail collection. + """ + if on_github_actions(): + return False + try: + from cthreads.gpu.compiler.translation.spirv import compile_glsl_to_spirv + + compile_glsl_to_spirv( + "#version 450\nlayout(local_size_x = 1) in;\nvoid main() {}\n" + ) + return True + except Exception: + return False diff --git a/tests/unit/.gitignore b/tests/unit/.gitignore new file mode 100644 index 0000000..0bf994e --- /dev/null +++ b/tests/unit/.gitignore @@ -0,0 +1,11 @@ +# >>> cthreads (auto) +__Thread__/ +__Threadable__/ +__Gpu__/ +.cthreads_cache.json +cthreads_kernels.dll +cthreads_kernels.so +cthreads_kernels.lib +libcthreads_kernels.so +libcthreads_kernels.dylib +# <<< cthreads (auto) diff --git a/tests/unit/test_cache.py b/tests/unit/test_cache.py index 1ea9a22..303c16f 100644 --- a/tests/unit/test_cache.py +++ b/tests/unit/test_cache.py @@ -47,12 +47,14 @@ def test_write_if_changed_writes_then_skips(tmp_path: Path): def test_load_cache_missing_and_corrupt(tmp_path: Path): empty = load_cache(tmp_path) assert empty["units"] == {} + assert empty["gpu_units"] == {} assert empty["link_hash"] is None bad = tmp_path / CACHE_FILENAME bad.write_text("{not json", encoding="utf-8") recovered = load_cache(tmp_path) assert recovered["units"] == {} + assert recovered["gpu_units"] == {} def test_load_cache_wrong_version(tmp_path: Path): @@ -63,12 +65,39 @@ def test_load_cache_wrong_version(tmp_path: Path): ) data = load_cache(tmp_path) assert data["units"] == {} + assert data["gpu_units"] == {} + + +def test_load_cache_fills_missing_gpu_units(tmp_path: Path): + path = cache_path_for_root(tmp_path) + path.write_text( + json.dumps( + { + "version": REGISTRY.VERSION, + "units": {"move": {"src_hash": "abc"}}, + "link_hash": None, + "binary": None, + } + ), + encoding="utf-8", + ) + data = load_cache(tmp_path) + assert data["units"]["move"]["src_hash"] == "abc" + assert data["gpu_units"] == {} def test_save_and_load_cache_roundtrip(tmp_path: Path): - save_cache(tmp_path, {"units": {"move": {"src_hash": "abc"}}, "link_hash": "L"}) + save_cache( + tmp_path, + { + "units": {"move": {"src_hash": "abc"}}, + "gpu_units": {"saxpy": {"src_hash": "def"}}, + "link_hash": "L", + }, + ) data = load_cache(tmp_path) assert data["units"]["move"]["src_hash"] == "abc" + assert data["gpu_units"]["saxpy"]["src_hash"] == "def" assert data["link_hash"] == "L" assert data["version"] == REGISTRY.VERSION @@ -89,6 +118,7 @@ def test_ensure_gitignore_creates_and_is_idempotent(tmp_path: Path): assert ensure_gitignore(tmp_path) is True text = (tmp_path / ".gitignore").read_text(encoding="utf-8") assert "__Thread__/" in text + assert "__Gpu__/" in text assert ".cthreads_cache.json" in text assert "cthreads_kernels.dll" in text assert ensure_gitignore(tmp_path) is False diff --git a/tests/unit/test_gpu_arena.py b/tests/unit/test_gpu_arena.py new file mode 100644 index 0000000..ab0011c --- /dev/null +++ b/tests/unit/test_gpu_arena.py @@ -0,0 +1,100 @@ +"""GpuArena residency: bind once, relaunch without re-upload, sync download.""" + +from __future__ import annotations + +import pytest + +from helpers_gpu import prepare_module + +from cthreads.frontend.Registry import REGISTRY +from cthreads.gpu import ( + GlobalIdx, + Gpu, + GpuArena, + GpuInvalidArgument, + available, + gpu, + shutdown, +) + + +pytestmark = pytest.mark.skipif( + not available(), + reason="GPU / Vulkan not available", +) + + +@pytest.fixture(autouse=True) +def _reset(): + prepare_mod = prepare_module() + REGISTRY.clear() + prepare_mod._gpu_prepared = False + yield + REGISTRY.clear() + try: + shutdown() + except Exception: + pass + prepare_mod._gpu_prepared = False + + +def test_arena_bind_reuse_and_sync(): + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [10.0, 20.0, 30.0, 40.0] + with GpuArena() as arena: + arena.bind(x=x, y=y) + gpu(saxpy, len(x), 2.0, x, y).join(download=False) + gpu(saxpy, len(x), 2.0, x, y).join(download=False) + # Host lists unchanged until sync (device authoritative). + assert y == [10.0, 20.0, 30.0, 40.0] + arena.sync() + assert y == pytest.approx([14.0, 28.0, 42.0, 56.0]) + + +def test_arena_join_download_true_still_writeback(): + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [1.0, 2.0] + y = [0.0, 0.0] + with GpuArena() as arena: + arena.bind(x=x, y=y) + gpu(saxpy, len(x), 3.0, x, y).join(download=True) + assert y == pytest.approx([3.0, 6.0]) + + +def test_arena_length_mismatch_raises(): + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [1.0, 2.0, 3.0] + y = [0.0, 0.0, 0.0] + with GpuArena() as arena: + arena.bind(x=x, y=y) + x.append(4.0) + with pytest.raises(GpuInvalidArgument, match="length changed"): + gpu(saxpy, len(x), 1.0, x, y) + + +def test_arena_duplicate_list_bind_raises(): + x = [1.0, 2.0] + with GpuArena() as a1: + a1.bind(x=x) + with GpuArena() as a2: + with pytest.raises(GpuInvalidArgument, match="already bound"): + a2.bind(x=x) diff --git a/tests/unit/test_gpu_assemble.py b/tests/unit/test_gpu_assemble.py new file mode 100644 index 0000000..50ab1b2 --- /dev/null +++ b/tests/unit/test_gpu_assemble.py @@ -0,0 +1,50 @@ +"""Unit tests for assemble_comp.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu.compiler.translation.assemble import assemble_comp + + +def test_assemble_wraps_version_and_main(): + pre = "layout(local_size_x = 64) in;\n" + body = " int i = 0;\n" + src = assemble_comp(pre, body) + assert src.startswith("#version 450\n") + assert "layout(local_size_x = 64) in;" in src + assert "void main() {" in src + assert " int i = 0;" in src + assert src.rstrip().endswith("}") + + +@pytest.mark.parametrize("ver", [450, 460, 430]) +def test_assemble_custom_version(ver): + src = assemble_comp("", " return;\n", version=ver) + assert src.startswith(f"#version {ver}\n") + + +def test_assemble_empty_body(): + src = assemble_comp("layout(local_size_x = 1) in;\n", "") + assert "void main() {" in src + assert src.rstrip().endswith("}") + + +def test_assemble_strips_trailing_whitespace_on_parts(): + src = assemble_comp("preamble\n\n", " body;\n\n") + assert "\n\n\nvoid main" not in src + assert "void main() {" in src + + +def test_assemble_multiline_body_preserved(): + body = " int i = 0;\n i += 1;\n return;\n" + src = assemble_comp("layout(local_size_x = 1) in;", body) + assert "int i = 0;" in src + assert "i += 1;" in src + assert "return;" in src + + +def test_assemble_order_version_preamble_main(): + src = assemble_comp("AAA\n", " BBB;\n") + assert src.index("#version") < src.index("AAA") < src.index("void main") + assert src.index("void main") < src.index("BBB") diff --git a/tests/unit/test_gpu_context.py b/tests/unit/test_gpu_context.py new file mode 100644 index 0000000..4a7fe53 --- /dev/null +++ b/tests/unit/test_gpu_context.py @@ -0,0 +1,254 @@ +""" +Issue GPU-01: cthreads.gpu Vulkan context probe. + +Always-safe paths run everywhere. Live Vulkan checks skip when the +extension was built without CTHREADS_GPU or when no compute device +is available (CI / headless / no driver). +""" + +from __future__ import annotations + +import pytest + +from cthreads import gpu +from cthreads.gpu.frontend.errors import ( + CThreadsGPUError, + GPUNotAvailable, + GpuInvalidArgument, + GpuUseAfterDestroy, + VulkanInitFailed, + VulkanLoaderNotFound, + VulkanNoDevice, + VulkanNotBuiltError, + VulkanOutOfMemory, +) + + +# --------------------------------------------------------------------------- +# Error types (no native / GPU required) +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "cls", + [ + CThreadsGPUError, + VulkanNotBuiltError, + VulkanLoaderNotFound, + VulkanNoDevice, + VulkanInitFailed, + VulkanOutOfMemory, + GpuInvalidArgument, + GpuUseAfterDestroy, + GPUNotAvailable, + ], +) +def test_error_is_cthreads_gpu_error(cls): + err = cls() + assert isinstance(err, CThreadsGPUError) + assert isinstance(err, Exception) + assert "cthreads gpu error" in str(err) + assert err.detail + + +def test_error_custom_detail(): + err = VulkanInitFailed("cthreads.gpu.VulkanInitFailed: boom") + assert err.detail == "cthreads.gpu.VulkanInitFailed: boom" + assert "boom" in str(err) + + +def test_gpu_module_exports(): + for name in ( + "available", + "device_name", + "init", + "shutdown", + "CThreadsGPUError", + "VulkanNotBuiltError", + "VulkanLoaderNotFound", + "VulkanNoDevice", + "VulkanInitFailed", + "VulkanOutOfMemory", + "GpuInvalidArgument", + "GpuUseAfterDestroy", + "GPUNotAvailable", + ): + assert hasattr(gpu, name) + + +# --------------------------------------------------------------------------- +# Soft path: extension built without CTHREADS_GPU (_gpu is None) +# --------------------------------------------------------------------------- + + +def test_not_built_available_false(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", None) + assert gpu.available() is False + + +def test_not_built_device_name_raises(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", None) + with pytest.raises(VulkanNotBuiltError, match="CTHREADS_GPU"): + gpu.device_name() + + +def test_not_built_init_raises(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", None) + with pytest.raises(VulkanNotBuiltError, match="CTHREADS_GPU"): + gpu.init() + + +def test_not_built_shutdown_noop(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", None) + gpu.shutdown() # must not raise + + +# --------------------------------------------------------------------------- +# Error mapping from C++ message prefixes (fake _ext.gpu) +# --------------------------------------------------------------------------- + + +class _FakeGpu: + def __init__(self, exc: BaseException | None = None, ready: bool = False): + self._exc = exc + self._ready = ready + self.init_calls = 0 + self.shutdown_calls = 0 + + def available(self) -> bool: + if self._exc is not None: + raise self._exc + return self._ready + + def device_name(self) -> str: + if self._exc is not None: + raise self._exc + return "FakeGPU" + + def init(self) -> None: + self.init_calls += 1 + if self._exc is not None: + raise self._exc + + def shutdown(self) -> None: + self.shutdown_calls += 1 + + +@pytest.mark.parametrize( + "msg,exc_type", + [ + ("cthreads.gpu.VulkanLoaderNotFound: vulkan-1.dll not found", VulkanLoaderNotFound), + ("cthreads.gpu.VulkanNoDevice: no physical devices", VulkanNoDevice), + ("cthreads.gpu.VulkanInitFailed: vkCreateInstance failed", VulkanInitFailed), + ("cthreads.gpu.VulkanOutOfMemory: vkAllocateMemory failed", VulkanOutOfMemory), + ("cthreads.gpu.GpuInvalidArgument: size invalid", GpuInvalidArgument), + ("cthreads.gpu.GpuUseAfterDestroy: scalar buffer is not initialized", GpuUseAfterDestroy), + ("cthreads.gpu.VulkanNotBuilt: should not happen from C++", VulkanNotBuiltError), + ("something else entirely", VulkanInitFailed), + ], +) +def test_map_error_via_device_name(monkeypatch, msg, exc_type): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", _FakeGpu(RuntimeError(msg))) + with pytest.raises(exc_type) as ei: + gpu.device_name() + assert msg in str(ei.value) + + +def test_map_error_via_init(monkeypatch): + monkeypatch.setattr( + gpu._ext_gpu_api, + "_gpu", + _FakeGpu(RuntimeError("cthreads.gpu.VulkanLoaderNotFound: missing")), + ) + with pytest.raises(VulkanLoaderNotFound): + gpu.init() + + +def test_fake_available_false_without_raising(monkeypatch): + """Mirrors C++ available(): False when init would fail, no exception.""" + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", _FakeGpu(ready=False)) + assert gpu.available() is False + + +def test_fake_ready_device_name(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", _FakeGpu(ready=True)) + assert gpu.available() is True + assert gpu.device_name() == "FakeGPU" + gpu.init() + gpu.shutdown() + assert gpu._ext_gpu_api._gpu.shutdown_calls == 1 + + +# --------------------------------------------------------------------------- +# Live Vulkan (skip when not built / no device) +# --------------------------------------------------------------------------- + + +def _ext_gpu_built() -> bool: + return gpu._ext_gpu_api._gpu is not None + + +def _require_gpu(): + if not _ext_gpu_built(): + pytest.skip("cthreads built without CTHREADS_GPU (_ext.gpu missing)") + if not gpu.available(): + pytest.skip("Vulkan loader/device not available in this environment") + + +def test_live_available_is_bool(): + """available() must never raise; False is fine without GPU.""" + assert isinstance(gpu.available(), bool) + + +def test_live_shutdown_safe_without_init(): + """shutdown is always safe (no-op if not built / not ready).""" + gpu.shutdown() + + +def test_live_unavailable_device_name_raises_mapped(): + """When built but init fails, device_name raises a mapped CThreadsGPUError.""" + if not _ext_gpu_built(): + pytest.skip("cthreads built without CTHREADS_GPU") + if gpu.available(): + pytest.skip("GPU is available — covered by live success tests") + with pytest.raises(CThreadsGPUError): + gpu.device_name() + + +def test_live_unavailable_init_raises_mapped(): + if not _ext_gpu_built(): + pytest.skip("cthreads built without CTHREADS_GPU") + if gpu.available(): + pytest.skip("GPU is available — covered by live success tests") + with pytest.raises(CThreadsGPUError): + gpu.init() + + +def test_live_device_name_nonempty(): + _require_gpu() + name = gpu.device_name() + assert isinstance(name, str) + assert len(name.strip()) > 0 + + +def test_live_init_idempotent_then_shutdown_reinit(): + _require_gpu() + try: + gpu.init() + gpu.init() # second call no-ops + name1 = gpu.device_name() + gpu.shutdown() + assert gpu.available() is True # re-inits + name2 = gpu.device_name() + assert name1 == name2 + finally: + gpu.shutdown() + + +def test_live_available_true_matches_device_name(): + _require_gpu() + try: + assert gpu.available() is True + assert gpu.device_name() + finally: + gpu.shutdown() diff --git a/tests/unit/test_gpu_decorator.py b/tests/unit/test_gpu_decorator.py new file mode 100644 index 0000000..6232105 --- /dev/null +++ b/tests/unit/test_gpu_decorator.py @@ -0,0 +1,101 @@ +"""@Gpu decorator registration (no launch / emit).""" + +from __future__ import annotations + +import pytest + +from cthreads.frontend.Registry import REGISTRY +from cthreads.gpu.frontend import Gpu +from cthreads.gpu.frontend.errors import GPUNotAvailable +from cthreads.gpu import frontend as gpu_frontend + + +@pytest.fixture(autouse=True) +def _clear_registry(): + REGISTRY.clear() + yield + REGISTRY.clear() + + +def test_gpu_decorator_requires_available(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: False) + + with pytest.raises(GPUNotAvailable): + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + +def test_gpu_decorator_registers_and_attaches_meta(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + assert saxpy.__gpu__ is True + assert saxpy.__gpu_version__ == REGISTRY.VERSION + assert saxpy.__qualname__ in REGISTRY.gpu_functions + meta = saxpy.__gpu_kernel_meta__ + assert meta["symbol"] == "saxpy" + assert meta["binding_count"] == 3 + assert meta["scalar_bytes"] == 8 + # Still a normal Python function. + assert saxpy(4, 2.0, [1.0], [0.0]) is None + + +def test_gpu_decorator_log_factory(monkeypatch, capsys): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + monkeypatch.setattr(gpu_frontend.wrapper, "device_name", lambda: "FakeGPU") + + @Gpu(log=True) + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + assert saxpy.__gpu__ is True + assert "FakeGPU" in capsys.readouterr().out + + +def test_gpu_decorator_rejects_non_none_return(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + + with pytest.raises(TypeError, match="return must be None"): + + @Gpu + def bad(x: int) -> int: + return x + + +def test_gpu_decorator_rejects_dict(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + + with pytest.raises(TypeError): + + @Gpu + def bad(d: dict[str, int]) -> None: + pass + + +def test_gpu_decorator_rejects_varargs(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + + with pytest.raises(TypeError): + + @Gpu + def bad(*args: int) -> None: + pass + + +def test_gpu_decorator_bool_and_int_list(monkeypatch): + monkeypatch.setattr(gpu_frontend.wrapper, "available", lambda: True) + + @Gpu + def k(n: int, flag: bool, xs: list[int], ys: list[bool]) -> None: + pass + + meta = k.__gpu_kernel_meta__ + assert meta["binding_count"] == 3 + assert meta["scalar_bytes"] == 8 + kinds = [p["kind"] for p in meta["params"]] + assert kinds == ["int", "bool", "list", "list"] diff --git a/tests/unit/test_gpu_errors.py b/tests/unit/test_gpu_errors.py new file mode 100644 index 0000000..2e5c5c8 --- /dev/null +++ b/tests/unit/test_gpu_errors.py @@ -0,0 +1,51 @@ +"""Unit tests for GPU error mapping and types.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu.frontend.errors import ( + CThreadsGPUError, + GPUNotAvailable, + GpuInvalidArgument, + GpuUseAfterDestroy, + VulkanInitFailed, + VulkanLoaderNotFound, + VulkanNoDevice, + VulkanNotBuiltError, + VulkanOutOfMemory, + _map_error, +) + + +@pytest.mark.parametrize( + "msg, cls", + [ + ("VulkanLoaderNotFound: missing", VulkanLoaderNotFound), + ("VulkanNoDevice: none", VulkanNoDevice), + ("VulkanOutOfMemory: oom", VulkanOutOfMemory), + ("GpuUseAfterDestroy: gone", GpuUseAfterDestroy), + ("GpuInvalidArgument: bad", GpuInvalidArgument), + ("VulkanNotBuilt: off", VulkanNotBuiltError), + ("VulkanInitFailed: boom", VulkanInitFailed), + ("something else entirely", VulkanInitFailed), + ], +) +def test_map_error_prefixes(msg, cls): + out = _map_error(RuntimeError(msg)) + assert isinstance(out, cls) + assert isinstance(out, CThreadsGPUError) + assert msg in str(out) + + +def test_error_str_contains_prefix(): + err = GPUNotAvailable("no gpu") + assert "cthreads gpu error" in str(err) + assert "no gpu" in str(err) + assert err.detail == "no gpu" + + +def test_default_details(): + assert "Vulkan loader not found" in VulkanLoaderNotFound().detail + assert "No Vulkan compute device" in VulkanNoDevice().detail + assert "out of memory" in VulkanOutOfMemory().detail.lower() diff --git a/tests/unit/test_gpu_glsl.py b/tests/unit/test_gpu_glsl.py new file mode 100644 index 0000000..3d75ab1 --- /dev/null +++ b/tests/unit/test_gpu_glsl.py @@ -0,0 +1,99 @@ +"""Unit tests for GPU Glsl type helpers.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu.compiler.translation.Glsl import ( + align_bytes, + elem_type_name, + size_bytes, + type_name, +) +from cthreads.types import PyBool, PyDict, PyFloat, PyInt, PyList, PyString, PyThreadable + + +@pytest.mark.parametrize( + "py, name, nbytes", + [ + (PyInt(), "int", 4), + (PyFloat(), "float", 4), + (PyBool(), "bool", 4), + ], +) +def test_scalar_type_name_and_size(py, name, nbytes): + assert type_name(py) == name + assert size_bytes(py) == nbytes + assert align_bytes(py) == nbytes + + +def test_float_is_glsl_float_not_double(): + assert type_name(PyFloat()) == "float" + assert size_bytes(PyFloat()) == 4 + + +@pytest.mark.parametrize( + "inner, elem", + [ + (PyInt(), "int"), + (PyFloat(), "float"), + (PyBool(), "bool"), + ], +) +def test_elem_type_name(inner, elem): + assert elem_type_name(PyList(inner)) == elem + + +def test_type_name_rejects_list(): + with pytest.raises(TypeError, match="list is not a scalar"): + type_name(PyList(PyFloat())) + + +def test_size_bytes_rejects_list(): + with pytest.raises(TypeError, match="list has no single scalar size"): + size_bytes(PyList(PyInt())) + + +def test_align_bytes_rejects_list(): + with pytest.raises(TypeError, match="list has no single scalar size"): + align_bytes(PyList(PyBool())) + + +def test_elem_type_name_rejects_non_list(): + with pytest.raises(TypeError, match="expects PyList"): + elem_type_name(PyInt()) # type: ignore[arg-type] + + +@pytest.mark.parametrize( + "bad", + [ + PyString(), + PyDict(PyString(), PyInt()), + PyThreadable("T"), + ], +) +def test_unsupported_scalar_types(bad): + with pytest.raises(TypeError, match="not supported"): + type_name(bad) + + +@pytest.mark.parametrize( + "bad", + [ + PyString(), + PyDict(PyString(), PyInt()), + PyThreadable("T"), + ], +) +def test_unsupported_size_bytes(bad): + with pytest.raises(TypeError, match="unsupported|not supported"): + size_bytes(bad) + + +def test_nested_list_elem_rejected(): + with pytest.raises(TypeError): + elem_type_name(PyList(PyList(PyInt()))) + + +def test_type_name_inner_of_list_ok(): + assert type_name(PyList(PyFloat()).inner_type) == "float" diff --git a/tests/unit/test_gpu_indexes.py b/tests/unit/test_gpu_indexes.py new file mode 100644 index 0000000..3d683af --- /dev/null +++ b/tests/unit/test_gpu_indexes.py @@ -0,0 +1,47 @@ +"""Unit tests for GPU index frontend markers.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu import BlockDim, BlockIdx, GlobalIdx, GridDim, ThreadIdx +from cthreads.gpu.frontend.indexes import GpuIndexBuiltin, _Axis + + +@pytest.mark.parametrize( + "cls, base", + [ + (ThreadIdx, "gl_LocalInvocationID"), + (BlockIdx, "gl_WorkGroupID"), + (BlockDim, "gl_WorkGroupSize"), + (GridDim, "gl_NumWorkGroups"), + (GlobalIdx, "gl_GlobalInvocationID"), + ], +) +def test_index_classes_glsl_base(cls, base): + assert issubclass(cls, GpuIndexBuiltin) + assert cls._glsl_base == base + assert isinstance(cls.x, _Axis) + assert isinstance(cls.y, _Axis) + assert isinstance(cls.z, _Axis) + + +@pytest.mark.parametrize( + "cls", + [ThreadIdx, BlockIdx, BlockDim, GridDim, GlobalIdx], +) +@pytest.mark.parametrize("axis", ["x", "y", "z"]) +def test_axis_repr(cls, axis): + obj = getattr(cls, axis) + assert isinstance(obj, _Axis) + assert f"{cls.__name__}.{axis}" in repr(obj) + + +def test_indexes_reexported_from_package(): + import cthreads.gpu as g + + assert g.GlobalIdx is GlobalIdx + assert g.ThreadIdx is ThreadIdx + assert g.BlockIdx is BlockIdx + assert g.BlockDim is BlockDim + assert g.GridDim is GridDim diff --git a/tests/unit/test_gpu_kernel_meta.py b/tests/unit/test_gpu_kernel_meta.py new file mode 100644 index 0000000..dbf37bd --- /dev/null +++ b/tests/unit/test_gpu_kernel_meta.py @@ -0,0 +1,190 @@ +"""Unit tests for build_gpu_kernel_meta and schemas.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu.gpu_kernel_meta import ( + GPU_KERNELS, + GpuParamMeta, + GpuTypeSchema, + build_gpu_kernel_meta, + pytype_to_gpu_schema, +) +from cthreads.types import PyBool, PyDict, PyFloat, PyInt, PyList, PyString + + +@pytest.fixture(autouse=True) +def _clear_kernels(): + GPU_KERNELS.clear() + yield + GPU_KERNELS.clear() + + +def test_saxpy_meta_shape(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + meta = build_gpu_kernel_meta(saxpy) + assert meta.symbol == "saxpy" + assert meta.binding_count == 3 + assert meta.scalar_bytes == 8 + assert meta.local_size_x == 64 + assert [p.name for p in meta.params] == ["n", "a", "x", "y"] + assert meta.params[0].pass_as == "value" + assert meta.params[2].pass_as == "ref" + assert meta.params[2].elem_kind == "float" + assert meta.params[2].elem_bytes == 4 + d = meta.to_dict() + assert d["symbol"] == "saxpy" + assert saxpy.__gpu_kernel_meta__["binding_count"] == 3 # type: ignore[attr-defined] + assert GPU_KERNELS["saxpy"] is meta + + +def test_custom_symbol_and_local_size(): + def k(n: int) -> None: + pass + + meta = build_gpu_kernel_meta(k, symbol="my_k", local_size_x=32) + assert meta.symbol == "my_k" + assert meta.local_size_x == 32 + assert "my_k" in GPU_KERNELS + + +@pytest.mark.parametrize( + "fn_src, binding, scalar", + [ + ("def k(a: int) -> None: pass", 1, 4), + ("def k(a: float, b: bool) -> None: pass", 1, 8), + ("def k(x: list[int]) -> None: pass", 1, 0), + ("def k(x: list[float], y: list[bool]) -> None: pass", 2, 0), + ("def k(n: int, x: list[int]) -> None: pass", 2, 4), + ], +) +def test_binding_and_scalar_bytes(fn_src, binding, scalar): + ns: dict = {} + exec(fn_src, ns) + meta = build_gpu_kernel_meta(ns["k"]) + assert meta.binding_count == binding + assert meta.scalar_bytes == scalar + + +def test_rejects_return_int(): + def k(n: int) -> int: + return n + + with pytest.raises(TypeError, match="return must be None"): + build_gpu_kernel_meta(k) + + +def test_rejects_missing_annotation(): + def k(n: int, a) -> None: # type: ignore[no-untyped-def] + pass + + with pytest.raises(TypeError, match="type annotation"): + build_gpu_kernel_meta(k) + + +def test_rejects_varargs(): + def k(*args: int) -> None: + pass + + with pytest.raises(TypeError, match=r"\*args"): + build_gpu_kernel_meta(k) + + +def test_rejects_kwargs(): + def k(**kwargs: int) -> None: + pass + + with pytest.raises(TypeError, match=r"\*\*kwargs"): + build_gpu_kernel_meta(k) + + +def test_rejects_kwonly(): + def k(*, n: int) -> None: + pass + + with pytest.raises(TypeError, match="keyword-only"): + build_gpu_kernel_meta(k) + + +def test_rejects_str_param(): + def k(x: str) -> None: + pass + + with pytest.raises(TypeError): + build_gpu_kernel_meta(k) + + +def test_rejects_dict_param(): + def k(x: dict[str, int]) -> None: + pass + + with pytest.raises(TypeError): + build_gpu_kernel_meta(k) + + +def test_rejects_list_str(): + def k(x: list[str]) -> None: + pass + + with pytest.raises(TypeError): + build_gpu_kernel_meta(k) + + +def test_rejects_nested_list(): + def k(x: list[list[int]]) -> None: + pass + + with pytest.raises(TypeError): + build_gpu_kernel_meta(k) + + +def test_pytype_to_gpu_schema_scalars(): + assert pytype_to_gpu_schema(PyInt()).kind == "int" + assert pytype_to_gpu_schema(PyFloat()).spirv_type == "float" + assert pytype_to_gpu_schema(PyBool()).elem_bytes == 4 + + +def test_pytype_to_gpu_schema_list(): + s = pytype_to_gpu_schema(PyList(PyFloat())) + assert s.kind == "list" + assert s.spirv_type == "float[]" + assert s.inner is not None + assert s.inner.kind == "float" + + +def test_pytype_rejects_dict_string(): + with pytest.raises(TypeError): + pytype_to_gpu_schema(PyDict(PyString(), PyInt())) + with pytest.raises(TypeError): + pytype_to_gpu_schema(PyString()) + + +def test_param_meta_pass_as_invalid(): + with pytest.raises(TypeError, match="pass_as"): + GpuParamMeta( + name="x", + pass_as="inout", # type: ignore[arg-type] + schema=GpuTypeSchema(kind="int", spirv_type="int", elem_bytes=4), + ) + + +def test_list_schema_to_dict_requires_inner(): + bad = GpuTypeSchema(kind="list", spirv_type="int[]", elem_bytes=4, inner=None) + with pytest.raises(ValueError, match="inner"): + bad.to_dict() + + +def test_param_to_dict_list_flattens_elem(): + p = GpuParamMeta( + name="xs", + pass_as="ref", + schema=pytype_to_gpu_schema(PyList(PyInt())), + ) + d = p.to_dict() + assert d["kind"] == "list" + assert d["elem_kind"] == "int" + assert d["elem_bytes"] == 4 + assert "schema" in d diff --git a/tests/unit/test_gpu_marshal.py b/tests/unit/test_gpu_marshal.py new file mode 100644 index 0000000..89311f6 --- /dev/null +++ b/tests/unit/test_gpu_marshal.py @@ -0,0 +1,153 @@ +"""Unit tests for gpu_marshal helpers.""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu.gpu_kernel_meta import build_gpu_kernel_meta +from cthreads.gpu.gpu_marshal import infer_group_count_x, ordered_values_for_meta + + +def _saxpy_meta(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + return build_gpu_kernel_meta(saxpy).to_dict() + + +def test_ordered_values_ok(): + meta = _saxpy_meta() + args = (4, 2.0, [1.0], [0.0]) + out = ordered_values_for_meta(meta, args) + assert out == [4, 2.0, [1.0], [0.0]] + assert out[2] is args[2] + + +def test_ordered_values_arity(): + meta = _saxpy_meta() + with pytest.raises(TypeError, match="expected 4"): + ordered_values_for_meta(meta, (1, 2.0)) + + +@pytest.mark.parametrize("extra", [0, 1, 5, 10]) +def test_ordered_values_too_many(extra): + meta = _saxpy_meta() + args = (1, 2.0, [0.0], [0.0]) + (0,) * extra + if extra == 0: + assert ordered_values_for_meta(meta, args) == list(args) + else: + with pytest.raises(TypeError, match="expected 4"): + ordered_values_for_meta(meta, args) + + +def test_ordered_values_missing_params(): + with pytest.raises(TypeError, match="params"): + ordered_values_for_meta({"symbol": "k"}, (1,)) + + +def test_ordered_values_params_not_list(): + with pytest.raises(TypeError, match="params"): + ordered_values_for_meta({"params": "bad"}, (1,)) + + +@pytest.mark.parametrize( + "n, local, expect", + [ + (1, 64, 1), + (63, 64, 1), + (64, 64, 1), + (65, 64, 2), + (128, 64, 2), + (129, 64, 3), + (130, 64, 3), + (200, 64, 4), + (1, 1, 1), + (10, 1, 10), + (10, 8, 2), + (16, 8, 2), + (17, 8, 3), + ], +) +def test_infer_group_count_table(n, local, expect): + meta = _saxpy_meta() + meta["local_size_x"] = local + args = [n, 1.0, [0.0] * n, [0.0] * n] + assert infer_group_count_x(meta, args) == expect + + +def test_infer_group_count_from_n(): + meta = _saxpy_meta() + meta["local_size_x"] = 64 + args = [130, 1.0, [0.0] * 130, [0.0] * 130] + assert infer_group_count_x(meta, args) == 3 + + +def test_infer_group_count_exact_multiple(): + meta = _saxpy_meta() + meta["local_size_x"] = 64 + args = [128, 1.0, [0.0] * 128, [0.0] * 128] + assert infer_group_count_x(meta, args) == 2 + + +def test_infer_group_count_from_list_len_without_n(): + def k(x: list[float], y: list[float]) -> None: + pass + + meta = build_gpu_kernel_meta(k).to_dict() + meta["local_size_x"] = 64 + assert infer_group_count_x(meta, [[0.0] * 100, [0.0] * 50]) == 2 + + +def test_infer_group_count_empty_defaults_one(): + def k(a: float) -> None: + pass + + meta = build_gpu_kernel_meta(k).to_dict() + assert infer_group_count_x(meta, [1.0]) == 1 + + +def test_infer_group_count_n_zero_defaults_one(): + meta = _saxpy_meta() + assert infer_group_count_x(meta, [0, 1.0, [], []]) == 1 + + +def test_infer_group_count_n_negative_defaults_one(): + meta = _saxpy_meta() + assert infer_group_count_x(meta, [-5, 1.0, [], []]) == 1 + + +def test_infer_group_count_bad_local_size_falls_back_64(): + meta = _saxpy_meta() + meta["local_size_x"] = 0 + # n=130 / 64 => 3 + assert infer_group_count_x(meta, [130, 1.0, [0.0] * 130, [0.0] * 130]) == 3 + + +def test_infer_group_count_missing_local_size_defaults_64(): + meta = _saxpy_meta() + meta.pop("local_size_x", None) + assert infer_group_count_x(meta, [64, 1.0, [0.0] * 64, [0.0] * 64]) == 1 + + +def test_infer_group_count_ignores_non_n_int_named_differently(): + def k(count: int, x: list[float]) -> None: + pass + + meta = build_gpu_kernel_meta(k).to_dict() + meta["local_size_x"] = 64 + # No param named `n` -> use longest list (100) => ceil(100/64)=2 + # even though count is 1 + assert infer_group_count_x(meta, [1, [0.0] * 100]) == 2 + + +def test_infer_group_count_no_params_returns_one(): + assert infer_group_count_x({"local_size_x": 64}, []) == 1 + + +def test_infer_group_count_skips_non_dict_params(): + meta = { + "local_size_x": 64, + "params": ["bad", {"name": "n", "kind": "int"}], + } + # Index of `n` is 1 — args must be long enough for that slot. + assert infer_group_count_x(meta, [None, 130]) == 3 diff --git a/tests/unit/test_gpu_pack.py b/tests/unit/test_gpu_pack.py new file mode 100644 index 0000000..a35e0b3 --- /dev/null +++ b/tests/unit/test_gpu_pack.py @@ -0,0 +1,173 @@ +""" +Issue GPU-02: GpuPack upload/download round-trips (test-only _ext.gpu.testing). + +Does not use public cthreads.gpu pack APIs (there are none). Live checks skip +when the extension was built without CTHREADS_GPU or when no device is available. +""" + +from __future__ import annotations + +import struct + +import pytest + +from cthreads import gpu +from cthreads.gpu.frontend.errors import ( + GpuInvalidArgument, + GpuUseAfterDestroy, +) + + +def _ext_gpu(): + return gpu._ext_gpu_api._gpu + + +def _require_gpu_testing(): + ext = _ext_gpu() + if ext is None: + pytest.skip("cthreads built without CTHREADS_GPU (_ext.gpu missing)") + if not hasattr(ext, "testing"): + pytest.skip( + "_ext.gpu.testing missing (local gpu/testing/ not in tree; " + "optional gitignored helpers)" + ) + if not gpu.available(): + pytest.skip("Vulkan loader/device not available in this environment") + return ext.testing + + +def _map_probe(exc: BaseException): + return gpu._map_error(exc) + + +def test_public_gpu_has_no_pack_roundtrip_exports(): + """Product package must not re-export test-only pack helpers.""" + assert not hasattr(gpu, "roundtrip_float_pack") + assert not hasattr(gpu, "roundtrip_int_pack") + assert not hasattr(gpu, "testing") + assert "roundtrip_float_pack" not in gpu.__all__ + for name in ("GpuInvalidArgument", "GpuUseAfterDestroy", "VulkanOutOfMemory"): + assert name in gpu.__all__ + + +def test_testing_submodule_absent_when_not_built(monkeypatch): + monkeypatch.setattr(gpu._ext_gpu_api, "_gpu", None) + assert _ext_gpu() is None + + +def test_live_roundtrip_float_scalars_and_lists(): + testing = _require_gpu_testing() + try: + scalar = struct.pack(" SPIR-V -> register -> launch via gpu(). + +Modular pieces are covered elsewhere; this file stresses end-to-end behavior +and edge cases across the public API. +""" + +from __future__ import annotations + +import math + +import pytest + +from helpers_gpu import glsl_compiler_available, prepare_module + +from cthreads.frontend.Registry import REGISTRY +from cthreads.gpu import GlobalIdx, Gpu, ThreadIdx, available, gpu, shutdown +from cthreads.gpu.compiler.translation.translate import translate_function_for_gpu + + +@pytest.fixture(autouse=True) +def _reset(): + prepare_mod = prepare_module() + REGISTRY.clear() + prepare_mod._gpu_prepared = False + yield + REGISTRY.clear() + try: + shutdown() + except Exception: + pass + prepare_mod._gpu_prepared = False + + +@pytest.mark.skipif( + not glsl_compiler_available(), + reason="no GLSL compiler (skipped on GitHub Actions / CPU-only builds)", +) +def test_pipeline_translate_source_is_shaderc_ready(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + r = translate_function_for_gpu(saxpy, compile_spirv=True) + assert r.spirv is not None + assert "layout(set = 0, binding = 0" in r.source + assert "void main()" in r.source + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +@pytest.mark.parametrize( + "n,a", + [ + (1, 1.0), + (2, -0.5), + (4, 2.0), + (8, 0.25), + (16, -2.0), + (31, 1.5), + (32, 0.0), + (33, 7.0), + (63, 0.5), + (64, -1.0), + (65, 3.25), + (96, 1.0), + (127, -3.0), + (128, 0.0), + (129, 2.5), + (200, math.pi), + (256, math.e), + (300, 1.125), + ], +) +def test_pipeline_saxpy_sizes(n, a): + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [float(i + 1) for i in range(n)] + y = [float(i) for i in range(n)] + expect = [a * xi + yi for xi, yi in zip(x, y)] + gpu(saxpy, n, a, x, y).join() + assert y == pytest.approx(expect) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +@pytest.mark.parametrize("n", [1, 2, 3, 7, 15, 16, 31, 32, 63, 64, 65, 100, 128, 200]) +def test_pipeline_int_list(n): + @Gpu + def add(n: int, x: list[int], y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] + y[i] + + x = list(range(n)) + y = [10] * n + expect = [x[i] + 10 for i in range(n)] + gpu(add, n, x, y).join() + assert y == expect + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +@pytest.mark.parametrize("n", [1, 4, 17, 64, 100]) +def test_pipeline_bool_list(n): + @Gpu + def invert(n: int, flags: list[bool], ys: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + if flags[i]: + ys[i] = 1 + else: + ys[i] = 0 + + flags = [(i % 2) == 0 for i in range(n)] + ys = [9] * n + gpu(invert, n, flags, ys).join() + assert ys == [1 if f else 0 for f in flags] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_bool_scalar(): + @Gpu + def set_if(n: int, flag: bool, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + if flag: + y[i] = 1.0 + else: + y[i] = 0.0 + + y = [9.0, 9.0, 9.0] + gpu(set_if, 3, True, y).join() + assert y == [1.0, 1.0, 1.0] + gpu(set_if, 3, False, y).join() + assert y == [0.0, 0.0, 0.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_early_return_leaves_tail_untouched(): + @Gpu + def partial(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 5.0 + + y = [0.0] * 10 + gpu(partial, 3, y).join() + assert y[:3] == [5.0, 5.0, 5.0] + assert y[3:] == [0.0] * 7 + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_two_kernels_same_process(): + @Gpu + def fill(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 1.0 + + @Gpu + def double(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = y[i] * 2.0 + + y = [0.0, 0.0, 0.0, 0.0] + gpu(fill, 4, y).join() + gpu(double, 4, y).join() + assert y == [2.0, 2.0, 2.0, 2.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_inplace_identity_lists(): + @Gpu + def copy_x_to_y(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] + + x = [1.5, 2.5, 3.5] + y = [0.0, 0.0, 0.0] + x_id = id(x) + y_id = id(y) + gpu(copy_x_to_y, 3, x, y).join() + assert id(x) == x_id and id(y) == y_id + assert y == x + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_n_less_than_list_len_only_updates_prefix(): + @Gpu + def mark(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 1.0 + + y = [0.0] * 8 + gpu(mark, 2, y).join() + assert y == [1.0, 1.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_three_lists(): + @Gpu + def madd(n: int, a: list[float], b: list[float], c: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + c[i] = a[i] * b[i] + c[i] + + n = 16 + a = [float(i) for i in range(n)] + b = [2.0] * n + c = [1.0] * n + expect = [a[i] * 2.0 + 1.0 for i in range(n)] + gpu(madd, n, a, b, c).join() + assert c == pytest.approx(expect) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_local_temps_and_augassign(): + @Gpu + def scale_add(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + t: float = a * x[i] + t += y[i] + y[i] = t + + x = [1.0, 2.0, 3.0] + y = [10.0, 20.0, 30.0] + gpu(scale_add, 3, 2.0, x, y).join() + assert y == pytest.approx([12.0, 24.0, 36.0]) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_for_range_reduction_into_slot(): + @Gpu + def sum_prefix(n: int, x: list[int], ys: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + s: int = 0 + for j in range(i + 1): + s += x[j] + ys[i] = s + + x = [1, 2, 3, 4] + ys = [0, 0, 0, 0] + gpu(sum_prefix, 4, x, ys).join() + assert ys == [1, 3, 6, 10] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_if_else_branch(): + @Gpu + def abs_copy(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + if x[i] < 0.0: + y[i] = 0.0 - x[i] + else: + y[i] = x[i] + + x = [-2.0, 0.0, 3.5] + y = [0.0, 0.0, 0.0] + gpu(abs_copy, 3, x, y).join() + assert y == pytest.approx([2.0, 0.0, 3.5]) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_threadidx_local_only_first_lane_writes(): + """ThreadIdx.x == 0 within each workgroup; GlobalIdx still unique.""" + + @Gpu + def mark_lane0(n: int, y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + lid: int = ThreadIdx.x + if lid == 0: + y[i] = 1 + else: + y[i] = 0 + + n = 130 + y = [9] * n + gpu(mark_lane0, n, y).join() + # Workgroup size 64: indices 0, 64, 128 are lane 0 in each group. + for i in range(n): + expect = 1 if (i % 64) == 0 else 0 + assert y[i] == expect, f"i={i}" + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_empty_n_no_write(): + @Gpu + def mark(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 1.0 + + y = [0.0, 0.0, 0.0] + gpu(mark, 0, y).join() + assert y == [0.0, 0.0, 0.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_scalar_only_kernel_runs(): + @Gpu + def noop(n: int) -> None: + i: int = GlobalIdx.x + if i >= n: + return + + gpu(noop, 8).join() + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_lists_only_no_scalars(): + @Gpu + def copy_lists(x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= 4: + return + y[i] = x[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [0.0, 0.0, 0.0, 0.0] + gpu(copy_lists, x, y).join() + assert y == x + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_lists_with_dummy_n(): + """Workaround path: always pass `n` so binding 0 exists.""" + + @Gpu + def copy_n(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [0.0, 0.0, 0.0, 0.0] + gpu(copy_n, 4, x, y).join() + assert y == x + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +@pytest.mark.parametrize( + "ops", + [ + ("add", lambda a, b: a + b), + ("sub", lambda a, b: a - b), + ("mul", lambda a, b: a * b), + ], +) +def test_pipeline_binop_variants(ops): + name, py_op = ops + + if name == "add": + + @Gpu + def kern(n: int, x: list[float], y: list[float], z: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + z[i] = x[i] + y[i] + + elif name == "sub": + + @Gpu + def kern(n: int, x: list[float], y: list[float], z: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + z[i] = x[i] - y[i] + + else: + + @Gpu + def kern(n: int, x: list[float], y: list[float], z: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + z[i] = x[i] * y[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [4.0, 3.0, 2.0, 1.0] + z = [0.0] * 4 + gpu(kern, 4, x, y, z).join() + assert z == pytest.approx([py_op(a, b) for a, b in zip(x, y)]) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_compare_chain_guard(): + @Gpu + def clamp_mark(n: int, lo: int, hi: int, y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + if lo <= i < hi: + y[i] = 1 + else: + y[i] = 0 + + y = [9] * 10 + gpu(clamp_mark, 10, 3, 7, y).join() + assert y == [0, 0, 0, 1, 1, 1, 1, 0, 0, 0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_many_launches_same_kernel(): + @Gpu + def add_k(n: int, k: float, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = y[i] + k + + y = [0.0] * 8 + for _ in range(10): + gpu(add_k, 8, 0.5, y).join() + assert y == pytest.approx([5.0] * 8) + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_pipeline_int_and_float_mixed_scalars(): + @Gpu + def mix(n: int, a: float, b: float, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + # i is int; promote via add with float literal (no float() call plugin yet) + t: float = a * b + t += 0.0 + 0.0 # keep float + if i == 0: + y[i] = t + elif i == 1: + y[i] = t + 1.0 + else: + y[i] = t + 2.0 + + y = [0.0, 0.0, 0.0] + gpu(mix, 3, 1.5, 2.0, y).join() + assert y == pytest.approx([3.0, 4.0, 5.0]) diff --git a/tests/unit/test_gpu_prepare.py b/tests/unit/test_gpu_prepare.py new file mode 100644 index 0000000..67fe1bf --- /dev/null +++ b/tests/unit/test_gpu_prepare.py @@ -0,0 +1,270 @@ +"""Unit / live tests for prepare() and gpu().""" + +from __future__ import annotations + +import pytest + +from helpers_gpu import prepare_module + +from cthreads.frontend.Registry import REGISTRY +from cthreads.gpu import GlobalIdx, Gpu, available, gpu, prepare, shutdown +from cthreads.gpu.frontend.errors import GPUNotAvailable +from cthreads.gpu.runtime import GpuJob + + +@pytest.fixture(autouse=True) +def _reset_gpu_prepare_state(): + prepare_mod = prepare_module() + REGISTRY.clear() + prepare_mod._gpu_prepared = False + yield + REGISTRY.clear() + # Isolation: shutdown releases ShaderCache; lib clears `_gpu_prepared`. + try: + shutdown() + except Exception: + pass + prepare_mod._gpu_prepared = False + + +def test_prepare_function_and_runtime_module_coexist(): + """Public API is the function; internals live on `cthreads.gpu.runtime`.""" + import cthreads.gpu.runtime as runtime_mod + from cthreads.gpu import prepare as prepare_fn + + assert callable(prepare_fn) + assert prepare_fn is runtime_mod.prepare + assert hasattr(runtime_mod, "_gpu_prepared") + assert prepare_module() is runtime_mod + + +def test_gpu_rejects_non_gpu_fn(): + def plain(n: int) -> None: + pass + + with pytest.raises(TypeError, match="@Gpu"): + gpu(plain, 1) + + +def test_gpu_rejects_non_callable(): + with pytest.raises(TypeError, match="callable"): + gpu(None) # type: ignore[arg-type] + + +def test_gpu_rejects_kwargs(): + if not available(): + pytest.skip("GPU not available") + + @Gpu + def k(n: int) -> None: + i: int = GlobalIdx.x + if i >= n: + return + + with pytest.raises(TypeError, match="keyword"): + gpu(k, 1, force=False, extra=1) + + +def test_gpu_rejects_arity_mismatch(): + if not available(): + pytest.skip("GPU not available") + + @Gpu + def k(n: int, a: float) -> None: + pass + + with pytest.raises(TypeError, match="expected 2"): + gpu(k, 1) + + +def test_prepare_raises_when_nothing_registered(): + if not available(): + pytest.skip("GPU not available") + with pytest.raises(RuntimeError, match="Nothing registered"): + prepare() + + +def test_prepare_unavailable(monkeypatch): + prepare_mod = prepare_module() + monkeypatch.setattr(prepare_mod._ext_gpu_api, "available", lambda: False) + with pytest.raises(GPUNotAvailable): + prepare() + + +def test_gpu_unavailable(monkeypatch): + prepare_mod = prepare_module() + monkeypatch.setattr(prepare_mod._ext_gpu_api, "available", lambda: False) + + def fake_gpu_fn(n: int) -> None: + pass + + fake_gpu_fn.__gpu__ = True # type: ignore[attr-defined] + with pytest.raises(GPUNotAvailable): + gpu(fake_gpu_fn, 1) + + +def test_gpu_job_result_is_none(): + job = GpuJob( + type( + "R", + (), + { + "start": lambda self: None, + "done": lambda self: True, + "join": lambda self: None, + }, + )() + ) + assert job.result() is None + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_live_prepare_and_gpu_saxpy(): + prepare_mod = prepare_module() + + @Gpu + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + x = [1.0, 2.0, 3.0, 4.0] + y = [10.0, 20.0, 30.0, 40.0] + a = 2.0 + expect = [a * xi + yi for xi, yi in zip(x, y)] + + info = prepare() + assert "rewritten" in info + assert prepare_mod._gpu_prepared is True + + job = gpu(saxpy, len(x), a, x, y) + assert isinstance(job, GpuJob) + job.join() + assert y == expect + assert job.done() + assert job.result() is None + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_live_gpu_auto_prepare(): + @Gpu + def fill(n: int, ys: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + ys[i] = 1.0 + + ys = [0.0, 0.0, 0.0, 0.0] + prepare_module()._gpu_prepared = False + gpu(fill, 4, ys).join() + assert ys == [1.0, 1.0, 1.0, 1.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_lib_reserved_param_name_out(): + """Live path: reserved `out` fails at GLSL compile with a clear hint.""" + + @Gpu + def fill(n: int, out: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + out[i] = 1.0 + + ys = [0.0, 0.0] + with pytest.raises(RuntimeError, match="reserved words|GLSL compile failed"): + gpu(fill, 2, ys).join() + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_live_gpu_second_launch_same_kernel(): + @Gpu + def add1(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = y[i] + 1.0 + + y = [0.0, 0.0, 0.0] + gpu(add1, 3, y).join() + gpu(add1, 3, y).join() + assert y == [2.0, 2.0, 2.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_live_larger_than_one_workgroup(): + @Gpu + def scale(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + + n = 200 + x = [float(i) for i in range(n)] + y = [0.0] * n + gpu(scale, n, 3.0, x, y).join() + assert y == [3.0 * float(i) for i in range(n)] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_live_force_reprepare(): + @Gpu + def k(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 7.0 + + y = [0.0, 0.0] + gpu(k, 2, y).join() + y2 = [0.0, 0.0] + gpu(k, 2, y2, force=True).join() + assert y2 == [7.0, 7.0] + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_compile_sets_prepared_flag(): + from cthreads.gpu import compile + + prepare_mod = prepare_module() + + @Gpu + def k(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 1.0 + + prepare_mod._gpu_prepared = False + info = compile() + assert prepare_mod._gpu_prepared is True + assert "rewritten" in info + + +@pytest.mark.skipif(not available(), reason="Vulkan GPU not available") +def test_shutdown_then_gpu_reregisters(): + """ + shutdown() releases ShaderCache; next gpu() must rewalk registry and work. + """ + prepare_mod = prepare_module() + + @Gpu + def fill(n: int, y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = 3.0 + + y = [0.0, 0.0] + gpu(fill, 2, y).join() + assert y == [3.0, 3.0] + + shutdown() + assert prepare_mod._gpu_prepared is False + + y2 = [0.0, 0.0] + gpu(fill, 2, y2).join() + assert y2 == [3.0, 3.0] + assert prepare_mod._gpu_prepared is True diff --git a/tests/unit/test_gpu_reserved_names.py b/tests/unit/test_gpu_reserved_names.py new file mode 100644 index 0000000..485ac46 --- /dev/null +++ b/tests/unit/test_gpu_reserved_names.py @@ -0,0 +1,67 @@ +"""Unit tests for GLSL reserved-identifier / keyword collisions.""" + +from __future__ import annotations + +import pytest + +from helpers_gpu import glsl_compiler_available + +from cthreads.gpu.compiler.translation.spirv import compile_glsl_to_spirv +from cthreads.gpu.compiler.translation.translate import translate_function_for_gpu + +pytestmark = pytest.mark.skipif( + not glsl_compiler_available(), + reason="no GLSL compiler (skipped on GitHub Actions / CPU-only builds)", +) + + +@pytest.mark.parametrize( + "param", + [ + "out", + "uniform", + "buffer", + "flat", + "smooth", + "shared", + ], +) +def test_reserved_list_param_names_fail_with_hint(param, tmp_module): + """No keyword denylist: glslang fails; we wrap with an identifier hint.""" + mod = tmp_module( + f""" + from cthreads.gpu import GlobalIdx + + def k(n: int, {param}: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + {param}[i] = 1.0 + """, + name=f"reserved_{param}", + ) + with pytest.raises(RuntimeError, match="reserved words|GLSL compile failed"): + translate_function_for_gpu(mod.k, compile_spirv=True) + + +def test_safe_param_names_compile(tmp_module): + mod = tmp_module( + """ + from cthreads.gpu import GlobalIdx + + def k(n: int, ys: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + ys[i] = 1.0 + """, + name="safe_param_names", + ) + r = translate_function_for_gpu(mod.k, compile_spirv=True) + assert r.spirv is not None + assert "} ys;" in r.source + + +def test_compile_error_includes_hint(): + with pytest.raises(RuntimeError, match="Hint:.*reserved"): + compile_glsl_to_spirv("#version 450\nvoid main() { not_a_type x; }\n") diff --git a/tests/unit/test_gpu_shader.py b/tests/unit/test_gpu_shader.py new file mode 100644 index 0000000..0400156 --- /dev/null +++ b/tests/unit/test_gpu_shader.py @@ -0,0 +1,150 @@ +""" +GPU shader / launch tests. + +Substrate smokes stay on `_ext.gpu.testing`. Launch + join use the product +path: `_ext.gpu.launch_gpu_kernel` / `GpuJob` (via `_ext_gpu_api`). +""" + +from __future__ import annotations + +import pytest + +from cthreads import gpu +from cthreads.gpu import _ext_gpu_api +from cthreads.gpu.frontend.errors import GpuInvalidArgument +from cthreads.gpu.gpu_kernel_meta import build_gpu_kernel_meta + + +def _ext_gpu(): + return gpu._ext_gpu_api._gpu + + +def _require_gpu_testing(): + ext = _ext_gpu() + if ext is None: + pytest.skip("cthreads built without CTHREADS_GPU (_ext.gpu missing)") + if not hasattr(ext, "testing"): + pytest.skip( + "_ext.gpu.testing missing (local gpu/testing/ not in tree; " + "optional gitignored helpers)" + ) + if not gpu.available(): + pytest.skip("Vulkan loader/device not available in this environment") + return ext.testing + + +def _map_probe(exc: BaseException): + return gpu._map_error(exc) + + +def test_public_gpu_has_no_shader_smoke_exports(): + assert not hasattr(gpu, "smoke_create_entry") + assert not hasattr(gpu, "smoke_update_descriptors") + assert not hasattr(gpu, "smoke_launch_saxpy") + assert not hasattr(gpu, "register_smoke_saxpy") + assert not hasattr(gpu, "testing") + + +def test_ext_gpu_exports_launch_api(): + ext = _ext_gpu() + if ext is None: + pytest.skip("cthreads built without CTHREADS_GPU (_ext.gpu missing)") + assert hasattr(ext, "launch_gpu_kernel") + assert hasattr(ext, "GpuJob") + + +def test_live_smoke_create_entry(): + testing = _require_gpu_testing() + try: + testing.smoke_create_entry() + finally: + gpu.shutdown() + + +def test_live_smoke_update_descriptors(): + testing = _require_gpu_testing() + try: + testing.smoke_update_descriptors() + finally: + gpu.shutdown() + + +def test_live_smoke_cache_register_and_get(): + testing = _require_gpu_testing() + try: + testing.smoke_cache_register_and_get() + finally: + gpu.shutdown() + + +def test_live_launch_saxpy_product_path(): + """ + Register smoke SPIR-V (test-only), then launch/join via product bindings. + """ + testing = _require_gpu_testing() + if not hasattr(testing, "register_smoke_saxpy"): + pytest.skip("rebuild with register_smoke_saxpy") + if not hasattr(_ext_gpu(), "launch_gpu_kernel"): + pytest.skip("rebuild with product launch_gpu_kernel") + + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + try: + symbol = testing.register_smoke_saxpy() + meta = build_gpu_kernel_meta(saxpy, symbol=symbol).to_dict() + meta["group_count_x"] = 1 + meta["group_count_y"] = 1 + meta["group_count_z"] = 1 + + n = 4 + a = 2.0 + x = [1.0, 2.0, 3.0, 4.0] + y = [10.0, 20.0, 30.0, 40.0] + expect = [a * xi + yi for xi, yi in zip(x, y)] + + job = _ext_gpu_api.launch_gpu_kernel(meta, [n, a, x, y]) + job.join() + assert y == expect + assert job.done() + finally: + if hasattr(testing, "clear_shader_cache"): + try: + testing.clear_shader_cache() + except Exception: + pass + gpu.shutdown() + + +def test_live_probe_cache_duplicate_add(): + testing = _require_gpu_testing() + try: + with pytest.raises(Exception) as ei: + testing.probe_cache_duplicate_add() + mapped = _map_probe(ei.value) + assert isinstance(mapped, GpuInvalidArgument) + assert "already exists" in str(mapped) or "GpuInvalidArgument" in str(mapped) + finally: + gpu.shutdown() + + +def test_live_probe_update_empty_list_slot(): + testing = _require_gpu_testing() + try: + with pytest.raises(Exception) as ei: + testing.probe_update_empty_list_slot() + mapped = _map_probe(ei.value) + assert isinstance(mapped, GpuInvalidArgument) + finally: + gpu.shutdown() + + +def test_live_probe_create_entry_zero_bindings(): + testing = _require_gpu_testing() + try: + with pytest.raises(Exception) as ei: + testing.probe_create_entry_zero_bindings() + mapped = _map_probe(ei.value) + assert isinstance(mapped, GpuInvalidArgument) + finally: + gpu.shutdown() diff --git a/tests/unit/test_gpu_signature.py b/tests/unit/test_gpu_signature.py new file mode 100644 index 0000000..2df9323 --- /dev/null +++ b/tests/unit/test_gpu_signature.py @@ -0,0 +1,214 @@ +"""Unit tests for GpuSignature preamble emission.""" + +from __future__ import annotations + +import pytest + +from cthreads.compiler.translation.Source import Source +from cthreads.gpu.compiler.translation.Signature import GpuSignature +from cthreads.gpu.compiler.translation.context import GpuTranslationContext +from cthreads.gpu.gpu_kernel_meta import build_gpu_kernel_meta +from cthreads.types import PyBool, PyFloat, PyInt, PyList + + +def _sig(fn, local_size_x: int = 64): + ctx = GpuTranslationContext(fn=fn, local_size_x=local_size_x) + result = GpuSignature.translate(Source.parse_function(fn), ctx) + return ctx, result + + +def test_saxpy_preamble_matches_meta_and_smoke_shape(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + pass + + meta = build_gpu_kernel_meta(saxpy) + ctx, sig = _sig(saxpy) + assert sig.binding_count == meta.binding_count == 3 + assert sig.scalar_bytes == meta.scalar_bytes == 8 + assert sig.scalar_fields == [("n", "int"), ("a", "float")] + assert sig.list_fields == [(1, "x", "float"), (2, "y", "float")] + assert "layout(local_size_x = 64)" in sig.preamble + assert "binding = 0" in sig.preamble + assert "int n;" in sig.preamble and "float a;" in sig.preamble + assert "binding = 1" in sig.preamble and "float data[];" in sig.preamble + assert "} x;" in sig.preamble and "} y;" in sig.preamble + assert "n" in ctx.scalar_params and "a" in ctx.scalar_params + assert "x" in ctx.list_params and "y" in ctx.list_params + assert isinstance(ctx.symbols["n"], PyInt) + assert isinstance(ctx.symbols["x"], PyList) + + +def test_lists_only_no_binding_zero_scalars(): + def k(x: list[int], y: list[int]) -> None: + pass + + _, sig = _sig(k) + assert sig.binding_count == 2 + assert sig.scalar_bytes == 0 + assert sig.scalar_fields == [] + assert "binding = 0" in sig.preamble and "binding = 1" in sig.preamble + assert "buffer Scalars" not in sig.preamble + assert sig.list_fields == [(0, "x", "int"), (1, "y", "int")] + + +def test_scalars_only_binding_zero(): + def k(a: int, b: float, c: bool) -> None: + pass + + _, sig = _sig(k) + assert sig.binding_count == 1 + assert sig.scalar_bytes == 12 + assert "bool c;" in sig.preamble + assert "data[]" not in sig.preamble + + +@pytest.mark.parametrize("local", [1, 8, 32, 64, 128, 256]) +def test_custom_local_size(local): + def k(n: int) -> None: + pass + + _, sig = _sig(k, local_size_x=local) + assert f"layout(local_size_x = {local})" in sig.preamble + assert sig.local_size_x == local + + +def test_param_order_preserved_in_scalar_block(): + def k(b: float, a: int) -> None: + pass + + _, sig = _sig(k) + assert sig.scalar_fields == [("b", "float"), ("a", "int")] + pos_b = sig.preamble.index("float b;") + pos_a = sig.preamble.index("int a;") + assert pos_b < pos_a + + +def test_list_binding_order_follows_params(): + def k(n: int, z: list[float], a: list[int], b: list[bool]) -> None: + pass + + _, sig = _sig(k) + assert sig.list_fields == [ + (1, "z", "float"), + (2, "a", "int"), + (3, "b", "bool"), + ] + assert sig.binding_count == 4 + + +def test_missing_annotation_raises(): + def k(n: int, a) -> None: # type: ignore[no-untyped-def] + pass + + with pytest.raises(TypeError, match="type annotation"): + _sig(k) + + +def test_vararg_rejected(): + def k(*args: int) -> None: + pass + + with pytest.raises(TypeError, match=r"\*args"): + _sig(k) + + +def test_kwargs_rejected(): + def k(**kwargs: int) -> None: + pass + + with pytest.raises(TypeError, match=r"\*\*kwargs"): + _sig(k) + + +def test_kwonly_rejected(): + def k(*, n: int) -> None: + pass + + with pytest.raises(TypeError, match="kw-only"): + _sig(k) + + +def test_non_none_return_rejected(): + def k(n: int) -> int: + return n + + with pytest.raises(TypeError, match="return must be None"): + _sig(k) + + +def test_none_return_ok(): + def k(n: int) -> None: + pass + + _, sig = _sig(k) + assert sig.func_name == "k" + + +def test_unsupported_dict_param(): + def k(d: dict[str, int]) -> None: + pass + + with pytest.raises(TypeError): + _sig(k) + + +def test_unsupported_str_param(): + def k(s: str) -> None: + pass + + with pytest.raises(TypeError): + _sig(k) + + +def test_unsupported_nested_list(): + def k(x: list[list[int]]) -> None: + pass + + with pytest.raises(TypeError): + _sig(k) + + +def test_list_block_name_capitalized(): + def k(values: list[float]) -> None: + pass + + _, sig = _sig(k) + assert "buffer Values" in sig.preamble + assert "} values;" in sig.preamble + + +def test_single_list_binding_zero(): + def k(xs: list[int]) -> None: + pass + + _, sig = _sig(k) + assert sig.binding_count == 1 + assert "binding = 0" in sig.preamble + assert "buffer Scalars" not in sig.preamble + + +def test_bool_list_elem(): + def k(flags: list[bool]) -> None: + pass + + _, sig = _sig(k) + assert "bool data[];" in sig.preamble + + +def test_ctx_symbols_typed(): + def k(n: int, flag: bool, a: float, xs: list[int]) -> None: + pass + + ctx, _ = _sig(k) + assert isinstance(ctx.symbols["n"], PyInt) + assert isinstance(ctx.symbols["flag"], PyBool) + assert isinstance(ctx.symbols["a"], PyFloat) + assert isinstance(ctx.symbols["xs"], PyList) + + +def test_std430_mentioned_on_buffers(): + def k(n: int, x: list[float]) -> None: + pass + + _, sig = _sig(k) + assert sig.preamble.count("std430") >= 2 diff --git a/tests/unit/test_gpu_spirv.py b/tests/unit/test_gpu_spirv.py new file mode 100644 index 0000000..ccad9a1 --- /dev/null +++ b/tests/unit/test_gpu_spirv.py @@ -0,0 +1,111 @@ +"""Unit tests for GLSL -> SPIR-V (native glslang or glslc fallback).""" + +from __future__ import annotations + +import pytest + +from helpers_gpu import glsl_compiler_available + +from cthreads.gpu import GlobalIdx +from cthreads.gpu.compiler.translation.spirv import compile_glsl_to_spirv +from cthreads.gpu.compiler.translation.translate import translate_function_for_gpu + +_MIN_COMP = """#version 450 +layout(local_size_x = 64) in; +layout(set = 0, binding = 0, std430) buffer Scalars { int n; } scalars; +void main() { + int i = int(gl_GlobalInvocationID.x); + if (i >= scalars.n) { return; } +} +""" + + +pytestmark = pytest.mark.skipif( + not glsl_compiler_available(), + reason="no GLSL compiler (skipped on GitHub Actions / CPU-only builds)", +) + + +def test_compile_min_comp_magic_and_alignment(): + data = compile_glsl_to_spirv(_MIN_COMP) + assert len(data) >= 20 + assert len(data) % 4 == 0 + assert data[:4] == b"\x03\x02\x23\x07" + + +def test_compile_empty_raises(): + with pytest.raises(Exception): + compile_glsl_to_spirv("") + + +def test_compile_bad_glsl_raises(): + with pytest.raises(Exception): + compile_glsl_to_spirv("#version 450\nvoid main() { not_a_type x; }\n") + + +def test_compile_rejects_uint_to_int_without_cast(): + bad = """#version 450 +layout(local_size_x = 1) in; +void main() { int i = gl_GlobalInvocationID.x; } +""" + with pytest.raises(Exception): + compile_glsl_to_spirv(bad) + + +def test_compile_with_cast_ok(): + ok = """#version 450 +layout(local_size_x = 1) in; +void main() { int i = int(gl_GlobalInvocationID.x); } +""" + data = compile_glsl_to_spirv(ok) + assert data[:4] == b"\x03\x02\x23\x07" + + +@pytest.mark.parametrize("local", [1, 8, 64, 256]) +def test_compile_local_sizes(local): + src = f"""#version 450 +layout(local_size_x = {local}) in; +void main() {{}} +""" + data = compile_glsl_to_spirv(src) + assert len(data) % 4 == 0 + + +def test_translate_compile_spirv_flag(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + r0 = translate_function_for_gpu(saxpy, compile_spirv=False) + assert r0.spirv is None + r1 = translate_function_for_gpu(saxpy, compile_spirv=True) + assert r1.spirv is not None + assert r1.spirv[:4] == b"\x03\x02\x23\x07" + assert "#version 450" in r1.source + assert "void main()" in r1.source + + +def test_translate_compile_many_kernels(): + def add(n: int, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = x[i] + y[i] + + def fill(n: int, y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = i + + for fn in (add, fill): + r = translate_function_for_gpu(fn, compile_spirv=True) + assert r.spirv is not None + assert r.spirv[:4] == b"\x03\x02\x23\x07" + + +def test_compile_missing_version_raises(): + with pytest.raises(Exception): + compile_glsl_to_spirv("void main() {}\n") diff --git a/tests/unit/test_gpu_syntax.py b/tests/unit/test_gpu_syntax.py new file mode 100644 index 0000000..3b7546a --- /dev/null +++ b/tests/unit/test_gpu_syntax.py @@ -0,0 +1,729 @@ +"""Unit tests for GPU Syntax leaf translators and index AttrPlugin.""" + +from __future__ import annotations + +import ast + +import pytest + +from cthreads.compiler.translation.Source import Source +from cthreads.gpu import BlockDim, BlockIdx, GlobalIdx, GridDim, ThreadIdx +from cthreads.gpu.compiler.translation.Signature import GpuSignature +from cthreads.gpu.compiler.translation.context import GpuTranslationContext +from cthreads.gpu.compiler.translation.syntax.Syntax import GpuSyntax +from cthreads.types import PyFloat, PyInt, PyList + + +def _ctx_for(fn): + ctx = GpuTranslationContext(fn=fn) + GpuSignature.translate(Source.parse_function(fn), ctx) + return ctx + + +def _expr(src: str, ctx: GpuTranslationContext) -> str: + tree = ast.parse(src, mode="eval") + assert isinstance(tree, ast.Expression) + return GpuSyntax.expr(tree.body, ctx) + + +def _stmt(src: str, ctx: GpuTranslationContext) -> list[str]: + tree = ast.parse(src) + assert len(tree.body) == 1 + return GpuSyntax.stmt(tree.body[0], ctx) + + +def _inject_indexes(fn): + fn.__globals__.update( + { + "GlobalIdx": GlobalIdx, + "ThreadIdx": ThreadIdx, + "BlockIdx": BlockIdx, + "BlockDim": BlockDim, + "GridDim": GridDim, + } + ) + + +# --- Name ------------------------------------------------------------------- + + +def test_name_scalar_rewrites_to_scalars_block(): + def k(n: int, a: float, x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + assert _expr("n", ctx) == "scalars.n" + assert _expr("a", ctx) == "scalars.a" + + +def test_name_list_stays_bare(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + assert _expr("x", ctx) == "x" + + +def test_name_local_bare(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + assert _expr("i", ctx) == "i" + + +def test_name_unknown_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unknown name"): + _expr("missing", ctx) + + +def test_name_self_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="self"): + _expr("self", ctx) + + +# --- Index ------------------------------------------------------------------ + + +def test_index_list_to_data(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + assert _expr("x[i]", ctx) == "(x.data[i])" + + +def test_index_nested_expr_index(): + def k(x: list[float], n: int) -> None: + pass + + ctx = _ctx_for(k) + assert _expr("x[n]", ctx) == "(x.data[scalars.n])" + + +def test_index_rejects_slice(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="slice"): + _expr("x[1:2]", ctx) + + +def test_index_rejects_non_list(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="list parameters"): + _expr("n[0]", ctx) + + +def test_index_literal(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + assert _expr("x[0]", ctx) == "(x.data[0])" + + +# --- Op --------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "src, needle", + [ + ("a + b", "+"), + ("a - b", "-"), + ("a * b", "*"), + ("a / b", "/"), + ("a // b", "/"), + ("a % b", "%"), + ("a << b", "<<"), + ("a >> b", ">>"), + ("a | b", "|"), + ("a ^ b", "^"), + ("a & b", "&"), + ], +) +def test_binop_table(src, needle): + def k(a: int, b: int) -> None: + pass + + ctx = _ctx_for(k) + out = _expr(src, ctx) + assert needle in out + assert "scalars.a" in out and "scalars.b" in out + + +def test_binop_mul_add(): + def k(a: float, x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + out = _expr("a * x[i] + x[i]", ctx) + assert "scalars.a" in out + assert "x.data[i]" in out + assert "*" in out and "+" in out + + +@pytest.mark.parametrize( + "src, op", + [ + ("a == b", "=="), + ("a != b", "!="), + ("a < b", "<"), + ("a <= b", "<="), + ("a > b", ">"), + ("a >= b", ">="), + ], +) +def test_compare_ops(src, op): + def k(a: int, b: int) -> None: + pass + + ctx = _ctx_for(k) + assert op in _expr(src, ctx) + + +def test_compare_and_bool(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + out = _expr("i >= n and True", ctx) + assert ">=" in out and "&&" in out + assert "scalars.n" in out + + +def test_bool_or(): + def k(a: bool, b: bool) -> None: + pass + + ctx = _ctx_for(k) + assert "||" in _expr("a or b", ctx) + + +def test_unary_not_neg(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + assert "!" in _expr("not n", ctx) + assert "-" in _expr("-n", ctx) + assert "+" in _expr("+n", ctx) + assert "~" in _expr("~n", ctx) + + +def test_pow_rejected(): + def k(a: float) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match=r"\*\*|pow"): + _expr("a ** 2", ctx) + + +def test_chained_compare(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + out = _expr("0 <= i < n", ctx) + assert "&&" in out + + +def test_unsupported_call_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unsupported call"): + _expr("abs(n)", ctx) + + +def test_float_call_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unsupported call"): + _expr("float(n)", ctx) + + +# --- Assign ----------------------------------------------------------------- + + +def test_assign_list_element(): + def k(a: float, x: list[float], y: list[float]) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + lines = _stmt("y[i] = a * x[i] + y[i]", ctx) + assert len(lines) == 1 + assert "y.data[i]" in lines[0] + assert "scalars.a" in lines[0] + + +def test_ann_assign_local(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("i: int = n", ctx) + assert lines == [" int i = scalars.n;"] + assert isinstance(ctx.symbols["i"], PyInt) + assert "i" not in ctx.scalar_params + + +def test_ann_assign_from_global_idx(): + def k(n: int) -> None: + pass + + _inject_indexes(k) + ctx = _ctx_for(k) + lines = _stmt("i: int = GlobalIdx.x", ctx) + assert "int i =" in lines[0] + assert "gl_GlobalInvocationID.x" in lines[0] + + +def test_ann_assign_uninitialized(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + # Python allows `i: int` without value via AnnAssign value=None in AST — + # constructing via exec-style source needs a value; use explicit None-ish + # by building AST. + tree = ast.parse("i: int") + assert isinstance(tree.body[0], ast.AnnAssign) + lines = GpuSyntax.stmt(tree.body[0], ctx) + assert lines == [" int i;"] + + +def test_ann_assign_redeclare_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + _stmt("i: int = 0", ctx) + with pytest.raises(TypeError, match="redeclaration"): + _stmt("i: int = 1", ctx) + + +def test_assign_unknown_name_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unknown name"): + _stmt("i = n", ctx) + + +def test_assign_bare_list_rejected(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="list parameter"): + _stmt("x = x", ctx) + + +def test_assign_multi_target_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + _stmt("i: int = 0", ctx) + _stmt("j: int = 0", ctx) + with pytest.raises(TypeError, match="single-target"): + _stmt("i = j = n", ctx) + + +def test_aug_assign(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + _stmt("i: int = 0", ctx) + lines = _stmt("i += n", ctx) + assert lines == [" i += scalars.n;"] + + +@pytest.mark.parametrize( + "src, op", + [ + ("i += n", "+="), + ("i -= n", "-="), + ("i *= n", "*="), + ("i %= n", "%="), + ("i &= n", "&="), + ("i |= n", "|="), + ("i ^= n", "^="), + ("i <<= n", "<<="), + ("i >>= n", ">>="), + ], +) +def test_aug_assign_ops(src, op): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + _stmt("i: int = 0", ctx) + assert op in _stmt(src, ctx)[0] + + +def test_aug_assign_pow_rejected(): + def k(a: float) -> None: + pass + + ctx = _ctx_for(k) + _stmt("t: float = a", ctx) + with pytest.raises(TypeError, match=r"\*\*|pow"): + _stmt("t **= 2", ctx) + + +def test_aug_assign_list_elem(): + def k(y: list[float]) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + lines = _stmt("y[i] += 1.0", ctx) + assert "y.data[i]" in lines[0] + assert "+=" in lines[0] + + +def test_ann_assign_local_list_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="local list"): + _stmt("xs: list[float] = []", ctx) + + +def test_slice_assign_rejected(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="slice"): + _stmt("x[1:2] = x[0:1]", ctx) + + +# --- Flow ------------------------------------------------------------------- + + +def test_if_return(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + ctx.symbols["i"] = PyInt() + lines = _stmt("if i >= n:\n return", ctx) + assert lines[0].startswith(" if ") + assert any("return;" in L for L in lines) + + +def test_return_always_bare(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + assert _stmt("return", ctx) == [" return;"] + assert _stmt("return 1", ctx) == [" return;"] + + +def test_pass_break_continue(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + assert _stmt("pass", ctx) == [] + assert _stmt("break", ctx) == [" break;"] + assert _stmt("continue", ctx) == [" continue;"] + + +@pytest.mark.parametrize( + "src, needles", + [ + ("for i in range(n):\n pass", ["for (int i = 0;", "scalars.n", "i += 1"]), + ( + "for i in range(1, n):\n pass", + ["for (int i = 1;", "scalars.n", "i += 1"], + ), + ( + "for i in range(0, n, 2):\n pass", + ["for (int i = 0;", "scalars.n", "i += 2"], + ), + ], +) +def test_for_range_forms(src, needles): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt(src, ctx) + joined = "\n".join(lines) + for needle in needles: + assert needle in joined + assert "i" not in ctx.symbols + + +def test_for_range(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("for i in range(n):\n pass", ctx) + assert "for (int i = 0;" in lines[0] + assert "scalars.n" in lines[0] + assert "i" not in ctx.symbols + + +def test_for_list_rejected(): + def k(x: list[float]) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="range"): + _stmt("for v in x:\n pass", ctx) + + +def test_for_else_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="for/else"): + _stmt("for i in range(n):\n pass\nelse:\n pass", ctx) + + +def test_while_else_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="while/else"): + _stmt("while n > 0:\n break\nelse:\n pass", ctx) + + +def test_for_rebind_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + _stmt("i: int = 0", ctx) + with pytest.raises(TypeError, match="rebinds"): + _stmt("for i in range(n):\n pass", ctx) + + +def test_for_range_zero_args_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="range"): + _stmt("for i in range():\n pass", ctx) + + +def test_for_range_four_args_rejected(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="range"): + _stmt("for i in range(0, n, 1, 2):\n pass", ctx) + + +def test_while(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("while n > 0:\n break", ctx) + assert lines[0].startswith(" while ") + + +def test_if_else(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("if n > 0:\n pass\nelse:\n return", ctx) + assert any("else" in L for L in lines) + + +def test_if_elif_lowers_as_nested_else(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt( + "if n > 0:\n return\nelif n < 0:\n return\nelse:\n pass", + ctx, + ) + joined = "\n".join(lines) + assert "if (" in joined + assert "else" in joined + + +def test_docstring_expr_ignored(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + assert _stmt('"doc"', ctx) == [] + + +def test_unsupported_expr_stmt_comment(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("n", ctx) + assert lines[0].startswith(" // unsupported") + + +def test_sync_threads_expr_stmt_lowers_to_barrier(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("__sync_threads()", ctx) + assert any("barrier()" in line for line in lines) + assert any("memoryBarrierShared()" in line for line in lines) + + +def test_barrier_arrive_and_wait_expr_stmt_same_barrier(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt("Barrier.arrive_and_wait()", ctx) + assert any("barrier()" in line for line in lines) + + +def test_barrier_constructor_rejected_in_gpu(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="Barrier\\(\\.\\.\\.\\) construction"): + _stmt("Barrier(4)", ctx) + + +def test_unsupported_stmt_comment(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + tree = ast.parse("raise RuntimeError()") + lines = GpuSyntax.stmt(tree.body[0], ctx) + assert "unsupported statement" in lines[0] + + +# --- Index builtins / AttrPlugin -------------------------------------------- + + +@pytest.mark.parametrize( + "expr, glsl", + [ + ("GlobalIdx.x", "int(gl_GlobalInvocationID.x)"), + ("GlobalIdx.y", "int(gl_GlobalInvocationID.y)"), + ("GlobalIdx.z", "int(gl_GlobalInvocationID.z)"), + ("ThreadIdx.x", "int(gl_LocalInvocationID.x)"), + ("ThreadIdx.y", "int(gl_LocalInvocationID.y)"), + ("ThreadIdx.z", "int(gl_LocalInvocationID.z)"), + ("BlockIdx.x", "int(gl_WorkGroupID.x)"), + ("BlockIdx.y", "int(gl_WorkGroupID.y)"), + ("BlockIdx.z", "int(gl_WorkGroupID.z)"), + ("BlockDim.x", "int(gl_WorkGroupSize.x)"), + ("BlockDim.y", "int(gl_WorkGroupSize.y)"), + ("BlockDim.z", "int(gl_WorkGroupSize.z)"), + ("GridDim.x", "int(gl_NumWorkGroups.x)"), + ("GridDim.y", "int(gl_NumWorkGroups.y)"), + ("GridDim.z", "int(gl_NumWorkGroups.z)"), + ], +) +def test_index_builtins(expr, glsl): + def k(n: int) -> None: + pass + + _inject_indexes(k) + ctx = _ctx_for(k) + assert _expr(expr, ctx) == glsl + + +def test_module_qualified_global_idx(): + import cthreads.gpu as gpu_mod + + def k(n: int) -> None: + pass + + k.__globals__["gpu"] = gpu_mod + ctx = _ctx_for(k) + assert _expr("gpu.GlobalIdx.x", ctx) == "int(gl_GlobalInvocationID.x)" + + +def test_unknown_attr_raises(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unsupported attribute"): + _expr("n.imag", ctx) + + +def test_index_builtin_bad_axis_raises(): + def k(n: int) -> None: + pass + + _inject_indexes(k) + ctx = _ctx_for(k) + with pytest.raises(TypeError, match="unsupported attribute"): + _expr("GlobalIdx.w", ctx) + + +def test_literal_constants(): + def k(n: int) -> None: + pass + + ctx = _ctx_for(k) + assert _expr("2", ctx) == "2" + assert _expr("True", ctx) == "true" + assert _expr("False", ctx) == "false" + assert _expr("1.5", ctx) == "1.5" + assert _expr("0.0", ctx) == "0.0" + + +def test_nested_if_in_for(): + def k(n: int, y: list[int]) -> None: + pass + + ctx = _ctx_for(k) + lines = _stmt( + "for i in range(n):\n" + " if i > 0:\n" + " y[i] = i\n", + ctx, + ) + joined = "\n".join(lines) + assert "for (int i = 0;" in joined + assert "if (" in joined + assert "y.data[i]" in joined diff --git a/tests/unit/test_gpu_translate.py b/tests/unit/test_gpu_translate.py new file mode 100644 index 0000000..99a735f --- /dev/null +++ b/tests/unit/test_gpu_translate.py @@ -0,0 +1,135 @@ +"""Unit tests for translate_function_for_gpu (no device required).""" + +from __future__ import annotations + +import pytest + +from cthreads.gpu import BlockIdx, GlobalIdx, ThreadIdx +from cthreads.gpu.compiler.translation.translate import translate_function_for_gpu +from cthreads.gpu.gpu_kernel_meta import build_gpu_kernel_meta + + +def test_translate_saxpy_source_shape(): + def saxpy(n: int, a: float, x: list[float], y: list[float]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + y[i] = a * x[i] + y[i] + + r = translate_function_for_gpu(saxpy) + meta = build_gpu_kernel_meta(saxpy) + assert r.func_name == "saxpy" + assert r.binding_count == meta.binding_count + assert r.scalar_bytes == meta.scalar_bytes + assert r.source.startswith("#version 450") + assert "gl_GlobalInvocationID.x" in r.source + assert "y.data[" in r.source + assert r.spirv is None + + +@pytest.mark.parametrize("local", [1, 32, 64, 128]) +def test_translate_custom_local_size(local): + def k(n: int) -> None: + i: int = GlobalIdx.x + if i >= n: + return + + r = translate_function_for_gpu(k, local_size_x=local) + assert f"layout(local_size_x = {local})" in r.source + assert r.local_size_x == local + + +def test_translate_rejects_bad_signature(): + def bad(x: dict[str, int]) -> None: + pass + + with pytest.raises(TypeError): + translate_function_for_gpu(bad) + + +def test_translate_pass_only_body(): + def k(n: int) -> None: + pass + + r = translate_function_for_gpu(k) + assert "void main()" in r.source + assert "binding = 0" in r.source + + +def test_translate_lists_only(): + def k(x: list[int], y: list[int]) -> None: + i: int = GlobalIdx.x + y[i] = x[i] + + r = translate_function_for_gpu(k) + assert "buffer Scalars" not in r.source + assert "binding = 0" in r.source + assert "binding = 1" in r.source + assert r.scalar_bytes == 0 + assert r.binding_count == 2 + + +def test_translate_for_range_and_while(): + def k(n: int, y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + s: int = 0 + for j in range(n): + s += j + while s > 0: + s -= 1 + break + y[i] = s + + r = translate_function_for_gpu(k) + assert "for (int j =" in r.source + assert "while (" in r.source + + +def test_translate_thread_and_block_idx(): + def k(n: int, y: list[int]) -> None: + i: int = GlobalIdx.x + t: int = ThreadIdx.x + b: int = BlockIdx.x + if i >= n: + return + y[i] = t + b + + r = translate_function_for_gpu(k) + assert "gl_LocalInvocationID.x" in r.source + assert "gl_WorkGroupID.x" in r.source + + +def test_translate_bool_and_or(): + def k(n: int, flag: bool, y: list[int]) -> None: + i: int = GlobalIdx.x + if i >= n: + return + if flag and i > 0 or i < 0: + y[i] = 1 + else: + y[i] = 0 + + r = translate_function_for_gpu(k) + assert "&&" in r.source and "||" in r.source + + +def test_translate_pow_rejected(): + def k(a: float, y: list[float]) -> None: + i: int = GlobalIdx.x + y[i] = a ** 2.0 + + with pytest.raises(TypeError, match=r"\*\*|pow"): + translate_function_for_gpu(k) + + +def test_translate_result_fields(): + def k(n: int) -> None: + pass + + r = translate_function_for_gpu(k) + assert r.func_name == "k" + assert isinstance(r.source, str) + assert r.local_size_x == 64 + assert r.spirv is None