Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 17 additions & 4 deletions .github/workflows/cpu.yml
Original file line number Diff line number Diff line change
Expand Up @@ -8,18 +8,27 @@ on:

jobs:
cpu:
runs-on: ubuntu-24.04
strategy:
fail-fast: false
matrix:
os: [ubuntu-24.04, macos-26]
runs-on: ${{ matrix.os }}
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: '3.11'
- name: Install dependencies
- name: Install Linux dependencies
if: runner.os == 'Linux'
run: |
sudo apt-get update
sudo apt-get install -y cmake ninja-build pkg-config liburing-dev libzstd-dev libblosc-dev
python -m pip install uv pytest pytest-cov numpy
- name: Install macOS dependencies
if: runner.os == 'macOS'
run: brew install cmake ninja pkg-config zstd c-blosc
- name: Install Python dependencies
run: python -m pip install uv pytest pytest-cov numpy
- name: Build without CUDA
run: |
cmake --preset cpu -DDAMACY_PYTHON=ON -DPython_EXECUTABLE="$(command -v python)" -DCMAKE_DISABLE_FIND_PACKAGE_CUDAToolkit=ON
Expand All @@ -32,9 +41,13 @@ jobs:
run: |
python - <<'PY'
import subprocess
import sys
from damacy import _native
assert not _native.CUDA_ENABLED
dependencies = subprocess.check_output(['ldd', _native.__file__], text=True).lower()
command = ['otool', '-L'] if sys.platform == 'darwin' else ['ldd']
dependencies = subprocess.check_output([*command, _native.__file__], text=True).lower()
if sys.platform == 'darwin':
assert 'liburing' not in dependencies
print(dependencies)
assert all(name not in dependencies for name in ('libcuda', 'libcudart', 'libnvcomp'))
PY
15 changes: 13 additions & 2 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -15,14 +15,25 @@ include(Warnings)
include(Helpers)
include(Fuzz) # declares DAMACY_FUZZ and applies fuzz-mode flags

option(DAMACY_CUDA "Build the CUDA executor" ON)
# CUDA is unavailable on current macOS toolchains; keep Linux's default.
if(APPLE)
set(DAMACY_CUDA_DEFAULT OFF)
else()
set(DAMACY_CUDA_DEFAULT ON)
endif()
option(DAMACY_CUDA "Build the CUDA executor" ${DAMACY_CUDA_DEFAULT})
if(DAMACY_FUZZ)
set(DAMACY_CUDA OFF)
endif()
if(APPLE AND DAMACY_CUDA)
message(FATAL_ERROR "macOS supports the CPU executor; set DAMACY_CUDA=OFF")
endif()

find_package(Threads REQUIRED)
find_package(PkgConfig REQUIRED)
pkg_check_modules(LIBURING REQUIRED IMPORTED_TARGET liburing)
if(NOT APPLE)
pkg_check_modules(LIBURING REQUIRED IMPORTED_TARGET liburing)
endif()

if(DAMACY_CUDA)
if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
Expand Down
17 changes: 11 additions & 6 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,8 @@ while allowing the pool to reuse its storage.

See [Pipeline composition](docs/pipeline.md) for the complete component,
limit, ownership, and build contracts. Build a CPU Python package from source
with the Linux dependencies `liburing`, `libzstd`, and `libblosc` installed:
with `libzstd` and `libblosc` installed (plus `liburing` on Linux). On macOS,
install dependencies with `brew install cmake ninja pkg-config zstd c-blosc`:

```sh
pip install . --config-settings=cmake.define.DAMACY_CUDA=OFF
Expand Down Expand Up @@ -161,23 +162,27 @@ If you have data that uses one of the unsupported codecs and you'd like it added

## Runtime dependencies

All builds require Linux async metadata I/O and CPU codec libraries. CUDA
builds additionally link the NVIDIA driver and nvCOMP. Build with
CPU builds support Linux and macOS and require the CPU codec libraries.
Linux uses io_uring for async metadata I/O; macOS uses a POSIX worker pool.
CUDA builds additionally link the NVIDIA driver and nvCOMP. Build with
`DAMACY_CUDA=OFF` to import and run on a host without a CUDA driver.

| Library | Used by | How it is loaded |
|---|---|---|
| `liburing` | CPU and CUDA: async metadata I/O | normal dynamic loader |
| `liburing` | Linux async metadata I/O | normal dynamic loader |
| `libzstd`, `libblosc` | CPU decoding, included in both builds | normal dynamic loader |
| `libcuda.so.1`, nvCOMP | CUDA builds | driver loader; nvCOMP may be linked statically |
| `libnuma.so.1` | Optional CUDA placement and host affinity | `dlopen`; absence disables placement |
| `libcufile.so.0` | Optional CUDA GPUDirect Storage | `dlopen`; requires `DAMACY_ENABLE_GDS=ON` |
| `libmount.so.1`, `libudev.so.1` | cuFile initialization when GDS is used | dynamic loader |

Metadata reads require a Linux kernel with the io_uring operations damacy uses:
On Linux, metadata reads require a kernel with the io_uring operations damacy uses:
`STATX`, `OPENAT2`, `READ`, and `CLOSE`. If the kernel does not advertise
those operations, pipeline construction fails instead of falling back to a thread
pool.
pool. On macOS, `metadata_io_concurrency` sets the number of metadata workers;
bulk reads use the shared POSIX file backend. NUMA placement and CPU affinity
are unavailable. CUDA defaults off on macOS and cannot be enabled there.
See [native build instructions](docs/pipeline.md#build-without-cuda).

GDS notes:

Expand Down
21 changes: 17 additions & 4 deletions docs/pipeline.md
Original file line number Diff line number Diff line change
Expand Up @@ -180,6 +180,14 @@ On Linux, install C/C++ build tools, CMake, Ninja, pkg-config, liburing, zstd,
and C-Blosc development packages. For example, Ubuntu packages are
`build-essential cmake ninja-build pkg-config python3-dev liburing-dev libzstd-dev libblosc-dev`.

On macOS, install the native dependencies with Homebrew:

```sh
brew install cmake ninja pkg-config zstd c-blosc uv
```

Then build and test on either platform:

```sh
cmake --preset cpu
cmake --build build
Expand All @@ -196,10 +204,15 @@ pip install . --config-settings=cmake.define.DAMACY_CUDA=OFF
```

The extension has no CUDA or nvCOMP dependency in this configuration.
`CudaExecutor` reports that CUDA support was not built. CPU builds retain the
Linux io_uring metadata requirements. CUDA support is enabled by default;
turning it on builds both executors and requires the CUDA toolkit, nvCOMP, and
a runtime NVIDIA driver. GDS requires a CUDA build.
`CudaExecutor` reports that CUDA support was not built. Linux retains its
io_uring metadata requirements. macOS uses a POSIX metadata worker pool,
with one worker per `metadata_io_concurrency`, and shared POSIX bulk reads.
NUMA placement and CPU affinity are unavailable. CUDA defaults off on macOS
and cannot be enabled. On either platform, `FileReader(workers=...)` and the
legacy `Config.n_io_threads` cannot exceed the online CPU count; larger values
raise `InvalidArgument`.
On Linux CUDA defaults on, builds both executors, and requires the CUDA toolkit,
nvCOMP, and a runtime NVIDIA driver. GDS requires a CUDA build.

## Future queries

Expand Down
7 changes: 6 additions & 1 deletion docs/prefetch.md
Original file line number Diff line number Diff line change
Expand Up @@ -155,7 +155,12 @@ sets the metadata request-concurrency budget; the ring allocates enough
entries internally for multi-step requests and driver wakeups. At startup the
driver requires kernel support for `IORING_OP_STATX`, `IORING_OP_OPENAT2`,
`IORING_OP_READ`, and `IORING_OP_CLOSE`; there is no thread-pool fallback in
the current build.
the Linux build.

On macOS, CMake selects a POSIX worker-pool backend instead of io_uring.
`metadata_io_concurrency` is the worker count; each worker completes one
stat/open/read/close request at a time. Shutdown drains accepted requests.
Operation histograms measure syscall duration on macOS, excluding queue wait.

The default metadata concurrency is 32. Treat much deeper values as storage
tuning: they are useful when metadata operations have real latency, but each
Expand Down
6 changes: 4 additions & 2 deletions docs/troubleshooting.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,12 +17,14 @@ the calling thread yet. Two fixes:

## Pipeline construction fails during metadata I/O setup

The CPU and CUDA metadata paths use io_uring for Zarr metadata and shard indexes. At construction damacy requires kernel support for
On Linux, the CPU and CUDA metadata paths use io_uring for Zarr metadata and shard indexes. At construction damacy requires kernel support for
`IORING_OP_STATX`, `IORING_OP_OPENAT2`, `IORING_OP_READ`, and
`IORING_OP_CLOSE`. If ring creation or the operation probe fails,
pipeline construction fails without a thread-pool fallback. Check the native log for the exact io_uring failure.

On supported kernels, an unusually high `metadata_io_concurrency` can also
macOS uses POSIX metadata worker threads and has no io_uring dependency.
An unusually high `metadata_io_concurrency` can exhaust thread resources on
macOS. On either platform it can also
stress process file-descriptor limits because each in-flight metadata read can
hold an open fd. The default is 64; for much deeper settings, check
`ulimit -n` and remember to multiply by ranks per node.
Expand Down
3 changes: 2 additions & 1 deletion python/damacy/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -501,7 +501,8 @@ class Config:
dtype: Destination dtype for assembled batches.
lookahead_samples: User-side push-queue depth in samples. Defaults
to two full output batches.
n_io_threads: Bulk data IO worker threads (>= 1). Defaults to 64.
n_io_threads: Bulk data IO worker threads, from 1 to the number of
online CPUs. Defaults to 64, so hosts with fewer CPUs must lower it.
metadata_io_concurrency: Async metadata request concurrency (>= 1).
n_array_meta_cache: LRU cap for zarr-metadata entries. Must be
``>= lookahead_samples + 2 * samples_per_batch`` so the in-flight
Expand Down
16 changes: 13 additions & 3 deletions src/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,8 @@ add_src_lib(lru SOURCES util/lru.c util/lru.h LINKS log)
add_src_lib(pool SOURCES util/pool.c util/pool.h LINKS log platform)
add_src_lib(path_intern SOURCES util/path_intern.c util/path_intern.h LINKS hash)

# Platform abstraction (posix-only for now; chucky-style platform-split lives here).
# Platform abstraction: POSIX sources on Linux and macOS. macOS has no NUMA
# or CPU affinity controls, so it builds the numa.darwin.c stub instead.
# platform.posix.c uses dlopen for platform_dl*; numa.posix.c uses libnuma via
# platform_dl* so a single binary works on hosts with or without libnuma installed.
add_src_lib(platform
Expand Down Expand Up @@ -188,10 +189,19 @@ else()
target_sources(store PRIVATE store/store_fs_gds.h store/store_fs_gds_stub.c)
endif()

if(APPLE)
set(METADATA_STORE_IMPL store/metadata_store_async.posix.c)
else()
set(METADATA_STORE_IMPL store/metadata_store_async.c)
endif()
add_src_lib(metadata_store_async
SOURCES store/metadata_store_async.h store/metadata_store_async.c
LINKS store log platform PkgConfig::LIBURING
SOURCES store/metadata_store_async.h store/metadata_store_async_common.h
store/metadata_store_async_common.c ${METADATA_STORE_IMPL}
LINKS store log platform
)
if(NOT APPLE)
target_link_libraries(metadata_store_async PUBLIC PkgConfig::LIBURING)
endif()
if(TARGET numa)
target_link_libraries(metadata_store_async PUBLIC numa)
target_compile_definitions(
Expand Down
20 changes: 11 additions & 9 deletions src/damacy.h
Original file line number Diff line number Diff line change
Expand Up @@ -108,14 +108,15 @@ extern "C"
// Required: must be in [1, DAMACY_HARD_MAX_SUBSTREAMS_PER_CHUNK].
uint32_t max_substreams_per_chunk;

// Bulk chunk-read worker threads. Wave IO uses this queue. Blocking
// IO workers may exceed the CPU count. Required: must be in
// [1, DAMACY_MAX_IO_THREADS].
// Bulk chunk-read worker threads. Wave IO uses this queue. Required:
// must be in [1, DAMACY_MAX_IO_THREADS] and no larger than the host's
// online CPU count; creation fails with DAMACY_INVAL otherwise.
uint32_t n_io_threads;
// Metadata request concurrency for array metadata, shard indexes, and
// chunk-layout probes. The Linux metadata path uses this as an io_uring
// request-depth budget, not as a host thread count. Required: must be > 0
// and no larger than DAMACY_MAX_METADATA_IO_CONCURRENCY.
// request-depth budget; macOS uses this many metadata worker threads.
// Required: must be > 0 and no larger than
// DAMACY_MAX_METADATA_IO_CONCURRENCY.
uint32_t metadata_io_concurrency;

uint32_t n_array_meta_cache;
Expand Down Expand Up @@ -341,10 +342,11 @@ extern "C"
uint64_t read_active;
uint64_t read_max_active;
} metadata_backend;
// Real, measured submit->completion latency of the io_uring metadata ops,
// broken out by op kind (statx/open/read/close). Distinct from the injected
// synthetic latency in metadata_latency above. Buckets are log2-scale on ns
// (bucket i: floor(log2(ns)) == i); percentiles are derived from them.
// Measured metadata-operation latency: submit-to-completion on Linux,
// syscall duration on macOS, by kind (stat/open/read/close). Distinct from
// the injected synthetic latency in metadata_latency above. Buckets are
// log2-scale on ns (bucket i: floor(log2(ns)) == i); percentiles are
// derived from them.
struct
{
uint64_t count;
Expand Down
9 changes: 9 additions & 0 deletions src/pipeline/components.c
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,15 @@ damacy_file_reader_create(uint32_t workers,
*out = NULL;
if (!workers || workers > DAMACY_MAX_IO_THREADS || !max_inflight_reads)
return DAMACY_INVAL;
int cpus = platform_default_thread_count();
if (workers > (uint32_t)cpus) {
log_error("file reader workers (n_io_threads)=%u exceeds the %d online "
"CPUs; set it to at most %d",
workers,
cpus,
cpus);
return DAMACY_INVAL;
}
struct damacy_reader* reader = calloc(1, sizeof(*reader));
if (!reader)
return DAMACY_OOM;
Expand Down
61 changes: 61 additions & 0 deletions src/platform/numa.darwin.c
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
// macOS does not expose Linux NUMA-node or CPU-mask affinity controls.
#include "platform/numa.h"
#include <stddef.h>
#include <string.h>

int
platform_numa_available(void)
{
return 0;
}
int
platform_numa_max_node(void)
{
return -1;
}
int
platform_numa_node_cpu_mask(int node, struct platform_cpu_mask* out)
{
(void)node;
if (out)
memset(out, 0, sizeof(*out));
return 1;
}
int
platform_thread_affinity_get(struct platform_cpu_mask* out)
{
if (out)
memset(out, 0, sizeof(*out));
return 1;
}
int
platform_thread_affinity_set(const struct platform_cpu_mask* mask)
{
(void)mask;
return 1;
}
int
platform_cpu_mask_is_empty(const struct platform_cpu_mask* mask)
{
if (!mask)
return 1;
for (size_t i = 0; i < sizeof(mask->bytes); ++i)
if (mask->bytes[i])
return 0;
return 1;
}
int
platform_cpu_mask_describe(const struct platform_cpu_mask* mask,
int* first,
int* last,
int* count)
{
(void)mask;
if (first)
*first = -1;
if (last)
*last = -1;
if (count)
*count = 0;
return 1;
}
1 change: 0 additions & 1 deletion src/platform/platform.h
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,6 @@ extern "C"
size_t platform_page_alignment(void);
void* platform_aligned_alloc(size_t alignment, size_t size);
void platform_aligned_free(void* ptr);
size_t platform_available_memory(void);

struct platform_clock
{
Expand Down
10 changes: 0 additions & 10 deletions src/platform/platform.posix.c
Original file line number Diff line number Diff line change
Expand Up @@ -22,16 +22,6 @@ platform_page_alignment(void)
return ps > 0 ? ps : 4096;
}

size_t
platform_available_memory(void)
{
long pages = sysconf(_SC_AVPHYS_PAGES);
long page_sz = sysconf(_SC_PAGESIZE);
if (pages > 0 && page_sz > 0)
return (size_t)pages * (size_t)page_sz;
return 0;
}

void*
platform_aligned_alloc(size_t alignment, size_t size)
{
Expand Down
Loading
Loading