Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 52 additions & 0 deletions .github/workflows/classify-scalar-sync.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
name: Sync classify_scalar
on:
workflow_dispatch:
inputs:
upstream_ref:
description: 'Reviewed upstream tag or commit to vendor'
required: true
type: string
permissions:
contents: write
pull-requests: write
concurrency:
group: classify-scalar-sync
cancel-in-progress: false
jobs:
sync:
runs-on: ubuntu-latest
steps:
# An optional bot token allows PR checks to start without approval.
- uses: actions/checkout@v5
with:
ref: master
token: ${{ secrets.VENDOR_SYNC_TOKEN || github.token }}
- name: Import and verify upstream header
env:
UPSTREAM_REF: ${{ inputs.upstream_ref }}
run: |
python3 tools/sync_classify_scalar.py --ref "$UPSTREAM_REF"
python3 tools/sync_classify_scalar.py --check
- name: Open update PR
env:
GH_TOKEN: ${{ secrets.VENDOR_SYNC_TOKEN || github.token }}
run: |
if git diff --quiet; then
echo 'Already synced.'
exit 0
fi
version=$(python3 -c 'import json; print(json.load(open("include/external/classify_scalar.json"))["version"])')
revision=$(python3 -c 'import json; print(json.load(open("include/external/classify_scalar.json"))["commit"])')
branch="codex/sync-classify-scalar-${revision:0:12}"
if [ -n "$(git ls-remote --heads origin "$branch")" ]; then
echo "An update branch already exists: $branch"
exit 0
fi
git switch -c "$branch"
git config user.name 'classify-scalar-sync[bot]'
git config user.email 'classify-scalar-sync[bot]@users.noreply.github.com'
git add include/external/classify_scalar.hpp include/external/classify_scalar.json
git commit -m "Sync classify_scalar $version"
git push --set-upstream origin "$branch"
printf 'Import the unchanged upstream header from https://github.com/vincentlaucsb/classify_scalar/commit/%s.\n\nVersion: %s. Provenance and checksum verified; normal CI must pass before merging.\n' "$revision" "$version" > "$RUNNER_TEMP/pr-body.md"
gh pr create --base master --head "$branch" --title "Sync classify_scalar $version" --body-file "$RUNNER_TEMP/pr-body.md"
17 changes: 17 additions & 0 deletions .github/workflows/vendor-integrity.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
name: Vendored header integrity
on:
push:
branches: [master]
paths: ['include/external/classify_scalar.*', 'tools/sync_classify_scalar.py', '.github/workflows/vendor-integrity.yml', '.gitattributes']
pull_request:
branches: [master]
paths: ['include/external/classify_scalar.*', 'tools/sync_classify_scalar.py', '.github/workflows/vendor-integrity.yml', '.gitattributes']
workflow_dispatch:
permissions:
contents: read
jobs:
verify:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
- run: python3 tools/sync_classify_scalar.py --check
2 changes: 2 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,8 @@ For Codecov/API-based coverage review workflow, see `CODECOV_AGENTS.md`.
See `tests/AGENTS.md` for test strategy, checklist, and conventions.

### Rules for Coding
Keep `include/external/classify_scalar.hpp` identical to its pinned upstream source. Fix it in `vincentlaucsb/classify_scalar`, then use `python tools/sync_classify_scalar.py --ref <tag-or-commit>` and `--check`; do not patch the vendored copy. See `include/external/README.md` for the release/sync workflow.

1. **Use compatibility macros defined in `common.hpp`** for cross-compiler or cross-standard concerns. If it doesn't exist, consider creating one.
2. **Compatibility macros defined in `common.hpp` MUST be referenced only after including `common.hpp`** to ensure correctness.
3. **Prefer compile time control flow and assertions where possible**. For example, if a branch may be safely written with `if constexpr`, then use the `IF_CONSTEXPR` macro (from `common.hpp`) to ensure C++11 compatibility while ensuring optimal control flow for C++17 and later users.
Expand Down
1 change: 1 addition & 0 deletions ARCHITECTURE.md
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ Subsystem deep-dive:
Operational/testing guidance:
- AGENTS.md
- tests/AGENTS.md
- include/external/README.md — upstream releases, pinned vendoring, and sync automation

Notes:
- Internal architecture content lives under include/internal to stay close to implementation.
Expand Down
2 changes: 2 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,8 @@ set(CSV_TEST_DIR ${CMAKE_CURRENT_LIST_DIR}/tests)

include_directories(${CSV_INCLUDE_DIR})

include(${CMAKE_CURRENT_LIST_DIR}/cmake/CheckFeatures.cmake)

## Load developer specific CMake settings
if (CMAKE_SOURCE_DIR STREQUAL CMAKE_CURRENT_SOURCE_DIR)
SET(CSV_DEVELOPER TRUE)
Expand Down
15 changes: 15 additions & 0 deletions cmake/CheckFeatures.cmake
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
# CheckFeatures.cmake

include(CheckCXXSourceCompiles)

unset(CSV_HAS_STD_FLOATING_POINT_FROM_CHARS CACHE)

check_cxx_source_compiles("
#include <charconv>
int main()
{
double value;
const char* text = \"3.14\";
return std::from_chars(text, text + 4, value).ec != std::errc();
}
" CSV_HAS_STD_FLOATING_POINT_FROM_CHARS)
33 changes: 33 additions & 0 deletions include/external/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
# Vendored classify_scalar

`classify_scalar.hpp` is an unchanged copy of the upstream public header.
Do not patch it here. Fix `vincentlaucsb/classify_scalar` first, run its
`python tools/version.py --patch` command, and review/test the upstream change.

Then sync an approved tag or commit from this repository's root:

```sh
python tools/sync_classify_scalar.py --ref <upstream-tag-or-commit>
python tools/sync_classify_scalar.py --check
```

The command resolves tags to a full commit, copies the header verbatim, and
records its version and SHA-256 in `classify_scalar.json`. CI compares both
the local checksum and the upstream bytes. `--check --offline` checks local
metadata without network access. `--source-repo <path>` reads committed bytes
from a local upstream clone instead of downloading them.

The **Sync classify_scalar** GitHub Actions workflow accepts the same ref and
opens an update PR without auto-merging. Enable **Allow GitHub Actions to create
and approve pull requests** in the repository's Actions settings. With the default
workflow token, GitHub may require **Approve workflows to run** on the resulting
PR. For checks that start automatically, optionally add a fine-grained bot token
as `VENDOR_SYNC_TOKEN`, scoped to this repository with Contents and Pull requests
write permissions. See [GitHub's workflow trigger rules](https://docs.github.com/en/actions/how-tos/write-workflows/choose-when-workflows-run/trigger-a-workflow).
No token is needed for the local sync command or the integrity check against
this public upstream repository.

The build-system adapter lives in `include/internal/CMakeLists.txt`: it passes
the CSV CMake probe's boolean result through the upstream-owned
`CLASSIFY_SCALAR_USE_STD_FLOAT_FROM_CHARS` macro. This keeps csv-specific names
out of the upstream header.
18 changes: 13 additions & 5 deletions include/external/classify_scalar.hpp
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
/*
classify_scalar, version 1.1.0
classify_scalar, version 1.1.1
https://github.com/vincentlaucsb/classify_scalar

MIT License
Expand Down Expand Up @@ -28,16 +28,16 @@ SOFTWARE.
#pragma once

#if defined(CLASSIFY_SCALAR_VERSION)
#if CLASSIFY_SCALAR_VERSION >= 10100
#if CLASSIFY_SCALAR_VERSION >= 10101
#define CLASSIFY_SCALAR_SKIP_HEADER
#else
#error "A newer classify_scalar.hpp was included after an older copy. Include the newest copy first."
#endif
#else
#define CLASSIFY_SCALAR_VERSION_MAJOR 1
#define CLASSIFY_SCALAR_VERSION_MINOR 1
#define CLASSIFY_SCALAR_VERSION_PATCH 0
#define CLASSIFY_SCALAR_VERSION 10100
#define CLASSIFY_SCALAR_VERSION_PATCH 1
#define CLASSIFY_SCALAR_VERSION 10101
#endif

#ifndef CLASSIFY_SCALAR_SKIP_HEADER
Expand Down Expand Up @@ -132,9 +132,17 @@ SOFTWARE.
#include <system_error>
#endif

#if defined(CLASSIFY_SCALAR_HAS_CXX17) && !defined(_LIBCPP_VERSION) && !defined(CLASSIFY_SCALAR_DISABLE_STD_FLOAT_FROM_CHARS)
#if defined(CLASSIFY_SCALAR_HAS_CXX17) && !defined(CLASSIFY_SCALAR_DISABLE_STD_FLOAT_FROM_CHARS)
// A build-system compile/link probe may override the standard feature macro.
// A negative probe must not fall through to the feature-macro fallback.
#if defined(CLASSIFY_SCALAR_USE_STD_FLOAT_FROM_CHARS)
#if CLASSIFY_SCALAR_USE_STD_FLOAT_FROM_CHARS
#define CLASSIFY_SCALAR_HAS_STD_FLOAT_FROM_CHARS
#endif
#elif defined(__cpp_lib_to_chars) && __cpp_lib_to_chars >= 201611L
#define CLASSIFY_SCALAR_HAS_STD_FLOAT_FROM_CHARS
#endif
#endif

namespace classify_scalar {

Expand Down
7 changes: 7 additions & 0 deletions include/external/classify_scalar.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
{
"repository": "vincentlaucsb/classify_scalar",
"path": "include/classify_scalar.hpp",
"commit": "42652a1c0159719cb731e098e04ddc683c2d72f2",
"version": "1.1.1",
"sha256": "559c9accb2a3346642d4061e7748d9686933297334b0d9198c8b7687c20b36fc"
}
5 changes: 5 additions & 0 deletions include/internal/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,9 @@ if(CSV_NO_SIMD)
target_compile_definitions(csv PUBLIC CSV_NO_SIMD=1)
endif()

target_compile_definitions(csv PUBLIC
CLASSIFY_SCALAR_USE_STD_FLOAT_FROM_CHARS=$<BOOL:${CSV_HAS_STD_FLOATING_POINT_FROM_CHARS}>)

if(CSV_ENABLE_THREADS)
target_compile_definitions(csv PUBLIC CSV_ENABLE_THREADS=1)
target_link_libraries(csv PRIVATE Threads::Threads)
Expand All @@ -69,6 +72,8 @@ get_target_property(_csv_sources csv SOURCES)
target_sources(csv_no_simd PRIVATE ${_csv_sources})
set_target_properties(csv_no_simd PROPERTIES LINKER_LANGUAGE CXX)
target_compile_definitions(csv_no_simd PUBLIC CSV_NO_SIMD=1)
target_compile_definitions(csv_no_simd PUBLIC
CLASSIFY_SCALAR_USE_STD_FLOAT_FROM_CHARS=$<BOOL:${CSV_HAS_STD_FLOATING_POINT_FROM_CHARS}>)
if(CSV_ENABLE_THREADS)
target_compile_definitions(csv_no_simd PUBLIC CSV_ENABLE_THREADS=1)
target_link_libraries(csv_no_simd PRIVATE Threads::Threads)
Expand Down
6 changes: 6 additions & 0 deletions tests/AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,12 @@ TEST_CASE("My test") {

### Testing Conventions

Tests tagged `[stress]` run in normal and sanitizer configurations but are
excluded from CTest runs when `ENABLE_CODE_COVERAGE=ON`. Repeated timing-sensitive
loops add little line coverage and can exceed their deadlines under coverage
instrumentation. Keep focused regression tests untagged so coverage still
exercises the underlying behavior.

#### Tests Should Expose Bugs, Not Assert Them

When writing a test for a known bug, assert correct behavior (even if it currently fails), not buggy behavior.
Expand Down
9 changes: 8 additions & 1 deletion tests/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -52,9 +52,16 @@ function(add_csv_parser_test_target target_name)
)
endif()

set(_csv_test_filters)
if(ENABLE_CODE_COVERAGE)
# Repeated stress loops add little line coverage and become brittle
# under instrumentation. Normal and sanitizer jobs still run them.
list(APPEND _csv_test_filters "~[stress]")
endif()

add_test(
NAME ${target_name}
COMMAND ${target_name} --verbosity high --durations yes
COMMAND ${target_name} ${_csv_test_filters} --verbosity high --durations yes
WORKING_DIRECTORY ${CSV_ROOT_DIR}
)
endfunction()
Expand Down
2 changes: 1 addition & 1 deletion tests/test_threadsafe_deque_race.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -112,7 +112,7 @@ TEST_CASE("ThreadSafeDeque kill_all race condition - small file iterator",
}

TEST_CASE("ThreadSafeDeque concurrent stress test",
"[threading][race_condition]") {
"[threading][race_condition][stress]") {
// Stress test: rapidly create and iterate many small CSVs
// to maximize the chance of hitting the race window
SECTION("Rapid sequential small CSV parsing") {
Expand Down
94 changes: 94 additions & 0 deletions tools/sync_classify_scalar.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,94 @@
#!/usr/bin/env python3
"""Copy classify_scalar verbatim from a pinned upstream commit and verify provenance."""
import argparse
import hashlib
import json
from pathlib import Path
import re
import subprocess
from urllib.parse import quote
from urllib.request import Request, urlopen

ROOT = Path(__file__).resolve().parents[1]
REPOSITORY = 'vincentlaucsb/classify_scalar'
SOURCE = 'include/classify_scalar.hpp'
HEADER = ROOT / 'include/external/classify_scalar.hpp'
LOCK = ROOT / 'include/external/classify_scalar.json'


def fetch(url):
request = Request(url, headers={'User-Agent': 'csv-parser-vendor-sync'})
with urlopen(request, timeout=30) as response:
return response.read()


def version(data):
text = data.decode('utf-8')
parts = []
for part in ('MAJOR', 'MINOR', 'PATCH'):
match = re.search(r'^#define CLASSIFY_SCALAR_VERSION_' + part + r' (\d+)$', text, re.M)
if not match:
raise ValueError('Missing version macro: ' + part)
parts.append(int(match.group(1)))
dotted = '.'.join(map(str, parts))
numeric = parts[0] * 10000 + parts[1] * 100 + parts[2]
for pattern in (rf'^#define CLASSIFY_SCALAR_VERSION {numeric}$',
rf'^#if CLASSIFY_SCALAR_VERSION >= {numeric}$',
rf'^classify_scalar, version {re.escape(dotted)}$'):
if not re.search(pattern, text, re.M):
raise ValueError('Inconsistent upstream version metadata')
return dotted


def upstream(revision, local=None):
if local:
sha = subprocess.check_output(
['git', '-C', str(local), 'rev-parse', '--verify', revision + '^{commit}'], text=True).strip()
data = subprocess.check_output(['git', '-C', str(local), 'show', sha + ':' + SOURCE])
else:
info = json.loads(fetch(f'https://api.github.com/repos/{REPOSITORY}/commits/{quote(revision, safe="")}'))
sha = info['sha']
if not re.fullmatch(r'[0-9a-f]{40}', sha):
raise ValueError('Invalid upstream commit ID')
data = fetch(f'https://raw.githubusercontent.com/{REPOSITORY}/{sha}/{SOURCE}')
return sha, data


def main():
parser = argparse.ArgumentParser(description=__doc__)
action = parser.add_mutually_exclusive_group(required=True)
action.add_argument('--ref', help='Upstream tag or commit to import')
action.add_argument('--check', action='store_true', help='Verify local bytes and pinned upstream bytes')
parser.add_argument('--source-repo', type=Path, help='Read committed data from a local upstream checkout')
parser.add_argument('--offline', action='store_true', help='With --check, verify local metadata/checksum only')
args = parser.parse_args()
if args.offline and not args.check:
parser.error('--offline requires --check')
try:
if args.check:
lock = json.loads(LOCK.read_text(encoding='utf-8'))
if lock['repository'] != REPOSITORY or lock['path'] != SOURCE:
raise ValueError('Unexpected upstream source in provenance file')
if not re.fullmatch(r'[0-9a-f]{40}', lock['commit']):
raise ValueError('Provenance must pin a full commit ID')
data = HEADER.read_bytes()
if version(data) != lock['version'] or hashlib.sha256(data).hexdigest() != lock['sha256']:
raise ValueError('Vendored header differs from its provenance; fix upstream and run the sync command')
if not args.offline:
sha, original = upstream(lock['commit'], args.source_repo)
if sha != lock['commit'] or original != data:
raise ValueError('Vendored header does not match the pinned upstream commit')
print(f"Verified classify_scalar {lock['version']} at {lock['commit']}")
else:
sha, data = upstream(args.ref, args.source_repo)
lock = {'repository': REPOSITORY, 'path': SOURCE, 'commit': sha,
'version': version(data), 'sha256': hashlib.sha256(data).hexdigest()}
HEADER.write_bytes(data)
LOCK.write_text(json.dumps(lock, indent=2) + '\n', encoding='utf-8', newline='\n')
print(f"Synced classify_scalar {lock['version']} at {sha}")
except (ValueError, KeyError, OSError, subprocess.CalledProcessError) as error:
parser.exit(1, f'{error}\n')


if __name__ == '__main__':
main()
Loading