diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index adf2660..aff6f1f 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -1,15 +1,40 @@ # This workflow will install Python dependencies, run tests and lint with a variety of Python versions # For more information see: https://docs.github.com/en/actions/automating-builds-and-tests/building-and-testing-python -# TODO: -# - cache this directory $HOME/.cache/pyensembl/ -# - update coveralls -# - get a badge for tests passing -# - download binary dependencies from conda name: Tests on: [push, pull_request] jobs: + packaging: + runs-on: ubuntu-latest + strategy: + matrix: + include: + - python-version: "3.9" + setuptools: "setuptools==77.0.3" + - python-version: "3.11" + setuptools: "setuptools" + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + - name: Install build and test dependencies + run: | + python -m pip install --upgrade pip + python -m pip install --upgrade build twine wheel ruff pytest pytest-cov '${{ matrix.setuptools }}' -r requirements.txt + - name: Build and validate distributions + run: | + python -m build --no-isolation + python -m twine check --strict dist/* + - name: Run lint and tests from the source distribution + run: | + archive_dir=$(mktemp -d) + tar -xzf dist/gtfparse-*.tar.gz -C "$archive_dir" + cd "$archive_dir"/gtfparse-* + ./lint.sh + ./test.sh -W error::DeprecationWarning + build: runs-on: ubuntu-latest strategy: diff --git a/.gitignore b/.gitignore index ba74660..65637cb 100644 --- a/.gitignore +++ b/.gitignore @@ -39,6 +39,7 @@ htmlcov/ .coverage .coverage.* .cache +.hypothesis/ nosetests.xml coverage.xml *,cover diff --git a/.hypothesis/unicode_data/13.0.0/charmap.json.gz b/.hypothesis/unicode_data/13.0.0/charmap.json.gz deleted file mode 100644 index 67b0e0b..0000000 Binary files a/.hypothesis/unicode_data/13.0.0/charmap.json.gz and /dev/null differ diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 0000000..ba435b8 --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,3 @@ +include lint.sh test.sh +recursive-include tests *.py +recursive-include tests/data *.gtf *.gtf.gz diff --git a/gtfparse/__init__.py b/gtfparse/__init__.py index a3dfadd..b0b9c25 100644 --- a/gtfparse/__init__.py +++ b/gtfparse/__init__.py @@ -24,7 +24,7 @@ ) from .write_gtf import write_gtf -__version__ = "2.9.0" +__version__ = "2.9.1" __all__ = [ "GENCODE_BIOTYPE_ALIASES", diff --git a/gtfparse/read_gtf.py b/gtfparse/read_gtf.py index 6b1f12c..eef49ac 100644 --- a/gtfparse/read_gtf.py +++ b/gtfparse/read_gtf.py @@ -197,11 +197,11 @@ def parse_gtf_and_expand_attributes( filepath_or_buffer, restrict_attribute_columns=None, features=None, *, progress_callback=None ): """ - Parse lines into column->values dictionary and then expand + Parse a GTF into a Polars DataFrame and then expand the 'attribute' column into multiple columns. This expansion happens by replacing strings of semi-colon separated key-value values in the 'attribute' column with one column per distinct key, with a list of - values for each row (using None for rows where key didn't occur). + values for each row (using empty strings for rows where the key didn't occur). Parameters ---------- @@ -324,7 +324,7 @@ def read_gtf( expand_attribute_column : bool Replace strings of semi-colon separated key-value values in the 'attribute' column with one column per distinct key, with a list of - values for each row (using None for rows where key didn't occur). + values for each row (using empty strings for rows where the key didn't occur). infer_biotype_column : bool Due to the annoying ambiguity of the second GTF column across multiple @@ -338,7 +338,7 @@ def read_gtf( column_cast_types : dict, optional Dictionary mapping column names to dtypes. Will cast columns to given - Polars types. + pandas-compatible types. usecols : list of str or None Restrict which columns are loaded to the give set. If None, then diff --git a/gtfparse/write_gtf.py b/gtfparse/write_gtf.py index a404740..af7a62e 100644 --- a/gtfparse/write_gtf.py +++ b/gtfparse/write_gtf.py @@ -59,8 +59,7 @@ def _attribute_expr(attribute_columns: list[str]) -> polars.Expr: empty string (string columns) and cannot distinguish absent from present-but-empty, so we treat both null and "" as absent. Values that are merely falsy but non-empty -- notably the string "0" -- are written and - survive a round trip (this is the regression that the naive ``if value:`` - check in earlier drafts got wrong). + survive a round trip. """ if not attribute_columns: return polars.lit("") diff --git a/pyproject.toml b/pyproject.toml index 3f10af2..f9c5c1b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,5 +1,5 @@ [build-system] -requires = ["setuptools>=61.0", "wheel"] +requires = ["setuptools>=77.0.3", "wheel"] build-backend = "setuptools.build_meta" [project] @@ -7,12 +7,13 @@ name = "gtfparse" requires-python = ">=3.9" authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ] description = "Parsing library for extracting data frames of genomic features from GTF files" +license = "Apache-2.0" +license-files = ["LICENSE"] classifiers = [ "Development Status :: 4 - Beta", "Environment :: Console", "Operating System :: OS Independent", "Intended Audience :: Science/Research", - "License :: OSI Approved :: Apache Software License", "Programming Language :: Python", "Programming Language :: Python :: 3", "Programming Language :: Python :: 3.9", @@ -71,13 +72,13 @@ select = [ ] ignore = [ "E501", # line too long (handled by formatter) - "E741", # ambiguous variable name (pre-existing in codebase) - "B006", # mutable default args (pre-existing; behavior-risky to change) + "E741", # ambiguous variable names + "B006", # mutable default arguments "B008", # do not perform function calls in argument defaults "B905", # zip() without explicit strict "SIM108", # use ternary operator instead of if-else "UP007", # use X | Y for type unions (need Python 3.10+) - "UP031", # %-format strings (pre-existing; out of scope for config PR) + "UP031", # %-format strings ] [tool.ruff.lint.per-file-ignores] diff --git a/tasks/todo.md b/tasks/todo.md index b5b2c81..7737ed7 100644 --- a/tasks/todo.md +++ b/tasks/todo.md @@ -137,3 +137,58 @@ file I/O and output conversion; see the PR validation record for its result. Release completion is recorded in the PR after CI, merge, and deployment so the deployment checkout remains clean. + +# Packaging and cleanup: #53, #76, #83 + +## Specification + +The 2.9.0 source distribution contains test modules but omits `tests/__init__.py`, +`tests/data.py`, and the GTF fixtures. Reproduced six collection errors from an +extracted release tarball. Add a manifest that ships all test Python modules, +GTF/gzip fixtures, and lint/test entry scripts in the source distribution. +Keep tests out of the runtime wheel and do not include generated caches or the +unused 10 MB parquet artifact. + +Use SPDX `Apache-2.0` project license metadata, explicitly include `LICENSE`, +and raise the setuptools build requirement to 77.0.3, which supports PEP 639. +Remove the deprecated license classifier. Validate wheel and sdist metadata +and the embedded license text with both the minimum and current build backend. + +Add a packaging CI job on Python 3.9/minimum setuptools and Python 3.11/current +setuptools. Build with the selected backend, check distributions with twine, +extract the tarball in a temporary directory, and run the shipped `lint.sh` and +`test.sh` there. The existing runtime test matrix and coverage aggregation stay +in place. No unit tests for comment-only edits; real archive execution guards +the packaging bug. + +Complete #76's remaining comment/docstring cleanup and remove the two tracked +Hypothesis cache files; ignore future Hypothesis caches. Preserve lint rules +and script behavior. Bump 2.9.0 to 2.9.1; leave the existing README PR #74 alone. + +## Plan + +- [x] Read issues, repository guidance, package contents, and setuptools docs. +- [x] Reproduce the release-tarball failure and create a feature branch. +- [x] Check in with the packaging fix and verification plan. +- [x] Apply packaging metadata, manifest, CI, and comment/cache changes. +- [x] Run `./lint.sh` and `./test.sh` with deprecations as errors. +- [x] Verify both build backends, distribution metadata, extracted-archive + lint/tests, and wheel contents. +- [ ] Open PR, pass CI, merge, run `./deploy.sh` from clean master, and verify + the published wheel and source archive. + +## Review + +`./lint.sh` and all 143 tests pass with deprecations treated as errors in the +checkout and in extracted source archives built by setuptools 77.0.3 and 84.0.0. +Both backends build wheels from their source archives successfully, and all +four distributions pass `twine check --strict`. Verified that source archives +contain the helper, initializer, GTF fixtures, and lint/test scripts, but no +Hypothesis/bytecode caches or unused parquet file. Wheels contain no tests. +Wheel and source metadata both declare `License-Expression: Apache-2.0` and +`License-File: LICENSE`; included license bytes match the repository file. +Neither build emits the old license-classifier warning. Workflow YAML parses, +and git ignores Hypothesis caches at both previously tracked paths. + +Release verification will be recorded on the PR after deployment, keeping the +master checkout clean. diff --git a/test.sh b/test.sh index 97556bc..e40eb95 100755 --- a/test.sh +++ b/test.sh @@ -1,7 +1,6 @@ #!/usr/bin/env bash # Run the gtfparse test suite with a memory- and CPU-aware pytest-xdist -# worker count. See ~/code/trufflepig/test.sh for the rationale: running -# several sibling repos' suites concurrently can fork-bomb the laptop, +# worker count. Running several test suites concurrently can exhaust resources, # so we cap workers at min(cpu_reserve, available_RAM / PER_WORKER_GB). # xdist is optional — fall back to serial pytest when it isn't installed. # diff --git a/tests/.hypothesis/unicode_data/13.0.0/charmap.json.gz b/tests/.hypothesis/unicode_data/13.0.0/charmap.json.gz deleted file mode 100644 index 414a302..0000000 Binary files a/tests/.hypothesis/unicode_data/13.0.0/charmap.json.gz and /dev/null differ diff --git a/tests/data.py b/tests/data.py index 23aaacb..bca618e 100644 --- a/tests/data.py +++ b/tests/data.py @@ -3,7 +3,6 @@ def data_path(name): """ - Return the absolute path to a file in the varcode/test/data directory. - The name specified should be relative to varcode/test/data. + Return the absolute path to a fixture relative to tests/data. """ return os.path.join(os.path.dirname(__file__), "data", name) diff --git a/tests/test_expand_attribute_column_false.py b/tests/test_expand_attribute_column_false.py index b06a206..ab74485 100644 --- a/tests/test_expand_attribute_column_false.py +++ b/tests/test_expand_attribute_column_false.py @@ -1,7 +1,4 @@ -"""Regression tests for #56: read_gtf(expand_attribute_column=False) -used to raise NameError because the else branch referenced `result_df` -before it had been assigned. -""" +"""Reading a GTF without expanding its raw attribute column.""" import pandas as pd diff --git a/tests/test_gencode_gtf.py b/tests/test_gencode_gtf.py index 061928a..a344847 100644 --- a/tests/test_gencode_gtf.py +++ b/tests/test_gencode_gtf.py @@ -1,4 +1,4 @@ -"""Tests for GENCODE attribute aliases (#63) and version-column casting (#64).""" +"""Tests for GENCODE attribute aliases and version-column casting.""" import gzip import logging @@ -14,7 +14,7 @@ # ----------------------------- -# attribute_aliases (#63) +# Attribute aliases # ----------------------------- @@ -99,7 +99,7 @@ def test_attribute_aliases_with_empty_dict_is_noop(): # ----------------------------- -# cast_version_columns (#64) +# Version-column casting # ----------------------------- @@ -157,7 +157,7 @@ def test_version_columns_missing_when_attribute_absent(): # ----------------------------- -# Interaction between #63 and #64 +# Interaction between aliases and version-column casting # -----------------------------