Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 30 additions & 5 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
@@ -1,15 +1,40 @@
# This workflow will install Python dependencies, run tests and lint with a variety of Python versions
# For more information see: https://docs.github.com/en/actions/automating-builds-and-tests/building-and-testing-python

# TODO:
# - cache this directory $HOME/.cache/pyensembl/
# - update coveralls
# - get a badge for tests passing
# - download binary dependencies from conda
name: Tests
on: [push, pull_request]

jobs:
packaging:
runs-on: ubuntu-latest
strategy:
matrix:
include:
- python-version: "3.9"
setuptools: "setuptools==77.0.3"
- python-version: "3.11"
setuptools: "setuptools"
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}
- name: Install build and test dependencies
run: |
python -m pip install --upgrade pip
python -m pip install --upgrade build twine wheel ruff pytest pytest-cov '${{ matrix.setuptools }}' -r requirements.txt
- name: Build and validate distributions
run: |
python -m build --no-isolation
python -m twine check --strict dist/*
- name: Run lint and tests from the source distribution
run: |
archive_dir=$(mktemp -d)
tar -xzf dist/gtfparse-*.tar.gz -C "$archive_dir"
cd "$archive_dir"/gtfparse-*
./lint.sh
./test.sh -W error::DeprecationWarning

build:
runs-on: ubuntu-latest
strategy:
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ htmlcov/
.coverage
.coverage.*
.cache
.hypothesis/
nosetests.xml
coverage.xml
*,cover
Expand Down
Binary file removed .hypothesis/unicode_data/13.0.0/charmap.json.gz
Binary file not shown.
3 changes: 3 additions & 0 deletions MANIFEST.in
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
include lint.sh test.sh
recursive-include tests *.py
recursive-include tests/data *.gtf *.gtf.gz
2 changes: 1 addition & 1 deletion gtfparse/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@
)
from .write_gtf import write_gtf

__version__ = "2.9.0"
__version__ = "2.9.1"

__all__ = [
"GENCODE_BIOTYPE_ALIASES",
Expand Down
8 changes: 4 additions & 4 deletions gtfparse/read_gtf.py
Original file line number Diff line number Diff line change
Expand Up @@ -197,11 +197,11 @@ def parse_gtf_and_expand_attributes(
filepath_or_buffer, restrict_attribute_columns=None, features=None, *, progress_callback=None
):
"""
Parse lines into column->values dictionary and then expand
Parse a GTF into a Polars DataFrame and then expand
the 'attribute' column into multiple columns. This expansion happens
by replacing strings of semi-colon separated key-value values in the
'attribute' column with one column per distinct key, with a list of
values for each row (using None for rows where key didn't occur).
values for each row (using empty strings for rows where the key didn't occur).

Parameters
----------
Expand Down Expand Up @@ -324,7 +324,7 @@ def read_gtf(
expand_attribute_column : bool
Replace strings of semi-colon separated key-value values in the
'attribute' column with one column per distinct key, with a list of
values for each row (using None for rows where key didn't occur).
values for each row (using empty strings for rows where the key didn't occur).

infer_biotype_column : bool
Due to the annoying ambiguity of the second GTF column across multiple
Expand All @@ -338,7 +338,7 @@ def read_gtf(

column_cast_types : dict, optional
Dictionary mapping column names to dtypes. Will cast columns to given
Polars types.
pandas-compatible types.

usecols : list of str or None
Restrict which columns are loaded to the give set. If None, then
Expand Down
3 changes: 1 addition & 2 deletions gtfparse/write_gtf.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,8 +59,7 @@ def _attribute_expr(attribute_columns: list[str]) -> polars.Expr:
empty string (string columns) and cannot distinguish absent from
present-but-empty, so we treat both null and "" as absent. Values that are
merely falsy but non-empty -- notably the string "0" -- are written and
survive a round trip (this is the regression that the naive ``if value:``
check in earlier drafts got wrong).
survive a round trip.
"""
if not attribute_columns:
return polars.lit("")
Expand Down
11 changes: 6 additions & 5 deletions pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,18 +1,19 @@
[build-system]
requires = ["setuptools>=61.0", "wheel"]
requires = ["setuptools>=77.0.3", "wheel"]
build-backend = "setuptools.build_meta"

[project]
name = "gtfparse"
requires-python = ">=3.9"
authors = [ {name="Alex Rubinsteyn", email="alex.rubinsteyn@unc.edu" } ]
description = "Parsing library for extracting data frames of genomic features from GTF files"
license = "Apache-2.0"
license-files = ["LICENSE"]
classifiers = [
"Development Status :: 4 - Beta",
"Environment :: Console",
"Operating System :: OS Independent",
"Intended Audience :: Science/Research",
"License :: OSI Approved :: Apache Software License",
"Programming Language :: Python",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.9",
Expand Down Expand Up @@ -71,13 +72,13 @@ select = [
]
ignore = [
"E501", # line too long (handled by formatter)
"E741", # ambiguous variable name (pre-existing in codebase)
"B006", # mutable default args (pre-existing; behavior-risky to change)
"E741", # ambiguous variable names
"B006", # mutable default arguments
"B008", # do not perform function calls in argument defaults
"B905", # zip() without explicit strict
"SIM108", # use ternary operator instead of if-else
"UP007", # use X | Y for type unions (need Python 3.10+)
"UP031", # %-format strings (pre-existing; out of scope for config PR)
"UP031", # %-format strings
]

[tool.ruff.lint.per-file-ignores]
Expand Down
55 changes: 55 additions & 0 deletions tasks/todo.md
Original file line number Diff line number Diff line change
Expand Up @@ -137,3 +137,58 @@ file I/O and output conversion; see the PR validation record for its result.

Release completion is recorded in the PR after CI, merge, and deployment so
the deployment checkout remains clean.

# Packaging and cleanup: #53, #76, #83

## Specification

The 2.9.0 source distribution contains test modules but omits `tests/__init__.py`,
`tests/data.py`, and the GTF fixtures. Reproduced six collection errors from an
extracted release tarball. Add a manifest that ships all test Python modules,
GTF/gzip fixtures, and lint/test entry scripts in the source distribution.
Keep tests out of the runtime wheel and do not include generated caches or the
unused 10 MB parquet artifact.

Use SPDX `Apache-2.0` project license metadata, explicitly include `LICENSE`,
and raise the setuptools build requirement to 77.0.3, which supports PEP 639.
Remove the deprecated license classifier. Validate wheel and sdist metadata
and the embedded license text with both the minimum and current build backend.

Add a packaging CI job on Python 3.9/minimum setuptools and Python 3.11/current
setuptools. Build with the selected backend, check distributions with twine,
extract the tarball in a temporary directory, and run the shipped `lint.sh` and
`test.sh` there. The existing runtime test matrix and coverage aggregation stay
in place. No unit tests for comment-only edits; real archive execution guards
the packaging bug.

Complete #76's remaining comment/docstring cleanup and remove the two tracked
Hypothesis cache files; ignore future Hypothesis caches. Preserve lint rules
and script behavior. Bump 2.9.0 to 2.9.1; leave the existing README PR #74 alone.

## Plan

- [x] Read issues, repository guidance, package contents, and setuptools docs.
- [x] Reproduce the release-tarball failure and create a feature branch.
- [x] Check in with the packaging fix and verification plan.
- [x] Apply packaging metadata, manifest, CI, and comment/cache changes.
- [x] Run `./lint.sh` and `./test.sh` with deprecations as errors.
- [x] Verify both build backends, distribution metadata, extracted-archive
lint/tests, and wheel contents.
- [ ] Open PR, pass CI, merge, run `./deploy.sh` from clean master, and verify
the published wheel and source archive.

## Review

`./lint.sh` and all 143 tests pass with deprecations treated as errors in the
checkout and in extracted source archives built by setuptools 77.0.3 and 84.0.0.
Both backends build wheels from their source archives successfully, and all
four distributions pass `twine check --strict`. Verified that source archives
contain the helper, initializer, GTF fixtures, and lint/test scripts, but no
Hypothesis/bytecode caches or unused parquet file. Wheels contain no tests.
Wheel and source metadata both declare `License-Expression: Apache-2.0` and
`License-File: LICENSE`; included license bytes match the repository file.
Neither build emits the old license-classifier warning. Workflow YAML parses,
and git ignores Hypothesis caches at both previously tracked paths.

Release verification will be recorded on the PR after deployment, keeping the
master checkout clean.
3 changes: 1 addition & 2 deletions test.sh
Original file line number Diff line number Diff line change
@@ -1,7 +1,6 @@
#!/usr/bin/env bash
# Run the gtfparse test suite with a memory- and CPU-aware pytest-xdist
# worker count. See ~/code/trufflepig/test.sh for the rationale: running
# several sibling repos' suites concurrently can fork-bomb the laptop,
# worker count. Running several test suites concurrently can exhaust resources,
# so we cap workers at min(cpu_reserve, available_RAM / PER_WORKER_GB).
# xdist is optional — fall back to serial pytest when it isn't installed.
#
Expand Down
Binary file not shown.
3 changes: 1 addition & 2 deletions tests/data.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@

def data_path(name):
"""
Return the absolute path to a file in the varcode/test/data directory.
The name specified should be relative to varcode/test/data.
Return the absolute path to a fixture relative to tests/data.
"""
return os.path.join(os.path.dirname(__file__), "data", name)
5 changes: 1 addition & 4 deletions tests/test_expand_attribute_column_false.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,4 @@
"""Regression tests for #56: read_gtf(expand_attribute_column=False)
used to raise NameError because the else branch referenced `result_df`
before it had been assigned.
"""
"""Reading a GTF without expanding its raw attribute column."""

import pandas as pd

Expand Down
8 changes: 4 additions & 4 deletions tests/test_gencode_gtf.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
"""Tests for GENCODE attribute aliases (#63) and version-column casting (#64)."""
"""Tests for GENCODE attribute aliases and version-column casting."""

import gzip
import logging
Expand All @@ -14,7 +14,7 @@


# -----------------------------
# attribute_aliases (#63)
# Attribute aliases
# -----------------------------


Expand Down Expand Up @@ -99,7 +99,7 @@ def test_attribute_aliases_with_empty_dict_is_noop():


# -----------------------------
# cast_version_columns (#64)
# Version-column casting
# -----------------------------


Expand Down Expand Up @@ -157,7 +157,7 @@ def test_version_columns_missing_when_attribute_absent():


# -----------------------------
# Interaction between #63 and #64
# Interaction between aliases and version-column casting
# -----------------------------


Expand Down
Loading