From b3e4d5146ec23de3176ab0664a91e6588d0c971d Mon Sep 17 00:00:00 2001 From: Alex Rubinsteyn Date: Tue, 29 Sep 2026 22:07:26 -0400 Subject: [PATCH 1/2] Inherit native PyEnsembl reference DNA without losing genome identity (10.11.0) --- CHANGELOG.md | 13 ++ docs/api_variants.md | 39 ++++++ docs/transforms.md | 4 +- tests/test_native_reference_dna.py | 206 +++++++++++++++++++++++++++++ varcode/genome.py | 124 ++++++++++------- varcode/genome_sequence.py | 14 +- varcode/reference.py | 22 +-- varcode/version.py | 2 +- 8 files changed, 358 insertions(+), 66 deletions(-) create mode 100644 tests/test_native_reference_dna.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 9778bb5a..070bb522 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,18 @@ # Change Log +## [v10.11.0](https://github.com/openvax/varcode/tree/v10.11.0) (2026-09-29) + +- `Genome(native_genome)` inherits optional native PyEnsembl reference DNA + instead of hiding it (#488), enabling intronic/intergenic reference lookups. + Construction and rewrapping remain lazy; explicit `fasta=` takes precedence. +- Native `sequence()` preserves PyEnsembl's errors for uninstalled DNA and + invalid intervals. Tiered reference lookups retain their cDNA fallback. + Reads never download DNA. `close()` closes only explicit path readers opened + by that wrapper, leaving native genomes and borrowed readers usable. +- Preserve the identity of supplied genome objects during inference (#565). + Equal annotation genomes can carry different DNA; caching them by equality + could silently select the wrong reference file. + ## [v10.10.0](https://github.com/openvax/varcode/tree/v10.10.0) (2026-09-29) - Add `reference_completion_hypotheses` for mapped Exacto and other RNA diff --git a/docs/api_variants.md b/docs/api_variants.md index d743f2b0..d6752f51 100644 --- a/docs/api_variants.md +++ b/docs/api_variants.md @@ -5,6 +5,45 @@ For examples, see [file loading](getting_started.md#load-a-file), ## Reference genomes +`Genome` inherits reference DNA configured on a PyEnsembl genome (PyEnsembl +2.11 or later). This makes intronic and intergenic bases available to Varcode's +reference lookups as well as features such as indel left alignment. + +```python +from pyensembl import EnsemblRelease +from varcode import Genome + +native = EnsemblRelease(81, genome_fasta_path="/data/GRCh38.fa") +genome = Genome(native) +bases = genome.reference_range("7", 117_480_000, 117_480_050) +``` + +Construction and rewrapping do not open or download native DNA. The first +read can index a local FASTA or decompress it into PyEnsembl's cache. For a +remote source, install DNA explicitly before querying: + +```python +native = EnsemblRelease(81, download_genome_fasta=True) +native.download_genome_fasta() # explicit, potentially large download +native.index_genome_fasta() +genome = Genome(native) +``` + +An explicit `Genome(native, fasta="/data/other.fa")` overrides native DNA; +the supplied reference must match the annotation assembly. Without DNA, +`reference_base` and `reference_range` fall back to transcript cDNA and return +`""` for uncovered positions. `sequence` reads only chromosome DNA: configured +native DNA preserves PyEnsembl's missing-DNA and invalid-interval errors, +while unconfigured genomes return `""`. Explicit FASTA overrides retain their +legacy permissive lookup behavior. Coordinates are 1-based inclusive, and +returned bases are uppercase on the genomic plus strand. + +The native PyEnsembl genome owns its reader. Calling `genome.close()` leaves +native DNA and caller-provided reader objects open; it closes only readers +that this wrapper opened from an explicit `fasta=` path. Rewrapping borrows an +explicit reader, so its owning wrapper must stay open. Close the native genome +or a caller-provided reader only after all its borrowers have finished. + ::: varcode.Genome ## Variants diff --git a/docs/transforms.md b/docs/transforms.md index 6f4113d4..98dd183e 100644 --- a/docs/transforms.md +++ b/docs/transforms.md @@ -138,7 +138,7 @@ leftmost equivalent position. from varcode import load_vcf, Genome from varcode.transforms import left_align_indels -# Default (transcript-cDNA coverage only) — exonic indels normalize, +# Without installed reference DNA — exonic indels normalize, # intronic/intergenic pass through unchanged. vc = load_vcf("tumor.vcf", genome="GRCh38") vc = left_align_indels(vc) @@ -153,6 +153,8 @@ No `reference` parameter — `left_align_indels` reads bases via the genome the variants already carry (see [varcode.Genome](api_variants.md#varcode.Genome)). Coverage depends on which genome shape was passed. +Native PyEnsembl reference DNA is also inherited when wrapping a configured +genome; see [reference genomes](api_variants.md#reference-genomes) for setup. ### Behavior diff --git a/tests/test_native_reference_dna.py b/tests/test_native_reference_dna.py new file mode 100644 index 00000000..89d9c8d8 --- /dev/null +++ b/tests/test_native_reference_dna.py @@ -0,0 +1,206 @@ +"""Native PyEnsembl DNA stays available through varcode.Genome (#488).""" + +import inspect + +import pytest +from pyensembl import Genome as AnnotationGenome + +from varcode import Genome, Variant +from varcode.genome_sequence import reference_base, reference_range +from varcode.reference import infer_genome + + +@pytest.fixture +def native_genome(tmp_path): + if "genome_fasta_path_or_url" not in inspect.signature(AnnotationGenome).parameters: + pytest.skip("Native reference DNA requires PyEnsembl >= 2.11") + fasta = tmp_path / "reference.fa" + fasta.write_text(">1\naacgttcaggtaccgattcgaacgt\n") + genome = AnnotationGenome( + reference_name="GRCh38", annotation_name="native_dna_test", + genome_fasta_path_or_url=str(fasta), + cache_directory_path=str(tmp_path / "cache")) + yield genome + genome.close() + + +def test_native_intronic_and_intergenic_queries(native_genome, tmp_path): + gtf = tmp_path / "tiny.gtf" + attributes = ( + 'gene_id "g1"; gene_name "G1"; transcript_id "t1"; ' + 'transcript_name "T1"; transcript_biotype "lncRNA";') + gtf.write_text("".join( + '1\ttest\t%s\t%d\t%d\t.\t-\t.\t%s\n' % ( + feature, start, end, attributes + extra) + for feature, start, end, extra in [ + ("gene", 2, 13, ""), ("transcript", 2, 13, ""), + ("exon", 2, 5, ' exon_id "e1"; exon_number "2";'), + ("exon", 10, 13, ' exon_id "e2"; exon_number "1";')])) + native = AnnotationGenome( + reference_name="GRCh38", annotation_name="native_dna_annotation", + gtf_path_or_url=str(gtf), + genome_fasta_path_or_url=str(tmp_path / "reference.fa"), + cache_directory_path=str(tmp_path / "annotated_cache")) + try: + native.index() + wrapped = Genome(native) + # This intron belongs to a minus-strand transcript; genomic reads + # must still return plus-strand DNA, with inclusive endpoints. + transcript = native.transcript_by_id("t1") + assert transcript.contains("1", 6, 9) + assert all(not exon.overlaps("1", 6, 9) for exon in transcript.exons) + assert not native.transcripts_at_locus("1", 16, 20) + assert wrapped.sequence("1", 6, 9) == "TCAG" + assert wrapped.reference_range("1", 6, 9) == "TCAG" + assert reference_base(wrapped, "1", 7) == "C" + assert reference_range(wrapped, "1", 16, 20) == "ATTCG" + assert wrapped.fasta is native.fasta + finally: + native.close() + + +def test_construction_inference_and_repr_do_not_open_native_dna( + native_genome, monkeypatch): + def unexpected_open(self): + pytest.fail("Native FASTA opened before a sequence request") + + monkeypatch.setattr(AnnotationGenome, "fasta", property(unexpected_open)) + wrapped = Genome(native_genome) + rewrapped = Genome(wrapped) + assert infer_genome(native_genome) == (native_genome, False) + assert infer_genome(wrapped) == (wrapped, False) + assert Variant("1", 7, "C", "A", genome=rewrapped).genome is rewrapped + assert "native FASTA configured" in repr(wrapped) + wrapped.close() + + +def test_rewrap_tracks_native_reader_refresh(native_genome): + wrapped = Genome(native_genome) + rewrapped = Genome(wrapped) + original = wrapped.fasta + assert original is not None + native_genome.clear_cache() + assert wrapped.fasta is not original + assert rewrapped.fasta is wrapped.fasta is native_genome.fasta + assert str(original["1"][5:9]).upper() == "TCAG" + original.close() + + +@pytest.mark.parametrize("rewrap", [False, True]) +def test_closing_borrowing_wrapper_keeps_native_reader_usable(native_genome, rewrap): + first = Genome(native_genome) + second = Genome(first if rewrap else native_genome) + reader = native_genome.fasta + first.close() + first.close() + assert str(reader["1"][5:9]).upper() == "TCAG" + assert second.sequence("1", 6, 9) == "TCAG" + assert native_genome.fasta is reader + + +def test_explicit_override_and_rewrap_precede_native_dna(native_genome, tmp_path): + override = tmp_path / "override.fa" + override.write_text(">1\n" + "a" * 24 + "\n") + wrapped = Genome(native_genome, fasta=override, verify=False) + rewrapped = Genome(wrapped) + assert rewrapped.fasta is wrapped.fasta + assert rewrapped.sequence("1", 6, 9) == "AAAA" + assert rewrapped.reference_range("1", 6, 9) == "AAAA" + assert native_genome.sequence("1", 6, 9) == "TCAG" + reader = wrapped.fasta + rewrapped.close() + assert not reader.faidx.file.closed + assert wrapped.reference_base("1", 6) == "A" + wrapped.close() + assert reader.faidx.file.closed + + +def test_caller_provided_reader_is_borrowed(native_genome, tmp_path): + from pyfaidx import Fasta + + path = tmp_path / "borrowed.fa" + path.write_text(">1\ncccc\n") + with Fasta(str(path)) as reader: + wrapped = Genome(native_genome, fasta=reader, verify=False) + wrapped.close() + assert wrapped.sequence("1", 1, 4) == "CCCC" + assert not reader.faidx.file.closed + + +def test_assigning_fasta_releases_owned_reader(native_genome, tmp_path): + path = tmp_path / "override.fa" + path.write_text(">1\ncccc\n") + wrapped = Genome(native_genome, fasta=path, verify=False) + reader = wrapped.fasta + wrapped.fasta = None + assert reader.faidx.file.closed + assert wrapped.sequence("1", 6, 9) == "TCAG" + + +@pytest.mark.parametrize("remote", [False, True]) +def test_uninstalled_native_dna_never_downloads(native_genome, tmp_path, monkeypatch, remote): + from pyensembl.genome_fasta import GenomeFasta, MissingGenomeFastaError + + def unexpected_download(*args, **kwargs): + pytest.fail("Reference lookup attempted a download") + + monkeypatch.setattr(GenomeFasta, "_download", unexpected_download) + source = ("https://example.invalid/reference.fa" if remote + else str(tmp_path / "missing.fa")) + native = AnnotationGenome( + reference_name="GRCh38", annotation_name="missing_native_dna", + genome_fasta_path_or_url=source, + cache_directory_path=str(tmp_path / "missing_cache")) + monkeypatch.setattr(native, "transcripts_at_locus", lambda *args, **kwargs: []) + wrapped = Genome(native) + assert wrapped.fasta is None + assert wrapped.reference_base("1", 7) == "" + assert wrapped.reference_range("1", 6, 9) == "" + with pytest.raises(MissingGenomeFastaError): + wrapped.sequence("1", 6, 9) + wrapped.close() + + +def test_equal_native_genomes_keep_distinct_dna(native_genome, tmp_path): + path = tmp_path / "other.fa" + path.write_text(">1\n" + "c" * 24 + "\n") + other = AnnotationGenome( + reference_name=native_genome.reference_name, + annotation_name=native_genome.annotation_name, + genome_fasta_path_or_url=str(path), + cache_directory_path=str(tmp_path / "other_cache")) + try: + assert native_genome == other + assert infer_genome(native_genome)[0] is native_genome + assert infer_genome(other)[0] is other + assert Genome(native_genome).sequence("1", 6, 9) == "TCAG" + assert Genome(other).sequence("1", 6, 9) == "CCCC" + assert Variant("1", 7, "C", "A", genome=other).genome is other + finally: + other.close() + + +@pytest.mark.parametrize("start,end", [(0, 3), (3, 2), (20, 30)]) +def test_native_sequence_preserves_coordinate_errors(native_genome, start, end): + with pytest.raises(ValueError): + Genome(native_genome).sequence("1", start, end) + + +def test_native_sequence_preserves_missing_contig_error(native_genome): + with pytest.raises(ValueError, match="Contig"): + Genome(native_genome).sequence("missing", 1, 4) + + +def test_old_pyensembl_without_native_api(monkeypatch, tmp_path): + # Simulate supported PyEnsembl versions before native DNA was added. + monkeypatch.delattr(AnnotationGenome, "fasta", raising=False) + monkeypatch.delattr(AnnotationGenome, "requires_genome_fasta", raising=False) + native = AnnotationGenome( + reference_name="GRCh38", annotation_name="old_api", + cache_directory_path=str(tmp_path / "old_cache")) + monkeypatch.setattr(native, "transcripts_at_locus", lambda *args, **kwargs: []) + wrapped = Genome(native) + assert wrapped.fasta is None + assert wrapped.sequence("1", 1, 4) == "" + assert wrapped.reference_base("1", 1) == "" + wrapped.close() diff --git a/varcode/genome.py b/varcode/genome.py index bff72e1a..daa6ed42 100644 --- a/varcode/genome.py +++ b/varcode/genome.py @@ -12,27 +12,17 @@ """Declarative wrapper for a pyensembl ``Genome`` + optional chromosome FASTA. -varcode's reference is the pyensembl ``Genome`` object. By default it -ships transcript and protein FASTAs but not the chromosome FASTA, so -features needing raw genomic bases (intronic, intergenic, flanking) -fall back to transcript-cDNA coverage only. :class:`Genome` is the -construction-time composition point: pass a release identifier plus -an optional ``fasta``, and downstream lookups see both sources. +Varcode inherits optional native reference DNA from the wrapped +pyensembl ``Genome``. An explicit ``fasta=`` overrides that source. +Without chromosome DNA, tiered lookups fall back to transcript cDNA. +Construction never opens or downloads native reference DNA. Acts as a drop-in for ``pyensembl.Genome`` — every pyensembl attribute is accessible via :py:meth:`__getattr__` delegation, so existing code that calls ``genome.transcripts_at_locus(...)``, ``genome.reference_name``, etc. works unchanged. -This module is a stopgap until ``pyensembl.Genome`` natively supports -the chromosome FASTA (tracked in openvax/pyensembl#337). Once that -lands, this wrapper either becomes a one-line delegate or evaporates. -Designed so the migration is straightforward — the API surface here -(``Genome.sequence()``, ``Genome.fasta``, the construction-time -composition pattern) is intentionally identical to the proposed -pyensembl-native API. - -Tracked in openvax/varcode#372. +Tracked in openvax/varcode#372 and #488. """ import os @@ -66,11 +56,11 @@ class Genome: The two methods have intentionally different fall-through semantics: - * :meth:`sequence` — chromosome FASTA *only*. Returns ``""`` when - no FASTA is attached. Mirrors the proposed - ``pyensembl.Genome.sequence()`` shape from - openvax/pyensembl#337 so the eventual upstream migration is - mechanical. + * :meth:`sequence` — chromosome FASTA *only*. Delegates to native + PyEnsembl DNA when configured, including its missing-DNA and + invalid-interval errors. Returns ``""`` for genomes without DNA + configured. Explicit FASTA overrides retain the legacy permissive + lookup behavior. * :meth:`reference_base` / :meth:`reference_range` — tiered. FASTA first when attached; otherwise transcript cDNA. Use these when you want "whatever varcode can tell you" about a position. @@ -90,20 +80,20 @@ class Genome: * ``pyfaidx.Fasta`` — used as-is. * Any object supporting ``fa[contig][start:end]`` and returning a string or an object with ``.seq``. - * ``None`` — no chromosome-level access; features fall back to - transcript cDNA. + * ``None`` — inherit native PyEnsembl DNA (or a rewrapped + Genome's override). Without DNA, features fall back to cDNA. - varcode does not take ownership of the FASTA object — when the - caller passes a pre-opened ``pyfaidx.Fasta``, the caller is - responsible for its lifetime. Closing the FASTA after passing - it to ``Genome`` will cause subsequent lookups to fail. + Native and pre-opened readers are borrowed; their owner controls + their lifetime. :meth:`close` closes only a reader this wrapper + opened from an explicit path. Rewrapping borrows that reader, so + keep its owning wrapper open while using it. verify : When True (default) and ``fasta`` is provided, spot-check a few exonic positions against pyensembl's transcript cDNA to catch mislabeled FASTAs (e.g. GRCh37 attached to a GRCh38 release). Iterates ``self.transcripts()`` which on a fresh process may trigger lazy DB construction; pass ``verify=False`` to defer - that cost. + that cost. Inherited native DNA is not opened or spot-checked. Examples -------- @@ -136,14 +126,12 @@ def __init__( fasta: Optional[Any] = None, verify: bool = True): fasta_was_provided = fasta is not None + self._fasta_override = None + self._owns_fasta = False if isinstance(ensembl_release, Genome): - # Idempotent rewrap. Inherits ._ensembl always; inherits - # .fasta unless the caller is providing a fresh one. self._ensembl = ensembl_release._ensembl - if fasta_was_provided: - self.fasta = _resolve_fasta(fasta, self._ensembl) - else: - self.fasta = ensembl_release.fasta + # Copy only the override, without opening a lazy native reader. + self._fasta_override = ensembl_release._fasta_override else: if ensembl_release is None: raise ValueError( @@ -151,8 +139,10 @@ def __init__( "(int release number, reference-name string, or a " "pyensembl.Genome / varcode.Genome instance).") self._ensembl, _ = infer_genome(ensembl_release) - self.fasta = (_resolve_fasta(fasta, self._ensembl) - if fasta_was_provided else None) + + if fasta_was_provided: + self._fasta_override = _resolve_fasta(fasta, self._ensembl) + self._owns_fasta = isinstance(fasta, (str, bytes, os.PathLike)) # Verify only when the caller freshly provided a FASTA. # Inheriting a verified FASTA on rewrap doesn't need a re-check. @@ -166,6 +156,36 @@ def __init__( # lifetime — no module-level cache to invalidate. self._missing_reference_warned = False + @property + def fasta(self): + """Explicit FASTA override, otherwise the lazy native DNA reader. + + Native access may index a local FASTA, but never downloads DNA. + Returns None when native DNA is unconfigured or uninstalled. + Assigning None removes the override and resumes native access. + """ + if self._fasta_override is not None: + return self._fasta_override + return getattr(self._ensembl, "fasta", None) + + @fasta.setter + def fasta(self, reader): + if reader is self._fasta_override: + return + self.close() + self._fasta_override = reader + + def close(self): + """Close only a FASTA opened by this wrapper from an explicit path. + + Native genomes and borrowed readers remain open. Repeated calls + are harmless. After closing an owned reader, assign a new reader + (or None to resume native access) before making further lookups. + """ + if self._owns_fasta: + self._fasta_override.close() + self._owns_fasta = False + def __getattr__(self, name): """Delegate everything else to the wrapped pyensembl Genome. @@ -189,7 +209,12 @@ def __getattr__(self, name): def __repr__(self) -> str: ref = getattr(self._ensembl, "reference_name", "?") - fasta_repr = "no FASTA" if self.fasta is None else "FASTA attached" + if self._fasta_override is not None: + fasta_repr = "FASTA attached" + elif getattr(self._ensembl, "requires_genome_fasta", False): + fasta_repr = "native FASTA configured" + else: + fasta_repr = "no FASTA" return "varcode.Genome(reference_name=%r, %s)" % (ref, fasta_repr) def __dir__(self): @@ -202,24 +227,29 @@ def __dir__(self): own.update(dir(ensembl)) return sorted(own) - # -- chromosome FASTA API (mirrors openvax/pyensembl#337) ---------- + # -- chromosome FASTA API ------------------------------------------ def sequence(self, contig: str, start: int, end: int) -> str: """Chromosome FASTA sequence on the ``+`` strand. - 1-based inclusive coordinates. Returns ``""`` when no FASTA is - attached or the contig / range isn't covered. + Uses 1-based inclusive coordinates and returns uppercase DNA. + Configured native DNA delegates to PyEnsembl: missing DNA raises + ``MissingGenomeFastaError``; invalid coordinates or absent contigs + raise ``ValueError``. Install native DNA explicitly with + ``download_genome_fasta()`` before querying remote sources. - FASTA-only — does **not** fall back to transcript cDNA. Use - :meth:`reference_base` / :meth:`reference_range` for the - tiered lookup. The split is deliberate: this method mirrors - the proposed ``pyensembl.Genome.sequence()`` API so callers - who want raw chromosome bases (and the upstream migration - path) can use it unambiguously. + Without configured DNA, returns ``""``. Explicit overrides use + the legacy permissive reader (unreadable intervals return ``""``). + Does not fall back to cDNA; use :meth:`reference_base` or + :meth:`reference_range` for tiered lookup. """ - if self.fasta is None: + if (self._fasta_override is None + and getattr(self._ensembl, "requires_genome_fasta", False)): + return self._ensembl.sequence(contig, start, end) + fasta = self.fasta + if fasta is None: return "" - return _fasta_range(self.fasta, contig, start, end) + return _fasta_range(fasta, contig, start, end) # -- tiered lookup convenience ------------------------------------- diff --git a/varcode/genome_sequence.py b/varcode/genome_sequence.py index 10c3a7e1..b5956967 100644 --- a/varcode/genome_sequence.py +++ b/varcode/genome_sequence.py @@ -16,10 +16,11 @@ varcode has two possible sources of reference bases at a genomic position, and falls back between them in this order: -1. **Chromosome FASTA** (when attached) — read directly from the - FASTA file via the ``.fasta`` attribute on :class:`varcode.Genome`. - Covers every base on every contig the FASTA contains. Pass a - FASTA when constructing the genome to enable this source. +1. **Chromosome FASTA** (when available) — read via ``.fasta`` on either + native PyEnsembl genomes or :class:`varcode.Genome`. The wrapper + inherits native DNA unless an explicit ``fasta=`` overrides it. + Covers every base on every contig the FASTA contains. Missing native + DNA falls through to cDNA; lookups never download reference DNA. 2. **Transcript cDNA** — fall back to pyensembl's transcript sequences via ``transcript.spliced_offset()`` for any transcript covering the position. Reverse-complements for ``-`` strand @@ -35,8 +36,7 @@ methods (or its construction-time ``fasta=`` kwarg) and the same module-level functions for ad-hoc use. -Tracked in openvax/varcode#372. Upstream pyensembl support for the -chromosome FASTA itself is tracked in openvax/pyensembl#337. +Tracked in openvax/varcode#372 and #488. """ import logging @@ -60,7 +60,7 @@ def reference_base(genome: Any, contig: str, position: int) -> str: Works on any genome shape — :class:`varcode.Genome` (chromosome FASTA available via ``.fasta``), bare ``pyensembl.Genome`` - (transcript cDNA only), or anything duck-typed for + (optional native FASTA plus transcript cDNA), or anything duck-typed for ``transcripts_at_locus``. See module docstring for the full fallback order. diff --git a/varcode/reference.py b/varcode/reference.py index f8f2d216..23e9e1aa 100644 --- a/varcode/reference.py +++ b/varcode/reference.py @@ -373,7 +373,6 @@ def infer_genome_for_reference_name(reference_name): genome = cached_ensembl_release(release, species=species) return genome, converted_ucsc_to_ensembl -@memoize def infer_genome(genome_object_string_or_int): """ If given an integer, get the human EnsemblRelease object for that @@ -384,21 +383,24 @@ def infer_genome(genome_object_string_or_int): (e.g. "GRCh38:93"). If the given name is a UCSC genome (e.g. hg19) then convert it to the equivalent Ensembl reference (e.g. GRCh37). - If given a PyEnsembl Genome, simply use it. + If given a PyEnsembl or Varcode Genome, return that same object. Do not + memoize object pass-through: native genomes with different attached DNA + can compare equal. Integer/string resolvers below already cache results. Returns a pair of (Genome, bool) where the bool corresponds to whether the input requested a UCSC genome (e.g. "hg19") and an Ensembl (e.g. GRCh37) was returned as a substitute. """ + # Import at call time to avoid the Genome wrapper's import of this module. + from .genome import Genome as VarcodeGenome + converted_ucsc_to_ensembl = False - # varcode.Genome wraps a pyensembl Genome with optional FASTA. - # Recognize it via duck-typing (has ``transcripts_at_locus`` and a - # ``fasta`` attribute) so we don't have to import it here and - # introduce a cycle. - if (hasattr(genome_object_string_or_int, "transcripts_at_locus") - and hasattr(genome_object_string_or_int, "fasta")): + # Known genome types must bypass duck-typing: probing .fasta can open + # and index native DNA before a caller has asked for any sequence. + if isinstance(genome_object_string_or_int, (Genome, VarcodeGenome)): genome = genome_object_string_or_int - elif isinstance(genome_object_string_or_int, Genome): + elif (hasattr(genome_object_string_or_int, "transcripts_at_locus") + and hasattr(genome_object_string_or_int, "fasta")): genome = genome_object_string_or_int elif is_integer(genome_object_string_or_int): genome = cached_ensembl_release(genome_object_string_or_int) @@ -411,4 +413,4 @@ def infer_genome(genome_object_string_or_int): "instance, got %s : %s") % ( str(genome_object_string_or_int), type(genome_object_string_or_int))) - return genome, converted_ucsc_to_ensembl \ No newline at end of file + return genome, converted_ucsc_to_ensembl diff --git a/varcode/version.py b/varcode/version.py index 1a947473..120f8138 100644 --- a/varcode/version.py +++ b/varcode/version.py @@ -1 +1 @@ -__version__ = "10.10.2" +__version__ = "10.11.0" From d4a61866889d5bd85c63dae136df0716351b7d17 Mon Sep 17 00:00:00 2001 From: Alex Rubinsteyn Date: Tue, 29 Sep 2026 22:12:53 -0400 Subject: [PATCH 2/2] Preserve previously pickled Genome wrappers --- tests/test_genome_sequence.py | 20 ++++++++++++++++++++ varcode/genome.py | 7 +++++++ 2 files changed, 27 insertions(+) diff --git a/tests/test_genome_sequence.py b/tests/test_genome_sequence.py index 2c66a35c..ca534a0f 100644 --- a/tests/test_genome_sequence.py +++ b/tests/test_genome_sequence.py @@ -31,6 +31,7 @@ """ import copy +import pickle import warnings import pytest @@ -152,6 +153,25 @@ def test_genome_rewrap_overrides_fasta(ensembl, cftr_position): assert g2.fasta is fasta2 +@pytest.mark.parametrize("with_override", [False, True]) +def test_restore_legacy_pickled_genome(ensembl, with_override): + # The public Genome's pickle state through 10.10.x stored .fasta + # directly. Emulate that old state without the new private fields. + legacy = Genome.__new__(Genome) + legacy.__dict__.update( + _ensembl=ensembl, + fasta=_DictBackedFasta({"1": "acgt"}) if with_override else None, + _missing_reference_warned=False) + restored = pickle.loads(pickle.dumps(legacy)) + assert restored.reference_name == ensembl.reference_name + assert restored.sequence("1", 1, 4) == ("ACGT" if with_override else "") + assert "FASTA" in repr(restored) + assert Genome(restored).sequence("1", 1, 4) == restored.sequence("1", 1, 4) + restored.close() + # Legacy explicit readers are borrowed, like caller-provided readers. + assert restored.sequence("1", 1, 4) == ("ACGT" if with_override else "") + + def test_genome_with_integer_fasta_raises_typeerror(ensembl): with pytest.raises(TypeError, match="pyfaidx-style protocol"): Genome(ensembl, fasta=42, verify=False) diff --git a/varcode/genome.py b/varcode/genome.py index daa6ed42..d4af3854 100644 --- a/varcode/genome.py +++ b/varcode/genome.py @@ -186,6 +186,13 @@ def close(self): self._fasta_override.close() self._owns_fasta = False + def __setstate__(self, state): + """Restore wrappers saved before native DNA inheritance as borrowers.""" + self.__dict__.update(state) + if "fasta" in state: + self._fasta_override = self.__dict__.pop("fasta") + self._owns_fasta = False + def __getattr__(self, name): """Delegate everything else to the wrapped pyensembl Genome.