diff --git a/src/datasets/packaged_modules/genbank/genbank.py b/src/datasets/packaged_modules/genbank/genbank.py index ff014ff954d..e508e57d08d 100644 --- a/src/datasets/packaged_modules/genbank/genbank.py +++ b/src/datasets/packaged_modules/genbank/genbank.py @@ -310,10 +310,17 @@ def _parse_location_node(cls, text: str) -> dict: located = [part for part in parts if "start" in part] if located: location["parts"] = [[part["start"], part["end"]] for part in located] - location["start"] = located[0]["start"] - location["end"] = located[-1]["end"] - location["start_partial"] = located[0]["start_partial"] - location["end_partial"] = located[-1]["end_partial"] + # Parts are listed in the file's (transcription) order, which is descending + # genomic order for minus-strand features (NCBI/Biopython write a two-exon + # minus-strand CDS as ``complement(join(..,..))``). + # The aggregate span is therefore min-start/max-end over all parts, matching + # Biopython's CompoundLocation.start/.end, not the first/last part positionally. + start_part = min(located, key=lambda part: part["start"]) + end_part = max(located, key=lambda part: part["end"]) + location["start"] = start_part["start"] + location["end"] = end_part["end"] + location["start_partial"] = start_part["start_partial"] + location["end_partial"] = end_part["end_partial"] return location location = {"strand": 1} diff --git a/tests/packaged_modules/test_genbank.py b/tests/packaged_modules/test_genbank.py index c5eb9f7c2a8..346388d4afd 100644 --- a/tests/packaged_modules/test_genbank.py +++ b/tests/packaged_modules/test_genbank.py @@ -1098,3 +1098,18 @@ def test_genbank_features_subset_selects_columns(tmp_path): features = Features({"locus_name": Value("string"), "sequence": Value("large_string")}) table = next(iter(GenBank(features=features)._generate_tables([[str(filename)]])))[1] assert table.column_names == ["locus_name", "sequence"] + + +def test_genbank_minus_strand_multi_exon_join_span_is_min_max(tmp_path): + """A minus-strand multi-exon feature is written by NCBI/Biopython as + ``complement(join(..,..))`` (exons in transcription order, + i.e. descending genomic order). The aggregate span must still be min-start/max-end, + as Biopython's CompoundLocation reports, not first-part-start/last-part-end.""" + _, features = _parse_one(tmp_path, _HDR + " CDS complement(join(121..180,11..60))\n" + _ORIGIN) + location = features[0]["location"] + assert location["strand"] == -1 + assert location["operator"] == "join" + assert location["parts"] == [[121, 180], [11, 60]] + assert location["start"] == 11 + assert location["end"] == 180 + assert location["start"] <= location["end"]