Skip to content
Merged
Show file tree
Hide file tree
Changes from 4 commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@ jobs:
runs-on: ubuntu-latest
strategy:
matrix:
python-version: ["3.8", "3.9", "3.10", "3.11", "3.12"]
python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"]

steps:
- uses: actions/checkout@v2
Expand Down
2 changes: 1 addition & 1 deletion docs/install.rst
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ The VAtools suite is written for Linux and Mac OS X.
If you are using Windows you will need to set up a
Linux environment, for example by setting up a virtual machine.

VAtools requires Python 3.8 or above. Before running any
VAtools requires Python 3.10 or above. Before running any
installation steps, check the Python version installed on your system:

.. code-block:: none
Expand Down
9 changes: 7 additions & 2 deletions setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,8 +15,9 @@
"ref-transcript-mismatch-reporter = vatools.ref_transcript_mismatch_reporter:main",
]
},
python_requires='>=3.10',
install_requires=[
'vcfpy==0.13.8',
'vcfpy==0.14.2',
'pysam',
'pandas',
'gtfparse==1.3.0',
Expand All @@ -29,7 +30,11 @@
'Intended Audience :: Science/Research',
'Topic :: Scientific/Engineering :: Bio-Informatics',

"Programming Language :: Python :: 3.5",
"Programming Language :: Python :: 3.10",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Programming Language :: Python :: 3.14",

'License :: OSI Approved :: MIT License',
],
Expand Down
24 changes: 24 additions & 0 deletions vatools/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,3 +9,27 @@ def open_maybe_gz(path, mode='r'):
if magic == b'\x1f\x8b':
return gzip.open(path, mode + 't' if 'b' not in mode else mode)
return open(path, mode)

# vcfpy>=0.14 requires every FORMAT field with Number != 1 to have a list
# value (even when data is missing) for every sample in the record. Some tools
# only populate a field for one sample, so these need to backfill an
# empty list for other samples, which will ultimately be inserted in the
# VCF as '.'. If a sample has a pre-existing scalar value for a field that
# used to be declared Number=1 in the input header (e.g. AF) but is now
# being redeclared as multi-value here, we should keep that value, not discard it.
def fill_missing_multi_value_format_fields(entry, header):
for field in entry.FORMAT:
field_info = header.get_format_field_info(field)
if field_info is None or field_info.number == 1:
continue
for call in entry.calls:
if field not in call.data:
call.data[field] = []
elif not isinstance(call.data[field], list):
call.data[field] = [call.data[field]]

# make sure that missing format fields (in all samples) are filled correctly
# before writing the VCF record.
def write_record(entry, vcf_writer):
fill_missing_multi_value_format_fields(entry, vcf_writer.header)
vcf_writer.write_record(entry)
5 changes: 3 additions & 2 deletions vatools/vcf_expression_annotator.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
from collections import OrderedDict
from gtfparse import read_gtf
import logging
from vatools.utils import write_record

def resolve_id_column(args):
if args.format == 'cufflinks':
Expand Down Expand Up @@ -203,7 +204,7 @@ def main(args_input = sys.argv[1:]):
genes = set()
if 'CSQ' not in entry.INFO:
logging.warning("Variant is missing VEP annotation. INFO column doesn't contain CSQ field for variant {}".format(entry))
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue
for transcript in entry.INFO['CSQ']:
for key, value in zip(csq_format, transcript.split('|')):
Expand All @@ -222,7 +223,7 @@ def main(args_input = sys.argv[1:]):
if len(transcript_ids) > 0:
(entry, missing_expressions_count) = add_expressions(entry, is_multi_sample, args.sample_name, df, transcript_ids, 'TX', id_column, expression_column, args.ignore_ensembl_id_version, missing_expressions_count)
entry_count += len(transcript_ids)
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)

vcf_reader.close()
vcf_writer.close()
Expand Down
3 changes: 2 additions & 1 deletion vatools/vcf_genotype_annotator.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import vcfpy
import csv
from collections import OrderedDict
from vatools.utils import write_record

def create_vcf_reader(args):
vcf_reader = vcfpy.Reader.from_path(args.input_vcf)
Expand Down Expand Up @@ -82,7 +83,7 @@ def main(args_input = sys.argv[1:]):
else:
entry.calls = [new_sample_call]
entry.call_for_sample = {call.sample: call for call in entry.calls}
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)

vcf_reader.close()
vcf_writer.close()
Expand Down
26 changes: 8 additions & 18 deletions vatools/vcf_readcount_annotator.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
import csv
from collections import OrderedDict, namedtuple
import logging
from vatools.utils import open_maybe_gz
from vatools.utils import open_maybe_gz, write_record

# parse the input params
def define_parser():
Expand Down Expand Up @@ -336,33 +336,23 @@ def main(args_input = sys.argv[1:]):
reference = entry.REF
alts = entry.ALT

#The NUMBER of the AD and AF fields in the input VCF might be 1,
#but we are now writing R/A number fields which are is-many
#In that case the values of these fields need to be changed
#from a single number to an array or else the writer will throw an error
for sample in vcf_reader.header.samples.names:
sample_data = entry.call_for_sample[sample].data
for field in [count_field, forward_count_field, reverse_count_field, frequency_field]:
if field in sample_data and (not isinstance(sample_data[field], list)):
sample_data[field] = [sample_data[field]]

#If we limit the annotations to only SNVs and the entry contains an InDel, skip it
if args.variant_type == 'snv' and has_indel(entry):
if has_snv(entry):
logging.warning("Running in `snv` variant type mode but VCF entry for chr {} pos {} ref {} alts {} contains both SNVs and InDels. Skipping.".format(chromosome, entry.POS, reference, alts))
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

#If we limit the annotations to only InDels and the entry contains a SNV, skip it
if args.variant_type == 'indel' and has_snv(entry):
if has_indel(entry):
logging.warning("Running in `indel` variant type mode but VCF entry for chr {} pos {} ref {} alts {} contains both SNVs and InDels. Skipping.".format(chromosome, entry.POS, reference, alts))
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

#If the entry contains a complex variant, skip it
if has_complex_variant(entry):
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

(bam_readcount_position, ref_base, var_base) = parse_to_bam_readcount(start, reference, alts[0].serialize(), entry.POS)
Expand All @@ -377,12 +367,12 @@ def main(args_input = sys.argv[1:]):
entry.FORMAT += [count_field]
ads = [0] * (len(alts) + 1)
entry.call_for_sample[sample_name].data[count_field] = ads
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

#Discrepant bam-readcount entries; none of the fields should be written.
if isinstance(brct, list):
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

if 'depth' not in brct:
Expand All @@ -396,7 +386,7 @@ def main(args_input = sys.argv[1:]):
#been a duplicate bam-readcount entry where only the depths matched.
#The only field to write is depth; frequency and count fields should not be written.
if len(brct.keys()) == 1 and list(brct.keys())[0] == 'depth':
vcf_writer.write_record(entry)
write_record(entry, vcf_writer)
continue

primary_brct = brct
Expand Down Expand Up @@ -471,7 +461,7 @@ def main(args_input = sys.argv[1:]):
val = None
add_format_value(entry, sample_name, f.tag, val)

vcf_writer.write_record(entry)
write_record(entry, vcf_writer)

vcf_writer.close()
vcf_reader.close()
Expand Down
Loading