INNER CODE UNIT · Python
remove_duplicate_sequences_and_sequences_missing_too_much_data
nickjcroucher/gubbins · python/gubbins/PreProcessFasta.py:82
def remove_duplicate_sequences_and_sequences_missing_too_much_data(self, output_filename,
remove_identical_sequences=None):
if not remove_identical_sequences:
taxa_to_remove = self.taxa_missing_too_much_data()
else:
taxa_to_remove = self.taxa_of_duplicate_sequences() + self.taxa_missing_too_much_data()
with open(self.input_filename) as input_handle:
alignments = AlignIO.parse(input_handle, "fasta")
output_alignments = []
number_of_included_alignments = 0
for alignment in alignments:
for record in alignment:
if record.id not in taxa_to_remove:
output_alignments.append(record)
number_of_included_alignments += 1
if number_of_included_alignments <= 1: