INNER CODE UNIT · Python
taxa_of_duplicate_sequences
nickjcroucher/gubbins · python/gubbins/PreProcessFasta.py:70
def taxa_of_duplicate_sequences(self):
taxa_to_remove = []
for sequence_hash, taxa in sorted(self.hash_sequences().items()):
if len(taxa) > 1:
taxon_to_keep = taxa.pop()
for taxon in taxa:
print("Sequences in " + taxon + " and " + taxon_to_keep + " are identical, removing " + taxon +
" from analysis")
taxa_to_remove.append(taxon)
return taxa_to_remove
def remove_duplicate_sequences_and_sequences_missing_too_much_data(self, output_filename,
remove_identical_sequences=None):
if not remove_identical_sequences:
taxa_to_remove = self.taxa_missing_too_much_data()
else: