INNER CODE UNIT · Python
hash_sequences
nickjcroucher/gubbins · python/gubbins/PreProcessFasta.py:18
def hash_sequences(self):
sequence_hash_to_taxa = defaultdict(list)
with open(self.input_filename) as input_handle:
alignments = AlignIO.parse(input_handle, "fasta")
for alignment in alignments:
for record in alignment:
sequence_hash = hashlib.md5()
sequence_hash.update(str(record.seq).encode('utf-8'))
hash_of_sequence = sequence_hash.digest()
sequence_hash_to_taxa[hash_of_sequence].append(record.id)
if self.verbose:
print("Sample " + str(record.id) + " has a hash of " + str(hash_of_sequence))
input_handle.close()
return sequence_hash_to_taxa
def calculate_sequences_missing_data_percentage(self):
sequences_to_missing_data = {}