INNER CODE UNIT · Python

hash_sequences

nickjcroucher/gubbins · python/gubbins/PreProcessFasta.py:18

    def hash_sequences(self):
        sequence_hash_to_taxa = defaultdict(list)
        with open(self.input_filename) as input_handle:
            alignments = AlignIO.parse(input_handle, "fasta")
            for alignment in alignments:
                for record in alignment:
                    sequence_hash = hashlib.md5()
                    sequence_hash.update(str(record.seq).encode('utf-8'))
                    hash_of_sequence = sequence_hash.digest()
                    sequence_hash_to_taxa[hash_of_sequence].append(record.id)

                    if self.verbose:
                        print("Sample " + str(record.id) + " has a hash of " + str(hash_of_sequence))
        input_handle.close()
        return sequence_hash_to_taxa

    def calculate_sequences_missing_data_percentage(self):
        sequences_to_missing_data = {}

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…