INNER CODE UNIT · Python
generate_sequence_features_single
BigDataBiology/SemiBin · SemiBin/main.py:798
def generate_sequence_features_single(logger, contig_fasta,
bams, binned_length,
must_link_threshold, num_process, output, abundances, only_kmer=False):
"""
Generate data.csv and data_split.csv for training and clustering of single-sample and co-assembly binning mode.
data.csv has the features(kmer and abundance) for original contigs.
data_split.csv has the features(kmer and abundace) for contigs that are breaked up as must-link pair.
"""
import pandas as pd
if bams is None and abundances is None and not only_kmer:
logger.error(
"You need to specify input BAM files or abundance files to calculate coverage features.")
sys.exit(1)
if (bams is not None or abundances is not None) and only_kmer:
logger.info('We will only calculate k-mer features.')