INNER CODE UNIT · Python

pul_section

linnabrown/run_dbcan · dbcan/cli/cgc_process_json.py:72

    def pul_section(self):
        """
        Processes the dataframe and yields structured genetic data.

        This method groups the dataframe by 'cgc_id' and processes each group to yield a dictionary containing detailed genetic information for each 'cgc_id'.

        Yields
        ------
            dict: A dictionary containing genetic and protein information structured by 'cgc_id'.
        """
        for (cgc_id), df_pul_grouped in self.df.groupby("cgc_id"):
            datalist = list(self.cluster_section(df_pul_grouped))
            gene_str = self.extract_gs(datalist)
            yield {
                cgc_id: {
                    "changelog": [],
                    "Cluster_ID": cgc_id,
                    "Gene_String": gene_str,

View source record →

📰 Research Paper
Loading…
⏳ Fetching content…