INNER CODE UNIT · Python
pul_section
linnabrown/run_dbcan · dbcan/cli/cgc_process_json.py:72
def pul_section(self):
"""
Processes the dataframe and yields structured genetic data.
This method groups the dataframe by 'cgc_id' and processes each group to yield a dictionary containing detailed genetic information for each 'cgc_id'.
Yields
------
dict: A dictionary containing genetic and protein information structured by 'cgc_id'.
"""
for (cgc_id), df_pul_grouped in self.df.groupby("cgc_id"):
datalist = list(self.cluster_section(df_pul_grouped))
gene_str = self.extract_gs(datalist)
yield {
cgc_id: {
"changelog": [],
"Cluster_ID": cgc_id,
"Gene_String": gene_str,