{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nAnalyse train data, in particular desription and genes. \nVersion 2 - added preparing saving features \n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-17T17:19:09.398932Z","iopub.execute_input":"2023-05-17T17:19:09.399375Z","iopub.status.idle":"2023-05-17T17:19:09.417005Z","shell.execute_reply.started":"2023-05-17T17:19:09.399334Z","shell.execute_reply":"2023-05-17T17:19:09.415534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:09.419939Z","iopub.execute_input":"2023-05-17T17:19:09.421246Z","iopub.status.idle":"2023-05-17T17:19:09.434849Z","shell.execute_reply.started":"2023-05-17T17:19:09.421172Z","shell.execute_reply":"2023-05-17T17:19:09.433572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \nfrom Bio import SeqIO\n\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = SeqIO.parse(fn, \"fasta\")\nl = []\nfor seq in sequences:\n    l.append(len(seq.seq))\ndisplay( pd.Series(l).describe()  )\nprint()\n\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\n# print(\"Sequence example:\\n\\n\", next(iter(SeqIO.parse(fn, \"fasta\"))))\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_seq_id_test = [seq.id for seq in sequences ]\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_seq_len = [ len(seq.seq) for seq in sequences ]\nprint(len(list_seq_id) , list_seq_id[:10] )\nsr = pd.Series(list_seq_len).value_counts().sort_index()\nfig = plt.figure(figsize = (20,3 ))\nplt.plot(sr.head(1500))\nplt.title('Test data',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\nfig = plt.figure(figsize = (20,3 ))\nplt.plot(sr.head(200))\nplt.title('Test data',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\nprint( (np.array(list_seq_len)<60).sum() )\npd.Series(list_seq_len).describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:09.436491Z","iopub.execute_input":"2023-05-17T17:19:09.437147Z","iopub.status.idle":"2023-05-17T17:19:17.23559Z","shell.execute_reply.started":"2023-05-17T17:19:09.437108Z","shell.execute_reply":"2023-05-17T17:19:17.234041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main - create dataframe with info on train ","metadata":{}},{"cell_type":"code","source":"%%time \nfrom Bio import SeqIO\nimport time\nt0 = time.time()\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = SeqIO.parse(fn, \"fasta\")\nl1=[]; l2=[]; l3=[]; l4 = []; l5 = []\nfor seq in sequences:\n#     df_seq.loc[IX, 'id'] = \n    l1.append(seq.id)\n#     df_seq.loc[IX, 'length'] = \n    l2.append( len(seq.seq) )\n#     df_seq.loc[IX, 'description'] = \n    l3.append( seq.description.split('|')[-1] )\n#     df_seq.loc[IX, 'dbase'] = \n    l4.append( seq.description.split(' ')[1].split('|')[0] )\n#     df_seq.loc[IX, 'sequence'] = \n    l5.append( str(seq.seq) )\n#     IX += 1 \n\nprint('%.3f'%(time.time() - t0  ) )\n\ndf_seq = pd.DataFrame(data = [l1,l2,l3,l4,l5 ]).T\nprint('%.3f'%(time.time() - t0  ) )\ndf_seq.columns = ['id', 'length', 'description', 'dbase','sequence']\ndf_seq['length'] =  df_seq['length'].astype(int)\ndf_seq = df_seq.reset_index()\ndf_seq = df_seq.set_index('id')\n\n### ------------------------ Fragment -------------------------------------------------\n\ndf_seq['in test'] = (df_seq.index.isin( list_seq_id_test ))\n\n\n\n### ------------------------ Fragment -------------------------------------------------\n\nl = ['Fragment'.lower() in t.lower() for t in l3] \ndf_seq['fragment'] = l\n\n\n#### ------- organism - extracted from \"OX\" (if no - from the second term) --------------------------------\nl = []\nfor t in l3:\n    if 'OS=' in t:\n        s = t.split('OS=')[1].split('OX=')[0]\n        if s[-1] == ' ': s = s[:-1]\n    elif 'HUMAN' in t: # Manually checked that only for humans it can happen\n        s = 'Homo sapiens'\n    else: \n        # that cannot happen - manually checked \n        print('Problem !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!')\n        s = t.split('_')[1].split(' ')[0]\n    l.append(s)\ndf_seq['organism'] = l\nprint(len(s), l[:30])\n\n#### ------- organism symbol ------------ ----------------\n# Fails on : 'CAB85509.1 OS=Arabidopsis thaliana OX=3702 GN=F8F6_100 PE=4 SV=1'\n# df_seq['organism symbol'] = [t.split(' ')[0].split('_')[1] if '_' in t else 'None' for t in l3 ]\n# print(len(s), l[:30])\n\n\n#### ------- organism taxon id ---------------- ----------------\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv'\ndf = pd.read_csv(fn, sep = '\\t', index_col = 0)\ndf_seq = df_seq.join(df, how = 'left' )\n\n\n##### ---------------- gene ------------------------------------------------------------------------------------------------ \nl = []; ll = []\nfor t in l3:\n    if 'GN=' in t:\n        s = t.split('GN=')[-1].split(' ')[0]\n    else:\n        s = t.split('_')[0]\n    l.append(s); ll.append(s.lower() )\nprint(len(l), l[:30]) \ndf_seq['gene name'] = l\ndf_seq['gene name lower'] = ll\n\n\n##### ---------------- PE SV ------------------------------------------------------------------------------------------------ \nl = [ int(t.split('SV=')[-1]) if 'SV=' in t else 0  for t in l3]\ndf_seq['sequence version'] = l\n    \nl = [ int(t.split('PE=')[1].split(' ')[0] ) if 'PE=' in t else 0 for t in l3]\ndf_seq['protein existence'] = l\n\n\n##### ---------------- Some analysis ------------------------------------------------------------------------------------------------ \n\nprint(df_seq.shape )\ndisplay(df_seq['length'].describe() )    \nprint();\ndisplay(df_seq['organism'].value_counts().head(30) )    \nprint();\ndisplay(df_seq['dbase'].value_counts().head(30) )    \nprint();\nprint(df_seq.shape )\ndisplay(df_seq )\n\nprint( df_seq['fragment'].value_counts() )\ndf_seq.groupby( 'fragment' )['length'].mean()\n\n##### ------------------ Brief Look on frequent genes ----------------------------\nprint()\nv = df_seq['gene name lower'].value_counts()\ndisplay( v.head(5) )\nm = df_seq['gene name lower'] == 'ank2'\nprint(df_seq[m]['organism'].value_counts()  )\nfig = plt.figure(figsize = (20, 3))\nax = plt.plot(v.iloc[0:75], '*-')\nplt.xticks(rotation = 90, fontsize = 15)\nplt.show()\nv.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:17.237422Z","iopub.execute_input":"2023-05-17T17:19:17.237882Z","iopub.status.idle":"2023-05-17T17:19:30.72757Z","shell.execute_reply.started":"2023-05-17T17:19:17.237815Z","shell.execute_reply":"2023-05-17T17:19:30.726028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq.loc['A0A8I6GHU0','description']","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.732465Z","iopub.execute_input":"2023-05-17T17:19:30.733665Z","iopub.status.idle":"2023-05-17T17:19:30.743162Z","shell.execute_reply.started":"2023-05-17T17:19:30.733609Z","shell.execute_reply":"2023-05-17T17:19:30.741643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Protein existence and sequence version and fragment\n\nhttps://www.uniprot.org/help/protein_existence\n\n    1. Experimental evidence at protein level\n    2. Experimental evidence at transcript level\n    3. Protein inferred from homology\n    4. Protein predicted\n    5. Protein uncertain\n    \n    0 - not found in train fasta file \n    ","metadata":{}},{"cell_type":"code","source":"df_seq['protein existence'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.744638Z","iopub.execute_input":"2023-05-17T17:19:30.745071Z","iopub.status.idle":"2023-05-17T17:19:30.762008Z","shell.execute_reply.started":"2023-05-17T17:19:30.745031Z","shell.execute_reply":"2023-05-17T17:19:30.760652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq['sequence version'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.763785Z","iopub.execute_input":"2023-05-17T17:19:30.764196Z","iopub.status.idle":"2023-05-17T17:19:30.789253Z","shell.execute_reply.started":"2023-05-17T17:19:30.764157Z","shell.execute_reply":"2023-05-17T17:19:30.787737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq['fragment'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.791176Z","iopub.execute_input":"2023-05-17T17:19:30.791587Z","iopub.status.idle":"2023-05-17T17:19:30.809718Z","shell.execute_reply.started":"2023-05-17T17:19:30.791546Z","shell.execute_reply":"2023-05-17T17:19:30.808371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq['in test'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.811481Z","iopub.execute_input":"2023-05-17T17:19:30.811899Z","iopub.status.idle":"2023-05-17T17:19:30.830476Z","shell.execute_reply.started":"2023-05-17T17:19:30.81185Z","shell.execute_reply":"2023-05-17T17:19:30.829001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Top frequent genes ","metadata":{}},{"cell_type":"code","source":"m = df_seq['organism'] == 'Homo sapiens'\nprint( m.sum() )\ndf_seq[m]['gene name lower'].nunique()\ndf_seq[m].groupby('gene name lower')['gene name lower'].count().sort_values(ascending = False).head(20)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.832287Z","iopub.execute_input":"2023-05-17T17:19:30.833379Z","iopub.status.idle":"2023-05-17T17:19:30.953659Z","shell.execute_reply.started":"2023-05-17T17:19:30.833332Z","shell.execute_reply":"2023-05-17T17:19:30.952635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print()\nv = df_seq['gene name lower'].value_counts()\ndisplay( v.head(5) )\nm = df_seq['gene name lower'] == 'ank2'\nprint(df_seq[m]['organism'].value_counts()  )\nfig = plt.figure(figsize = (20, 3))\nax = plt.plot(v.iloc[0:75], '*-')\nplt.xticks(rotation = 90, fontsize = 15)\nplt.show()\nv.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:30.954998Z","iopub.execute_input":"2023-05-17T17:19:30.956165Z","iopub.status.idle":"2023-05-17T17:19:32.01852Z","shell.execute_reply.started":"2023-05-17T17:19:30.956123Z","shell.execute_reply":"2023-05-17T17:19:32.017245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( (v==1).sum(), (v==2).sum() ) \nfor t in range(1,20):\n    print(t, (v>=t).sum() )","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.020175Z","iopub.execute_input":"2023-05-17T17:19:32.020533Z","iopub.status.idle":"2023-05-17T17:19:32.039627Z","shell.execute_reply.started":"2023-05-17T17:19:32.020497Z","shell.execute_reply":"2023-05-17T17:19:32.037925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('max_colwidth', 200)\npd.set_option('display.max_rows', 100)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.041151Z","iopub.execute_input":"2023-05-17T17:19:32.041759Z","iopub.status.idle":"2023-05-17T17:19:32.048682Z","shell.execute_reply.started":"2023-05-17T17:19:32.041572Z","shell.execute_reply":"2023-05-17T17:19:32.047266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.054814Z","iopub.execute_input":"2023-05-17T17:19:32.055324Z","iopub.status.idle":"2023-05-17T17:19:32.068056Z","shell.execute_reply.started":"2023-05-17T17:19:32.055279Z","shell.execute_reply":"2023-05-17T17:19:32.06671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_seq['gene name lower'] == 'ank2'\nprint(m.sum())\nm2 = df_seq['organism'] == 'Homo sapiens'\nprint( (m&m2).sum() )\ndict(df_seq[m&m2]['description'] )\ndf_seq[m&m2][['index', 'length', 'description', 'dbase',  'organism',\n       'taxonomyID', 'gene name', 'gene name lower','in test']].head(100)","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.069765Z","iopub.execute_input":"2023-05-17T17:19:32.070306Z","iopub.status.idle":"2023-05-17T17:19:32.20308Z","shell.execute_reply.started":"2023-05-17T17:19:32.070257Z","shell.execute_reply":"2023-05-17T17:19:32.201748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m2 = df_seq['organism'] == 'Rattus norvegicus'\ndict(df_seq[m&m2]['description'] )\ndf_seq[m&m2][['index', 'length', 'description', 'dbase',  'organism',\n       'taxonomyID', 'gene name', 'gene name lower', 'in test' ]]","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.204867Z","iopub.execute_input":"2023-05-17T17:19:32.205367Z","iopub.status.idle":"2023-05-17T17:19:32.256023Z","shell.execute_reply.started":"2023-05-17T17:19:32.205316Z","shell.execute_reply":"2023-05-17T17:19:32.254484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lengths ","metadata":{}},{"cell_type":"code","source":"m = df_seq['length'] < 60\nprint( m.sum() )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.258067Z","iopub.execute_input":"2023-05-17T17:19:32.258671Z","iopub.status.idle":"2023-05-17T17:19:32.268006Z","shell.execute_reply.started":"2023-05-17T17:19:32.258618Z","shell.execute_reply":"2023-05-17T17:19:32.26663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_seq['length'] < 10\nprint( df_seq[m].index.tolist() )\nprint( df_seq[m]['gene name'].tolist() )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.270287Z","iopub.execute_input":"2023-05-17T17:19:32.270958Z","iopub.status.idle":"2023-05-17T17:19:32.286827Z","shell.execute_reply.started":"2023-05-17T17:19:32.270901Z","shell.execute_reply":"2023-05-17T17:19:32.28554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr = df_seq['length'].value_counts().sort_index()\n\nfig = plt.figure(figsize = (20,3) )\nplt.plot(sr.head(1500))\nplt.title('Train data - first 1500',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\n\nfig = plt.figure(figsize = (20,3) )\nplt.plot(sr.head(150))\nplt.title('Train data - first 150 ',fontsize = 20 )\nplt.xlabel('Sequence length',fontsize = 20)\nplt.ylabel('Count',fontsize = 20)\nplt.show()\n\nm = df_seq['length'] < 60\nprint('count length < 60 : ', m.sum() )\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.289272Z","iopub.execute_input":"2023-05-17T17:19:32.289979Z","iopub.status.idle":"2023-05-17T17:19:32.839704Z","shell.execute_reply.started":"2023-05-17T17:19:32.289897Z","shell.execute_reply":"2023-05-17T17:19:32.838247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in range(8):\n    m = df_seq['length'] == k\n    if m.sum() == 0 : continue \n    print(k, m.sum() )\n#     print(dict(df_seq2[m]['description']))\n    print(dict(df_seq[m]['organism']))    \n    display(df_seq[m])\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:19:32.841395Z","iopub.execute_input":"2023-05-17T17:19:32.841933Z","iopub.status.idle":"2023-05-17T17:19:32.92227Z","shell.execute_reply.started":"2023-05-17T17:19:32.841875Z","shell.execute_reply":"2023-05-17T17:19:32.920894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_seq.select_dtypes('number').corr()\nnumerics = ['bool','float',  'int16', 'int32', 'int64', 'float16', 'float32', 'float64']\ndf.select_dtypes(include=numerics)\nnewdf = pd.concat([ df_seq.select_dtypes('number'), df_seq.select_dtypes('bool') ], axis = 1 )   #  \nnewdf\ndisplay(newdf.corr().round(1))\ndisplay(newdf[['index', 'protein existence' ,'in test','fragment']].corr().round(1))\nsns.clustermap( newdf.corr().abs() )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:20:29.980365Z","iopub.execute_input":"2023-05-17T17:20:29.981001Z","iopub.status.idle":"2023-05-17T17:20:31.005297Z","shell.execute_reply.started":"2023-05-17T17:20:29.980944Z","shell.execute_reply":"2023-05-17T17:20:31.003927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add targets and analyse ","metadata":{}},{"cell_type":"code","source":"# fn =  str(path) + '/Train/train_terms.tsv'\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv'\ndf = pd.read_csv(fn , sep = '\\t')\nprint(df.shape)\ndisplay(df)\nsr2 = df.groupby('EntryID')['term'].apply(lambda x: len(list(x)))\nsr2.name = 'Terms Count'\nprint(sr2.shape)\ndisplay(sr2.head(2))\n\ndf_seq2 = df_seq.join(sr2, how = 'left')\nprint(df_seq2.shape)\ndisplay(df_seq2.head(2))\n\n# df_seq.select_dtypes('number').corr()\n# numerics = ['bool','float',  'int16', 'int32', 'int64', 'float16', 'float32', 'float64']\ndf.select_dtypes(include=numerics)\nnewdf = pd.concat([ df_seq2.select_dtypes('number'), df_seq2.select_dtypes('bool') ], axis = 1 )   #  \nnewdf\ndisplay(newdf.corr().round(1))\n# display(newdf[['index', 'protein existence' ,'in test','fragment']].corr().round(1))\nsns.clustermap( newdf.corr().abs() )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:20:44.557293Z","iopub.execute_input":"2023-05-17T17:20:44.558407Z","iopub.status.idle":"2023-05-17T17:20:55.234824Z","shell.execute_reply.started":"2023-05-17T17:20:44.558224Z","shell.execute_reply":"2023-05-17T17:20:55.233017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['aspect'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:20:55.238204Z","iopub.execute_input":"2023-05-17T17:20:55.238639Z","iopub.status.idle":"2023-05-17T17:20:56.171645Z","shell.execute_reply.started":"2023-05-17T17:20:55.238596Z","shell.execute_reply":"2023-05-17T17:20:56.169491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['aspect'].unique() )\nfor a in ['BPO',  'CCO' , 'MFO']:\n    m = df['aspect'] == a\n    sr2 = df[m].groupby('EntryID')['term'].apply(lambda x: len(list(x)))\n    sr2.name = 'Terms Count ' + a\n    sr2 = sr2.fillna(0)\n    print(sr2.shape)\n#     display(sr2.head(2))\n\n    df_seq2 = df_seq2.join(sr2, how = 'left')\n    df_seq2[sr2.name] = df_seq2[sr2.name].fillna(0) # df_seq.join(sr2, how = 'left')\n    print(df_seq2.shape)\n#     display(df_seq2.head(2))\n\n# df_seq.select_dtypes('number').corr()\n# numerics = ['bool','float',  'int16', 'int32', 'int64', 'float16', 'float32', 'float64']\ndf.select_dtypes(include=numerics)\nnewdf = pd.concat([ df_seq2.select_dtypes('number'), df_seq2.select_dtypes('bool') ], axis = 1 )   #  \nnewdf\ndisplay(newdf.corr().round(1))\n# display(newdf[['index', 'protein existence' ,'in test','fragment']].corr().round(1))\nsns.clustermap( newdf.corr().abs() )\nplt.show()    ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:20:56.175119Z","iopub.execute_input":"2023-05-17T17:20:56.175592Z","iopub.status.idle":"2023-05-17T17:21:10.368286Z","shell.execute_reply.started":"2023-05-17T17:20:56.175552Z","shell.execute_reply":"2023-05-17T17:21:10.366399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_seq2.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:21:10.37Z","iopub.execute_input":"2023-05-17T17:21:10.370827Z","iopub.status.idle":"2023-05-17T17:21:10.469633Z","shell.execute_reply.started":"2023-05-17T17:21:10.370771Z","shell.execute_reply":"2023-05-17T17:21:10.467932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = []; i = 0\nfor t in l3:\n    if 'OX=' in t:\n        s1 = t.split('OS=')[0]\n        p = s1.find(' ')\n        s = s1[(p+1):]\n        l.append(s); i += 1\n    else:\n        l.append('')\nprint(i,len(l), l[:10])\ndf_seq2['Description Cleaned'] = l\nl =  [len(t) for t in l]\ndf_seq2['len description cleaned'] = l\ndisplay(df_seq2.head(3))","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:26:41.981846Z","iopub.execute_input":"2023-05-17T17:26:41.982394Z","iopub.status.idle":"2023-05-17T17:26:42.341494Z","shell.execute_reply.started":"2023-05-17T17:26:41.982345Z","shell.execute_reply":"2023-05-17T17:26:42.33992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nprint(df_seq2.memory_usage().sum()/1e6 )\ndf_seq2.to_csv('df_train_eda.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:27:00.450563Z","iopub.execute_input":"2023-05-17T17:27:00.451035Z","iopub.status.idle":"2023-05-17T17:27:06.001692Z","shell.execute_reply.started":"2023-05-17T17:27:00.450994Z","shell.execute_reply":"2023-05-17T17:27:06.000329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save features","metadata":{}},{"cell_type":"code","source":"df = df_seq2\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:21:39.355675Z","iopub.execute_input":"2023-05-17T17:21:39.3561Z","iopub.status.idle":"2023-05-17T17:21:39.366691Z","shell.execute_reply.started":"2023-05-17T17:21:39.356063Z","shell.execute_reply":"2023-05-17T17:21:39.364967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nd['sequence version'] = df['sequence version'].values\nd['protein existence'] = df['protein existence'].values\ndisplay(d.describe() )\nd.to_csv('train_features01_SV_PE.csv')\nnp.save( 'train_features01_SV_PE', d.values ) \nd","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:22:27.262365Z","iopub.execute_input":"2023-05-17T17:22:27.262863Z","iopub.status.idle":"2023-05-17T17:22:27.664101Z","shell.execute_reply.started":"2023-05-17T17:22:27.262807Z","shell.execute_reply":"2023-05-17T17:22:27.662889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['gene name','gene name lower' ]\nfor col in l:\n    d[col] = df[col].values\ndisplay(d.describe() )\nd.to_csv('train_features02_genes.csv')\nnp.save( 'train_features02_genes', d.values ) \nd","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:23:17.746427Z","iopub.execute_input":"2023-05-17T17:23:17.746919Z","iopub.status.idle":"2023-05-17T17:23:18.558359Z","shell.execute_reply.started":"2023-05-17T17:23:17.746865Z","shell.execute_reply":"2023-05-17T17:23:18.557053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['description']","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:28:27.235678Z","iopub.execute_input":"2023-05-17T17:28:27.236177Z","iopub.status.idle":"2023-05-17T17:28:27.247497Z","shell.execute_reply.started":"2023-05-17T17:28:27.236134Z","shell.execute_reply":"2023-05-17T17:28:27.246114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['Description Cleaned','description']\nfor col in l:\n    d[col] = df[col].values\nd.columns = ['Description Cleaned','Description' ]    \ndisplay(d.describe() )\nd.to_csv('train_features03_description.csv')\nnp.save( 'train_features03_description', d.values ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:29:15.73601Z","iopub.execute_input":"2023-05-17T17:29:15.736469Z","iopub.status.idle":"2023-05-17T17:29:17.314644Z","shell.execute_reply.started":"2023-05-17T17:29:15.736429Z","shell.execute_reply":"2023-05-17T17:29:17.313185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.DataFrame()\nl = ['dbase','fragment']\nfor col in l:\n    d[col] = df[col].values\ndisplay(d.describe() )\nd.to_csv('train_features04_dbase_fragment.csv')\nnp.save( 'train_features04_dbase_fragment', d.values ) ","metadata":{"execution":{"iopub.status.busy":"2023-05-17T17:31:14.773294Z","iopub.execute_input":"2023-05-17T17:31:14.773734Z","iopub.status.idle":"2023-05-17T17:31:15.302481Z","shell.execute_reply.started":"2023-05-17T17:31:14.773695Z","shell.execute_reply":"2023-05-17T17:31:15.300999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}