{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nWe generate features for CAFA5 challenge based on the following simple idea:\n\n    Take some N (say 100) protein sequences (for example choose them randomly).\n    And for ALL protein sequences calculate the Levenstein distances to these N selected seqeunces.\n    So one gets N features - distances to those N selected ones \n\nComputed results for N=100, 1000 (versrions 1,2 of the notebook) stored in the dataset: \nhttps://www.kaggle.com/datasets/alexandervc/cafa5-features-etc","metadata":{}},{"cell_type":"markdown","source":"# Key params","metadata":{}},{"cell_type":"code","source":"N_features_to_generate = 1000 ","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:15:57.488802Z","iopub.execute_input":"2023-04-26T20:15:57.489203Z","iopub.status.idle":"2023-04-26T20:15:57.517268Z","shell.execute_reply.started":"2023-04-26T20:15:57.489167Z","shell.execute_reply":"2023-04-26T20:15:57.51591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Install  python-Levenshtein \n\nHere is CPU way. \n\nFor the GPU way - see  Chris Deotte : https://www.kaggle.com/code/cdeotte/train-data-contains-mutations-like-test-data \n\nhttps://docs.rapids.ai/api/cudf/stable/api_docs/api/cudf.core.column.string.stringmethods.edit_distance_matrix/#cudf.core.column.string.StringMethods.edit_distance_matrix ","metadata":{}},{"cell_type":"code","source":"!pip install python-Levenshtein","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:15:57.519536Z","iopub.execute_input":"2023-04-26T20:15:57.520704Z","iopub.status.idle":"2023-04-26T20:16:11.357528Z","shell.execute_reply.started":"2023-04-26T20:15:57.52066Z","shell.execute_reply":"2023-04-26T20:16:11.355884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Levenshtein import distance\nedit_dist = distance(\"ah\", \"aho\")\nedit_dist","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:16:11.35972Z","iopub.execute_input":"2023-04-26T20:16:11.360132Z","iopub.status.idle":"2023-04-26T20:16:11.456832Z","shell.execute_reply.started":"2023-04-26T20:16:11.360094Z","shell.execute_reply":"2023-04-26T20:16:11.455703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start  = time.time()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-26T20:16:11.459705Z","iopub.execute_input":"2023-04-26T20:16:11.460074Z","iopub.status.idle":"2023-04-26T20:16:11.477474Z","shell.execute_reply.started":"2023-04-26T20:16:11.46004Z","shell.execute_reply":"2023-04-26T20:16:11.475919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load protein data ","metadata":{}},{"cell_type":"markdown","source":"## Example working fasta data","metadata":{}},{"cell_type":"code","source":"%%time \nfrom Bio import SeqIO\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nprint(\"Sequence example:\\n\", next(iter(SeqIO.parse(fn, \"fasta\"))))\nsequences = SeqIO.parse(fn, \"fasta\")\nnum_sequences = sum(1 for seq in sequences)\nprint()\nprint(\"Number of sequences in train:\", num_sequences)\n\nsequences = SeqIO.parse(fn, \"fasta\")\nseq = next(iter(sequences))\ngb = seq\nprint('\\nParsed Example:\\nLength of Sequence:', len(gb.seq), '\\nRecord ID:', gb.id,  'nName:', gb.name, 'Description:', gb.description,'Number of Annotations:', len(gb.annotations),\n      'Number of Features:', len(gb.features) )\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:16:11.479455Z","iopub.execute_input":"2023-04-26T20:16:11.479843Z","iopub.status.idle":"2023-04-26T20:16:13.831658Z","shell.execute_reply.started":"2023-04-26T20:16:11.479808Z","shell.execute_reply":"2023-04-26T20:16:13.830208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature generation  ","metadata":{}},{"cell_type":"code","source":"%%time \nimport matplotlib.pyplot as plt\nfrom Levenshtein import distance\n\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_seq = [str(seq.seq) for seq in sequences ]\nsequences = SeqIO.parse(fn, \"fasta\")\nlist_ids = [seq.id for seq in sequences ]\nprint(len(list_seq), list_ids[:10])\n\nN = N_features_to_generate\nlist_selected_indexes = [np.random.randint(0, len(list_seq ) ) for i in range(N) ] \n#list_selected_indexes =  [81609, 98401, 140030, 24786, 7949, 45596, 49783, 16145, 88492, 81662] # One possible random choice saved \nprint(len(list_selected_indexes), list_selected_indexes[:10])\n\n# list_selected_indexes =  [81609, 98401, 140030, 24786, 7949, 45596, 49783, 16145, 88492, 81662]\nlist_selected_seq = [ list_seq[i] for i in list_selected_indexes  ]\nprint(len(list_selected_seq), list_selected_seq[:2])\nlist_selected_ids = [ list_ids[i] for i in list_selected_indexes  ]\nprint(len(list_selected_ids), list_selected_ids[:10])\n\n\ndict_df_results = {}\nfor str_train_or_test in ['train','test']:\n    print()\n    if str_train_or_test == 'train':\n        fn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\n    else:\n        fn = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\n        \n    sequences = SeqIO.parse(fn, \"fasta\")\n    list_seq = [str(seq.seq) for seq in sequences ]\n    sequences = SeqIO.parse(fn, \"fasta\")\n    list_ids = [seq.id for seq in sequences ]\n    print(len(list_seq), list_ids[:10])\n    \n\n    df_distance_results = pd.DataFrame()\n\n    for i,s1 in enumerate(list_selected_seq):\n        print(i)\n        print(i,len(s1),  list_selected_ids[i]  ) ; \n        l = [ distance(s1,list_seq[j]) for j in range(len(list_seq))  ]\n        print(i, len(l)) \n        df_distance_results[i] = l\n        \n        if i < 10: # Show some statistcs \n            l2 = df_distance_results[i]\n            l2 = l2[l2 < 2*len(s1)]\n            plt.hist(l2, bins = 100)\n            plt.title(str(i ), fontsize = 20) \n            plt.show()\n            display( pd.Series(l2).describe() )\n\n            print( l2.value_counts().sort_index()[:20]     )\n\n    #     break \n\n    df_distance_results = df_distance_results.astype(int)   \n    df_distance_results.index = list_ids\n    i = df_distance_results.shape[1]; \n    df_distance_results.columns = list_selected_ids[:i]\n    print(df_distance_results.shape  )\n    display(df_distance_results)\n    display(df_distance_results.describe() )\n    \n    dict_df_results[ str_train_or_test ] = df_distance_results\n\n    \n    f2 = 'df_Levenshtein_distance_'+str(df_distance_results.shape[1])+'_features_'+str_train_or_test + '.csv'\n    print( df_distance_results.memory_usage().sum() )\n    df_distance_results.to_csv(f2)\n    print(f2)    ","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:16:13.833687Z","iopub.execute_input":"2023-04-26T20:16:13.834759Z","iopub.status.idle":"2023-04-26T20:30:00.884445Z","shell.execute_reply.started":"2023-04-26T20:16:13.834698Z","shell.execute_reply":"2023-04-26T20:30:00.882717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Show some statistics ","metadata":{}},{"cell_type":"code","source":"for k in dict_df_results.keys():\n    df_distance_results = dict_df_results[k]\n    l = []\n    for col in df_distance_results.columns:\n        mn = df_distance_results[col][ df_distance_results[col] > 0  ].min()\n        l.append(mn)\n    print('Min distances:', l )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:30:00.88682Z","iopub.execute_input":"2023-04-26T20:30:00.887301Z","iopub.status.idle":"2023-04-26T20:30:01.61036Z","shell.execute_reply.started":"2023-04-26T20:30:00.887261Z","shell.execute_reply":"2023-04-26T20:30:01.60833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T20:30:01.612203Z","iopub.execute_input":"2023-04-26T20:30:01.613624Z","iopub.status.idle":"2023-04-26T20:30:01.620076Z","shell.execute_reply.started":"2023-04-26T20:30:01.613567Z","shell.execute_reply":"2023-04-26T20:30:01.618316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}