{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nCreate as save one-hot encoding features  for 31 taxons selected by Anton. \n\nResults saved in CSV files and stored into the Kaggle dataset: https://www.kaggle.com/datasets/alexandervc/cafa5-data-selected.\n\n    /kaggle/input/cafa5-data-selected/features/test_features05_taxons31_onehot.csv\n    /kaggle/input/cafa5-data-selected/features/train_features05_taxons31_onehot.csv\n    For 43k:\n    /kaggle/input/cafa5-data-selected/embeds_43k_Y1850_etc/train_features05_taxons31_onehot_cut43k.csv\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-08T21:14:18.461957Z","iopub.execute_input":"2023-07-08T21:14:18.462372Z","iopub.status.idle":"2023-07-08T21:14:18.487744Z","shell.execute_reply.started":"2023-07-08T21:14:18.462342Z","shell.execute_reply":"2023-07-08T21:14:18.486545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '../input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv'\ndf = pd.read_csv(fn, sep='\\t', error_bad_lines=False, encoding= 'unicode_escape')\ndisplay(df)\n# display(df['Species'].value_counts().head(30) )\nprint(df['Species'].nunique() )","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:56.883331Z","iopub.execute_input":"2023-07-08T21:08:56.883883Z","iopub.status.idle":"2023-07-08T21:08:56.953406Z","shell.execute_reply.started":"2023-07-08T21:08:56.88385Z","shell.execute_reply":"2023-07-08T21:08:56.952321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tax_list = [\n    9606, 3702, 10090, 7955, 7227, 10116, 559292, 6239,\n    284812, 83333, 83332, 44689, 237561, 39947, 9031, 36329,\n    9913, 227321, 8355, 9823, 224308, 330879, 4577, 170187,\n    9615, 99287, 85962, 243232, 287, 235443, 8364\n]\n\ntax_dict = {x: n for (n, x) in enumerate(tax_list)}\ntax_dict","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:56.955431Z","iopub.execute_input":"2023-07-08T21:08:56.956224Z","iopub.status.idle":"2023-07-08T21:08:56.972916Z","shell.execute_reply.started":"2023-07-08T21:08:56.956179Z","shell.execute_reply":"2023-07-08T21:08:56.971595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df['ID'].isin(tax_list)\nprint(m.sum(), len(tax_list))\ndf_tax_names = df[m]\ndf_tax_names = df_tax_names.set_index('ID')\ndf_tax_names","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:56.97649Z","iopub.execute_input":"2023-07-08T21:08:56.977884Z","iopub.status.idle":"2023-07-08T21:08:57.021391Z","shell.execute_reply.started":"2023-07-08T21:08:56.977837Z","shell.execute_reply":"2023-07-08T21:08:57.019811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv'\ndf2 = pd.read_csv(fn, sep = '\\t')\nprint(df2.shape)\ndisplay(df2.head(2))\n","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:57.024102Z","iopub.execute_input":"2023-07-08T21:08:57.025138Z","iopub.status.idle":"2023-07-08T21:08:57.244228Z","shell.execute_reply.started":"2023-07-08T21:08:57.025088Z","shell.execute_reply":"2023-07-08T21:08:57.242928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df2['taxonomyID'].isin(tax_list)\nprint(m.sum() )\ndf2[m]\nprint(len(df2[m]['taxonomyID'].value_counts()))\ndf_tax_freq = df2[m]['taxonomyID'].value_counts().to_frame().join(df_tax_names)\ndf_tax_freq","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:57.246202Z","iopub.execute_input":"2023-07-08T21:08:57.246836Z","iopub.status.idle":"2023-07-08T21:08:57.304095Z","shell.execute_reply.started":"2023-07-08T21:08:57.246791Z","shell.execute_reply":"2023-07-08T21:08:57.30289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom Bio import SeqIO\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nprint(\"Sequence example:\\n\\n\", next(iter(SeqIO.parse(fn, \"fasta\"))))\nprint()\nsequences = SeqIO.parse(fn, \"fasta\")\n\nlist_ids =  [str(seq.id) for seq in sequences ]\nprint(len(list_ids), list_ids[:10])\ndf_train_features05_taxons = pd.DataFrame(index = list_ids, data = np.zeros( (len(list_ids), len(tax_dict)) , dtype = np.float32 ), columns = tax_dict.keys()  )\ndisplay(df_train_features05_taxons.head(2))\n\nsequences = SeqIO.parse(fn, \"fasta\")\nfor I1, seq in enumerate( sequences ):\n    s = seq.description\n#     print(s)\n    if 'OX=' in s: \n        s2 = s.split('OX=')[1].split(' ')[0]\n#         print(s2)\n        s2 = float(s2)\n        if s2 in tax_dict.keys():\n            IX = tax_dict[s2]\n            df_train_features05_taxons.iloc[I1,IX] = 1\n#     print(seq.id)\n    if I1%20_000 == 0: print(I1)\n\ndf_train_features05_taxons.describe()    ","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:08:57.306116Z","iopub.execute_input":"2023-07-08T21:08:57.306657Z","iopub.status.idle":"2023-07-08T21:09:13.688842Z","shell.execute_reply.started":"2023-07-08T21:08:57.306609Z","shell.execute_reply":"2023-07-08T21:09:13.687544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_features05_taxons.memory_usage().sum()/1e6","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:09:13.690829Z","iopub.execute_input":"2023-07-08T21:09:13.691282Z","iopub.status.idle":"2023-07-08T21:09:13.704541Z","shell.execute_reply.started":"2023-07-08T21:09:13.69124Z","shell.execute_reply":"2023-07-08T21:09:13.703185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_features05_taxons.to_csv('train_features05_taxons31_onehot.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:09:13.707179Z","iopub.execute_input":"2023-07-08T21:09:13.707792Z","iopub.status.idle":"2023-07-08T21:09:16.964362Z","shell.execute_reply.started":"2023-07-08T21:09:13.707747Z","shell.execute_reply":"2023-07-08T21:09:16.963426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom Bio import SeqIO\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta'\nprint(\"Sequence example:\\n\\n\", next(iter(SeqIO.parse(fn, \"fasta\"))))\nprint()\nsequences = SeqIO.parse(fn, \"fasta\")\n\nlist_ids =  [str(seq.id) for seq in sequences ]\nprint(len(list_ids), list_ids[:10])\n\ndf_test_features05_taxons = pd.DataFrame(index = list_ids, data = np.zeros( (len(list_ids), len(tax_dict)) , dtype = np.float32 ), columns = tax_dict.keys()  )\ndisplay(df_test_features05_taxons.head(2))\n\nsequences = SeqIO.parse(fn, \"fasta\")\nfor I1, seq in enumerate( sequences ):\n    s = seq.description\n    #print(s)\n    s2 = float(s.split('\\t')[1])\n#     break\n    if 1:\n        if s2 in tax_dict.keys():\n            IX = tax_dict[s2]\n            df_test_features05_taxons.iloc[I1,IX] = 1\n#     print(seq.id)\n    if I1%20_000 == 0: print(I1)\n\nprint(df_test_features05_taxons.shape)\ndf_test_features05_taxons.head(2)\ndf_test_features05_taxons.describe()    ","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:09:16.968537Z","iopub.execute_input":"2023-07-08T21:09:16.969793Z","iopub.status.idle":"2023-07-08T21:09:33.431487Z","shell.execute_reply.started":"2023-07-08T21:09:16.969734Z","shell.execute_reply":"2023-07-08T21:09:33.430343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_test_features05_taxons.shape)\ndf_test_features05_taxons.to_csv('test_features05_taxons31_onehot.csv')\ndf_test_features05_taxons","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:09:33.433228Z","iopub.execute_input":"2023-07-08T21:09:33.433638Z","iopub.status.idle":"2023-07-08T21:09:36.810077Z","shell.execute_reply.started":"2023-07-08T21:09:33.433606Z","shell.execute_reply":"2023-07-08T21:09:36.808752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids_43k = np.load('/kaggle/input/cafa5-data-selected/embeds_43k_Y1850_etc/train_ids_cut43k.npy')\nprint(len(train_ids_43k), train_ids_43k[:10])","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:14:51.185366Z","iopub.execute_input":"2023-07-08T21:14:51.185882Z","iopub.status.idle":"2023-07-08T21:14:51.210417Z","shell.execute_reply.started":"2023-07-08T21:14:51.18585Z","shell.execute_reply":"2023-07-08T21:14:51.209322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_train_features05_taxons.index.isin(  train_ids_43k)\nprint(m.sum() )\ndf_train_features05_taxons[m]","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:14:57.59864Z","iopub.execute_input":"2023-07-08T21:14:57.600289Z","iopub.status.idle":"2023-07-08T21:14:57.747738Z","shell.execute_reply.started":"2023-07-08T21:14:57.600241Z","shell.execute_reply":"2023-07-08T21:14:57.746214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(df_train_features05_taxons[m].index) == list(train_ids_43k)","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:15:27.051757Z","iopub.execute_input":"2023-07-08T21:15:27.052248Z","iopub.status.idle":"2023-07-08T21:15:27.099808Z","shell.execute_reply.started":"2023-07-08T21:15:27.052209Z","shell.execute_reply":"2023-07-08T21:15:27.098561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_features05_taxons[m].to_csv('train_features05_taxons31_onehot_cut43k.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-08T21:15:33.793631Z","iopub.execute_input":"2023-07-08T21:15:33.794011Z","iopub.status.idle":"2023-07-08T21:15:34.792872Z","shell.execute_reply.started":"2023-07-08T21:15:33.793983Z","shell.execute_reply":"2023-07-08T21:15:34.791453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}