{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#script to vectorize the GO_terms anotations in a one hot label\nk = 2000 #top k go terms to process\nfunction = \"CCO\" #None for all go terms","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-11T17:23:28.680871Z","iopub.execute_input":"2023-07-11T17:23:28.681272Z","iopub.status.idle":"2023-07-11T17:23:28.686338Z","shell.execute_reply.started":"2023-07-11T17:23:28.681242Z","shell.execute_reply":"2023-07-11T17:23:28.685382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars\nfrom Bio import SeqIO\nfrom tqdm import tqdm\n\nMAIN_DIR = \"/kaggle/input/cafa-5-protein-function-prediction\"\nclass config:\n    train_sequences_path = MAIN_DIR  + \"/Train/train_sequences.fasta\"\n    train_labels_path = MAIN_DIR + \"/Train/train_terms.tsv\"\n    test_sequences_path = MAIN_DIR + \"/Test (Targets)/testsuperset.fasta\"\n    \nfasta_train = SeqIO.parse(config.train_sequences_path, \"fasta\")\nstring_list = []\nid_list = []\nsequence_len = []\n\nfor item in tqdm(fasta_train):\n    seq = str(item.seq)\n    item_id = item.id \n\n    string_list.append(seq)\n    id_list.append(item_id)\n    sequence_len.append(len(seq))\n    \nprotein_sequences= np.array(string_list, dtype = object)\nsequence_lengths= np.array(sequence_len)","metadata":{"execution":{"iopub.status.busy":"2023-07-11T17:48:27.731667Z","iopub.execute_input":"2023-07-11T17:48:27.732466Z","iopub.status.idle":"2023-07-11T17:48:30.44546Z","shell.execute_reply.started":"2023-07-11T17:48:27.732433Z","shell.execute_reply":"2023-07-11T17:48:30.444259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the pos and neg vectors in order\npos_neg_ids = np.load('/kaggle/input/vectorize-go-labels-pc/training_ids.npy', allow_pickle=True).astype(str)\npos_idx_train = np.load('/kaggle/input/combine-pos-neg/pos_idx_train.npy').astype(int)\nneg_idx_train = np.load('/kaggle/input/combine-pos-neg/neg_idx_train.npy').astype(int)\npos_idx_test = np.load('/kaggle/input/pos-neg-compute-test-v2/pos_idx_test.npy').astype(int)\nneg_idx_test = np.load('/kaggle/input/pos-neg-compute-test-v2/neg_idx_test.npy').astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-07-11T17:48:19.333176Z","iopub.execute_input":"2023-07-11T17:48:19.333603Z","iopub.status.idle":"2023-07-11T17:48:20.246415Z","shell.execute_reply.started":"2023-07-11T17:48:19.333572Z","shell.execute_reply":"2023-07-11T17:48:20.245299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport polars\n\nlabels = polars.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", separator = \"\\t\")\nsorted_terms = labels.groupby('term').count().sort(['count'], descending = True)\n\nlabel_encoder = sorted_terms.join(labels.drop(\"EntryID\").unique(), on = 'term', how = 'left')\nif not function is None:\n    label_encoder = label_encoder.filter(polars.col('aspect')==function)\nlabel_encoder = label_encoder.with_columns(polars.Series(name=\"index\", values= range(len(label_encoder))))    \n    \nlabel_encoder = label_encoder[:k]\nprint(label_encoder)\nlabel_encoder = label_encoder.drop('count')\n\n\ntraining_ids = np.array(id_list, dtype = 'object')#labels['EntryID'].unique().to_numpy()#for whatever reason unique is not deterministic","metadata":{"execution":{"iopub.status.busy":"2023-07-11T17:48:22.983576Z","iopub.execute_input":"2023-07-11T17:48:22.983951Z","iopub.status.idle":"2023-07-11T17:48:24.082243Z","shell.execute_reply.started":"2023-07-11T17:48:22.983923Z","shell.execute_reply":"2023-07-11T17:48:24.081435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get the correct random indices\nfixed_indices = np.zeros(len(pos_neg_ids), dtype = int)\n\nfor i in tqdm(range(len(pos_neg_ids)) ):\n    fixed_indices[i]=np.where(training_ids == pos_neg_ids[i])[0][0]","metadata":{"execution":{"iopub.status.busy":"2023-07-11T17:48:32.789722Z","iopub.execute_input":"2023-07-11T17:48:32.790113Z","iopub.status.idle":"2023-07-11T18:07:46.556633Z","shell.execute_reply.started":"2023-07-11T17:48:32.790078Z","shell.execute_reply":"2023-07-11T18:07:46.555462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fix the indices to match pos_neg ids\n#random_index = np.random.choice(len(training_ids), len(training_ids), replace = False)\n\ntraining_ids=training_ids[fixed_indices]\nprotein_sequences=protein_sequences[fixed_indices]\nsequence_lengths=sequence_lengths[fixed_indices]\ndel fixed_indices #so we don't accidentally call it again and screw up the indices","metadata":{"execution":{"iopub.status.busy":"2023-07-11T18:10:42.105632Z","iopub.execute_input":"2023-07-11T18:10:42.106015Z","iopub.status.idle":"2023-07-11T18:10:42.526065Z","shell.execute_reply.started":"2023-07-11T18:10:42.105987Z","shell.execute_reply":"2023-07-11T18:10:42.524664Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_labels = np.zeros([len(training_ids), k],dtype=bool)\nfor i in tqdm(range(len(training_ids))):\n    if i % 1000 == 0:\n        print(i)\n    ids = training_ids[i]\n    id_GO_terms = labels.filter(polars.col('EntryID')  == ids)['term'].to_list()\n    indices = label_encoder.filter(polars.col('term').is_in(list(id_GO_terms)))['index'].to_numpy()\n    training_labels[i,indices] = 1","metadata":{"execution":{"iopub.status.busy":"2023-06-23T00:26:38.70771Z","iopub.execute_input":"2023-06-23T00:26:38.710882Z","iopub.status.idle":"2023-06-23T00:26:38.919225Z","shell.execute_reply.started":"2023-06-23T00:26:38.710765Z","shell.execute_reply":"2023-06-23T00:26:38.917265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = polars.read_csv(MAIN_DIR+\"/Train/train_terms.tsv\", separator = \"\\t\")\nlabels = np.unique(labels.filter(polars.col('aspect')== function)['term'].to_numpy())\ntrain_filter = np.isin(training_ids, labels)\n\nnp.save('training_ids.npy',training_ids[train_filter])\nnp.save('protein_sequences.npy',protein_sequences[train_filter])\nnp.save('sequence_lengths.npy',sequence_lengths[train_filter])\n\nnp.save('label_vectors.npy',training_labels[train_filter])\nnp.save('go_terms.npy',label_encoder['term'].to_numpy())","metadata":{"execution":{"iopub.status.busy":"2023-07-11T17:17:07.249933Z","iopub.execute_input":"2023-07-11T17:17:07.25115Z","iopub.status.idle":"2023-07-11T17:17:07.799812Z","shell.execute_reply.started":"2023-07-11T17:17:07.251104Z","shell.execute_reply":"2023-07-11T17:17:07.798281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ignore the rest of notebook. Mostly Data exploration","metadata":{}},{"cell_type":"code","source":"# import polars\n# ia_weights = polars.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\", separator = \"\\t\", has_header = False, new_columns = ['term',\"IA\"])","metadata":{"execution":{"iopub.status.busy":"2023-07-07T19:11:42.758652Z","iopub.execute_input":"2023-07-07T19:11:42.75919Z","iopub.status.idle":"2023-07-07T19:11:43.090007Z","shell.execute_reply.started":"2023-07-07T19:11:42.759152Z","shell.execute_reply":"2023-07-07T19:11:43.08895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_encoder[14000:]#.sort(['IA'], descending = True)","metadata":{"execution":{"iopub.status.busy":"2023-07-07T19:18:14.442686Z","iopub.execute_input":"2023-07-07T19:18:14.443122Z","iopub.status.idle":"2023-07-07T19:18:14.45378Z","shell.execute_reply.started":"2023-07-07T19:18:14.44309Z","shell.execute_reply":"2023-07-07T19:18:14.452699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pos_neg_test_ids = pos_neg_ids[-len(neg_idx_test):]\n# test_labels = labels.filter(polars.col('EntryID').is_in(pos_neg_test_ids))\n# test_label_encoder = test_labels.groupby('term').count().sort(['count'], descending = True).join(labels.drop(\"EntryID\").unique(), on = 'term', how = 'left').filter(polars.col('aspect')==function)\n# comparison = label_encoder.join(test_label_encoder.rename({\"count\": \"count_test\", 'aspect':'aspect_test'}), on = 'term', how = 'left')","metadata":{"execution":{"iopub.status.busy":"2023-07-11T04:35:09.03772Z","iopub.execute_input":"2023-07-11T04:35:09.038127Z","iopub.status.idle":"2023-07-11T04:35:09.307114Z","shell.execute_reply.started":"2023-07-11T04:35:09.038095Z","shell.execute_reply":"2023-07-11T04:35:09.306305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function = \"MFO\"\n#function = 'CCO'","metadata":{"execution":{"iopub.status.busy":"2023-07-11T04:35:02.207048Z","iopub.execute_input":"2023-07-11T04:35:02.207446Z","iopub.status.idle":"2023-07-11T04:35:02.212335Z","shell.execute_reply.started":"2023-07-11T04:35:02.207416Z","shell.execute_reply":"2023-07-11T04:35:02.211193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}