{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"The given code is a starter code that uses a naive approach for CAFA 5 Protein Function Prediction problem. The code reads a training dataset that contains protein sequences and their corresponding Gene Ontology (GO) terms, and extracts the most frequently occurring GO terms. These top GO terms are then used to make predictions for a test dataset that contains protein sequences. The predictions are made by assigning the top GO terms to each protein sequence in the test dataset with a confidence score that corresponds to the relative frequency of the term in the training dataset. ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom Bio import SeqIO\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-29T15:58:28.528332Z","iopub.execute_input":"2023-04-29T15:58:28.529242Z","iopub.status.idle":"2023-04-29T15:58:28.667714Z","shell.execute_reply.started":"2023-04-29T15:58:28.529189Z","shell.execute_reply":"2023-04-29T15:58:28.666752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_fasta(fastaPath):    \n    fasta_sequences = SeqIO.parse(open(fastaPath), 'fasta')\n    ids = []\n    sequences = []\n    for fasta in fasta_sequences:\n        ids.append(fasta.id)\n        sequences.append(str(fasta.seq))\n    return pd.DataFrame({'Id': ids, 'Sequence': sequences})\n\ndef get_top_go_terms(data, num_terms):\n    term_counts = data['term'].value_counts()\n    freq_counts = term_counts / len(data)\n    freq_top = freq_counts.nlargest(num_terms)\n    return freq_top","metadata":{"execution":{"iopub.status.busy":"2023-04-29T15:59:35.693966Z","iopub.execute_input":"2023-04-29T15:59:35.694365Z","iopub.status.idle":"2023-04-29T15:59:35.702334Z","shell.execute_reply.started":"2023-04-29T15:59:35.694331Z","shell.execute_reply":"2023-04-29T15:59:35.701023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_terms = pd.read_csv('/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv', sep='\\t')\ntop_terms = get_top_go_terms(train_terms, 10)\n\ntest_data = read_fasta('/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta')\n\nresults = []\nfor index, row in tqdm(test_data.iterrows(), total=test_data.shape[0], position=0):\n    for term, freq in top_terms.items():\n        results.append((row['Id'], term, freq))\n\nfinal_results = pd.DataFrame(results, columns=['Id', 'GO term', 'Confidence'])\nfinal_results.to_csv('submission.tsv', sep='\\t', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-29T15:59:49.543158Z","iopub.execute_input":"2023-04-29T15:59:49.54359Z","iopub.status.idle":"2023-04-29T16:00:16.175889Z","shell.execute_reply.started":"2023-04-29T15:59:49.543553Z","shell.execute_reply":"2023-04-29T16:00:16.174732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}