{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41875,"databundleVersionId":5521661,"sourceType":"competition"},{"sourceId":6046570,"sourceType":"datasetVersion","datasetId":3458902},{"sourceId":6046672,"sourceType":"datasetVersion","datasetId":3391266},{"sourceId":6136171,"sourceType":"datasetVersion","datasetId":3518517}],"dockerImageVersionId":30527,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Since, running Foldseek on Kaggle gives MMseq2, I directly used the foldseek test set submission. You can refer to this amazing notebook on Foldseek by :RAMAN for the entire code https://www.kaggle.com/code/samusram/leveraging-foldseek","metadata":{}},{"cell_type":"code","source":"!wget https://ftp.ncbi.nlm.nih.gov/blast/executables/LATEST/ncbi-blast-2.14.0+-x64-linux.tar.gz","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-26T17:39:44.683306Z","iopub.execute_input":"2023-06-26T17:39:44.683801Z","iopub.status.idle":"2023-06-26T17:40:02.461899Z","shell.execute_reply.started":"2023-06-26T17:39:44.683758Z","shell.execute_reply":"2023-06-26T17:40:02.460621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!tar zxvpf ncbi-blast-2.14.0+-x64-linux.tar.gz","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-26T17:40:02.464513Z","iopub.execute_input":"2023-06-26T17:40:02.465155Z","iopub.status.idle":"2023-06-26T17:40:11.56483Z","shell.execute_reply.started":"2023-06-26T17:40:02.46511Z","shell.execute_reply":"2023-06-26T17:40:11.563191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/working/ncbi-blast-2.14.0+/bin/* /opt/conda/bin","metadata":{"execution":{"iopub.status.busy":"2023-06-26T17:40:11.566962Z","iopub.execute_input":"2023-06-26T17:40:11.567495Z","iopub.status.idle":"2023-06-26T17:40:13.407831Z","shell.execute_reply.started":"2023-06-26T17:40:11.567438Z","shell.execute_reply":"2023-06-26T17:40:13.406591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install git+https://github.com/SamusRam/ProFun.git","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-26T17:40:13.411788Z","iopub.execute_input":"2023-06-26T17:40:13.412199Z","iopub.status.idle":"2023-06-26T17:40:35.762003Z","shell.execute_reply.started":"2023-06-26T17:40:13.412155Z","shell.execute_reply":"2023-06-26T17:40:35.760498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom pathlib import Path\nfrom tqdm.auto import tqdm, trange\nfrom Bio import SeqIO\nimport numpy as np\n\nfrom profun.models import BlastMatching, BlastConfig\nfrom profun.utils.project_info import ExperimentInfo","metadata":{"execution":{"iopub.status.busy":"2023-06-26T17:40:35.764179Z","iopub.execute_input":"2023-06-26T17:40:35.765305Z","iopub.status.idle":"2023-06-26T17:40:37.169684Z","shell.execute_reply.started":"2023-06-26T17:40:35.765246Z","shell.execute_reply":"2023-06-26T17:40:37.168204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Obtaining train data","metadata":{}},{"cell_type":"code","source":"data_root = Path('/kaggle/input/cafa-5-protein-function-prediction/')\ntrain_terms = pd.read_csv(data_root/\"Train/train_terms.tsv\",sep=\"\\t\")\n\nids = []\nseqs = []\nwith open(data_root/\"Train/train_sequences.fasta\") as handle:\n    for record in SeqIO.parse(handle, \"fasta\"):\n        ids.append(record.id)\n        seqs.append(str(record.seq))\ntrain_seqs_df = pd.DataFrame({'EntryID': ids, 'Seq': seqs})\ntrain_df_long = train_terms.merge(train_seqs_df, on='EntryID')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-26T17:40:37.171915Z","iopub.execute_input":"2023-06-26T17:40:37.172818Z","iopub.status.idle":"2023-06-26T17:40:45.833808Z","shell.execute_reply.started":"2023-06-26T17:40:37.172761Z","shell.execute_reply":"2023-06-26T17:40:45.832518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Init model","metadata":{"execution":{"iopub.status.busy":"2023-06-11T20:25:23.042425Z","iopub.execute_input":"2023-06-11T20:25:23.042957Z","iopub.status.idle":"2023-06-11T20:25:28.55935Z","shell.execute_reply.started":"2023-06-11T20:25:23.042919Z","shell.execute_reply":"2023-06-11T20:25:28.558633Z"}}},{"cell_type":"code","source":"experiment_info = ExperimentInfo(validation_schema='public_lb', \n                                 model_type='blast', model_version='1nn')\n\nconfig = BlastConfig(experiment_info=experiment_info, \n                     id_col_name='EntryID', \n                     target_col_name='term', \n                     seq_col_name='Seq', \n                     class_names=list(train_df_long['term'].unique()), \n                     optimize_hyperparams=False, \n                     n_calls_hyperparams_opt=None,\n                    hyperparam_dimensions=None,\n                    per_class_optimization=None,\n                    class_weights=None,\n                    n_neighbours=5,\n                    e_threshold=0.02,\n                     n_jobs=100,\n                     pred_batch_size=10\n                    )\n\nblast_model = BlastMatching(config)","metadata":{"execution":{"iopub.status.busy":"2023-06-26T17:40:45.835465Z","iopub.execute_input":"2023-06-26T17:40:45.835882Z","iopub.status.idle":"2023-06-26T17:40:46.479461Z","shell.execute_reply.started":"2023-06-26T17:40:45.83584Z","shell.execute_reply":"2023-06-26T17:40:46.478143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train model","metadata":{}},{"cell_type":"code","source":"#blast_model.fit(train_df_long)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-26T17:40:46.481467Z","iopub.execute_input":"2023-06-26T17:40:46.481964Z","iopub.status.idle":"2023-06-26T17:41:10.548913Z","shell.execute_reply.started":"2023-06-26T17:40:46.48191Z","shell.execute_reply":"2023-06-26T17:41:10.547386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#obtained by training offline\ntest_pred_df_blast = pd.read_csv('/kaggle/input/proteinet-best/blast_submission.tsv',\n    sep='\\t', header=None).drop(0, axis=1)\n\nsubmission_best_public = pd.read_csv('/kaggle/input/cafa5-055757-pred/submission.tsv',\n    sep='\\t', header=None, names=['Id', 'GO term', 'Confidence'])\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions_merged = submission_best_public.merge(test_pred_df_blast, left_on=['Id', 'GO term'], \n                                                  right_on=[1, 2], how='outer')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions_merged.drop([1, 2], axis=1, inplace=True)\nsubmissions_merged['confidence_combined'] = submissions_merged.apply(lambda row: row['Confidence'] if not np.isnan(row['Confidence']) else row[3], axis=1)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submissions_merged[['Id', 'GO term', 'confidence_combined']].to_csv('submission.tsv',\n    sep='\\t', header=False, index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}