{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Combining datasets\n\n### Here we work towards a method to combine datasets and outputs predictions\n\n#### We will use: SPROF Predictions, ProFun (https://github.com/SamusRam/ProFun) Predictions and QuickGo Annotations","metadata":{}},{"cell_type":"code","source":"#Create a dictionary to assign each go term to the roots (CCO, MFO, BPO)\n\nimport re\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport joblib\nimport pickle\nfrom tqdm import tqdm\nfrom Bio import SeqIO\nimport gc\n\ndef extract_go_terms_and_branches(file_path):\n    with open(file_path, 'r') as file:\n        content = file.read()\n        # Match each stanza with [Term] in the OBO file\n        stanzas = re.findall(r'\\[Term\\][\\s\\S]*?(?=\\n\\[|$)', content)\n\n    go_terms_dict = {}\n    for stanza in stanzas:\n        # Extract the GO term ID\n        go_id = re.search(r'^id: (GO:\\d+)', stanza, re.MULTILINE)\n        if go_id:\n            go_id = go_id.group(1)\n\n        # Extract the namespace (branch)\n        namespace = re.search(r'^namespace: (\\w+)', stanza, re.MULTILINE)\n        if namespace:\n            namespace = namespace.group(1)\n\n        if go_id and namespace:\n            # Map the branch abbreviation to the corresponding BPO, CCO, or MFO\n            branch_abbr = {'biological_process': 'BPO', 'cellular_component': 'CCO', 'molecular_function': 'MFO'}\n            go_terms_dict[go_id] = branch_abbr[namespace]\n\n    return go_terms_dict\n\nfile_path = '/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo'\ngo_terms_dict = extract_go_terms_and_branches(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T12:52:19.048584Z","iopub.execute_input":"2023-06-16T12:52:19.050324Z","iopub.status.idle":"2023-06-16T12:52:24.037978Z","shell.execute_reply.started":"2023-06-16T12:52:19.050242Z","shell.execute_reply":"2023-06-16T12:52:24.035755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a class to manage predictions for proteins.\n# The class keeps track of the highest score for each GO (Gene Ontology) term prediction.\n# Note: This assumes scores are comparable, which might not be the case.\n# A ranking-based selection could be more suitable.\n# Each branch outputs a maximum of 35 predictions for each protein after sorting predictions from highest to lowest.\n# There is an option to add a bonus to the score if the term is predicted by multiple methods.\n\nclass ProteinPredictions:\n    # Initialize an empty dictionary to store the predictions\n    def __init__(self):\n        self.predictions = {}\n\n    # Add a prediction to the storage, with optional bonus\n    # Arguments:\n    #   - protein: Identifier for the protein\n    #   - go_term: GO term that is being predicted\n    #   - score: Confidence score of the prediction\n    #   - branch: Branch of the Gene Ontology (e.g., 'CCO', 'MFO', 'BPO')\n    #   - bonus: Optional bonus to be added to the score\n    def add_prediction(self, protein, go_term, score, branch, bonus=0):\n        # If the protein is not already in the storage, initialize its structure\n        if protein not in self.predictions:\n            self.predictions[protein] = {'CCO': {}, 'MFO': {}, 'BPO': {}}\n        \n        # Convert the score to a float for comparison and calculation\n        score = float(score)\n\n        # If this GO term has already been predicted for this protein and branch,\n        # add the bonus to the score. Keep the highest score.\n        if go_term in self.predictions[protein][branch]:\n            if self.predictions[protein][branch][go_term] < score:\n                self.predictions[protein][branch][go_term] = score + bonus\n            else:\n                self.predictions[protein][branch][go_term] += bonus\n        # If this GO term has not been predicted yet, store it with the score\n        else:\n            self.predictions[protein][branch][go_term] = score\n\n        # Ensure that the score does not exceed 1\n        if self.predictions[protein][branch][go_term] > 1:\n            self.predictions[protein][branch][go_term] = 1\n\n    # Export the stored predictions to a file\n    # Arguments:\n    #   - output_file: File name for the exported predictions\n    #   - top: Number of top predictions to export for each protein and branch\n    def get_predictions(self, output_file='submission.tsv', top=35):\n        # Open the output file\n        with open(output_file, 'w') as f:\n            # Iterate through each protein and its branches\n            for protein, branches in self.predictions.items():\n                # For each branch, sort the GO terms by score in descending order and select the top ones\n                for branch, go_terms in branches.items():\n                    # Sort go_terms by score in descending order and take the top ones\n                    top_go_terms = sorted(go_terms.items(), key=lambda x: x[1], reverse=True)[:top]\n                    # Write each of the top predictions to the file\n                    for go_term, score in top_go_terms:\n                        f.write(f\"{protein}\\t{go_term}\\t{score:.3f}\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-16T12:52:24.041811Z","iopub.execute_input":"2023-06-16T12:52:24.042355Z","iopub.status.idle":"2023-06-16T12:52:24.060669Z","shell.execute_reply.started":"2023-06-16T12:52:24.042301Z","shell.execute_reply":"2023-06-16T12:52:24.059225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Define a class to manage predictions for proteins.\n# # The class keeps track of the highest score for each GO (Gene Ontology) term prediction.\n# # Note: This assumes scores are comparable, which might not be the case.\n# # A ranking-based selection could be more suitable.\n# # Each branch outputs a maximum of 35 predictions for each protein after sorting predictions from highest to lowest.\n# # There is an option to add a bonus to the score if the term is predicted by multiple methods.\n\n# class ProteinPredictions:\n#     # Initialize an empty dictionary to store the predictions\n#     def __init__(self):\n#         self.predictions = {}\n\n#     # Add a prediction to the storage, with optional bonus\n#     # Arguments:\n#     #   - protein: Identifier for the protein\n#     #   - go_term: GO term that is being predicted\n#     #   - score: Confidence score of the prediction\n#     #   - branch: Branch of the Gene Ontology (e.g., 'CCO', 'MFO', 'BPO')\n#     #   - bonus: Optional bonus to be added to the score\n#     def add_prediction(self, protein, go_term, score, branch, bonus=0, hist=0, inc=1):\n#         # If the protein is not already in the storage, initialize its structure\n#         if protein not in self.predictions:\n#             self.predictions[protein] = {'CCO': {}, 'MFO': {}, 'BPO': {}}\n        \n#         # Convert the score to a float for comparison and calculation\n#         score = float(score)\n\n#         # If this GO term has already been predicted for this protein and branch,\n#         # add the bonus to the score. Keep the highest score.\n#         if go_term in self.predictions[protein][branch]:\n#             if self.predictions[protein][branch][go_term] < score:\n#                 self.predictions[protein][branch][go_term] = (score*inc+self.predictions[protein][branch][go_term]*hist)/(hist+inc)    + bonus\n#             else:\n#                 self.predictions[protein][branch][go_term]  = (score*inc+self.predictions[protein][branch][go_term]*hist)/(hist+inc) + bonus\n#         # If this GO term has not been predicted yet, store it with the score\n#         else:\n#             self.predictions[protein][branch][go_term] = score\n\n#         # Ensure that the score does not exceed 1\n#         if self.predictions[protein][branch][go_term] > 1:\n#             self.predictions[protein][branch][go_term] = 1\n\n#     # Export the stored predictions to a file\n#     # Arguments:\n#     #   - output_file: File name for the exported predictions\n#     #   - top: Number of top predictions to export for each protein and branch\n#     def get_predictions(self, output_file='submission.tsv', top=40):\n#         # Open the output file\n#         with open(output_file, 'w') as f:\n#             # Iterate through each protein and its branches\n#             for protein, branches in self.predictions.items():     \n#                 # For each branch, sort the GO terms by score in descending order and select the top ones\n#                 for branch, go_terms in branches.items():\n#                     if branch =='CCO': \n#                         top = 39\n#                     if branch =='MFO': \n#                         top = 39\n#                     if branch =='BPO': \n#                         top = 39     \n#                     # Sort go_terms by score in descending order and take the top ones\n#                     top_go_terms = sorted(go_terms.items(), key=lambda x: x[1], reverse=True)[:top]\n#                     # Write each of the top predictions to the file\n#                     for go_term, score in top_go_terms:\n#                         f.write(f\"{protein}\\t{go_term}\\t{score:.3f}\\n\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_predictions = ProteinPredictions()","metadata":{"execution":{"iopub.status.busy":"2023-06-16T13:02:46.838216Z","iopub.execute_input":"2023-06-16T13:02:46.83882Z","iopub.status.idle":"2023-06-16T13:02:46.84601Z","shell.execute_reply.started":"2023-06-16T13:02:46.838767Z","shell.execute_reply":"2023-06-16T13:02:46.84445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for l in tqdm(open('/kaggle/input/quick-go-2022-03-02/quickgo.tsv')):\n    item_list = l.split('\\t')\n    temp_id = item_list[1]\n    go=item_list[2].strip()\n    score = float(1)\n    if go in go_terms_dict:\n        root = go_terms_dict[go]\n        #branch = item_list[3].strip()\n        protein_predictions.add_prediction(temp_id, go, score, root)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-16T13:02:47.682448Z","iopub.execute_input":"2023-06-16T13:02:47.683016Z","iopub.status.idle":"2023-06-16T13:03:08.04466Z","shell.execute_reply.started":"2023-06-16T13:02:47.682962Z","shell.execute_reply":"2023-06-16T13:03:08.043229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for l in tqdm(open('/kaggle/input/merge-datasets/submission.tsv')):\n#     item_list = l.split('\\t')\n#     temp_id = item_list[0]\n#     go=item_list[1]\n#     score = float(item_list[2].strip())\n#     if go in go_terms_dict:\n#         root = go_terms_dict[go]\n#         #branch = item_list[3].strip()\n#         protein_predictions.add_prediction(temp_id, go, score, root, 0, 1, 1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for l in tqdm(open('/kaggle/input/proteinet-best/blast_submission.tsv')):\n#     item_list = l.split('\\t')\n#     temp_id = item_list[1]\n#     go=item_list[2]\n#     score = float(item_list[3].strip())\n#     if go in go_terms_dict:\n#         root = go_terms_dict[go]\n#         #branch = item_list[3].strip()\n#         protein_predictions.add_prediction(temp_id, go, score, root)","metadata":{"execution":{"iopub.status.busy":"2023-06-16T13:03:08.049514Z","iopub.execute_input":"2023-06-16T13:03:08.050857Z","iopub.status.idle":"2023-06-16T13:04:12.289387Z","shell.execute_reply.started":"2023-06-16T13:03:08.050789Z","shell.execute_reply":"2023-06-16T13:04:12.287901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = [#'/kaggle/input/proteinet-best/submission_best_public_merge.tsv' , \n    '/kaggle/input/cafa-datasets/submission_best_LB.tsv',\n#     '/kaggle/input/merge-datasets/submission.tsv',\n#     '/kaggle/input/f-t-15pytorch-keras-etc-3-blend-cafa-metric-etc0/submission.tsv',\n#     '/kaggle/input/gpu-pytorch-keras-etc-3-blend-cafa-metric-e/submission.tsv',\n#     '/kaggle/input/k/kamilla1158/pytorch-keras-etc-3-blend-cafa-metric-etc/submission.tsv',\n#     '/kaggle/input/pytorch-keras-etc-3-blend-cafa-metric-etc-1f8461/submission.tsv',\n#     '/kaggle/input/pytorch-keras-etc-3-blend-cafa-metric-etc-18011b/submission.tsv',      \n#     '/kaggle/input/k/olegpush/pytorch-keras-etc-3-blend-cafa-metric-etc/submission.tsv',      \n#     #'/kaggle/input/fork-of-pytorch-01-basics-3a6104/submission.tsv',\n#     #'/kaggle/input/pytorch-01-basics-3a6104/submission.tsv',\n#     #'/kaggle/input/pytorch-2-cv-focal-sophia-etc/submission.tsv',\n#     #'/kaggle/input/pytorch-01-basics/submission.tsv',\n#     '/kaggle/input/cafa-5-t5-embeds-ensemble/submission.tsv'\n]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submit = [#'/kaggle/input/proteinet-best/submission_best_public_merge.tsv' , \n#     '/kaggle/input/merge-datasets/submission.tsv',\n# #     '/kaggle/input/f-t-15pytorch-keras-etc-3-blend-cafa-metric-etc0/submission.tsv',\n# #     '/kaggle/input/gpu-pytorch-keras-etc-3-blend-cafa-metric-e/submission.tsv',\n# #     '/kaggle/input/k/kamilla1158/pytorch-keras-etc-3-blend-cafa-metric-etc/submission.tsv',\n# #     '/kaggle/input/pytorch-keras-etc-3-blend-cafa-metric-etc-1f8461/submission.tsv',\n# #     '/kaggle/input/pytorch-keras-etc-3-blend-cafa-metric-etc-18011b/submission.tsv',      \n# #     '/kaggle/input/k/olegpush/pytorch-keras-etc-3-blend-cafa-metric-etc/submission.tsv',      \n# #     #'/kaggle/input/fork-of-pytorch-01-basics-3a6104/submission.tsv',\n# #     #'/kaggle/input/pytorch-01-basics-3a6104/submission.tsv',\n# #     #'/kaggle/input/pytorch-2-cv-focal-sophia-etc/submission.tsv',\n# #     #'/kaggle/input/pytorch-01-basics/submission.tsv',\n# #     '/kaggle/input/cafa-5-t5-embeds-ensemble/submission.tsv'\n# ]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fil in submit:\n    for l in tqdm(open(fil)):\n        item_list = l.split('\\t')\n        temp_id = item_list[0]\n        go=item_list[1]\n        score = float(item_list[2].strip())\n        if go in go_terms_dict:\n            root = go_terms_dict[go]\n            #branch = item_list[3].strip()\n            protein_predictions.add_prediction(temp_id, go, score, root)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_predictions.get_predictions()","metadata":{"execution":{"iopub.status.busy":"2023-06-16T13:04:29.982196Z","iopub.execute_input":"2023-06-16T13:04:29.982575Z","iopub.status.idle":"2023-06-16T13:04:46.239453Z","shell.execute_reply.started":"2023-06-16T13:04:29.982538Z","shell.execute_reply":"2023-06-16T13:04:46.237721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head -n 200 'submission.tsv'","metadata":{"execution":{"iopub.status.busy":"2023-06-16T13:04:46.241421Z","iopub.execute_input":"2023-06-16T13:04:46.241877Z","iopub.status.idle":"2023-06-16T13:04:47.484094Z","shell.execute_reply.started":"2023-06-16T13:04:46.241832Z","shell.execute_reply":"2023-06-16T13:04:47.48217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}