{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import Libraries","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from dataclasses import dataclass\nfrom pathlib import Path\nimport pandas as pd\nimport requests\nfrom tqdm.auto import tqdm\nfrom tqdm.contrib.concurrent import thread_map\nfrom bs4 import BeautifulSoup\nfrom requests.adapters import HTTPAdapter\nfrom requests.packages.urllib3.util.retry import Retry\nfrom concurrent.futures import ThreadPoolExecutor\nimport os\nimport logging\nfrom Bio import SeqIO\ntqdm.pandas()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load Paths","metadata":{}},{"cell_type":"code","source":"test_table = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset-taxon-list.tsv\")\ntest_sequences = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\")\ntrain_table = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_taxonomy.tsv\")\ntrain_sequences = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\")\ntrain_terms = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\")\ngo = Path(\"/kaggle/input/cafa-5-protein-function-prediction/Train/go-basic.obo\") # gene ontology -> pyobo\nia = Path(\"/kaggle/input/cafa-5-protein-function-prediction/IA.txt\") # information accretion\nsample_submission = Path(\"/kaggle/input/cafa-5-protein-function-prediction/sample_submission.tsv\")\nsubmission = Path(\"/kaggle/input/cafa-5-protein-function-prediction/submission.tsv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_df_from_fasta(path):\n    ids = []\n    seqs = []\n    with open(path) as handle: \n        for record in SeqIO.parse(handle, \"fasta\"):\n            ids.append(record.id)\n            seqs.append(str(record.seq))\n    df = pd.DataFrame({'EntryID' : ids, 'Seq':seqs})\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = get_df_from_fasta(test_sequences)\ntrain_df = get_df_from_fasta(train_sequences)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Approach\n* Use Uniprot Rest API and retreive the 3D structure databases available for all protein accessions that we have.<br>\n    *NOTE:* Each protein accession can have more than 1 3D structure database\n* After retreiving the DBs we need to analyse where the structures are, and see if there is a DB that contains all of them, if not we may have to use multiple DBs.","metadata":{}},{"cell_type":"code","source":"STRUCTURE_DB_NAMES=(\"PDB\", \"PDBsum\", \"SMR\", \"BioGRID-3D\", \"SWISS-MODEL-Workspace\", \"PDBe\", \"PDBj\", \"PMDB\", \"ModBase\", \n                          \"RCSB-PDB\", \"PDBe-KB\", \"BMRB\", \"PCDDB\", \"SASBDB\", \"AlphaFoldDB\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ProteinStructureDBReader():\n    def __init__(self,structure_db_names:tuple):\n        self.structure_db_names = structure_db_names\n        # Initialize a Session\n        self.session = requests.Session()\n\n        # Set a retry strategy\n        retry_strategy = Retry(\n            total=3,\n            status_forcelist=[429, 500, 502, 503, 504],\n            allowed_methods=[\"HEAD\", \"GET\", \"OPTIONS\"],\n            backoff_factor=1\n        )\n        adapter = HTTPAdapter(max_retries=retry_strategy)\n        self.session.mount(\"https://\", adapter)\n        self.session.mount(\"http://\", adapter)\n        \n    def get_uniprot_3d_structure_databases(self, uniprot_id):\n        base_url = f\"https://rest.uniprot.org/uniprotkb/{uniprot_id}?format=xml\"\n\n        try:\n            response = self.session.get(base_url, timeout=10)\n            response.raise_for_status()  # Check if the requests were successful\n\n        except requests.exceptions.HTTPError as errh:\n            print (\"HTTP Error:\", errh)\n        except requests.exceptions.ConnectionError as errc:\n            print (\"Error Connecting:\", errc)\n        except requests.exceptions.Timeout as errt:\n            print (\"Timeout Error:\", errt)\n        except requests.exceptions.RequestException as err:\n            print (\"Something went wrong\", err)\n\n        # parse the XML response\n        soup = BeautifulSoup(response.content, 'lxml')\n\n        # find all the 'dbreference' tags\n        db_references = soup.find_all('dbreference')\n\n        # we will store the 3D structure database references here\n        structure_db_refs = {name: None for name in self.structure_db_names}\n\n        for db_reference in db_references:\n            # if this is a 3D structure database, add it to the list\n            if db_reference['type'] in self.structure_db_names:\n                structure_db_refs[db_reference['type']] = db_reference['id']\n\n        return structure_db_refs\n        \n    def get_protein_structure_df(self, df, max_workers=5):\n        # Wrap the executor with tqdm for a progress bar\n        structure_db_series = thread_map(self.get_uniprot_3d_structure_databases, df['EntryID'], total=len(df), max_workers=max_workers)\n\n        # Convert the result to a DataFrame.\n        df_structure_db = pd.DataFrame(structure_db_series, index=df.index)\n\n        # Join the original DataFrame with the new DataFrame.\n        df = df.join(df_structure_db)\n\n        return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reader = ProteinStructureDBReader(structure_db_names=STRUCTURE_DB_NAMES)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cores = os.cpu_count()\nnum_workers = 4 * num_cores","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_structure_df = reader.get_protein_structure_df(train_df,max_workers=num_workers)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_structure_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_structure_df = reader.get_protein_structure_df(test_df,max_workers=num_workers)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_structure_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_structure_df.to_csv(\"protein_structure_dbs_train.csv\",index=False)\ntest_structure_df.to_csv(\"protein_structure_dbs_test.csv\",index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}