{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-08T18:30:20.728075Z","iopub.execute_input":"2023-06-08T18:30:20.72845Z","iopub.status.idle":"2023-06-08T18:30:20.733314Z","shell.execute_reply.started":"2023-06-08T18:30:20.728415Z","shell.execute_reply":"2023-06-08T18:30:20.732314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_inputs=pd.read_csv(\"/kaggle/input/machine-learning-challenge-2-prediction/training_set_rna.csv\", index_col=0).T\n\ntrain_cite_inputs","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:20.754428Z","iopub.execute_input":"2023-06-08T18:30:20.754818Z","iopub.status.idle":"2023-06-08T18:30:21.28077Z","shell.execute_reply.started":"2023-06-08T18:30:20.754786Z","shell.execute_reply":"2023-06-08T18:30:21.279456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import SeqIO\nfn = '/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta'\nsequences = open(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_sequences.fasta\")\nrecords = list(SeqIO.parse(sequences, \"fasta\"))\nsequences.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:21.282619Z","iopub.execute_input":"2023-06-08T18:30:21.282905Z","iopub.status.idle":"2023-06-08T18:30:24.272112Z","shell.execute_reply.started":"2023-06-08T18:30:21.282878Z","shell.execute_reply":"2023-06-08T18:30:24.270463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(records)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:24.273978Z","iopub.execute_input":"2023-06-08T18:30:24.274427Z","iopub.status.idle":"2023-06-08T18:30:24.280176Z","shell.execute_reply.started":"2023-06-08T18:30:24.274383Z","shell.execute_reply":"2023-06-08T18:30:24.279506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids=[]\nfor i in range(len(records)):\n    ids.append(records[i].id)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:24.282023Z","iopub.execute_input":"2023-06-08T18:30:24.283153Z","iopub.status.idle":"2023-06-08T18:30:24.370116Z","shell.execute_reply.started":"2023-06-08T18:30:24.283124Z","shell.execute_reply":"2023-06-08T18:30:24.368576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"old_genes = train_cite_inputs.columns","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:24.37201Z","iopub.execute_input":"2023-06-08T18:30:24.372505Z","iopub.status.idle":"2023-06-08T18:30:24.378756Z","shell.execute_reply.started":"2023-06-08T18:30:24.372454Z","shell.execute_reply":"2023-06-08T18:30:24.377287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ids)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:24.380552Z","iopub.execute_input":"2023-06-08T18:30:24.380935Z","iopub.status.idle":"2023-06-08T18:30:24.395223Z","shell.execute_reply.started":"2023-06-08T18:30:24.3809Z","shell.execute_reply":"2023-06-08T18:30:24.393885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Перевод old_genes names в ids","metadata":{}},{"cell_type":"code","source":"old_ids=[]\nfrom tqdm import tqdm\nimport joblib\nimport requests, sys, json\n\nrequestURL = \"https://www.ebi.ac.uk/proteins/api/features?offset=0&size=100&gene=\"\n\nfor geneName in old_genes:\n    r = requests.get(f'{requestURL}{geneName}+&organism=HUMAN', headers={ \"Accept\" : \"application/json\"})\n    if not r.ok:\n        r.raise_for_status()\n        sys.exit()\n    responseBody = r.json()\n    for i in range(len(responseBody)):\n        old_ids.append(responseBody[i]['accession'])","metadata":{"execution":{"iopub.status.busy":"2023-06-08T18:30:24.396906Z","iopub.execute_input":"2023-06-08T18:30:24.397212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(old_ids)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"common_ids=[]\nfor i in old_ids:\n    for j in ids:\n        if i == j:\n            common_ids.append(i)\n            break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(common_ids)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Перевод общих ids в имена генов","metadata":{}},{"cell_type":"code","source":"genes_names=[]\nfrom tqdm import tqdm\nimport re\nimport joblib\nimport requests, sys, json\n\nWEBSITE_API = 'https://rest.uniprot.org'\n\ndef get_url(url, **kwargs):\n    response = requests.get(url, **kwargs)\n\n    if not response.ok:\n        response.raise_for_status()\n        sys.exit()\n\n    return response\n\nuniprot_data_proteins = {}\nfor id in common_ids:\n\n    r = get_url(f'{WEBSITE_API}/uniprotkb/{id}')\n    uniprot_data_proteins[str(id)] = r.json()\n    j = r.json()\n    if 'genes' in r.json():\n        if 'geneName' in r.json()['genes'][0]:\n            if r.json()['genes'][0]['geneName']['value'] not in genes_names:\n            \n                genes_names.append((r.json()['genes'][0]['geneName']['value']))\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"letters=[]\nnums=[]\nletter_names=[]\nno_dubli_list=[]\nfor l in range(len(genes_names)):\n    if genes_names[l].isalpha() is True:\n        letter_names.append(genes_names[l])\n    else:\n        r = re.split(r'[A-z]', genes_names[l], maxsplit=1)\n        nums.append(r[-1])\n        k =re.split(r'[0-9]', genes_names[l], maxsplit=1)\n    \n        letters.append(k[0])\n    \n    \ngenes_data = pd.DataFrame({'letters':letters, 'nums':nums})\ngenes_data = genes_data.drop_duplicates(['letters'])\ngenes_data['2'] = genes_data['letters']+genes_data['nums']\nno_dubli_list = genes_data['2'].tolist()\nnew_gene_list = no_dubli_list + letter_names\n    \ndf_genes = pd.DataFrame(new_gene_list)\ndf_genes.to_csv('genes_names')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}