{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-01T17:57:14.343543Z","iopub.execute_input":"2023-07-01T17:57:14.343965Z","iopub.status.idle":"2023-07-01T17:57:14.435248Z","shell.execute_reply.started":"2023-07-01T17:57:14.343921Z","shell.execute_reply":"2023-07-01T17:57:14.433928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import SeqIO\nimport pandas as pd\n\ndef parse_fasta_file(fasta_file):\n    sequences = []\n    for record in SeqIO.parse(fasta_file, \"fasta\"):\n        sequences.append([record.id, str(record.seq)])\n    return sequences\n\n# Path to your FASTA file\nfasta_file_path = \"/kaggle/input/cafa-5-protein-function-prediction/Test (Targets)/testsuperset.fasta\"\n\n# Parse the FASTA file and get sequences\nsequences = parse_fasta_file(fasta_file_path)\n\n# Create a DataFrame from the sequences\nprotein_df_train = pd.DataFrame(sequences, columns=[\"Sequence ID\", \"Sequence\"])\n\n# Print the DataFrame\nprint(protein_df_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-01T17:57:26.697367Z","iopub.execute_input":"2023-07-01T17:57:26.697798Z","iopub.status.idle":"2023-07-01T17:57:29.681355Z","shell.execute_reply.started":"2023-07-01T17:57:26.697768Z","shell.execute_reply":"2023-07-01T17:57:29.68022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import requests\nimport json\n\ndef get_citation_ids(protein_id):\n    url = f\"https://rest.uniprot.org/uniprotkb/{protein_id}/publications\"\n    response = requests.get(url)\n    data = response.json()\n\n    citation_ids = []\n    for result in data[\"results\"]:\n        citation = result[\"citation\"]\n        citation_id = citation[\"id\"]\n        citation_ids.append(citation_id)\n\n    return citation_ids\n\n# Example usage\nprotein_id = \"P54366\"\nids = get_citation_ids(protein_id)\n\n# Print the list of citation IDs\nprint(\"Citation IDs:\")\nfor citation_id in ids:\n    print(citation_id)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-29T07:15:41.042755Z","iopub.execute_input":"2023-06-29T07:15:41.043106Z","iopub.status.idle":"2023-06-29T07:15:42.020075Z","shell.execute_reply.started":"2023-06-29T07:15:41.043078Z","shell.execute_reply":"2023-06-29T07:15:42.018922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"protein_df_train['Citation IDs'] = protein_df_train['Sequence ID'].apply(get_citation_ids)\nprotein_df_train.to_csv('/kaggle/working/train_pubmed_ids.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-28T07:54:53.871797Z","iopub.execute_input":"2023-06-28T07:54:53.872234Z","iopub.status.idle":"2023-06-28T07:56:49.779207Z","shell.execute_reply.started":"2023-06-28T07:54:53.872203Z","shell.execute_reply":"2023-06-28T07:56:49.774756Z"}}},{"cell_type":"code","source":"from tqdm import tqdm\nimport time\ncount =120000\ncount_end = 150000\nprotein_ids = protein_df_train['Sequence ID'][count:count_end]\ncitation_ids_list = []\ncount = 0\nts = time.time()\n\nfor protein_id in protein_ids:\n    \n    if (count+1)%200==0:\n        print(count+1)\n        print(time.time()-ts)\n        ts = time.time()\n    try:\n        citation_ids = get_citation_ids(protein_id)\n        citation_ids_list.append(citation_ids)\n        count = count+1\n    except:\n        citation_ids_list.append([''])\n        count = count+1\n        pass\n\n# Add the citation IDs to the dataframe\nprotein_df = pd.DataFrame({'Sequence ID':protein_ids})\n\nprotein_df['Citation ID'] = citation_ids_list\n\n# Save the dataframe as a CSV file\nprotein_df.to_csv('/kaggle/working/test_pubmed_ids_60k_80k.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-29T07:16:12.98469Z","iopub.execute_input":"2023-06-29T07:16:12.985048Z","iopub.status.idle":"2023-06-29T07:16:28.552756Z","shell.execute_reply.started":"2023-06-29T07:16:12.98502Z","shell.execute_reply":"2023-06-29T07:16:28.551667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"protein_df","metadata":{"execution":{"iopub.status.busy":"2023-06-29T07:16:41.413393Z","iopub.execute_input":"2023-06-29T07:16:41.413746Z","iopub.status.idle":"2023-06-29T07:16:41.438401Z","shell.execute_reply.started":"2023-06-29T07:16:41.413719Z","shell.execute_reply":"2023-06-29T07:16:41.437436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import Entrez\n\n# Set your email address (required by NCBI)\nEntrez.email = 'your_email@example.com'\n\ndef fetch_abstract(pubmed_id):\n    handle = Entrez.efetch(db='pubmed', id=pubmed_id, retmode='xml')\n    record = Entrez.read(handle)\n    handle.close()\n    abstract = record['PubmedArticle'][0]['MedlineCitation']['Article']['Abstract']['AbstractText']\n    return ' '.join(abstract)\n\n# Example usage\npubmed_ids = ['2219722', 'CI-37OHSPBO5F2CO', '24603707', '26045555', '27599859', '16326701']\n\nfor pubmed_id in pubmed_ids:\n    try:\n        abstract_text = fetch_abstract(pubmed_id)\n        print(f\"PubMed ID: {pubmed_id}\\nAbstract: {abstract_text}\\n\")\n    except Exception as e:\n        print(f\"Error retrieving data for PubMed ID: {pubmed_id}\\n{str(e)}\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T04:17:52.145876Z","iopub.execute_input":"2023-06-30T04:17:52.14673Z","iopub.status.idle":"2023-06-30T04:17:55.021791Z","shell.execute_reply.started":"2023-06-30T04:17:52.146667Z","shell.execute_reply":"2023-06-30T04:17:55.020626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Bio import Entrez\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom gensim.models import Doc2Vec\nfrom gensim.utils import simple_preprocess\nfrom gensim.models.doc2vec import TaggedDocument\n\n# Set your email address (required by NCBI)\nEntrez.email = 'your_email@example.com'\n\ndef fetch_abstract(pubmed_id):\n    handle = Entrez.efetch(db='pubmed', id=pubmed_id, retmode='xml')\n    record = Entrez.read(handle)\n    handle.close()\n    abstract = record['PubmedArticle'][0]['MedlineCitation']['Article']['Abstract']['AbstractText']\n    return ' '.join(abstract)\n\n# Protein documents with PubMed IDs\nprotein_pubmed_ids = {\n    'P20536': ['2219722', 'CI-37OHSPBO5F2CO', '24603707', '26045555', '27599859', '16326701'],\n    'O73864': ['9507106', '9142986', '9007254', '9007240', '30559456', '28222105', '26459057', '24928507', '24346703', '24177263', '23201782', '22773843', '22357957', '22039507', '20843857', '20188722', '19595791', '19046963', '18971206', '18701549', '18538138', '17906624', '17689523', '17611617', '17412835'],\n    'O95231': ['11549314', '15164054', '15489334', '8617720', '8234276', '32573491', '29872044', '27888632', '27175592', '25416956', '24706756', '2405250', '22791709', '22178396', '21979375', '2183031', '21670496', '21325273', '20833819', '20452968', '20351173', '20028861', '19805069', '18555778', '18555777'],\n    'A0A0B4J1F4': ['19468303', 'CI-70VOB8UOSDVQL', '15489334', '19605364', '27462458', '35950500', '34916487', '34188787', '31541165', '24398455', '21677750', '16766796', '14610273', '12466851', '11217851'],\n    'P54366': ['8625850', '8670808', '10731132', '12537572', '9449676', '9335144', '9292972', '9175876', '35723254', '34389661', '33438579', '33163946', '32601058', '30247122', '29670218', '28860209', '28246214', '28092692', '27317776', '26550828', '26464018', '26053861', '25791631', '25779349', '25687947'],\n    'P33681': ['2794510', '1377173', '17953528', '16641997', '15489334', '1714935', '7527824', '10583602', '16920215', '23186163', '30918073', '10661405', '11279502', '9973466', '9822663', '9712716', '9694876', '9438848', '9032261', '8990121', '8946678', '8946277', '8657148', '8649453', '8638161']\n}\n\n# Fetch abstracts for each protein\nprotein_abstracts = {}\nfor protein_id, pubmed_ids in protein_pubmed_ids.items():\n    abstracts = []\n    for pubmed_id in pubmed_ids:\n        try:\n            abstract = fetch_abstract(pubmed_id)\n            abstracts.append(abstract)\n        except Exception as e:\n            print(f\"Error retrieving data for PubMed ID: {pubmed_id}\\n{str(e)}\\n\")\n    protein_abstracts[protein_id] = ' '.join(abstracts)\n\n# Concatenate abstracts and preprocess the text\nconcatenated_documents = [protein_abstracts[protein_id] for protein_id in protein_abstracts]\npreprocessed_documents = [simple_preprocess(doc) for doc in concatenated_documents]\n\n# Convert the documents into TF-IDF vectors\nvectorizer = TfidfVectorizer()\ntfidf_vectors = vectorizer.fit_transform([' '.join(doc) for doc in preprocessed_documents])\n\n# Get the feature names (terms) corresponding to the TF-IDF vectors\nfeature_names = vectorizer.get_feature_names_out()\n\n# Print TF-IDF vectors for each protein\n#for protein_id, vector in zip(protein_abstracts.keys(), tfidf_vectors):\n#    print(f\"Protein ID: {protein_id}\")\n#    for feature_index, feature_value in zip(vector.indices, vector.data):\n#        feature_name = feature_names[feature_index]\n#        print(f\"Term: {feature_name}, TF-IDF: {feature_value}\")\n#    print()\n    \n# Convert the documents into TaggedDocuments for Doc2Vec\ntagged_documents = [TaggedDocument(words=doc, tags=[protein_id])\n                    for protein_id, doc in zip(protein_abstracts.keys(), preprocessed_documents)]\n\n# Train the Doc2Vec model\nmodel = Doc2Vec(tagged_documents, vector_size=100, min_count=1, epochs=10)\n\n# Get the Doc2Vec vectors for each protein\ndoc2vec_vectors = {protein_id: model.infer_vector([protein_id])\n                   for protein_id in protein_abstracts}\n\n# Print Doc2Vec vectors for each protein\n#for protein_id, vector in doc2vec_vectors.items():\n#    print(f\"Protein ID: {protein_id}\")\n#    print(f\"Vector: {vector}\")\n#    print()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-30T04:18:04.319223Z","iopub.execute_input":"2023-06-30T04:18:04.319751Z","iopub.status.idle":"2023-06-30T04:19:02.118417Z","shell.execute_reply.started":"2023-06-30T04:18:04.319689Z","shell.execute_reply":"2023-06-30T04:19:02.117097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tfidf_vectors","metadata":{"execution":{"iopub.status.busy":"2023-06-29T07:55:00.477562Z","iopub.execute_input":"2023-06-29T07:55:00.477907Z","iopub.status.idle":"2023-06-29T07:55:00.484008Z","shell.execute_reply.started":"2023-06-29T07:55:00.47788Z","shell.execute_reply":"2023-06-29T07:55:00.483134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"doc2vec_vectors","metadata":{"execution":{"iopub.status.busy":"2023-06-29T07:57:09.706073Z","iopub.execute_input":"2023-06-29T07:57:09.706806Z","iopub.status.idle":"2023-06-29T07:57:09.723791Z","shell.execute_reply.started":"2023-06-29T07:57:09.706774Z","shell.execute_reply":"2023-06-29T07:57:09.72277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}