{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"# -----------------------\n# System and Utility Libraries\n# -----------------------\nimport os\nimport sys\nimport time\nimport gc\nimport pickle\nimport warnings\nimport csv\nfrom collections import Counter\nfrom tqdm import tqdm\n\n# -----------------------\n# Data Manipulation and Visualization Libraries\n# -----------------------\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom scipy.sparse import vstack\n\n# -----------------------\n# File and IO Operations\n# -----------------------\nimport glob\nfrom joblib import dump, load\n\n# -----------------------\n# Biological Sequence Processing\n# -----------------------\nfrom Bio import SeqIO\n\n# -----------------------\n# Machine Learning Libraries\n# -----------------------\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, MultiLabelBinarizer, StandardScaler\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.utils import class_weight\n\n# -----------------------\n# Imbalanced Learning\n# -----------------------\nfrom imblearn.over_sampling import RandomOverSampler, SMOTE\nfrom imblearn.keras import BalancedBatchGenerator\n\n# -----------------------\n# Deep Learning with TensorFlow and Keras\n# -----------------------\nimport tensorflow as tf\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization, Input, Flatten, LeakyReLU\nfrom tensorflow.keras.optimizers import Adam, RMSprop, schedules\nfrom tensorflow.keras.metrics import Precision, Recall\nfrom tensorflow.keras.callbacks import (EarlyStopping, ModelCheckpoint, \n                                        ReduceLROnPlateau, TensorBoard, LearningRateScheduler)\n\n# Setting up warning settings\nwarnings.filterwarnings('ignore')\nif not sys.warnoptions:\n    warnings.simplefilter(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:01:42.481267Z","iopub.execute_input":"2023-08-15T21:01:42.481845Z","iopub.status.idle":"2023-08-15T21:01:55.067956Z","shell.execute_reply.started":"2023-08-15T21:01:42.481792Z","shell.execute_reply":"2023-08-15T21:01:55.066602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/cafa-5-protein-competion-obonet-1-0-0/obonet-1.0.0-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:01:55.070607Z","iopub.execute_input":"2023-08-15T21:01:55.071422Z","iopub.status.idle":"2023-08-15T21:02:30.137634Z","shell.execute_reply.started":"2023-08-15T21:01:55.07138Z","shell.execute_reply":"2023-08-15T21:02:30.136558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nprint(sys.version)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:30.139502Z","iopub.execute_input":"2023-08-15T21:02:30.140089Z","iopub.status.idle":"2023-08-15T21:02:30.14739Z","shell.execute_reply.started":"2023-08-15T21:02:30.14004Z","shell.execute_reply":"2023-08-15T21:02:30.146036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gensim\nprint(gensim.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:30.149055Z","iopub.execute_input":"2023-08-15T21:02:30.149377Z","iopub.status.idle":"2023-08-15T21:02:30.362478Z","shell.execute_reply.started":"2023-08-15T21:02:30.14935Z","shell.execute_reply":"2023-08-15T21:02:30.361247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import obonet\nprint(obonet.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:30.366448Z","iopub.execute_input":"2023-08-15T21:02:30.366898Z","iopub.status.idle":"2023-08-15T21:02:30.708959Z","shell.execute_reply.started":"2023-08-15T21:02:30.366863Z","shell.execute_reply":"2023-08-15T21:02:30.707564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# The 'kaggle/input' folder is the root directory where all competition data is stored\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:30.710419Z","iopub.execute_input":"2023-08-15T21:02:30.710834Z","iopub.status.idle":"2023-08-15T21:02:30.762127Z","shell.execute_reply.started":"2023-08-15T21:02:30.710801Z","shell.execute_reply":"2023-08-15T21:02:30.761246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading and pre-processing","metadata":{}},{"cell_type":"code","source":"# Suppress specific warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\n# Define files and dirs\ndirs = {\n    \"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\",\n    \"output\": \"/kaggle/working/\"\n}\n\nfiles = {\n    \"train_sequences\": \"Train/train_sequences.fasta\",\n    \"train_terms\": \"Train/train_terms.tsv\",\n    \"train_taxonomy\": \"Train/train_taxonomy.tsv\",\n    \"go-basic\": \"go-basic.obo\",\n    \"testsuperset\": \"Test (Targets)/testsuperset.fasta\",  # Updated this line\n    \"testsuperset_taxon\": \"Test (Targets)/testsuperset-taxon-list.tsv\",  # You may need to update this too if this file is also in the \"Test (Targets)\" subdirectory\n    \"IA\": \"IA.txt\",\n    \"sample_submission\": \"sample_submission.tsv\",\n    \"submission\": \"submission.csv\"\n}\n\nstart = time.time()\n\ntry:\n    # Loading the data\n    train_sequences = {record.id: str(record.seq) for record in SeqIO.parse(dirs[\"input\"] + files[\"train_sequences\"], \"fasta\")}\n    train_sequences_df = pd.DataFrame.from_dict(train_sequences, orient='index', columns=['Sequence'])\n    train_sequences_df.reset_index(inplace=True)\n    train_sequences_df.rename(columns={\"index\": \"Id\"}, inplace=True)\n\n    train_terms = pd.read_csv(dirs[\"input\"] + files[\"train_terms\"], sep=\"\\t\", header=None, names=[\"Id\", \"Term\", \"Ontology\"])\n    train_taxonomy = pd.read_csv(dirs[\"input\"] + files[\"train_taxonomy\"], sep=\"\\t\", header=None, names=[\"Id\", \"Taxon\"])\n\n    # Adjusted file path for IA.txt\n    ia_file_path = dirs[\"input\"] + files[\"IA\"]\n    with open(ia_file_path, 'r') as file:\n        ia_content = file.read()\n        # Process the content of IA.txt as needed\n\n    # Adjusted file path for sample_submission.tsv\n    sample_submission_file_path = dirs[\"input\"] + files[\"sample_submission\"]\n    sample_submission = pd.read_csv(sample_submission_file_path)\n\n    end = time.time()\n    print(\"Execution time with original dataset:\", end - start)\n\nexcept Exception as e:\n    print(f\"An error occurred: {e}\")\n\nprint(f\"Number of training sequences: {len(train_sequences)}\")\nprint(f\"First few rows of train_terms:\\n{train_terms.head()}\")\nprint(f\"First few rows of train_taxonomy:\\n{train_taxonomy.head()}\")\nprint(f\"First few rows of sample_submission:\\n{sample_submission.head()}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:30.763089Z","iopub.execute_input":"2023-08-15T21:02:30.763398Z","iopub.status.idle":"2023-08-15T21:02:38.098788Z","shell.execute_reply.started":"2023-08-15T21:02:30.763372Z","shell.execute_reply":"2023-08-15T21:02:38.097539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Access individual files","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"IA\": \"IA.txt\"}\n\nia_file_path = dirs[\"input\"] + files[\"IA\"]\n\nwith open(ia_file_path, 'r') as file:\n    ia_content = file.read()\n\nprint(ia_content[:500])  # prints the first 500 characters","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:38.100228Z","iopub.execute_input":"2023-08-15T21:02:38.100987Z","iopub.status.idle":"2023-08-15T21:02:38.111958Z","shell.execute_reply.started":"2023-08-15T21:02:38.100949Z","shell.execute_reply":"2023-08-15T21:02:38.109465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training sequences","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"fasta\": \"Train/train_sequences.fasta\"}\n\n# Creating the complete path to the file\nfasta_file_path = dirs[\"input\"] + files[\"fasta\"]\n\n# Loading the fasta file\nsequences = list(SeqIO.parse(fasta_file_path, \"fasta\"))\n\n# Creating lists to store the IDs and the sequences\nsequence_ids = [sequence.id for sequence in sequences]\nsequence_seqs = [str(sequence.seq) for sequence in sequences]\n\n# Creating a DataFrame\ndf_sequences = pd.DataFrame({'ID': sequence_ids, 'Sequences': sequence_seqs})\n\nprint(df_sequences.head())","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:38.114426Z","iopub.execute_input":"2023-08-15T21:02:38.11527Z","iopub.status.idle":"2023-08-15T21:02:42.177377Z","shell.execute_reply.started":"2023-08-15T21:02:38.11522Z","shell.execute_reply":"2023-08-15T21:02:42.176199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sequence count:\n\nnum_sequences = len(sequences)\nprint(\"Total number of sequences:\", num_sequences)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:42.1789Z","iopub.execute_input":"2023-08-15T21:02:42.179939Z","iopub.status.idle":"2023-08-15T21:02:42.186273Z","shell.execute_reply.started":"2023-08-15T21:02:42.179894Z","shell.execute_reply":"2023-08-15T21:02:42.185151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# String size statistics:\n\nsizes = [len(sequence) for sequence in sequences]\naverage_size = sum(sizes) / len(sizes)\nmin_size = min(sizes)\nmax_size = max(sizes)\nprint(\"Average string size:\", average_size)\nprint(\"Minimum string size:\", min_size)\nprint(\"Maximum length of sequences:\", max_size)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:42.187908Z","iopub.execute_input":"2023-08-15T21:02:42.188632Z","iopub.status.idle":"2023-08-15T21:02:42.353443Z","shell.execute_reply.started":"2023-08-15T21:02:42.188565Z","shell.execute_reply":"2023-08-15T21:02:42.351993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Break each amino acid sequence into individual characters and count the frequency of each amino acid\namino_acid_counts = df_sequences['Sequences'].str.split('').explode().value_counts()\nprint(\"Most frequent amino acids:\")\nprint(amino_acid_counts)\n\n# This function will split a sequence of amino acids into triplets\ndef split_into_triplets(sequence):\n    return [sequence[i:i+3] for i in range(0, len(sequence), 3)]\n\n# Apply the function to each sequence of amino acids and count the frequency of each triplet\ntriplet_counts = df_sequences['Sequences'].apply(split_into_triplets).explode().value_counts()\nprint(\"Most frequent amino acid triplets:\")\nprint(triplet_counts)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:02:42.354989Z","iopub.execute_input":"2023-08-15T21:02:42.355369Z","iopub.status.idle":"2023-08-15T21:03:24.665949Z","shell.execute_reply.started":"2023-08-15T21:02:42.355337Z","shell.execute_reply":"2023-08-15T21:03:24.664536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Graph of the 10 most frequent amino acids\ntop_amino_acids = amino_acid_counts.head(10)\nplt.figure(figsize=(12, 6))\nsns.barplot(x=top_amino_acids.index, y=top_amino_acids.values)\nplt.xlabel('amino acid')\nplt.ylabel('Frequency')\nplt.title('Top 10 Most Common Amino Acids')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:24.667523Z","iopub.execute_input":"2023-08-15T21:03:24.66794Z","iopub.status.idle":"2023-08-15T21:03:25.076173Z","shell.execute_reply.started":"2023-08-15T21:03:24.667905Z","shell.execute_reply":"2023-08-15T21:03:25.075234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Chart of the 10 most common amino acid triplets\ntop_triplets = triplet_counts.head(10)\nplt.figure(figsize=(12, 6))\nsns.barplot(x=top_triplets.index, y=top_triplets.values)\nplt.xlabel('Trick of Amino Acids')\nplt.ylabel('Frequency')\nplt.title('Top 10 Most Common Amino Acid Tricks')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:25.081765Z","iopub.execute_input":"2023-08-15T21:03:25.082393Z","iopub.status.idle":"2023-08-15T21:03:25.450496Z","shell.execute_reply.started":"2023-08-15T21:03:25.082355Z","shell.execute_reply":"2023-08-15T21:03:25.449594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# String filtering:\nthreshold_size = 100  # Minimum desired size\nfiltered_sequences = [sequence for sequence in sequences if len(sequence) > threshold_size]\n\n# Limit to print\nprint_limit = 5\nprint_count = 0\n\nfor sequence in filtered_sequences:\n    if print_count < print_limit:\n        print(sequence.id)\n        print(sequence.seq)\n        print_count += 1\n    else:\n        break\n\n# Nucleotide content analysis:\nnucleotide_counts = {\"A\": 0, \"C\": 0, \"G\": 0, \"T\": 0}\ntotal_bases = 0\n\nfor sequence in filtered_sequences:\n    for base in sequence.seq:\n        if base in nucleotide_counts:\n            nucleotide_counts[base] += 1\n            total_bases += 1\n\nfrequencies = {base: count / total_bases for base, count in nucleotide_counts.items()}\n\nprint(\"base count:\", nucleotide_counts)\nprint(\"Relative frequencies of bases:\", frequencies)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:25.45167Z","iopub.execute_input":"2023-08-15T21:03:25.452821Z","iopub.status.idle":"2023-08-15T21:03:42.927266Z","shell.execute_reply.started":"2023-08-15T21:03:25.452773Z","shell.execute_reply":"2023-08-15T21:03:42.926028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Relative base frequencies\nbases = ['A', 'C', 'G', 'T']\nfrequencies = {base: frequencies[base] for base in bases}\n\n# Total length of strings\ntotal_size = sum(sizes)\n\n# Create the figure with custom size\nplt.figure(figsize=(12, 6))\n\n# Create the bar chart\nplt.bar(bases + ['Total'], [frequencies[base] for base in bases] + [1], color=['blue', 'green', 'orange', 'red', 'gray'])\nplt.xlabel('Nucleotide Bases')\nplt.ylabel('Relative Frequency / Total Size')\nplt.title('Relative Frequencies of Nucleotide Bases')\n\n# Display the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:42.929937Z","iopub.execute_input":"2023-08-15T21:03:42.930434Z","iopub.status.idle":"2023-08-15T21:03:43.270782Z","shell.execute_reply.started":"2023-08-15T21:03:42.930389Z","shell.execute_reply":"2023-08-15T21:03:43.268958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training labels","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"terms\": \"Train/train_terms.tsv\"}\n\n# Creating the complete path to the file\nterms_file_path = dirs[\"input\"] + files[\"terms\"]\n\n# Loading the TSV file with column names\ncolumn_names = [\"EntryID\", \"Term\", \"Aspect\"]\ntrain_labels = pd.read_csv(terms_file_path, sep='\\t', names=column_names)\n\n# Visualizing the first rows\ntrain_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:43.273081Z","iopub.execute_input":"2023-08-15T21:03:43.273656Z","iopub.status.idle":"2023-08-15T21:03:45.979184Z","shell.execute_reply.started":"2023-08-15T21:03:43.273607Z","shell.execute_reply":"2023-08-15T21:03:45.978222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.info()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:45.980487Z","iopub.execute_input":"2023-08-15T21:03:45.981262Z","iopub.status.idle":"2023-08-15T21:03:45.996477Z","shell.execute_reply.started":"2023-08-15T21:03:45.98123Z","shell.execute_reply":"2023-08-15T21:03:45.99559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\nmissing_values = train_labels.isnull().sum()\nprint(\"Missing values per column:\")\nprint(missing_values)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:45.99797Z","iopub.execute_input":"2023-08-15T21:03:45.998629Z","iopub.status.idle":"2023-08-15T21:03:51.156637Z","shell.execute_reply.started":"2023-08-15T21:03:45.998591Z","shell.execute_reply":"2023-08-15T21:03:51.155267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the quantity per column\nfor column in train_labels.columns:\n    counts = train_labels[column].value_counts()\n    print(f\"Number of values in the column '{column}':\")\n    print(counts)\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:51.157963Z","iopub.execute_input":"2023-08-15T21:03:51.159137Z","iopub.status.idle":"2023-08-15T21:03:54.050705Z","shell.execute_reply.started":"2023-08-15T21:03:51.159085Z","shell.execute_reply":"2023-08-15T21:03:54.049098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get top 20 categories\ntop_categories = train_labels['EntryID'].value_counts().nlargest(20)\n\nplt.figure(figsize=(10, 8))\nsns.barplot(x=top_categories.values, y=top_categories.index)\n\nplt.xlabel('Score')\nplt.ylabel('EntryID')\nplt.title('Top 20 EntryID')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:54.053937Z","iopub.execute_input":"2023-08-15T21:03:54.05432Z","iopub.status.idle":"2023-08-15T21:03:55.437925Z","shell.execute_reply.started":"2023-08-15T21:03:54.05429Z","shell.execute_reply":"2023-08-15T21:03:55.436247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get top 20 categories\ntop_categories = train_labels['Term'].value_counts().nlargest(20)\n\nplt.figure(figsize=(10, 8))\nsns.barplot(x=top_categories.values, y=top_categories.index)\n\nplt.xlabel('Score')\nplt.ylabel('Term')\nplt.title('Top 20 Terms')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:55.441779Z","iopub.execute_input":"2023-08-15T21:03:55.442212Z","iopub.status.idle":"2023-08-15T21:03:57.468453Z","shell.execute_reply.started":"2023-08-15T21:03:55.442176Z","shell.execute_reply":"2023-08-15T21:03:57.467527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get top 20 categories\ntop_categories = train_labels['Aspect'].value_counts().nlargest(20)\n\nplt.figure(figsize=(10, 8))\nsns.barplot(x=top_categories.values, y=top_categories.index)\n\nplt.xlabel('Score')\nplt.ylabel('Aspect')\nplt.title('Top 20 Aspects')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:57.469726Z","iopub.execute_input":"2023-08-15T21:03:57.470766Z","iopub.status.idle":"2023-08-15T21:03:58.613724Z","shell.execute_reply.started":"2023-08-15T21:03:57.470729Z","shell.execute_reply":"2023-08-15T21:03:58.611889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training taxonomy","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"taxonomy\": \"Train/train_taxonomy.tsv\"}\n\n# Creating the complete path to the file\ntaxonomy_file_path = dirs[\"input\"] + files[\"taxonomy\"]\n\n# Loading the TSV file\ntrain_taxonomy = pd.read_csv(taxonomy_file_path, sep='\\t')\n\n# Visualizing the first rows\ntrain_taxonomy.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:58.615435Z","iopub.execute_input":"2023-08-15T21:03:58.615909Z","iopub.status.idle":"2023-08-15T21:03:58.727559Z","shell.execute_reply.started":"2023-08-15T21:03:58.615874Z","shell.execute_reply":"2023-08-15T21:03:58.726129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check general information of the DataFrame\ntrain_taxonomy.info()\n\n# Count of unique values in column \"taxonomyID\"\nunique_taxonomies = train_taxonomy['taxonomyID'].nunique()\nprint(\"Number of unique taxonomies:\", unique_taxonomies)\n\n# Check taxonomies distribution\ntaxonomy_counts = train_taxonomy['taxonomyID'].value_counts()\nprint(\"taxonomies count:\")\nprint(taxonomy_counts)\n\n# Check the size of each taxonomy group\ngroup_sizes = train_taxonomy.groupby('taxonomyID').size()\nprint(\"Taxonomy group sizes:\")\nprint(group_sizes)\n\n# Check for missing values in DataFrame\nmissing_values = train_taxonomy.isnull().sum()\nprint(\"Missing values per column:\")\nprint(missing_values)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:58.728907Z","iopub.execute_input":"2023-08-15T21:03:58.729324Z","iopub.status.idle":"2023-08-15T21:03:58.857912Z","shell.execute_reply.started":"2023-08-15T21:03:58.729295Z","shell.execute_reply":"2023-08-15T21:03:58.856082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a Counter to count the frequency of taxonomies\ntaxonomy_counts = Counter(train_taxonomy['taxonomyID'])\n\n# Print the most common taxonomies\nprint(\"Most common taxonomies:\")\nfor taxonomy, count in taxonomy_counts.most_common(10):\n    print(f\"{taxonomy}: {count}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:58.859363Z","iopub.execute_input":"2023-08-15T21:03:58.85972Z","iopub.status.idle":"2023-08-15T21:03:58.90667Z","shell.execute_reply.started":"2023-08-15T21:03:58.859691Z","shell.execute_reply":"2023-08-15T21:03:58.905019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the 10 most common taxonomies\ntop_10_taxonomies = dict(taxonomy_counts.most_common(10))\n\n# Create a DataFrame for plotting\ndf_top_taxonomies = pd.DataFrame(top_10_taxonomies.items(), columns=['TaxonomyID', 'Count'])\n\n# Create a bar chart\nplt.figure(figsize=(12, 6))\nsns.barplot(x='TaxonomyID', y='Count', data=df_top_taxonomies)\nplt.title('Top 10 Most Frequent Taxonomies')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:58.908069Z","iopub.execute_input":"2023-08-15T21:03:58.908429Z","iopub.status.idle":"2023-08-15T21:03:59.238688Z","shell.execute_reply.started":"2023-08-15T21:03:58.908398Z","shell.execute_reply":"2023-08-15T21:03:59.237334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Structure of the Genetic Ontology","metadata":{}},{"cell_type":"code","source":"import obonet\n\ndirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"go-basic\": \"Train/go-basic.obo\"}\n\n# Creating the complete path to the file\ngo_basic_file_path = dirs[\"input\"] + files[\"go-basic\"]\n\n# Loading the ontology structure\ngraph = obonet.read_obo(go_basic_file_path)\n\n# Displaying the first nodes' information\nfor node, data in list(graph.nodes(data=True))[:5]:\n    print(f\"Node: {node}\")\n    print(f\"Data: {data}\")\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:03:59.240706Z","iopub.execute_input":"2023-08-15T21:03:59.241201Z","iopub.status.idle":"2023-08-15T21:04:19.183868Z","shell.execute_reply.started":"2023-08-15T21:03:59.241157Z","shell.execute_reply":"2023-08-15T21:04:19.182391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the first edge information\nfor edge in list(graph.edges(data=True))[:5]:\n    print(f\"Aresta: {edge}\")\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:19.18535Z","iopub.execute_input":"2023-08-15T21:04:19.185756Z","iopub.status.idle":"2023-08-15T21:04:19.393852Z","shell.execute_reply.started":"2023-08-15T21:04:19.185722Z","shell.execute_reply":"2023-08-15T21:04:19.392641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the total number of terms in the genetic ontology:\ntotal_terms = len(graph)\nprint(\"Total number of terms in the genetic ontology:\", total_terms)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:19.395612Z","iopub.execute_input":"2023-08-15T21:04:19.39596Z","iopub.status.idle":"2023-08-15T21:04:19.401598Z","shell.execute_reply.started":"2023-08-15T21:04:19.395929Z","shell.execute_reply":"2023-08-15T21:04:19.400484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"train_terms\": \"Train/train_terms.tsv\"}\n\n# Creating the complete path to the file\ntrain_terms_file_path = dirs[\"input\"] + files[\"train_terms\"]\n\n# Loading the TSV file\nprotein_to_go = pd.read_csv(train_terms_file_path, sep='\\t', header=None)\n\n# Adding column names\nprotein_to_go.columns = ['EntryID', 'Term', 'Aspect']\n\n# Previewing the first few rows\nprotein_to_go.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:19.403246Z","iopub.execute_input":"2023-08-15T21:04:19.403659Z","iopub.status.idle":"2023-08-15T21:04:22.091701Z","shell.execute_reply.started":"2023-08-15T21:04:19.403623Z","shell.execute_reply":"2023-08-15T21:04:22.090407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test sequences","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"testsuperset\": \"Test (Targets)/testsuperset.fasta\"}\n\n# Creating the complete path to the file\ntestsuperset_file_path = dirs[\"input\"] + files[\"testsuperset\"]\n\n# Loading the FASTA file\ntest_sequences = list(SeqIO.parse(testsuperset_file_path, \"fasta\"))\n\n# Printing the first 5 sequences\nfor sequence in test_sequences[:5]:\n    print(sequence.id)\n    print(sequence.seq)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:22.094726Z","iopub.execute_input":"2023-08-15T21:04:22.096515Z","iopub.status.idle":"2023-08-15T21:04:25.136937Z","shell.execute_reply.started":"2023-08-15T21:04:22.096463Z","shell.execute_reply":"2023-08-15T21:04:25.135667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test taxonomy","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"testsuperset_taxon\": \"Test (Targets)/testsuperset-taxon-list.tsv\"}\n\n# Creating the complete path to the file\ntestsuperset_taxon_file_path = dirs[\"input\"] + files[\"testsuperset_taxon\"]\n\n# Loading the TSV file with 'ISO-8859-1' encoding\ntest_taxonomy = pd.read_csv(testsuperset_taxon_file_path, sep='\\t', encoding='ISO-8859-1')\n\n# Printing the first few rows\ntest_taxonomy.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:25.141035Z","iopub.execute_input":"2023-08-15T21:04:25.141458Z","iopub.status.idle":"2023-08-15T21:04:25.160377Z","shell.execute_reply.started":"2023-08-15T21:04:25.141427Z","shell.execute_reply":"2023-08-15T21:04:25.158535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Addition of Information","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"IA\": \"IA.txt\"}\n\n# Creating the complete path to the file\nia_file_path = dirs[\"input\"] + files[\"IA\"]\n\n# Loading the TSV file (assuming it is tab-separated)\nia_weights = pd.read_csv(ia_file_path, sep='\\t')\n\n# Printing the first few rows\nia_weights.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:25.162019Z","iopub.execute_input":"2023-08-15T21:04:25.162425Z","iopub.status.idle":"2023-08-15T21:04:25.209704Z","shell.execute_reply.started":"2023-08-15T21:04:25.162392Z","shell.execute_reply":"2023-08-15T21:04:25.208395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and merging training data","metadata":{}},{"cell_type":"code","source":"dirs = {\"input\": \"/kaggle/input/cafa-5-protein-predict-competion/\"}\nfiles = {\"train_terms\": \"Train/train_terms.tsv\",\n         \"train_taxonomy\": \"Train/train_taxonomy.tsv\"}\n\n# Creating the complete path to the files\ntrain_terms_file_path = dirs[\"input\"] + files[\"train_terms\"]\ntrain_taxonomy_file_path = dirs[\"input\"] + files[\"train_taxonomy\"]\n\n# Loading the TSV file for protein_to_go\nprotein_to_go = pd.read_csv(train_terms_file_path, sep='\\t', header=None)\n\n# Adding column names\nprotein_to_go.columns = ['EntryID', 'Term', 'Aspect']\n\n# Loading the TSV file for train_taxonomy\ntrain_taxonomy = pd.read_csv(train_taxonomy_file_path, sep='\\t', header=None)\n\n# Adding column names\ntrain_taxonomy.columns = ['EntryID', 'Taxon']\n\n# Merging the dataframes\ndf_prot = pd.merge(train_taxonomy, protein_to_go, on=\"EntryID\")\n\n# Previewing the first few rows\ndf_prot.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:25.211289Z","iopub.execute_input":"2023-08-15T21:04:25.212457Z","iopub.status.idle":"2023-08-15T21:04:30.781036Z","shell.execute_reply.started":"2023-08-15T21:04:25.21241Z","shell.execute_reply":"2023-08-15T21:04:30.779766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = df_prot.value_counts()\n\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:30.782429Z","iopub.execute_input":"2023-08-15T21:04:30.782928Z","iopub.status.idle":"2023-08-15T21:04:36.161479Z","shell.execute_reply.started":"2023-08-15T21:04:30.782897Z","shell.execute_reply":"2023-08-15T21:04:36.160149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count unique values in column \"term\"\nterm_counts = protein_to_go['Term'].value_counts()\nprint(\"Most frequent GO terms:\")\nprint(term_counts.head())\n\n# Count unique values in column \"aspect\"\naspect_counts = protein_to_go['Aspect'].value_counts()\nprint(\"\\nMost frequent aspects:\")\nprint(aspect_counts)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:36.164659Z","iopub.execute_input":"2023-08-15T21:04:36.165366Z","iopub.status.idle":"2023-08-15T21:04:38.087407Z","shell.execute_reply.started":"2023-08-15T21:04:36.165328Z","shell.execute_reply":"2023-08-15T21:04:38.086167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Analyze the correlation between aspects and GO terms:\n\n# Group the data by aspect, term, and taxonomyID, and count the frequency of each pair\naspect_term_taxonomy_counts = df_prot.groupby(['Aspect', 'Term', 'Taxon']).size().reset_index(name='counts')\n\n# Display the most frequent aspect-term-taxonomyID pairs\nprint(aspect_term_taxonomy_counts.sort_values('counts', ascending=False).head())\n\n# Display the most frequent aspect-term pairs\nprint(aspect_term_taxonomy_counts.sort_values('counts', ascending=False).head())","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:38.089438Z","iopub.execute_input":"2023-08-15T21:04:38.089891Z","iopub.status.idle":"2023-08-15T21:04:40.512568Z","shell.execute_reply.started":"2023-08-15T21:04:38.089855Z","shell.execute_reply":"2023-08-15T21:04:40.511654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualization of the distribution of GO terms and aspects:\n\n# Plot a histogram of the distribution of GO terms\nplt.figure(figsize=(10, 6))\nterm_counts.plot(kind='hist')\nplt.title('Distribution of GO Terms')\nplt.xlabel('GO Terms')\nplt.ylabel('Score')\nplt.show()\n\n# Plot a bar chart of aspects\nplt.figure(figsize=(10, 6))\naspect_counts.plot(kind='bar')\nplt.title('Distribution of Aspects')\nplt.xlabel('Aspects')\nplt.ylabel('Score')\nplt.xticks(rotation=0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:40.513938Z","iopub.execute_input":"2023-08-15T21:04:40.514853Z","iopub.status.idle":"2023-08-15T21:04:41.179183Z","shell.execute_reply.started":"2023-08-15T21:04:40.514816Z","shell.execute_reply":"2023-08-15T21:04:41.17797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group the data by taxonomyID and term, and count the frequency of each pair\ntaxonomy_term_counts = df_prot.groupby(['Taxon', 'Term']).size().reset_index(name='counts')\n\n# Display the most frequent taxonomyID-term pairs\nprint(taxonomy_term_counts.sort_values('counts', ascending=False).head())","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:41.180832Z","iopub.execute_input":"2023-08-15T21:04:41.181227Z","iopub.status.idle":"2023-08-15T21:04:42.984063Z","shell.execute_reply.started":"2023-08-15T21:04:41.181194Z","shell.execute_reply":"2023-08-15T21:04:42.981718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading and exploring the Genetic Ontology","metadata":{}},{"cell_type":"code","source":"import obonet\n\n# Load the Genetic Ontology chart\ngraph = obonet.read_obo('/kaggle/input/cafa-5-protein-predict-competion/Train/go-basic.obo')\n\n# The roots of the three subontologies (Biological Process, Cellular Component, Molecular Function)\nsubontology_roots = {'BPO': 'GO:0008150', 'CCO': 'GO:0005575', 'MFO': 'GO:0003674'}\n\n# Iterate over the subontologies\nfor root, root_id in subontology_roots.items():\n    # Print the root node information\n    print(f'\\nRoot node for the subontology {root}:')\n    print(graph.nodes[root_id])\n\n    # Print the child nodes of the root node\n    print(f\"\\nNodes children of the root node {root}:\")\n    for edge in graph.out_edges(root_id, data=True):\n        print(edge)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:04:42.99767Z","iopub.execute_input":"2023-08-15T21:04:42.998144Z","iopub.status.idle":"2023-08-15T21:05:02.892408Z","shell.execute_reply.started":"2023-08-15T21:04:42.99811Z","shell.execute_reply":"2023-08-15T21:05:02.890856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for root, root_id in subontology_roots.items():\n    print(f'{root}: {len(list(graph.in_edges(root_id)))} parent nodes')","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:02.893873Z","iopub.execute_input":"2023-08-15T21:05:02.894231Z","iopub.status.idle":"2023-08-15T21:05:02.900707Z","shell.execute_reply.started":"2023-08-15T21:05:02.894202Z","shell.execute_reply":"2023-08-15T21:05:02.899414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_nodes = set(graph.nodes)\nprint('BPO children in all nodes:', any(node.startswith('GO:0008150') for node in all_nodes))\nprint('CCO children in all nodes:', any(node.startswith('GO:0005575') for node in all_nodes))\nprint('MFO children in all nodes:', any(node.startswith('GO:0003674') for node in all_nodes))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:02.902535Z","iopub.execute_input":"2023-08-15T21:05:02.903337Z","iopub.status.idle":"2023-08-15T21:05:02.950386Z","shell.execute_reply.started":"2023-08-15T21:05:02.903299Z","shell.execute_reply":"2023-08-15T21:05:02.94883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training and evaluation of a classification model using DNNs with class imbalance treatment","metadata":{}},{"cell_type":"code","source":"# Listing all the physical GPUs available\ngpus = tf.config.experimental.list_physical_devices('GPU')\n\n# If there are GPUs available\nif gpus:\n    try:\n        # Setting memory limit for the first GPU to 4096 MB\n        tf.config.experimental.set_virtual_device_configuration(\n            gpus[0],\n            [tf.config.experimental.VirtualDeviceConfiguration(memory_limit=4096)])\n    except RuntimeError as e:\n        # Printing any runtime errors that might occur during configuration\n        print(e)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:02.952043Z","iopub.execute_input":"2023-08-15T21:05:02.952514Z","iopub.status.idle":"2023-08-15T21:05:02.970561Z","shell.execute_reply.started":"2023-08-15T21:05:02.952467Z","shell.execute_reply":"2023-08-15T21:05:02.969331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the directory where the files are located\ndirectory = '/kaggle/input/t5embeds/'\n\n# Loading embeddings and IDS\ntrain_embeds = np.load(directory + 'train_embeds.npy').astype(np.float32)\ntrain_ids = np.load(directory + 'train_ids.npy')\ntest_embeds = np.load(directory + 'test_embeds.npy').astype(np.float32)\ntest_ids = np.load(directory + 'test_ids.npy')","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:02.972561Z","iopub.execute_input":"2023-08-15T21:05:02.973371Z","iopub.status.idle":"2023-08-15T21:05:25.705375Z","shell.execute_reply.started":"2023-08-15T21:05:02.973333Z","shell.execute_reply":"2023-08-15T21:05:25.704106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_embeds.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:25.707004Z","iopub.execute_input":"2023-08-15T21:05:25.707354Z","iopub.status.idle":"2023-08-15T21:05:25.714188Z","shell.execute_reply.started":"2023-08-15T21:05:25.707324Z","shell.execute_reply":"2023-08-15T21:05:25.712805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Normalize the embeddings","metadata":{}},{"cell_type":"code","source":"# Normalize the embeddings\nscaler = StandardScaler()\ntrain_embeds = scaler.fit_transform(train_embeds)\ntest_embeds = scaler.transform(test_embeds)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:25.715787Z","iopub.execute_input":"2023-08-15T21:05:25.716144Z","iopub.status.idle":"2023-08-15T21:05:28.618305Z","shell.execute_reply.started":"2023-08-15T21:05:25.716115Z","shell.execute_reply":"2023-08-15T21:05:28.616647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set the encoding dimension and model layers","metadata":{}},{"cell_type":"code","source":"# Define the input placeholder\ninput_data = Input(shape=(1024,))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:28.619565Z","iopub.execute_input":"2023-08-15T21:05:28.619964Z","iopub.status.idle":"2023-08-15T21:05:28.656951Z","shell.execute_reply.started":"2023-08-15T21:05:28.619932Z","shell.execute_reply":"2023-08-15T21:05:28.655633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding process\n# Layer 1: Reduce dimension to 768\nencoded = Dense(768, activation='relu')(input_data)\n# Layer 2: Reduce dimension to 512\nencoded = Dense(512, activation='relu')(encoded)\n# Encoding layer: Final encoded representation, reduce dimension to 256\nencoded_layer = Dense(256, activation='relu')(encoded)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:28.658371Z","iopub.execute_input":"2023-08-15T21:05:28.659067Z","iopub.status.idle":"2023-08-15T21:05:28.899518Z","shell.execute_reply.started":"2023-08-15T21:05:28.65903Z","shell.execute_reply":"2023-08-15T21:05:28.898047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decoding process\n# Layer 1: Increase dimension to 512\ndecoded = Dense(512, activation='relu')(encoded_layer)\n# Layer 2: Increase dimension to 768\ndecoded = Dense(768, activation='relu')(decoded)\n# Final layer: Restore original dimension of 1024\ndecoded = Dense(1024, activation='sigmoid')(decoded)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:28.901101Z","iopub.execute_input":"2023-08-15T21:05:28.90155Z","iopub.status.idle":"2023-08-15T21:05:28.972633Z","shell.execute_reply.started":"2023-08-15T21:05:28.9015Z","shell.execute_reply":"2023-08-15T21:05:28.971543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set the autoencoder and encoder model","metadata":{}},{"cell_type":"code","source":"# Assemble the autoencoder model\nautoencoder = Model(input_data, decoded)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:28.97399Z","iopub.execute_input":"2023-08-15T21:05:28.97456Z","iopub.status.idle":"2023-08-15T21:05:28.99012Z","shell.execute_reply.started":"2023-08-15T21:05:28.974524Z","shell.execute_reply":"2023-08-15T21:05:28.988605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the encoder model\nencoder = Model(input_data, encoded_layer)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:28.991937Z","iopub.execute_input":"2023-08-15T21:05:28.992316Z","iopub.status.idle":"2023-08-15T21:05:29.002964Z","shell.execute_reply.started":"2023-08-15T21:05:28.992284Z","shell.execute_reply":"2023-08-15T21:05:29.001319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and train the autoencoder","metadata":{}},{"cell_type":"code","source":"# Clear the previous session\ntf.keras.backend.clear_session()\n\n# Compile the autoencoder\nautoencoder.compile(optimizer='adam', loss='binary_crossentropy')\n\n# Train the model\nautoencoder.fit(train_embeds, train_embeds,\n                epochs=10,\n                batch_size=256,\n                shuffle=True,\n                validation_data=(test_embeds, test_embeds))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:05:29.004694Z","iopub.execute_input":"2023-08-15T21:05:29.005095Z","iopub.status.idle":"2023-08-15T21:12:53.733641Z","shell.execute_reply.started":"2023-08-15T21:05:29.005051Z","shell.execute_reply":"2023-08-15T21:12:53.73256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Use the encoder to get the reduced dimensionality representations","metadata":{}},{"cell_type":"code","source":"# Use the encoder to get the reduced dimensionality representations\ntrain_embeds_reduced = encoder.predict(train_embeds)\ntest_embeds_reduced = encoder.predict(test_embeds)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:12:53.735205Z","iopub.execute_input":"2023-08-15T21:12:53.736382Z","iopub.status.idle":"2023-08-15T21:13:35.221758Z","shell.execute_reply.started":"2023-08-15T21:12:53.736342Z","shell.execute_reply":"2023-08-15T21:13:35.218939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Create an ID dictionary for reduced embeddings","metadata":{}},{"cell_type":"code","source":"# Create an ID dictionary for reduced embeddings\nid_to_embedding_reduced = {id: embedding for id, embedding in zip(train_ids, train_embeds_reduced)}","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:35.224242Z","iopub.execute_input":"2023-08-15T21:13:35.224714Z","iopub.status.idle":"2023-08-15T21:13:35.392435Z","shell.execute_reply.started":"2023-08-15T21:13:35.224671Z","shell.execute_reply":"2023-08-15T21:13:35.391168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the TSV file\ndf_prot = pd.read_csv('/kaggle/input/cafa-5-protein-predict-competion/Train/train_terms.tsv', sep='\\t')","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:35.393929Z","iopub.execute_input":"2023-08-15T21:13:35.394407Z","iopub.status.idle":"2023-08-15T21:13:38.377781Z","shell.execute_reply.started":"2023-08-15T21:13:35.394372Z","shell.execute_reply":"2023-08-15T21:13:38.376623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting the IDs into a DataFrame\ndf_sequences = pd.DataFrame({\n    'EntryID': train_ids,\n})\n\n# Mesclar df_sequences e df_prot em 'EntryID'\ndf_merged = pd.merge(df_sequences, df_prot, on='EntryID', how='inner')","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:38.379546Z","iopub.execute_input":"2023-08-15T21:13:38.380224Z","iopub.status.idle":"2023-08-15T21:13:41.095371Z","shell.execute_reply.started":"2023-08-15T21:13:38.380189Z","shell.execute_reply":"2023-08-15T21:13:41.094017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Aggregating the 'Term' for each unique 'EntryID'\ndf_merged_grouped = df_merged.groupby(['EntryID'])['term'].apply(list).reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:41.09714Z","iopub.execute_input":"2023-08-15T21:13:41.09749Z","iopub.status.idle":"2023-08-15T21:13:48.10978Z","shell.execute_reply.started":"2023-08-15T21:13:41.09746Z","shell.execute_reply":"2023-08-15T21:13:48.108558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete df_merged to free up memory\ndel df_merged\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:48.1118Z","iopub.execute_input":"2023-08-15T21:13:48.112462Z","iopub.status.idle":"2023-08-15T21:13:49.550146Z","shell.execute_reply.started":"2023-08-15T21:13:48.112426Z","shell.execute_reply":"2023-08-15T21:13:49.548882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparation and splitting of training and testing data","metadata":{}},{"cell_type":"code","source":"# Splitting the data into training and testing\ndf_train, df_test = train_test_split(df_merged_grouped, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:49.552234Z","iopub.execute_input":"2023-08-15T21:13:49.553382Z","iopub.status.idle":"2023-08-15T21:13:49.600825Z","shell.execute_reply.started":"2023-08-15T21:13:49.55334Z","shell.execute_reply":"2023-08-15T21:13:49.599251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete df_merged_grouped to free up memory\ndel df_merged_grouped\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:49.602222Z","iopub.execute_input":"2023-08-15T21:13:49.602722Z","iopub.status.idle":"2023-08-15T21:13:50.435994Z","shell.execute_reply.started":"2023-08-15T21:13:49.602679Z","shell.execute_reply":"2023-08-15T21:13:50.435007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert `train_ids` to a numpy array for efficient searching\ntrain_ids = np.array(train_ids)\n\n# Extract EntryID values from the test dataframe\ntest_ids = df_test['EntryID'].values\n\n# Use `np.searchsorted` to find the indices of the 'EntryID's in the df_train dataframe within the sorted train_ids array\n# This approach is efficient for sorted arrays and returns the desired indices directly.\ntrain_indices = np.where(np.in1d(train_ids, df_train['EntryID']))[0] \n\n# Similarly, find the indices of the test IDs within the sorted train_ids array\ntest_indices = np.where(np.in1d(train_ids, test_ids))[0]\n\n# Extract the corresponding embeddings for the training set based on the computed indices\nX_train = train_embeds_reduced[train_indices]\n\n# Extract the corresponding embeddings for the test set based on the computed indices\nX_test = train_embeds_reduced[test_indices]","metadata":{"execution":{"iopub.status.busy":"2023-08-15T21:13:50.437227Z","iopub.execute_input":"2023-08-15T21:13:50.438096Z","iopub.status.idle":"2023-08-15T22:04:38.167097Z","shell.execute_reply.started":"2023-08-15T21:13:50.438047Z","shell.execute_reply":"2023-08-15T22:04:38.165437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if the training data has any NaN values\nhas_nan_train = np.isnan(X_train).any()\n\n# Checking if the testing data has any NaN values\nhas_nan_test = np.isnan(X_test).any()\n\n# Checking if the training data has any infinite values\nhas_inf_train = np.isinf(X_train).any()\n\n# Checking if the testing data has any infinite values\nhas_inf_test = np.isinf(X_test).any()\n\n# Printing the results:\n# Check and display if the training data contains NaN values\nprint(\"X_train has NaN values:\", has_nan_train)\n\n# Check and display if the testing data contains NaN values\nprint(\"X_test has NaN values:\", has_nan_test)\n\n# Check and display if the training data contains infinite values\nprint(\"X_train has infinite values:\", has_inf_train)\n\n# Check and display if the testing data contains infinite values\nprint(\"X_test has infinite values:\", has_inf_test)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:04:38.169051Z","iopub.execute_input":"2023-08-15T22:04:38.169518Z","iopub.status.idle":"2023-08-15T22:04:38.236817Z","shell.execute_reply.started":"2023-08-15T22:04:38.169473Z","shell.execute_reply":"2023-08-15T22:04:38.23526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete train_embeds_reduced to free up memory\ndel train_embeds_reduced\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:04:38.239766Z","iopub.execute_input":"2023-08-15T22:04:38.240167Z","iopub.status.idle":"2023-08-15T22:04:39.511256Z","shell.execute_reply.started":"2023-08-15T22:04:38.240133Z","shell.execute_reply":"2023-08-15T22:04:39.510007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the csr_matrix from scipy.sparse for efficient storage of sparse matrices\nfrom scipy.sparse import csr_matrix\n\n# Initialize the MultiLabelBinarizer\nmlb = MultiLabelBinarizer()\n\n# Fit the binarizer only on the terms of the training set.\n# This ensures the binarizer learns only the classes/labels present in the training data\nmlb.fit(df_train['term'])\n\n# After fitting, we can directly transform the terms for both training and testing sets.\n# This avoids the additional cost of combining and then splitting the datasets again.\n\n# Transform the training terms into binary format\ny_train = mlb.transform(df_train['term'])\n\n# Transform the testing terms into binary format\ny_test = mlb.transform(df_test['term'])\n\n# If the output is mostly zeros (sparse), then convert it to a sparse matrix for efficient storage and computation.\n# For instance, if more than 90% of the values are zero\nif (y_train.size - np.count_nonzero(y_train)) / y_train.size > 0.9:\n    y_train = csr_matrix(y_train)  # Convert training matrix to compressed sparse row format\n    y_test = csr_matrix(y_test)    # Convert testing matrix to compressed sparse row format","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:04:39.512845Z","iopub.execute_input":"2023-08-15T22:04:39.513189Z","iopub.status.idle":"2023-08-15T22:05:24.906909Z","shell.execute_reply.started":"2023-08-15T22:04:39.51316Z","shell.execute_reply":"2023-08-15T22:05:24.905364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the EntryID's for later use\ntest_EntryIDs = df_test['EntryID'].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:24.908977Z","iopub.execute_input":"2023-08-15T22:05:24.909717Z","iopub.status.idle":"2023-08-15T22:05:24.917503Z","shell.execute_reply.started":"2023-08-15T22:05:24.909662Z","shell.execute_reply":"2023-08-15T22:05:24.916287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Delete df_train and df_test to free up memory\ndel df_train, df_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:24.919069Z","iopub.execute_input":"2023-08-15T22:05:24.919716Z","iopub.status.idle":"2023-08-15T22:05:25.845802Z","shell.execute_reply.started":"2023-08-15T22:05:24.919672Z","shell.execute_reply":"2023-08-15T22:05:25.844417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of classes will be the dimension of the encoded labels\nn_classes = y_train.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:25.848731Z","iopub.execute_input":"2023-08-15T22:05:25.849135Z","iopub.status.idle":"2023-08-15T22:05:25.858082Z","shell.execute_reply.started":"2023-08-15T22:05:25.849092Z","shell.execute_reply":"2023-08-15T22:05:25.85682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creation of Data Generators","metadata":{}},{"cell_type":"code","source":"def simplified_data_generator(X, y, batch_size):\n    \"\"\"\n    Generator function to yield batches of data for training or testing.\n    \n    Parameters:\n    - X : Input data\n    - y : Corresponding labels or targets\n    - batch_size : Number of samples per batch\n    \n    Yields:\n    - batch_X : A batch of input data\n    - batch_y : Corresponding labels or targets for the batch\n    \"\"\"\n    \n    num_samples = len(X)\n    \n    while True:\n        # Shuffle indices for stochastic gradient descent (SGD)\n        indices = np.arange(num_samples)\n        np.random.shuffle(indices)\n        \n        for i in range(0, num_samples, batch_size):\n            end_idx = min(i + batch_size, num_samples)\n            \n            # Fetch a batch of data and labels\n            batch_X = X[indices[i:end_idx]]\n            batch_y = y[indices[i:end_idx]]\n            \n            # Convert sparse arrays/tensors to dense if necessary\n            if isinstance(batch_X, tf.SparseTensor) or hasattr(batch_X, \"toarray\"):\n                batch_X = batch_X.toarray() if hasattr(batch_X, \"toarray\") else tf.sparse.to_dense(batch_X)\n                \n            if isinstance(batch_y, tf.SparseTensor) or hasattr(batch_y, \"toarray\"):\n                batch_y = batch_y.toarray() if hasattr(batch_y, \"toarray\") else tf.sparse.to_dense(batch_y)\n            \n            yield batch_X, batch_y\n\n# Example usage\nbatch_size = 256\ntrain_generator = simplified_data_generator(X_train, y_train, batch_size)\ntest_generator = simplified_data_generator(X_test, y_test, batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:25.859674Z","iopub.execute_input":"2023-08-15T22:05:25.860152Z","iopub.status.idle":"2023-08-15T22:05:25.875935Z","shell.execute_reply.started":"2023-08-15T22:05:25.860107Z","shell.execute_reply":"2023-08-15T22:05:25.874459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create an instance of the data generator\nexample_generator = simplified_data_generator(X_train, y_train, batch_size)\n\n# Retrieve one batch of data from the generator\nbatch_X, batch_y = next(example_generator)\n\n# Print or inspect the batch to get a sense of its structure and contents\n\n# Print the shape (dimensions) of the batch data for features\nprint(\"Shape of batch_X:\", batch_X.shape)\n\n# Print the shape (dimensions) of the batch data for labels\nprint(\"Shape of batch_y:\", batch_y.shape)\n\n# Display some sample feature data from the batch\nprint(\"\\nSamples from batch_X:\", batch_X)\n\n# Display some sample label data from the batch\nprint(\"\\nSamples from batch_y:\", batch_y)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:25.877513Z","iopub.execute_input":"2023-08-15T22:05:25.877955Z","iopub.status.idle":"2023-08-15T22:05:25.914149Z","shell.execute_reply.started":"2023-08-15T22:05:25.877921Z","shell.execute_reply":"2023-08-15T22:05:25.912754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building and training","metadata":{}},{"cell_type":"code","source":"# Define a custom metric class for multi-label F1 score, which inherits from tf.keras.metrics.Metric\nclass MultiLabelF1Score(tf.keras.metrics.Metric):\n    # Initialize the custom metric class\n    def __init__(self, name='multi_label_f1_score', **kwargs):\n        # Initialize the parent class\n        super(MultiLabelF1Score, self).__init__(name=name, **kwargs)\n        \n        # Instantiate the Precision and Recall metrics which will be used to compute the F1 score\n        self.precision = Precision()\n        self.recall = Recall()\n\n    # This method is called for every batch of data, updating the state of the metric\n    def update_state(self, y_true, y_pred, sample_weight=None):\n        # Update the state of the precision and recall metrics with the current batch of data\n        self.precision.update_state(y_true, y_pred, sample_weight)\n        self.recall.update_state(y_true, y_pred, sample_weight)\n\n    # This method computes and returns the final metric value (in this case, the F1 score)\n    def result(self):\n        # Retrieve the computed precision and recall values\n        precision = self.precision.result()\n        recall = self.recall.result()\n        \n        # Calculate the F1 score using the formula: 2 * (precision * recall) / (precision + recall)\n        # A small value (1e-5) is added to the denominator to avoid division by zero\n        return 2 * ((precision * recall) / (precision + recall + 1e-5))\n  \n    # This method resets the state of the metric at the start of each epoch during training\n    def reset_states(self):\n        # Reset the state of the precision and recall metrics\n        self.precision.reset_states()\n        self.recall.reset_states()","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:25.915945Z","iopub.execute_input":"2023-08-15T22:05:25.91634Z","iopub.status.idle":"2023-08-15T22:05:25.927381Z","shell.execute_reply.started":"2023-08-15T22:05:25.916306Z","shell.execute_reply":"2023-08-15T22:05:25.926361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Building\n\n# Retrieve the number of features from the training dataset\ninput_dim = X_train.shape[1]\n\n# Retrieve the number of classes or labels from the training dataset\nn_classes = y_train.shape[1]\n\n# Define the model architecture using Keras Sequential API\nmodel = tf.keras.Sequential([\n    # First dense layer with 64 neurons and ReLU activation function. \n    # 'input_shape' is specified for the first layer to determine the number of input features.\n    Dense(64, activation='relu', input_shape=(input_dim,)),\n    \n    # Dropout layer to prevent overfitting by randomly setting a fraction (0.5) of input units to 0 during training.\n    Dropout(0.3),\n    \n    # Output layer with 'n_classes' neurons (one for each label) using sigmoid activation \n    # as it's a multi-label classification problem.\n    Dense(n_classes, activation='sigmoid')\n])\n\n# Model Compilation\n\n# Define the optimizer for the model. RMSprop is chosen here.\noptimizer = tf.keras.optimizers.RMSprop()\n\n# Compile the model with the chosen optimizer, loss function, and evaluation metrics\nmodel.compile(\n    optimizer=optimizer,\n    \n    # 'binary_crossentropy' is used as the loss function since it's a multi-label classification.\n    loss='binary_crossentropy',\n    \n    # Metrics to evaluate the model's performance during training and validation.\n    # It includes precision, recall, custom multi-label F1 score, and accuracy.\n    metrics=[Precision(), Recall(), MultiLabelF1Score(), 'accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:25.928967Z","iopub.execute_input":"2023-08-15T22:05:25.92931Z","iopub.status.idle":"2023-08-15T22:05:26.050127Z","shell.execute_reply.started":"2023-08-15T22:05:25.929282Z","shell.execute_reply":"2023-08-15T22:05:26.048695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the type of y_train data\nif isinstance(y_train, (np.ndarray, list)):\n    # If y_train is a numpy array or a list\n    print(\"y_train is a numpy array or a list.\")\n    print(\"First element of y_train:\", y_train[0])\nelse:\n    try:\n        # Try to import the sparse matrix type from scipy.sparse to check if y_train is a sparse matrix\n        from scipy.sparse import spmatrix\n        if isinstance(y_train, spmatrix):\n            print(\"y_train is a sparse matrix.\")\n            print(\"Shape of y_train:\", y_train.shape)\n            print(\"Data type of y_train:\", type(y_train))\n    except ImportError:\n        # If scipy is not installed, print a message indicating so\n        print(\"Scipy is not installed, so we can't check for sparse matrices.\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:26.051525Z","iopub.execute_input":"2023-08-15T22:05:26.051947Z","iopub.status.idle":"2023-08-15T22:05:26.061634Z","shell.execute_reply.started":"2023-08-15T22:05:26.051914Z","shell.execute_reply":"2023-08-15T22:05:26.060002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the type of X_train data\nif isinstance(X_train, (np.ndarray, list)):\n    # If X_train is a numpy array or a list\n    print(\"X_train is a numpy array or a list.\")\n    print(\"First element of X_train:\", X_train[0])\nelse:\n    try:\n        # Try to import the sparse matrix type from scipy.sparse to check if X_train is a sparse matrix\n        from scipy.sparse import spmatrix\n        if isinstance(X_train, spmatrix):\n            print(\"X_train is a sparse matrix.\")\n            print(\"Shape of X_train:\", X_train.shape)\n            print(\"Data type of X_train:\", type(X_train))\n    except ImportError:\n        # If scipy is not installed, print a message indicating so\n        print(\"Scipy is not installed, so we can't check for sparse matrices.\")","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:26.06322Z","iopub.execute_input":"2023-08-15T22:05:26.063563Z","iopub.status.idle":"2023-08-15T22:05:26.085321Z","shell.execute_reply.started":"2023-08-15T22:05:26.063536Z","shell.execute_reply":"2023-08-15T22:05:26.084309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\nimport numpy as np\nfrom scipy.sparse import csr_matrix\n\ndef compute_average_class_weights(y_labels, max_classes, chunk_size):\n    \"\"\"\n    Compute average class weights from the given labels.\n    \n    Parameters:\n    - y_labels : list\n        A list of labels.\n    - max_classes : int\n        Maximum number of classes.\n    - chunk_size : int\n        Number of labels to process in each chunk.\n    \n    Returns:\n    - class_weights : dict\n        Dictionary of average class weights with class indices as keys.\n    \"\"\"\n    # Initialize arrays to accumulate weights and count instances of each class\n    accum_weights = np.zeros(max_classes)\n    class_counts = np.zeros(max_classes)\n    \n    # Check if y_labels is a list or a csr_matrix\n    if isinstance(y_labels, csr_matrix):\n        y_labels = y_labels.toarray().tolist()\n\n    # Loop through the 'y_labels' in chunks\n    for i in range(0, len(y_labels), chunk_size):\n        y_chunk = y_labels[i:i+chunk_size]\n        unique_classes = np.unique(y_chunk)\n        \n        weights = compute_class_weight('balanced', classes=unique_classes, y=y_chunk)\n        \n        accum_weights[unique_classes] += weights\n        class_counts[unique_classes] += 1\n\n    # Calculate the average weights\n    avg_weights = accum_weights / np.maximum(class_counts, 1)\n    \n    return dict(enumerate(avg_weights))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:26.086639Z","iopub.execute_input":"2023-08-15T22:05:26.088134Z","iopub.status.idle":"2023-08-15T22:05:26.099114Z","shell.execute_reply.started":"2023-08-15T22:05:26.088015Z","shell.execute_reply":"2023-08-15T22:05:26.097699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert y_train to a list format suitable for the function\ny_labels_list = [row.indices for row in y_train]\ny_labels = [label for sublist in y_labels_list for label in sublist]\n\nchunk_size = 10000\nclass_weights = compute_average_class_weights(y_labels, max_classes=30325, chunk_size=chunk_size)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:26.101122Z","iopub.execute_input":"2023-08-15T22:05:26.101468Z","iopub.status.idle":"2023-08-15T22:05:37.053912Z","shell.execute_reply.started":"2023-08-15T22:05:26.10144Z","shell.execute_reply":"2023-08-15T22:05:37.052548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_classes = y_train.shape[1]\nprint(max_classes)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:37.055529Z","iopub.execute_input":"2023-08-15T22:05:37.056394Z","iopub.status.idle":"2023-08-15T22:05:37.062126Z","shell.execute_reply.started":"2023-08-15T22:05:37.056357Z","shell.execute_reply":"2023-08-15T22:05:37.060924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize arrays to accumulate weights and count instances of each class\nmax_classes = 30325\naccum_weights = np.zeros(max_classes)\nclass_counts = np.zeros(max_classes)\n\n# Loop through the 'y_labels' in chunks to process the data in smaller portions\nfor i in range(0, len(y_labels), chunk_size):\n    # Extract a chunk of labels\n    y_chunk = y_labels[i:i+chunk_size]\n    \n    # Identify the unique classes present in the current chunk\n    unique_classes = np.unique(y_chunk)\n    \n    # Compute the class weights for the unique classes in the chunk using scikit-learn's function\n    weights = compute_class_weight('balanced', classes=unique_classes, y=y_chunk)\n    \n    try:\n        # Accumulate the computed weights for the current chunk to the overall weights\n        accum_weights[unique_classes] += weights\n        # Update the count of occurrences for each class\n        class_counts[unique_classes] += 1\n    except IndexError:\n        # Handle possible IndexError\n        print(f\"Error in iteration starting at {i}\")\n        print(\"Unique classes:\", unique_classes)\n        print(\"Weights:\", weights)\n        print(\"Shape of weights:\", weights.shape)\n        # Break out of the loop on encountering an error\n        break","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:37.063557Z","iopub.execute_input":"2023-08-15T22:05:37.063935Z","iopub.status.idle":"2023-08-15T22:05:39.496778Z","shell.execute_reply.started":"2023-08-15T22:05:37.063904Z","shell.execute_reply":"2023-08-15T22:05:39.495556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training with Callbacks","metadata":{}},{"cell_type":"code","source":"x, y = next(iter(train_generator))\nprint(x.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:39.49841Z","iopub.execute_input":"2023-08-15T22:05:39.498809Z","iopub.status.idle":"2023-08-15T22:05:39.521278Z","shell.execute_reply.started":"2023-08-15T22:05:39.498776Z","shell.execute_reply":"2023-08-15T22:05:39.520054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Constants\nEPOCHS = 10\nBATCH_SIZE = 32  # Vou supor que o batch_size seja 32 para o exemplo\n\n# Calculate the number of steps\nsteps_per_epoch = len(X_train) // BATCH_SIZE\nvalidation_steps = len(X_test) // BATCH_SIZE\n\n# Callbacks\nearly_stop = EarlyStopping(monitor='val_loss', patience=15, verbose=1, restore_best_weights=True)\ncheckpoint = ModelCheckpoint('best_model.h5', monitor='val_loss', verbose=1, save_best_only=True, mode='min')\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', factor=0.2, patience=7, verbose=1, min_delta=0.0001)\n\n# TensorBoard setup for visualization\nlog_dir = \"./logs/fit/\" + time.strftime(\"%Y%m%d-%H%M%S\")\ntensorboard = TensorBoard(log_dir=log_dir)\n\n# Aggregating all the callbacks\ncallbacks = [early_stop, checkpoint, reduce_lr, tensorboard]\n\n# Optimizer with gradient clipping (RMSprop)\noptimizer = RMSprop(clipnorm=1.0)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:39.523386Z","iopub.execute_input":"2023-08-15T22:05:39.523859Z","iopub.status.idle":"2023-08-15T22:05:39.535543Z","shell.execute_reply.started":"2023-08-15T22:05:39.523815Z","shell.execute_reply":"2023-08-15T22:05:39.534593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\n\n# First dense layer with 256 neurons and input shape specified\nmodel.add(Dense(256, activation='relu', input_shape=(input_dim,)))\n\n# 30% dropout to prevent overfitting\nmodel.add(Dropout(0.3))\n\n# Second dense layer with 256 neurons\nmodel.add(Dense(256, activation='relu'))\n\n# Another 30% dropout\nmodel.add(Dropout(0.3))\n\n# Output layer with sigmoid activation for multi-label classification\nmodel.add(Dense(30325, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:39.536877Z","iopub.execute_input":"2023-08-15T22:05:39.537384Z","iopub.status.idle":"2023-08-15T22:05:39.711616Z","shell.execute_reply.started":"2023-08-15T22:05:39.537354Z","shell.execute_reply":"2023-08-15T22:05:39.710619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling the model with specified optimizer, loss function, and evaluation metrics\nmodel.compile(optimizer=optimizer,\n              loss='binary_crossentropy',\n              metrics=[Precision(), Recall(), 'accuracy'])\n\n# Training the model\n# - Using training generator to feed data\n# - Specifying the number of epochs to run\n# - Setting the steps per epoch\n# - Using test generator for validation\n# - Setting validation steps\n# - Using class weights to handle imbalance\n# - Specifying callbacks for specific actions during training (e.g., early stopping)\nhistory = model.fit(\n    train_generator,\n    epochs=EPOCHS,\n    steps_per_epoch=steps_per_epoch,\n    validation_data=test_generator,\n    validation_steps=validation_steps,\n    class_weight=class_weights,\n    callbacks=callbacks\n)\n\n# Saving the training history to a file for later analysis\nwith open('training_history.pkl', 'wb') as history_file:\n    pickle.dump(history.history, history_file)","metadata":{"execution":{"iopub.status.busy":"2023-08-15T22:05:39.713865Z","iopub.execute_input":"2023-08-15T22:05:39.714215Z","iopub.status.idle":"2023-08-16T02:05:47.163954Z","shell.execute_reply.started":"2023-08-15T22:05:39.714185Z","shell.execute_reply":"2023-08-16T02:05:47.162397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the number of steps needed\ntest_steps = len(X_test) // batch_size\n\n# Evaluate the model on the test data using the generator\nscore = model.evaluate(test_generator, steps=test_steps, verbose=0)\n\n# Print test loss and any other metrics\nprint(\"Test loss:\", score[0])\nprint(\"Test metrics:\", score[1:])","metadata":{"execution":{"iopub.status.busy":"2023-08-16T02:05:47.166676Z","iopub.execute_input":"2023-08-16T02:05:47.167141Z","iopub.status.idle":"2023-08-16T02:06:07.779104Z","shell.execute_reply.started":"2023-08-16T02:05:47.167099Z","shell.execute_reply":"2023-08-16T02:06:07.777679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Performance Visualization","metadata":{}},{"cell_type":"code","source":"# Load history\nwith open('training_history.pkl', 'rb') as file_pi:\n    history = pickle.load(file_pi)\n\n# Plot loss\nplt.figure(figsize=(12, 4))\nplt.subplot(1, 2, 1)\nplt.plot(history['loss'], label='Train Loss')\nplt.plot(history['val_loss'], label='Validation Loss')\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\n\n# Plot accuracy\nplt.subplot(1, 2, 2)\nplt.plot(history['accuracy'], label='Train Accuracy')\nplt.plot(history['val_accuracy'], label='Validation Accuracy')\nplt.title('Model Accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\nplt.tight_layout()\nplt.show()\n\n# Plot precision and recall\nplt.figure(figsize=(12, 4))\n\n# Plot precision\nplt.subplot(1, 2, 1)\nplt.plot(history['precision_2'], label='Train Precision')\nplt.plot(history['val_precision_2'], label='Validation Precision')\nplt.title('Model Precision')\nplt.ylabel('Precision')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\n\n# Plot recall\nplt.subplot(1, 2, 2)\nplt.plot(history['recall_2'], label='Train Recall')\nplt.plot(history['val_recall_2'], label='Validation Recall')\nplt.title('Model Recall')\nplt.ylabel('Recall')\nplt.xlabel('Epoch')\nplt.legend(['Train', 'Validation'], loc='upper left')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-16T02:32:56.071059Z","iopub.execute_input":"2023-08-16T02:32:56.071685Z","iopub.status.idle":"2023-08-16T02:32:57.561819Z","shell.execute_reply.started":"2023-08-16T02:32:56.071646Z","shell.execute_reply":"2023-08-16T02:32:57.56063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Memory Cleanup","metadata":{}},{"cell_type":"code","source":"# Cleaning embeddings and IDs\ndel train_embeds, train_ids, test_embeds, test_ids\n\n# Cleaning after training\ndel train_generator, test_generator, history\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-16T02:33:15.457775Z","iopub.execute_input":"2023-08-16T02:33:15.458251Z","iopub.status.idle":"2023-08-16T02:33:17.966078Z","shell.execute_reply.started":"2023-08-16T02:33:15.458216Z","shell.execute_reply":"2023-08-16T02:33:17.964797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Forecasting using a data generator","metadata":{}},{"cell_type":"code","source":"def prediction_data_generator(X, batch_size):\n    \"\"\"\n    Generator function to yield batches of data for prediction.\n    \n    Parameters:\n    - X : numpy array\n        Input data\n    - batch_size : int\n        Number of samples per batch\n    \n    Yields:\n    - batch_X : numpy array\n        A batch of input data\n    \"\"\"\n    indices = np.arange(len(X))\n    np.random.shuffle(indices)\n    \n    for i in range(0, len(X), batch_size):\n        yield X[indices[i:i+batch_size]]\n\nbatch_size = 128\npredictions = []\n\n# Calculate total batches and use it to limit tqdm's progress bar\ntotal_batches = int(np.ceil(len(X_test) / batch_size))\n\n# Predict in batches and append to the predictions list\nfor batch_X in tqdm(prediction_data_generator(X_test, batch_size), total=total_batches, desc=\"Predicting\"):\n    batch_predictions = model.predict(batch_X)\n    \n    # If the output is a multi-dimensional array, concatenate it\n    if isinstance(batch_predictions, np.ndarray):\n        predictions.append(batch_predictions)\n    else:\n        predictions.extend(batch_predictions)\n\n# If predictions is a list of arrays, convert it to a single array\nif isinstance(predictions[0], np.ndarray):\n    predictions = np.concatenate(predictions, axis=0)\n\n# Clear memory\ngc.collect()\n\n# Save predictions\nnp.save('predictions.npy', predictions)","metadata":{"execution":{"iopub.status.busy":"2023-08-16T02:33:20.884358Z","iopub.execute_input":"2023-08-16T02:33:20.884799Z","iopub.status.idle":"2023-08-16T02:34:09.568525Z","shell.execute_reply.started":"2023-08-16T02:33:20.884763Z","shell.execute_reply":"2023-08-16T02:34:09.567037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission File Generation","metadata":{}},{"cell_type":"code","source":"# Parameters\nthreshold = 0.001 \nchunksize = 500   # or any other desired value\n\n# Retrieve GO terms\ngo_terms = mlb.classes_\n\ndef save_predictions_to_file(chunk_predictions, protein_ids):\n    \"\"\"\n    Save the predictions for proteins to a TSV file.\n    \n    Parameters:\n    - chunk_predictions: Predictions corresponding to the current chunk.\n    - protein_ids: List of protein IDs in the current chunk.\n    \"\"\"\n    \n    with open('submission.tsv', 'a') as f:\n        # Iterate over current chunk of IDs\n        for protein_index, protein_id in enumerate(protein_ids):\n            protein_prediction = chunk_predictions[protein_index]\n\n            # Filter and sort terms based on prediction probability\n            predicted_terms = [(go_terms[k], prob) for k, prob in enumerate(protein_prediction) if prob > threshold]\n            sorted_terms = sorted(predicted_terms, key=lambda x: x[1], reverse=True)[:1500]\n\n            # Write the terms and their probabilities to the file\n            for term, prob in sorted_terms:\n                f.write(f\"{protein_id}\\t{term}\\t{prob}\\n\")\n\n# Before saving, make sure 'submission.tsv' doesn't already exist (or is empty) \n# to avoid appending results multiple times when you run the script.\nif os.path.exists('submission.tsv'):\n    os.remove('submission.tsv')\n\nfor i, test_chunk in enumerate(np.array_split(test_EntryIDs, len(test_EntryIDs) // chunksize)):\n    \n    start_index = i*chunksize\n    end_index = start_index + len(test_chunk)  # Use actual chunk size instead of the predefined chunksize\n    \n    # Load predictions for the current chunk\n    chunk_predictions = predictions[start_index:end_index]\n    \n    save_predictions_to_file(chunk_predictions, test_chunk)\n    \n    # Clear chunk data after processing\n    del test_chunk\n    del chunk_predictions\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-16T02:34:19.409039Z","iopub.execute_input":"2023-08-16T02:34:19.409453Z","iopub.status.idle":"2023-08-16T03:25:31.977213Z","shell.execute_reply.started":"2023-08-16T02:34:19.409421Z","shell.execute_reply":"2023-08-16T03:25:31.975621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preview the first few lines of the unified submission file\ndef preview_submission_file():\n    \"\"\"\n    Read and print the first few lines of the unified submission file \n    to verify its contents.\n    \"\"\"\n    submission_preview = pd.read_csv('submission.tsv', \n                                     sep='\\t', \n                                     header=None,\n                                     names=[\"Protein Id\", \"GO Term Id\", \"Prediction\"],  \n                                     nrows=5)\n    print(submission_preview)\n\n# Execute the function to preview the submission file\npreview_submission_file()","metadata":{"execution":{"iopub.status.busy":"2023-08-16T03:25:31.979787Z","iopub.execute_input":"2023-08-16T03:25:31.980185Z","iopub.status.idle":"2023-08-16T03:25:32.003519Z","shell.execute_reply.started":"2023-08-16T03:25:31.980144Z","shell.execute_reply":"2023-08-16T03:25:32.001045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}