{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Set the hyperparameters\nlayers = 3\nnodes = 512\nact_func = 'relu'\nbatch_norm = False\noptimizer = 'Adam'  # Non-functional\neta = 0.001\nl2 = None\ndropout = None\nbatch_size = 5120\nepochs = 20\nnum_of_predictions = None  # Set globally\nnum_of_go_terms = 1000\nnum_of_batches = 2\nis_debug = False","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:37:51.783808Z","iopub.execute_input":"2023-08-07T19:37:51.784193Z","iopub.status.idle":"2023-08-07T19:37:51.796868Z","shell.execute_reply.started":"2023-08-07T19:37:51.78416Z","shell.execute_reply":"2023-08-07T19:37:51.795818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load memory profiler\n%load_ext memory_profiler\n\n# Set up tensorflow\nimport os\nrandom_seed = 42\nos.environ['PYTHONHASHSEED'] = str(random_seed)\nos.environ['TF_FORCE_GPU_ALLOW_GROWTH'] = 'true'\nos.environ['TF_GPU_ALLOCATOR'] = 'cuda_malloc_async'\n\n# Import libraries\nimport sys\nimport time\nimport random\nimport gc\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport progressbar\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Flatten, Dense, Activation, BatchNormalization, Dropout\nfrom keras import backend as K\n\nimport cupy as cp\nimport cudf\n\n# Set random seeds\nrandom.seed(random_seed)\nnp.random.seed(random_seed)\ntf.random.set_seed(random_seed)\ncp.random.seed(0)","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:37:51.801478Z","iopub.execute_input":"2023-08-07T19:37:51.801754Z","iopub.status.idle":"2023-08-07T19:38:03.611401Z","shell.execute_reply.started":"2023-08-07T19:37:51.801731Z","shell.execute_reply":"2023-08-07T19:38:03.610409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to prepare the training features\ndef prepare_X_train():\n\n    # Display a status update\n    print(\"Preparing the training features\")\n\n    # Load the embeddings\n    train_embeddings = cp.load('/kaggle/input/t5embeds/train_embeds.npy')\n\n    # Create the training features from the embeddings\n    X_train = cudf.DataFrame.from_records(cp.asnumpy(train_embeddings))\n\n    # Convert column names to strings\n    X_train.columns = X_train.columns.astype(str)\n\n    # Convert to float32\n    X_train = X_train.astype('float32')\n\n    # Save the training features\n    X_train.to_parquet(f'/kaggle/working/X_train.parquet')\n    \n    # Clean up memory\n    del train_embeddings\n    del X_train\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.613401Z","iopub.execute_input":"2023-08-07T19:38:03.614224Z","iopub.status.idle":"2023-08-07T19:38:03.620751Z","shell.execute_reply.started":"2023-08-07T19:38:03.614187Z","shell.execute_reply":"2023-08-07T19:38:03.619837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to prepare the test features\ndef prepare_X_test():\n\n    # Display a status update\n    print(\"Preparing the test features\")\n\n    # Get the test embeddings\n    test_embeddings = cp.load('/kaggle/input/t5embeds/test_embeds.npy')\n\n    # Convert test_embeddings to dataframe\n    X_test = cudf.DataFrame(test_embeddings)\n\n    # Convert column names to strings\n    X_test.columns = X_test.columns.astype(str)\n\n    # Convert to float32\n    X_test = X_test.astype('float32')\n\n    # Save the test features\n    X_test.to_parquet(f'/kaggle/working/X_test.parquet')\n    \n    # Clean up memory\n    del test_embeddings\n    del X_test\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.622327Z","iopub.execute_input":"2023-08-07T19:38:03.623Z","iopub.status.idle":"2023-08-07T19:38:03.633591Z","shell.execute_reply.started":"2023-08-07T19:38:03.62296Z","shell.execute_reply":"2023-08-07T19:38:03.632548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to prepare the training labels\ndef prepare_y_train(batch):\n    \n    # Display a status update\n    print(f\"Preparing the training labels for batch {batch + 1}\")\n    \n    # Get the start GO term\n    start_term = batch * num_of_go_terms\n\n    # Get the end GO term\n    end_term = start_term + num_of_go_terms\n\n    # Load the protein IDs\n    train_protein_ids = np.load('/kaggle/input/t5embeds/train_ids.npy')\n\n    # Read the GO terms\n    train_terms = cudf.read_csv(\"/kaggle/input/cafa-5-protein-function-prediction/Train/train_terms.tsv\", sep=\"\\t\")\n\n    # Get the first M labels\n    go_terms = train_terms['term'].value_counts().index[start_term:end_term].to_arrow().to_pylist()\n\n    # Convert GO terms to array for performance\n    go_terms = np.array(go_terms)\n\n    # Get train_terms data for the top M labels only\n    train_terms_updated = train_terms.loc[train_terms['term'].isin(go_terms)]\n\n    # Get the number of rows (N)\n    train_size_N = train_protein_ids.shape[0]\n    train_size_M = len(go_terms)\n\n    # Create an empty matrix (N x M) for the labels\n    y_train = np.zeros((train_size_N, train_size_M))\n\n    # Convert from numpy to pandas series for better handling\n    train_protein_ids = pd.Series(train_protein_ids)\n\n    # Create the progress bar\n    bar = progressbar.ProgressBar(\n        maxval=num_of_go_terms,\n        widgets=[progressbar.Bar('=', '[', ']'), ' ', progressbar.Percentage()])\n\n\n    # Group train_terms_updated by 'term' and get the corresponding unique 'EntryID's for each label\n    go_terms_to_proteins_map = train_terms_updated.groupby('term')['EntryID'].unique()\n\n    # Create a matrix of proteins and Go terms\n    for i, label in enumerate(go_terms):\n\n        # Get the proteins related to the current GO term\n        go_term_related_proteins = go_terms_to_proteins_map.loc[label] if label in go_terms_to_proteins_map else []\n\n        # Fill the corresponding column in the matrix\n        y_train[:, i] = train_protein_ids.isin(go_term_related_proteins)\n\n        # Increment the counter\n        i += 1\n\n        # Update the progress bar\n        bar.update(i)\n\n    # End the progress bar \n    bar.finish()\n\n    # Convert labels into aa pandas dataframe\n    y_train = pd.DataFrame(data=y_train, columns=go_terms)\n\n    # Convert to float32\n    y_train = y_train.astype('float32')\n\n    # Save labels to disk\n    np.save('/kaggle/working/go_terms.npy', go_terms)\n\n    # Save the training data to disk\n    y_train.to_parquet(f'/kaggle/working/y_train.parquet')\n    \n    # Clean up memory\n    del train_terms\n    del train_protein_ids\n    del train_terms_updated\n    del go_terms\n    del go_terms_to_proteins_map\n    del go_term_related_proteins\n    del y_train\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.63743Z","iopub.execute_input":"2023-08-07T19:38:03.637696Z","iopub.status.idle":"2023-08-07T19:38:03.652718Z","shell.execute_reply.started":"2023-08-07T19:38:03.637673Z","shell.execute_reply":"2023-08-07T19:38:03.651695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to train the model\ndef train_model():\n\n    # Display a status update\n    print(\"Training the model\")\n\n    # Load features and labels from disk\n    X_train = pd.read_parquet('/kaggle/working/X_train.parquet')\n    y_train = pd.read_parquet('/kaggle/working/y_train.parquet')\n\n    # Convert to float32\n    X_train = X_train.astype('float32')\n    y_train = y_train.astype('float32')\n\n    # Get the input and output shapes\n    input_shape = [X_train.shape[1]]\n    output_shape = y_train.shape[1]\n\n    # Create the model\n    model = Sequential()\n\n    # Add input layer\n    model.add(BatchNormalization(input_shape=input_shape, name='input'))\n    if is_debug: print('Using batch normalization on input')\n\n    # Add the hidden layers\n    for layer in range(layers):\n\n        # Add the hidden layer\n        if l2 is not None:\n            model.add(Dense(\n                units=nodes,\n                kernel_regularizer=tf.keras.regularizers.l2(l2)),\n                name=f'hidden_{layer}')\n            if is_debug: print('Using L2 regularization')\n        else:\n            model.add(Dense(units=nodes, name=f'hidden_{layer}'))\n            if is_debug: print('No L2 regularization')\n\n        # Add batch normalization\n        if batch_norm:\n            model.add(BatchNormalization())\n            if is_debug: print('Using batch norm on hidden layer')\n\n        # Add the activation function\n        model.add(Activation(act_func, name=f'activation_{layer}'))\n\n        # Add dropout\n        if dropout is not None:\n            model.add(Dropout(dropout))\n            if is_debug: print('Using dropout in hidden layer')\n\n    # Add the output layer\n    model.add(Dense(\n        units=output_shape, \n        activation='sigmoid', \n        name='output'))\n\n    # Compile the model\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(\n            learning_rate=eta),\n        loss='binary_crossentropy')\n\n    # Summarize the model\n    if is_debug:\n        model.summary()\n\n    # Train the model\n    history = model.fit(\n        X_train, y_train,\n        #validation_split=0.2,\n        batch_size=batch_size,\n        epochs=epochs,\n        verbose=is_debug)\n\n    # Save the model\n    model.save(f'/kaggle/working/model.h5')\n    \n    # Clean up memory\n    del X_train\n    del y_train\n    del model\n    del history\n    tf.keras.backend.clear_session()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.65412Z","iopub.execute_input":"2023-08-07T19:38:03.654563Z","iopub.status.idle":"2023-08-07T19:38:03.670626Z","shell.execute_reply.started":"2023-08-07T19:38:03.654532Z","shell.execute_reply":"2023-08-07T19:38:03.66964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to make the predictions\ndef make_predictions():\n\n    # Display a status update\n    print(\"Making the predictions\")\n\n    # Load the model\n    model = tf.keras.models.load_model('/kaggle/working/model.h5')\n\n    # Load the test features\n    X_test = pd.read_parquet('/kaggle/working/X_test.parquet')\n\n    # Convert to float32\n    X_test = X_test.astype('float32')\n\n    # Make the predictions\n    predictions = model.predict(X_test, batch_size=1024)\n\n    # Convert to cupy array\n    predictions = cp.array(predictions)\n\n    # Convert to float32\n    predictions = predictions.astype('float16')\n    \n    # Save the predictions\n    cp.save('/kaggle/working/predictions.npy', predictions)\n    \n    # Set the number of predictions (globally)\n    global num_of_predictions\n    num_of_predictions = predictions.shape[0]\n    \n    # Clean up memory\n    del X_test\n    del model\n    del predictions\n    tf.keras.backend.clear_session()\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.672274Z","iopub.execute_input":"2023-08-07T19:38:03.672655Z","iopub.status.idle":"2023-08-07T19:38:03.686614Z","shell.execute_reply.started":"2023-08-07T19:38:03.672597Z","shell.execute_reply":"2023-08-07T19:38:03.685613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to create the submission\ndef create_submission():\n\n    # Display a status update\n    print(\"Creating the submission\")\n    \n    # Create the submission table\n    submission = cudf.DataFrame(columns = ['Protein Id', 'GO Term Id','Prediction'])\n    \n    \n    # Load the protein IDs\n    test_protein_ids = np.load('/kaggle/input/t5embeds/test_ids.npy')\n    \n    # Expand (broadcast) the list of protein IDs\n    protein_ids_list = []\n    for k in list(test_protein_ids):\n        protein_ids_list += [k] * num_of_go_terms\n\n    # Clean up memory\n    del test_protein_ids\n    gc.collect()\n\n    # Create protein ID column\n    submission['Protein Id'] = protein_ids_list\n\n    # Clean up memory\n    del protein_ids_list\n    gc.collect()\n    \n    \n    # Load the GO terms\n    go_terms = np.load('/kaggle/working/go_terms.npy')\n\n    # Convert GO terms to a list (for broadcasting)\n    go_terms = go_terms.tolist()\n    \n    # Create GO terms column\n    submission['GO Term Id'] = go_terms * num_of_predictions\n\n    # Clean up memory\n    del go_terms\n    gc.collect()\n    \n    \n    # Load the predictions\n    predictions = cp.load('/kaggle/working/predictions.npy')\n\n    # Ravel the prediction\n    predictions = predictions.ravel()\n\n    # Create the predictions column\n    submission['Prediction'] = predictions\n\n    # Convert to a decimal with 3 decimal places\n    submission['Prediction'] = submission['Prediction'].astype(cudf.Decimal32Dtype(4, 3))\n\n    # Clean up memory\n    del predictions\n    gc.collect()\n    \n    \n    # Save the submission file\n    submission.to_csv(\n        f'partial_submission.tsv',\n        sep='\\t',\n        index=False,\n        header=False,\n        chunksize=100000)\n\n    # Clean up memory\n    del submission\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.687916Z","iopub.execute_input":"2023-08-07T19:38:03.688387Z","iopub.status.idle":"2023-08-07T19:38:03.703986Z","shell.execute_reply.started":"2023-08-07T19:38:03.688356Z","shell.execute_reply":"2023-08-07T19:38:03.703069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to merge submissions\ndef merge_submission():\n    \n    # Display a status update\n    print(\"Merging the submission\")\n\n    # Merge the submission file\n    os.system(\"cat /kaggle/working/partial_submission.tsv >> /kaggle/working/submission.tsv\")\n\n    # Clean up the partial submission file\n    os.system(\"rm /kaggle/working/partial_submission.tsv\")","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.707386Z","iopub.execute_input":"2023-08-07T19:38:03.707655Z","iopub.status.idle":"2023-08-07T19:38:03.719876Z","shell.execute_reply.started":"2023-08-07T19:38:03.707633Z","shell.execute_reply":"2023-08-07T19:38:03.718957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to inspect memory usage\ndef inspect_memory():\n    \n    # Only display if in debug mode\n    if not is_debug:\n        return\n\n    # Display a status update\n    print(\"Inspecting memory usage\")\n\n    # Get a copy of the global variables dictionary\n    global_vars = globals().copy()\n\n    # Iterate over all variables in memory\n    for var_name, var_value in global_vars.items():\n        # Exclude special variables and modules\n        if not var_name.startswith('__') and not hasattr(var_value, '__call__'):\n            # Get the size of the variable\n            var_size = sys.getsizeof(var_value)\n            # Print the variable name and its size\n            print(f\"Variable: {var_name} | Size: {var_size} bytes\")","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.721285Z","iopub.execute_input":"2023-08-07T19:38:03.721814Z","iopub.status.idle":"2023-08-07T19:38:03.735889Z","shell.execute_reply.started":"2023-08-07T19:38:03.721784Z","shell.execute_reply":"2023-08-07T19:38:03.734816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n%%memit\n\nprepare_X_train()\nprepare_X_test()\n\nfor i in range(0, num_of_batches):\n    prepare_y_train(i)\n    train_model()\n    make_predictions()\n    create_submission()\n    merge_submission()\n    inspect_memory()\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-08-07T19:38:03.738889Z","iopub.execute_input":"2023-08-07T19:38:03.739337Z","iopub.status.idle":"2023-08-07T19:43:23.683769Z","shell.execute_reply.started":"2023-08-07T19:38:03.739312Z","shell.execute_reply":"2023-08-07T19:43:23.682731Z"},"trusted":true},"execution_count":null,"outputs":[]}]}