{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-success\"> \n        <h1 align=\"center\" style=\"color:darkcyan;\">UBC-OCEAN \n</h1>  \n     \n</div>","metadata":{"execution":{"iopub.status.busy":"2023-10-07T05:00:08.100543Z","iopub.execute_input":"2023-10-07T05:00:08.101117Z","iopub.status.idle":"2023-10-07T05:00:08.13043Z","shell.execute_reply.started":"2023-10-07T05:00:08.10109Z","shell.execute_reply":"2023-10-07T05:00:08.129334Z"}}},{"cell_type":"markdown","source":"<div style=\"border-radius:10px; border:#DEB887 solid; padding: 15px; background-color: #7FFFD4; font-size:100%; text-align:left\">\n\n<h3 align=\"center\"><font color='#DAA520'>💡 About The Dataset :</font></h3>\n    \n\n1. **Data Types**:\n   - The dataset contains microscopy images of two types: whole slide images (WSI) and tissue microarray (TMA).\n\n2. **Train/Test Split**:\n   - It's divided into a training set and a test set.\n   - The test set includes images from different sources to assess model generalization.\n\n3. **Labels**:\n   - In the training set, labels are provided for each image, representing ovarian cancer subtypes.\n   - The \"Other\" class is not present in the training set, making outlier detection a challenge.\n\n4. **Image Sizes**:\n   - Image dimensions (width and height) are provided in the CSV files.\n   - Some large test images may not fit entirely in GPU memory.\n\n5. **Additional Notes**:\n   - The dataset is substantial, with a size of 550 GB.\n   - It includes WSI and TMA images with variations in dimensions and quality.\n\n6. **Sample Submission**:\n   - A sample submission CSV file is provided with one row.\n\n7. **Usage Restrictions**:\n   - Participants are asked not to use the data for external projects or research until the competition paper is published.","metadata":{}},{"cell_type":"code","source":"import xml.etree.ElementTree as ET\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport warnings\nfrom pathlib import Path\nimport random\n\nimport tensorflow as tf\n\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:20.706285Z","iopub.execute_input":"2023-10-09T09:45:20.707646Z","iopub.status.idle":"2023-10-09T09:45:34.894117Z","shell.execute_reply.started":"2023-10-09T09:45:20.707592Z","shell.execute_reply":"2023-10-09T09:45:34.892933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\"> \n        <h1 align=\"center\" style=\"color:darkcyan;\">Reading The Data \n</h1>  \n     \n</div>","metadata":{}},{"cell_type":"code","source":"image_path = '/kaggle/input/UBC-OCEAN/train_thumbnails'\ntrain_path  = '/kaggle/input/UBC-OCEAN/train_images'","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:34.896034Z","iopub.execute_input":"2023-10-09T09:45:34.896774Z","iopub.status.idle":"2023-10-09T09:45:34.901805Z","shell.execute_reply.started":"2023-10-09T09:45:34.896738Z","shell.execute_reply":"2023-10-09T09:45:34.900588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\"> \n        <h1 align=\"center\" style=\"color:darkcyan;\">Loading the Thumbnails\n</h1>  \n     \n</div>","metadata":{}},{"cell_type":"markdown","source":"def display_image(image_folder):\n    \n    image_files = [f for f in (os.listdir(image_folder)) if f.endswith(('jpg','png','jpeg'))]\n    \n    #print(image_files)\n    \n    fig,axes = plt.subplots(3,3,figsize=(10,10))\n    \n    for i, file_name in enumerate(image_files):  \n        \n        if i>=9:\n            break\n        row = i//3\n        col = i%3\n        \n         # Load and display the image in the current subplot\n        image_path = os.path.join(image_folder, file_name)\n        image = cv2.imread(image_path)\n        axes[row, col].imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n        axes[row, col].set_title(file_name)\n        axes[row, col].axis('off')\n        \n        \n    # Ensure proper layout and show the plots\n    plt.tight_layout()\n    plt.show()\n    ","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-09T08:29:47.413958Z","iopub.execute_input":"2023-10-09T08:29:47.414254Z","iopub.status.idle":"2023-10-09T08:29:47.432318Z","shell.execute_reply.started":"2023-10-09T08:29:47.41423Z","shell.execute_reply":"2023-10-09T08:29:47.431491Z"}}},{"cell_type":"markdown","source":"# Specify the folder path containing the images\nimage_folder =image_path  # Replace with the path to your image folder\ndisplay_image(image_folder)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-09T08:29:47.436651Z","iopub.execute_input":"2023-10-09T08:29:47.437054Z","iopub.status.idle":"2023-10-09T08:29:57.819649Z","shell.execute_reply.started":"2023-10-09T08:29:47.437006Z","shell.execute_reply":"2023-10-09T08:29:57.818154Z"}}},{"cell_type":"markdown","source":"# List all image files in the folder\nimage_files = [os.path.join(image_folder, filename) for filename in os.listdir(image_folder)]\n\n# Randomly select 20 images from the list\nselected_images = random.sample(image_files, 10)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-09T08:29:57.821176Z","iopub.execute_input":"2023-10-09T08:29:57.821693Z","iopub.status.idle":"2023-10-09T08:29:57.83496Z","shell.execute_reply.started":"2023-10-09T08:29:57.821662Z","shell.execute_reply":"2023-10-09T08:29:57.834092Z"}}},{"cell_type":"markdown","source":"from matplotlib.animation import FuncAnimation\n\n# Create a function to update the animation frames\ndef update(frame):\n    plt.clf()\n    plt.imshow(cv2.cvtColor(cv2.imread(selected_images[frame]), cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n\n# Create the animation\nanimation = FuncAnimation(plt.figure(), update, frames=len(selected_images), interval=500)\n\n# Display the animation\nfrom IPython.display import HTML\nHTML(animation.to_jshtml())","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-09T08:29:57.836371Z","iopub.execute_input":"2023-10-09T08:29:57.836684Z","iopub.status.idle":"2023-10-09T08:30:09.94617Z","shell.execute_reply.started":"2023-10-09T08:29:57.836658Z","shell.execute_reply":"2023-10-09T08:30:09.944962Z"}}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndf_test = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')\ndf_sub = pd.read_csv('/kaggle/input/UBC-OCEAN/sample_submission.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:34.903506Z","iopub.execute_input":"2023-10-09T09:45:34.903853Z","iopub.status.idle":"2023-10-09T09:45:34.954112Z","shell.execute_reply.started":"2023-10-09T09:45:34.903826Z","shell.execute_reply":"2023-10-09T09:45:34.952845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:34.957715Z","iopub.execute_input":"2023-10-09T09:45:34.958597Z","iopub.status.idle":"2023-10-09T09:45:34.976913Z","shell.execute_reply.started":"2023-10-09T09:45:34.958538Z","shell.execute_reply":"2023-10-09T09:45:34.975803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:34.978137Z","iopub.execute_input":"2023-10-09T09:45:34.97915Z","iopub.status.idle":"2023-10-09T09:45:35.009316Z","shell.execute_reply.started":"2023-10-09T09:45:34.979096Z","shell.execute_reply":"2023-10-09T09:45:35.007342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dtypes,df_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.010971Z","iopub.execute_input":"2023-10-09T09:45:35.011675Z","iopub.status.idle":"2023-10-09T09:45:35.021535Z","shell.execute_reply.started":"2023-10-09T09:45:35.011616Z","shell.execute_reply":"2023-10-09T09:45:35.019934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.022831Z","iopub.execute_input":"2023-10-09T09:45:35.023547Z","iopub.status.idle":"2023-10-09T09:45:35.044596Z","shell.execute_reply.started":"2023-10-09T09:45:35.023501Z","shell.execute_reply":"2023-10-09T09:45:35.04275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.045952Z","iopub.execute_input":"2023-10-09T09:45:35.046344Z","iopub.status.idle":"2023-10-09T09:45:35.064641Z","shell.execute_reply.started":"2023-10-09T09:45:35.046314Z","shell.execute_reply":"2023-10-09T09:45:35.063699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Class Distribution\nclass_distribution = df_train['label'].value_counts()\n\n# Plot class distribution\nplt.figure(figsize=(10, 5))\nsns.countplot(data=df_train, x='label', order=class_distribution.index)\nplt.title('Class Distribution')\nplt.xticks(rotation=45)\nplt.xlabel('Ovarian Cancer Subtypes')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.066236Z","iopub.execute_input":"2023-10-09T09:45:35.066621Z","iopub.status.idle":"2023-10-09T09:45:35.376638Z","shell.execute_reply.started":"2023-10-09T09:45:35.06659Z","shell.execute_reply":"2023-10-09T09:45:35.374974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image Dimensions\nimage_width_stats = df_train.groupby('label')['image_width'].describe()\nimage_height_stats = df_train.groupby('label')['image_height'].describe()\n\n# Plot image dimensions per class\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nsns.boxplot(data=df_train, x='label', y='image_width', order=class_distribution.index)\nplt.title('Image Width by Class')\nplt.xticks(rotation=45)\nplt.xlabel('Ovarian Cancer Subtypes')\nplt.ylabel('Image Width (pixels)')\n\nplt.subplot(1, 2, 2)\nsns.boxplot(data=df_train, x='label', y='image_height', order=class_distribution.index)\nplt.title('Image Height by Class')\nplt.xticks(rotation=45)\nplt.xlabel('Ovarian Cancer Subtypes')\nplt.ylabel('Image Height (pixels)')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.381286Z","iopub.execute_input":"2023-10-09T09:45:35.38282Z","iopub.status.idle":"2023-10-09T09:45:35.925579Z","shell.execute_reply.started":"2023-10-09T09:45:35.382755Z","shell.execute_reply":"2023-10-09T09:45:35.923731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tissue Microarray (TMA)\ntma_proportion = df_train['is_tma'].value_counts(normalize=True)\n\n# Plot TMA proportion\nplt.figure(figsize=(6, 6))\nplt.pie(tma_proportion, labels=['Non-TMA', 'TMA'], autopct='%1.1f%%', startangle=90)\nplt.title('TMA Proportion')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:35.927513Z","iopub.execute_input":"2023-10-09T09:45:35.927915Z","iopub.status.idle":"2023-10-09T09:45:36.058248Z","shell.execute_reply.started":"2023-10-09T09:45:35.927885Z","shell.execute_reply":"2023-10-09T09:45:36.056264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 9. Exploration of TMA Slides\ntma_proportion_by_class = df_train.groupby('label')['is_tma'].value_counts(normalize=True).unstack(fill_value=0)\ntma_proportion_by_class.plot(kind='bar', stacked=True)\nplt.title('Proportion of TMA Slides by Class')\nplt.xlabel('Ovarian Cancer Subtypes')\nplt.ylabel('Proportion')\nplt.xticks(rotation=45)\nplt.legend(title='TMA')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.060808Z","iopub.execute_input":"2023-10-09T09:45:36.062332Z","iopub.status.idle":"2023-10-09T09:45:36.411007Z","shell.execute_reply.started":"2023-10-09T09:45:36.062251Z","shell.execute_reply":"2023-10-09T09:45:36.409832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.412603Z","iopub.execute_input":"2023-10-09T09:45:36.413805Z","iopub.status.idle":"2023-10-09T09:45:36.422774Z","shell.execute_reply.started":"2023-10-09T09:45:36.413769Z","shell.execute_reply":"2023-10-09T09:45:36.421581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\nlabel_encoder = LabelEncoder()\ndf_train['label'] = label_encoder.fit_transform(df_train['label'])","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.423902Z","iopub.execute_input":"2023-10-09T09:45:36.424309Z","iopub.status.idle":"2023-10-09T09:45:36.439294Z","shell.execute_reply.started":"2023-10-09T09:45:36.4242Z","shell.execute_reply":"2023-10-09T09:45:36.438475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.440531Z","iopub.execute_input":"2023-10-09T09:45:36.441023Z","iopub.status.idle":"2023-10-09T09:45:36.459098Z","shell.execute_reply.started":"2023-10-09T09:45:36.440994Z","shell.execute_reply":"2023-10-09T09:45:36.457765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['is_tma'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.460511Z","iopub.execute_input":"2023-10-09T09:45:36.460811Z","iopub.status.idle":"2023-10-09T09:45:36.477476Z","shell.execute_reply.started":"2023-10-09T09:45:36.460786Z","shell.execute_reply":"2023-10-09T09:45:36.476233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(['is_tma'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.479056Z","iopub.execute_input":"2023-10-09T09:45:36.479832Z","iopub.status.idle":"2023-10-09T09:45:36.497702Z","shell.execute_reply.started":"2023-10-09T09:45:36.479799Z","shell.execute_reply":"2023-10-09T09:45:36.496261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:36.498908Z","iopub.execute_input":"2023-10-09T09:45:36.499881Z","iopub.status.idle":"2023-10-09T09:45:36.523669Z","shell.execute_reply.started":"2023-10-09T09:45:36.499841Z","shell.execute_reply":"2023-10-09T09:45:36.522482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px; border:#DEB887 solid; padding: 15px; background-color: #00FA9A; font-size:100%; text-align:left\">\n\n<h3 align=\"center\"><font color='#DAA520'>💡We will use features and anlyze it using xgboost intially:</font></h3>\n    \n\n","metadata":{}},{"cell_type":"code","source":"# Separate features (X) and the target (y)\nX = df_train.drop(columns=['label'])\ny = df_train['label']\nnum_classes = 5\n\n# Apply SMOTE to balance the classes in the training set\nsmote = SMOTE(random_state=42)\nX_train_resampled, y_train_resampled = smote.fit_resample(X, y)\n\n# Split the resampled data into training and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_train_resampled, y_train_resampled, test_size=0.2, stratify=y_train_resampled, random_state=42)\n\n# Define the XGBoost model with hyperparameters to address overfitting\nmodel = xgb.XGBClassifier(\n    num_class=num_classes,\n    learning_rate=0.001,  # Adjust the learning rate\n    max_depth=7,         # Limit the maximum depth of trees\n    reg_alpha=0.1,       # Adjust the L1 regularization term\n    reg_lambda=0.01,      # Adjust the L2 regularization term\n    n_estimators=1000    # Increase the number of trees\n)\n\n# Fit the XGBoost model\nmodel.fit(X_train, y_train, eval_set=[(X_test, y_test)], early_stopping_rounds=10, eval_metric=\"mlogloss\")\n\n# Make predictions on the test set\ny_pred = model.predict(X_test)\n\n# Evaluate the model\naccuracy = accuracy_score(y_test, y_pred)\nprint(f'Accuracy: {accuracy:.2f}')\nprint(classification_report(y_test, y_pred))\n","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-10-09T09:45:36.525566Z","iopub.execute_input":"2023-10-09T09:45:36.526718Z","iopub.status.idle":"2023-10-09T09:45:47.522779Z","shell.execute_reply.started":"2023-10-09T09:45:36.526661Z","shell.execute_reply":"2023-10-09T09:45:47.521915Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ny_predx = model.predict(df_test)\n\n# Convert predicted labels back to original class labels\npredicted_labels = label_encoder.inverse_transform(y_predx.astype(int))\npredicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:47.526988Z","iopub.execute_input":"2023-10-09T09:45:47.527432Z","iopub.status.idle":"2023-10-09T09:45:47.545754Z","shell.execute_reply.started":"2023-10-09T09:45:47.527385Z","shell.execute_reply":"2023-10-09T09:45:47.54468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub['label'] =predicted_labels","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:47.547451Z","iopub.execute_input":"2023-10-09T09:45:47.550186Z","iopub.status.idle":"2023-10-09T09:45:47.555445Z","shell.execute_reply.started":"2023-10-09T09:45:47.550139Z","shell.execute_reply":"2023-10-09T09:45:47.554649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T09:45:47.556622Z","iopub.execute_input":"2023-10-09T09:45:47.558438Z","iopub.status.idle":"2023-10-09T09:45:47.576666Z","shell.execute_reply.started":"2023-10-09T09:45:47.558357Z","shell.execute_reply":"2023-10-09T09:45:47.574708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div style=\"border-radius:10px; border:#DEB887 solid; padding: 15px; background-color: #00FA9A; font-size:100%; text-align:left\">\n\n<h3 align=\"center\"><font color='#DAA520'>💡Feedback for improvement :</font></h3>\n    \n\n","metadata":{}}]}