{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:80%;letter-spacing:0.5px;margin:0\"><b> </b> Import Libraries</p></div>\n","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob\nimport matplotlib.image as mpimg\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:50:57.71883Z","iopub.execute_input":"2023-10-10T07:50:57.719191Z","iopub.status.idle":"2023-10-10T07:50:59.549046Z","shell.execute_reply.started":"2023-10-10T07:50:57.719164Z","shell.execute_reply":"2023-10-10T07:50:59.547743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Load Data</p></div>\n","metadata":{}},{"cell_type":"code","source":"# Load image file paths\ntrain_data = glob.glob('/kaggle/input/UBC-OCEAN/train_images/*.png')\n\n# Load CSV \ntrain = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\n\n# Create a DataFrame that combines file paths and labels\ncombined_data = pd.DataFrame({'file_path': train_data})\n\n# Extract image IDs from file paths (you might need to modify this depending on your file naming convention)\ncombined_data['image_id'] = combined_data['file_path'].apply(lambda x: x.split('/')[-1].split('.')[0])\n\n# Convert 'image_id' column in train DataFrame to string\ntrain['image_id'] = train['image_id'].astype(str)\n\n# Merge with the labels DataFrame based on the image IDs\ncombined_data = pd.merge(combined_data, train, how='inner', left_on='image_id', right_on='image_id')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:04.345986Z","iopub.execute_input":"2023-10-10T07:51:04.346567Z","iopub.status.idle":"2023-10-10T07:51:04.569507Z","shell.execute_reply.started":"2023-10-10T07:51:04.346531Z","shell.execute_reply":"2023-10-10T07:51:04.568305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_data.head().style.set_properties(**{'background-color':'royalblue','color':'white','border':'#8b8c8c'})","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:08.562142Z","iopub.execute_input":"2023-10-10T07:51:08.562647Z","iopub.status.idle":"2023-10-10T07:51:08.666148Z","shell.execute_reply.started":"2023-10-10T07:51:08.562603Z","shell.execute_reply":"2023-10-10T07:51:08.664989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:90%;letter-spacing:0.5px;margin:0\"><b> </b>Data Cleaning and Preprocessing</p></div>\n\n","metadata":{}},{"cell_type":"code","source":"combined_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:12.463367Z","iopub.execute_input":"2023-10-10T07:51:12.463881Z","iopub.status.idle":"2023-10-10T07:51:12.488144Z","shell.execute_reply.started":"2023-10-10T07:51:12.463854Z","shell.execute_reply":"2023-10-10T07:51:12.48703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"combined_data.describe().style.background_gradient(cmap='tab20c')","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:16.542886Z","iopub.execute_input":"2023-10-10T07:51:16.543237Z","iopub.status.idle":"2023-10-10T07:51:16.567017Z","shell.execute_reply.started":"2023-10-10T07:51:16.543193Z","shell.execute_reply":"2023-10-10T07:51:16.565916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's print out the unique values in the 'is_tma' column\nunique_is_tma = combined_data['is_tma'].unique()\n\nprint(\"\\nUnique is_tma values:\")\nprint(unique_is_tma)","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:20.324806Z","iopub.execute_input":"2023-10-10T07:51:20.325157Z","iopub.status.idle":"2023-10-10T07:51:20.334077Z","shell.execute_reply.started":"2023-10-10T07:51:20.325132Z","shell.execute_reply":"2023-10-10T07:51:20.332926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Class Distribution\nclass_distribution = combined_data['label'].value_counts()\n\nprint(\"\\nClass Distribution:\")\nprint(class_distribution)","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:24.255958Z","iopub.execute_input":"2023-10-10T07:51:24.256303Z","iopub.status.idle":"2023-10-10T07:51:24.264351Z","shell.execute_reply.started":"2023-10-10T07:51:24.256277Z","shell.execute_reply":"2023-10-10T07:51:24.263148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image Dimensions\nimage_dimensions = combined_data[['image_width', 'image_height']]\n\nprint(\"\\nImage Dimensions:\")\nprint(image_dimensions)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:51:27.726812Z","iopub.execute_input":"2023-10-10T07:51:27.727131Z","iopub.status.idle":"2023-10-10T07:51:27.73818Z","shell.execute_reply.started":"2023-10-10T07:51:27.727108Z","shell.execute_reply":"2023-10-10T07:51:27.737032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:80%;letter-spacing:0.5px;margin:0\"><b> </b>Exploratory Data Analysis (EDA)📊</p></div>","metadata":{}},{"cell_type":"code","source":"# Label Distribution\nplt.figure(figsize=(10, 6))\nsns.countplot(data=combined_data, x='label', palette='Set2')\nplt.title('Label Distribution', fontsize = 15, fontweight = 'bold', color = 'magenta')\nplt.xlabel('Label', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Count', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.savefig('Label Distribution.png')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:51:31.93852Z","iopub.execute_input":"2023-10-10T07:51:31.939241Z","iopub.status.idle":"2023-10-10T07:51:32.286322Z","shell.execute_reply.started":"2023-10-10T07:51:31.939179Z","shell.execute_reply":"2023-10-10T07:51:32.285279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:75%;letter-spacing:0.5px;margin:0\"><b> </b>The descriptions of each subtype of ovarian carcinoma:</p></div>\n\n### 1. HGSC (High-Grade Serous Carcinoma)\n### 2. LGSC (Low-Grade Serous Carcinoma)\n### 3. EC (Endometrioid Carcinoma)\n### 4. CC (Clear Cell Carcinoma)\n### 5. MC (Mucinous Carcinoma)\n","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"color:blue;display:inline-block;border-radius:5px;background-color:#fec8c1;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:blue;overflow:hidden;font-size:70%;letter-spacing:0.5px;margin:0\"><b> </b>HGSC (High-Grade Serous Carcinoma)</p></div>\n\n\n**HGSC (High-Grade Serous Carcinoma):** High-Grade Serous Carcinoma is an aggressive form of ovarian cancer that is typically diagnosed at an advanced stage.\n","metadata":{}},{"cell_type":"code","source":"# Define the file path you want to search for\nsearch_path = '/kaggle/input/UBC-OCEAN/train_images/8280.png' \n\n# Search for the image in the combined data\nresult = combined_data[combined_data['file_path'] == search_path]\n\n# Print the result\nprint(result)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:51:38.515591Z","iopub.execute_input":"2023-10-10T07:51:38.515976Z","iopub.status.idle":"2023-10-10T07:51:38.525812Z","shell.execute_reply.started":"2023-10-10T07:51:38.515946Z","shell.execute_reply":"2023-10-10T07:51:38.524628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the relevant information from the result\nfile_path = result['file_path'].values[0]\nimage_id = result['image_id'].values[0]\nlabel = result['label'].values[0]\nimage_width = result['image_width'].values[0]\nimage_height = result['image_height'].values[0]\nis_tma = result['is_tma'].values[0]\n\n# Plotting\nplt.figure(figsize=(8, 6))\n\n# Display the image (if available)\ntry:\n    img = mpimg.imread(file_path)\n    plt.imshow(img, extent=(0, image_width, 0, image_height), alpha=0.5)\nexcept FileNotFoundError:\n    pass\n\n# Add labels\nplt.text(0, image_height + 1000, f\"Image ID: {image_id}\", fontsize=12)\nplt.text(0, image_height + 2000, f\"Label: {label}\", fontsize=12)\nplt.text(0, image_height + 3000, f\"Image Width: {image_width}\", fontsize=12)\nplt.text(0, image_height + 4000, f\"Image Height: {image_height}\", fontsize=12)\nplt.text(0, image_height + 5000, f\"Is TMA: {is_tma}\", fontsize=12)\n\n# Set labels and title\nplt.xlabel('Image Width', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Image Height', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Image Information', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Image Information1.png')\n\n# Show plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:51:42.516488Z","iopub.execute_input":"2023-10-10T07:51:42.516841Z","iopub.status.idle":"2023-10-10T07:51:46.96014Z","shell.execute_reply.started":"2023-10-10T07:51:42.516815Z","shell.execute_reply":"2023-10-10T07:51:46.959314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:blue;display:inline-block;border-radius:5px;background-color: #fec8c1;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:blue;overflow:hidden;font-size:70%;letter-spacing:0.5px;margin:0\"><b> </b>LGSC (Low-Grade Serous Carcinoma)</p></div>\n\nLGSC (Low-Grade Serous Carcinoma):Low-Grade Serous Carcinoma is a less aggressive form of ovarian cancer, often diagnosed at an earlier stage.\n","metadata":{}},{"cell_type":"code","source":"# Define the file path you want to search for\nsearch_path = '/kaggle/input/UBC-OCEAN/train_images/11557.png'\n\n# Search for the image in the combined data\nresult = combined_data[combined_data['file_path'] == search_path]\n\n# Print the result\nprint(result)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:51:55.507276Z","iopub.execute_input":"2023-10-10T07:51:55.507612Z","iopub.status.idle":"2023-10-10T07:51:55.516006Z","shell.execute_reply.started":"2023-10-10T07:51:55.507588Z","shell.execute_reply":"2023-10-10T07:51:55.515269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the relevant information from the result\nfile_path = result['file_path'].values[0]\nimage_id = result['image_id'].values[0]\nlabel = result['label'].values[0]\nimage_width = result['image_width'].values[0]\nimage_height = result['image_height'].values[0]\nis_tma = result['is_tma'].values[0]\n\n# Plotting\nplt.figure(figsize=(8, 6))\n\n# Display the image (if available)\ntry:\n    img = mpimg.imread(file_path)\n    plt.imshow(img, extent=(0, image_width, 0, image_height), alpha=0.5)\nexcept FileNotFoundError:\n    pass\n\n# Add labels\nplt.text(0, image_height + 1000, f\"Image ID: {image_id}\", fontsize=12)\nplt.text(0, image_height + 2000, f\"Label: {label}\", fontsize=12)\nplt.text(0, image_height + 3000, f\"Image Width: {image_width}\", fontsize=12)\nplt.text(0, image_height + 4000, f\"Image Height: {image_height}\", fontsize=12)\nplt.text(0, image_height + 5000, f\"Is TMA: {is_tma}\", fontsize=12)\n\n# Set labels and title\nplt.xlabel('Image Width', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Image Height', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Image Information', fontsize = 14, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Image Information2.png')\n\n# Show plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:52:01.37936Z","iopub.execute_input":"2023-10-10T07:52:01.379708Z","iopub.status.idle":"2023-10-10T07:54:05.224317Z","shell.execute_reply.started":"2023-10-10T07:52:01.379682Z","shell.execute_reply":"2023-10-10T07:54:05.222634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:blue;display:inline-block;border-radius:5px;background-color: #fec8c1;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:blue;overflow:hidden;font-size:70%;letter-spacing:0.5px;margin:0\"><b> </b>EC (Endometrioid Carcinoma)</p></div>\n\n**EC (Endometrioid Carcinoma):** Endometrioid Carcinoma is a type of ovarian cancer that resembles the tissue lining the uterus (endometrium).\n","metadata":{}},{"cell_type":"code","source":"# Define the file path you want to search for \nsearch_path = '/kaggle/input/UBC-OCEAN/train_images/51215.png'\n\n# Search for the image in the combined data\nresult = combined_data[combined_data['file_path'] == search_path]\n\n# Print the result\nprint(result)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:54:14.276803Z","iopub.execute_input":"2023-10-10T07:54:14.277939Z","iopub.status.idle":"2023-10-10T07:54:14.292007Z","shell.execute_reply.started":"2023-10-10T07:54:14.277901Z","shell.execute_reply":"2023-10-10T07:54:14.290955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the relevant information from the result\nfile_path = result['file_path'].values[0]\nimage_id = result['image_id'].values[0]\nlabel = result['label'].values[0]\nimage_width = result['image_width'].values[0]\nimage_height = result['image_height'].values[0]\nis_tma = result['is_tma'].values[0]\n\n# Plotting\nplt.figure(figsize=(8, 6))\n\n# Display the image (if available)\ntry:\n    img = mpimg.imread(file_path)\n    plt.imshow(img, extent=(0, image_width, 0, image_height), alpha=0.5)\nexcept FileNotFoundError:\n    pass\n\n# Add labels\nplt.text(0, image_height + 1000, f\"Image ID: {image_id}\", fontsize=12)\nplt.text(0, image_height + 2000, f\"Label: {label}\", fontsize=12)\nplt.text(0, image_height + 3000, f\"Image Width: {image_width}\", fontsize=12)\nplt.text(0, image_height + 4000, f\"Image Height: {image_height}\", fontsize=12)\nplt.text(0, image_height + 5000, f\"Is TMA: {is_tma}\", fontsize=12)\n\n# Set labels and title\nplt.xlabel('Image Width', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Image Height', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Image Information', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Image Information3.png')\n\n# Show plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:54:18.354094Z","iopub.execute_input":"2023-10-10T07:54:18.354481Z","iopub.status.idle":"2023-10-10T07:54:42.510794Z","shell.execute_reply.started":"2023-10-10T07:54:18.354452Z","shell.execute_reply":"2023-10-10T07:54:42.509734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:blue;display:inline-block;border-radius:5px;background-color: #fec8c1;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:blue;overflow:hidden;font-size:70%;letter-spacing:0.5px;margin:0\"><b> </b>CC (Clear Cell Carcinoma)</p></div>\n\n**CC (Clear Cell Carcinoma):** Clear Cell Carcinoma is a type of ovarian cancer characterized by cells that appear clear under a microscope.\n","metadata":{}},{"cell_type":"code","source":"# Define the file path you want to search for\nsearch_path = '/kaggle/input/UBC-OCEAN/train_images/36302.png'\n\n# Search for the image in the combined data\nresult = combined_data[combined_data['file_path'] == search_path]\n\n# Print the result\nprint(result)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:54:51.665652Z","iopub.execute_input":"2023-10-10T07:54:51.665991Z","iopub.status.idle":"2023-10-10T07:54:51.674548Z","shell.execute_reply.started":"2023-10-10T07:54:51.665967Z","shell.execute_reply":"2023-10-10T07:54:51.673509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the relevant information from the result\nfile_path = result['file_path'].values[0]\nimage_id = result['image_id'].values[0]\nlabel = result['label'].values[0]\nimage_width = result['image_width'].values[0]\nimage_height = result['image_height'].values[0]\nis_tma = result['is_tma'].values[0]\n\n# Plotting\nplt.figure(figsize=(8, 6))\n\n# Display the image (if available)\ntry:\n    img = mpimg.imread(file_path)\n    plt.imshow(img, extent=(0, image_width, 0, image_height), alpha=0.5)\nexcept FileNotFoundError:\n    pass\n\n# Add labels\nplt.text(0, image_height + 1000, f\"Image ID: {image_id}\", fontsize=12)\nplt.text(0, image_height + 2000, f\"Label: {label}\", fontsize=12)\nplt.text(0, image_height + 3000, f\"Image Width: {image_width}\", fontsize=12)\nplt.text(0, image_height + 4000, f\"Image Height: {image_height}\", fontsize=12)\nplt.text(0, image_height + 5000, f\"Is TMA: {is_tma}\", fontsize=12)\n\n# Set labels and title\nplt.xlabel('Image Width', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Image Height', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Image Information', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Image Information4.png')\n\n\n# Show plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:54:56.06607Z","iopub.execute_input":"2023-10-10T07:54:56.066453Z","iopub.status.idle":"2023-10-10T07:55:01.673785Z","shell.execute_reply.started":"2023-10-10T07:54:56.066425Z","shell.execute_reply":"2023-10-10T07:55:01.672418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:blue;display:inline-block;border-radius:5px;background-color: #fec8c1;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:blue;overflow:hidden;font-size:70%;letter-spacing:0.5px;margin:0\"><b> </b>MC (Mucinous Carcinoma)</p></div>\n\n**MC (Mucinous Carcinoma):** Mucinous Carcinoma is a type of ovarian cancer that arises from cells that produce mucus.\n","metadata":{}},{"cell_type":"code","source":"# Define the file path you want to search for \nsearch_path = '/kaggle/input/UBC-OCEAN/train_images/9200.png' \n\n# Search for the image in the combined data\nresult = combined_data[combined_data['file_path'] == search_path]\n\n# Print the result\nprint(result)\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:55:09.58408Z","iopub.execute_input":"2023-10-10T07:55:09.584446Z","iopub.status.idle":"2023-10-10T07:55:09.593303Z","shell.execute_reply.started":"2023-10-10T07:55:09.584419Z","shell.execute_reply":"2023-10-10T07:55:09.592282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extract the relevant information from the result\nfile_path = result['file_path'].values[0]\nimage_id = result['image_id'].values[0]\nlabel = result['label'].values[0]\nimage_width = result['image_width'].values[0]\nimage_height = result['image_height'].values[0]\nis_tma = result['is_tma'].values[0]\n\n# Plotting\nplt.figure(figsize=(8, 6))\n\n# Display the image (if available)\ntry:\n    img = mpimg.imread(file_path)\n    plt.imshow(img, extent=(0, image_width, 0, image_height), alpha=0.5)\nexcept FileNotFoundError:\n    pass\n\n# Add labels\nplt.text(0, image_height + 1000, f\"Image ID: {image_id}\", fontsize=12)\nplt.text(0, image_height + 2000, f\"Label: {label}\", fontsize=12)\nplt.text(0, image_height + 3000, f\"Image Width: {image_width}\", fontsize=12)\nplt.text(0, image_height + 4000, f\"Image Height: {image_height}\", fontsize=12)\nplt.text(0, image_height + 5000, f\"Is TMA: {is_tma}\", fontsize=12)\n\n# Set labels and title\nplt.xlabel('Image Width', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.ylabel('Image Height', fontsize = 12, fontweight = 'bold', color = 'darkblue')\nplt.title('Image Information', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\n\n# Save the plot\nplt.savefig('Image Information5.png')\n\n\n# Show plot\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:55:13.882397Z","iopub.execute_input":"2023-10-10T07:55:13.88275Z","iopub.status.idle":"2023-10-10T07:55:19.456857Z","shell.execute_reply.started":"2023-10-10T07:55:13.882722Z","shell.execute_reply":"2023-10-10T07:55:19.455715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:75%;letter-spacing:0.5px;margin:0\"><b> </b>Image Augmentations</p></div>\n\n### There are several different types of image augmentations.\n\n💠 **Rotation:** You can rotate the image by a specified angle.\n\n💠 **Flipping:** You can flip the image horizontally or vertically.\n\n💠 **Scaling:** You can resize the image to a different width and height.\n\n💠 **Translation:** You can shift the image in the x and y directions.\n\n💠 **Shearing:** You can shear the image by tilting it in a certain direction.\n\n💠 **Zooming:** You can zoom in or out of the image.\n\n💠 **Brightness and Contrast Adjustments:** You can adjust the brightness, contrast, and gamma of the image.\n\n💠 **Noise:** You can add different types of noise to the image (e.g., Gaussian noise).\n\n💠 **Blurring:** You can apply different types of blurs to the image (e.g., Gaussian blur, median blur).\n\n💠 **Color Space Adjustments:** You can convert the image to different color spaces (e.g., grayscale, HSV, LAB).\n\n💠 **Histogram Equalization:** This can help to improve the contrast of the image.\n\n💠 **Random Cropping:** You can randomly crop a portion of the image.\n\n💠 **Elastic Deformations:** This can simulate deformations in the image.\n\n💠 **Augmentations Specific to Task:**\n\n  * For object detection tasks, you might perform bounding box adjustments along with the image.\n  * For semantic segmentation tasks, you can apply the same transformations to both the image and the corresponding mask.\n","metadata":{}},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\n\n# Load the original image\nfile_path = \"/kaggle/input/UBC-OCEAN/train_images/9200.png\"\noriginal_image = cv2.imread(file_path)\n\n# Display the original image\nplt.imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))\nplt.title('Original Image', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\nplt.show()\n\n# Define augmentation parameters\nrotation_angle = 30\nhorizontal_flip = True\nvertical_flip = True\nscaling_factor = 0.5\ntranslation_x = 50\ntranslation_y = 50\nshear_factor = 0.5\nzoom_factor = 1.5\nbrightness = 1.2\n\n# Apply rotation\nrows, cols, _ = original_image.shape\nrotation_matrix = cv2.getRotationMatrix2D((cols/2, rows/2), rotation_angle, 1)\nrotated_image = cv2.warpAffine(original_image, rotation_matrix, (cols, rows))\n\n# Apply horizontal flip\nflipped_horizontal = cv2.flip(original_image, 1)\n\n# Apply vertical flip\nflipped_vertical = cv2.flip(original_image, 0)\n\n# Apply scaling\nscaled_image = cv2.resize(original_image, None, fx=scaling_factor, fy=scaling_factor)\n\n# Apply translation\ntranslation_matrix = np.float32([[1, 0, translation_x], [0, 1, translation_y]])\ntranslated_image = cv2.warpAffine(original_image, translation_matrix, (cols, rows))\n\n# Apply shear\nshear_matrix = np.float32([[1, shear_factor, 0], [0, 1, 0]])\nsheared_image = cv2.warpAffine(original_image, shear_matrix, (cols, rows))\n\n# Apply zoom\nzoomed_image = cv2.resize(original_image, None, fx=zoom_factor, fy=zoom_factor)\n\n# Adjust brightness\nbrightened_image = np.clip(original_image * brightness, 0, 255).astype(np.uint8)\n\n# Display augmented images\nimages = [rotated_image, flipped_horizontal, flipped_vertical, scaled_image,\n          translated_image, sheared_image, zoomed_image, brightened_image]\ntitles = ['Rotated Image', 'Flipped Horizontal', 'Flipped Vertical', 'Scaled Image',\n          'Translated Image', 'Sheared Image', 'Zoomed Image', 'Brightened Image']\n\nfor i in range(len(images)):\n    plt.subplot(2, 4, i+1)\n    plt.imshow(cv2.cvtColor(images[i], cv2.COLOR_BGR2RGB))\n    plt.title(titles[i])\n    plt.axis('off')\n\nplt.tight_layout()\n\n# Save the plot\nplt.savefig('Titles.png')\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:55:28.4926Z","iopub.execute_input":"2023-10-10T07:55:28.492944Z","iopub.status.idle":"2023-10-10T07:55:51.148394Z","shell.execute_reply.started":"2023-10-10T07:55:28.492918Z","shell.execute_reply":"2023-10-10T07:55:51.147375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Explanation of Augmentations:\n\n🔵 **Rotation (rotation_angle=30):** The image is rotated by 30 degrees.\n\n🔵 **Horizontal Flip (horizontal_flip=True):** The image is flipped horizontally.\n\n🔵 **Vertical Flip (vertical_flip=True):** The image is flipped vertically.\n\n🔵 **Scaling (scaling_factor=0.5):** The image is scaled down by a factor of 0.5.\n\n🔵 **Translation (translation_x=50, translation_y=50):** The image is translated 50 pixels to the right and 50 pixels down.\n\n🔵 **Shear (shear_factor=0.5):** The image is sheared by a factor of 0.5.\n\n🔵 **Zoom (zoom_factor=1.5):** The image is zoomed in by a factor of 1.5.\n\n🔵 **Brightness Adjustment (brightness=1.2):** The brightness of the image is increased by a factor of 1.2.","metadata":{}},{"cell_type":"markdown","source":"🔵 **Gaussian noise** is a random variation in brightness or color information. It is generated here using a normal distribution with a mean of 0 and standard deviation of 1. This noise is then added to the original image.\n","metadata":{}},{"cell_type":"code","source":"# Add Gaussian Noise\nnoise = np.random.normal(0, 1, original_image.shape).astype('uint8')\nnoisy_image = cv2.add(original_image, noise)\nnoisy_image","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:55:59.709253Z","iopub.execute_input":"2023-10-10T07:55:59.7096Z","iopub.status.idle":"2023-10-10T07:56:00.934429Z","shell.execute_reply.started":"2023-10-10T07:55:59.709575Z","shell.execute_reply":"2023-10-10T07:56:00.933371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **Gaussian blur** is a type of image-blurring filter. It smoothens the image by averaging the pixel values in a local region. The (5, 5) parameter specifies the size of the kernel (an odd number is recommended).","metadata":{}},{"cell_type":"code","source":"# Apply Gaussian Blur\nblurred_image = cv2.GaussianBlur(original_image, (5, 5), 0)\nblurred_image","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:56:05.973066Z","iopub.execute_input":"2023-10-10T07:56:05.973483Z","iopub.status.idle":"2023-10-10T07:56:06.014753Z","shell.execute_reply.started":"2023-10-10T07:56:05.97345Z","shell.execute_reply":"2023-10-10T07:56:06.012755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **Histogram equalization** enhances the contrast of an image by redistributing the intensity levels. This is particularly useful for improving visibility of details.","metadata":{}},{"cell_type":"code","source":"# Apply Histogram Equalization (for grayscale images)\ngray_image = cv2.cvtColor(original_image, cv2.COLOR_BGR2GRAY)\nequalized_image = cv2.equalizeHist(gray_image)\nequalized_image","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:56:11.081058Z","iopub.execute_input":"2023-10-10T07:56:11.081463Z","iopub.status.idle":"2023-10-10T07:56:11.103698Z","shell.execute_reply.started":"2023-10-10T07:56:11.081433Z","shell.execute_reply":"2023-10-10T07:56:11.10241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **Random cropping** involves selecting a random region of the image. In this code, new_width and new_height are the desired dimensions of the cropped region.","metadata":{}},{"cell_type":"code","source":"# Assuming you have a desired new width and height\nnew_width = 100  # Specify your desired width\nnew_height = 100  # Specify your desired height\n\n# Randomly crop a portion of the image\ncrop_x = np.random.randint(0, cols - new_width)\ncrop_y = np.random.randint(0, rows - new_height)\ncropped_image = original_image[crop_y:crop_y+new_height, crop_x:crop_x+new_width]\ncropped_image","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:56:14.974372Z","iopub.execute_input":"2023-10-10T07:56:14.975262Z","iopub.status.idle":"2023-10-10T07:56:14.987133Z","shell.execute_reply.started":"2023-10-10T07:56:14.975184Z","shell.execute_reply":"2023-10-10T07:56:14.986269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **Elastic deformations** simulate the effect of stretching and squishing an image. In this code, `alpha` controls the intensity of the deformation, and `sigma` determines the smoothness of the deformation.","metadata":{}},{"cell_type":"code","source":"from scipy.ndimage import gaussian_filter\nfrom scipy.ndimage import gaussian_filter, map_coordinates\n\n# Assuming you have the shape of the original image\nshape = original_image.shape\n\n# Define parameters for elastic deformations\nalpha = 20\nsigma = 5\n\nrandom_state = np.random.RandomState(None)\n\n# Generate random displacement fields\ndx = gaussian_filter((random_state.rand(*shape) * 2 - 1), sigma, mode=\"constant\", cval=0) * alpha\ndy = gaussian_filter((random_state.rand(*shape) * 2 - 1), sigma, mode=\"constant\", cval=0) * alpha\ndz = np.zeros_like(dx)\n\n# Generate grid of indices\nx, y, z = np.meshgrid(np.arange(shape[1]), np.arange(shape[0]), np.arange(shape[2]))\n\n# Apply deformation to indices\nindices = np.reshape(y+dy, (-1, 1)), np.reshape(x+dx, (-1, 1)), np.reshape(z+dz, (-1, 1))\n\n# Map coordinates from original to distorted image\ndistorted_image = map_coordinates(original_image, indices, order=1, mode='reflect')\ndistorted_image = distorted_image.reshape(shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:56:20.811809Z","iopub.execute_input":"2023-10-10T07:56:20.812484Z","iopub.status.idle":"2023-10-10T07:56:33.714559Z","shell.execute_reply.started":"2023-10-10T07:56:20.812449Z","shell.execute_reply":"2023-10-10T07:56:33.713452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the original image\nplt.imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))\nplt.title('Original Image', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\n\n# Save the plot\nplt.savefig('Original Image.png')\nplt.show()\n\n# Display the distorted image\nplt.imshow(cv2.cvtColor(distorted_image, cv2.COLOR_BGR2RGB))\nplt.title('Distorted Image', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\n\n# Save the plot\nplt.savefig('Distorted Image.png')\n\nplt.show()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:56:41.304506Z","iopub.execute_input":"2023-10-10T07:56:41.305056Z","iopub.status.idle":"2023-10-10T07:56:47.32232Z","shell.execute_reply.started":"2023-10-10T07:56:41.305028Z","shell.execute_reply":"2023-10-10T07:56:47.32113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Object Detection: Bounding Box Adjustments","metadata":{}},{"cell_type":"code","source":"# Assuming bounding box coordinates are provided\nbounding_box = (100, 50, 200, 150)  # (x_min, y_min, x_max, y_max)\n\n# Extract bounding box coordinates\nx_min, y_min, x_max, y_max = bounding_box\n\n# Apply same augmentation to image and bounding box\n# For example, let's rotate both the image and bounding box by 30 degrees\nrotation_angle = 30\nrotation_matrix = cv2.getRotationMatrix2D(((x_min + x_max)/2, (y_min + y_max)/2), rotation_angle, 1)\nrotated_image = cv2.warpAffine(original_image, rotation_matrix, (cols, rows))\n\n# Adjust bounding box coordinates after rotation\nnew_x_min = int(rotation_matrix[0, 0] * x_min + rotation_matrix[0, 1] * y_min + rotation_matrix[0, 2])\nnew_y_min = int(rotation_matrix[1, 0] * x_min + rotation_matrix[1, 1] * y_min + rotation_matrix[1, 2])\nnew_x_max = int(rotation_matrix[0, 0] * x_max + rotation_matrix[0, 1] * y_max + rotation_matrix[0, 2])\nnew_y_max = int(rotation_matrix[1, 0] * x_max + rotation_matrix[1, 1] * y_max + rotation_matrix[1, 2])\n\n# Draw the adjusted bounding box on the rotated image\ncv2.rectangle(rotated_image, (new_x_min, new_y_min), (new_x_max, new_y_max), (0, 255, 0), 2)\n\n# Display the rotated image with bounding box\nplt.imshow(cv2.cvtColor(rotated_image, cv2.COLOR_BGR2RGB))\nplt.title('Rotated Image with Bounding Box', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\nplt.savefig('Rotated Image with Bounding Box.png')\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:56:52.722176Z","iopub.execute_input":"2023-10-10T07:56:52.722547Z","iopub.status.idle":"2023-10-10T07:56:55.745017Z","shell.execute_reply.started":"2023-10-10T07:56:52.722521Z","shell.execute_reply":"2023-10-10T07:56:55.743754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **In object detection, it's crucial to adjust bounding boxes along with the image transformations. Here, I've demonstrated how to rotate both the image and its bounding box. After rotation, the bounding box coordinates are adjusted accordingly.**\n","metadata":{}},{"cell_type":"markdown","source":"### Semantic Segmentation: Apply Transformations to Image and Mask\n","metadata":{}},{"cell_type":"code","source":"# Generate a simple binary mask\nmask = np.zeros_like(original_image)\nmask[50:150, 100:200, :] = 255  # Assuming region of interest\n\n# Apply same augmentation to image and mask\nblurred_image = cv2.GaussianBlur(original_image, (5, 5), 0)\nblurred_mask = cv2.GaussianBlur(mask, (5, 5), 0)\n\n# Display the original image and mask\nplt.figure(figsize=(10, 4))\n\nplt.subplot(1, 2, 1)\nplt.imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))\nplt.title('Original Image', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nplt.imshow(cv2.cvtColor(mask, cv2.COLOR_BGR2RGB))  # Display mask in color for visualization\nplt.title('Original Mask', fontsize = 15, fontweight = 'bold', color = 'darkred')\nplt.axis('off')\n\nplt.tight_layout()\nplt.show()\n\n# Display the augmented image and mask\nplt.figure(figsize=(10, 4))\n\nplt.subplot(1, 2, 1)\nplt.imshow(cv2.cvtColor(blurred_image, cv2.COLOR_BGR2RGB))\nplt.title('Augmented Image', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\n\nplt.subplot(1, 2, 2)\nplt.imshow(cv2.cvtColor(blurred_mask, cv2.COLOR_BGR2RGB))  # Display mask in color for visualization\nplt.title('Augmented Mask', fontsize = 15, fontweight = 'bold', color = 'darkred')\nplt.axis('off')\n\nplt.tight_layout()\n\n# Save the plot\nplt.savefig('Augmented.png')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:57:08.293064Z","iopub.execute_input":"2023-10-10T07:57:08.294045Z","iopub.status.idle":"2023-10-10T07:57:18.629949Z","shell.execute_reply.started":"2023-10-10T07:57:08.29401Z","shell.execute_reply":"2023-10-10T07:57:18.629127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"🔵 **For semantic segmentation tasks, it's crucial to apply the same transformations to both the image and its corresponding mask. Here, I've demonstrated how to apply Gaussian Blur to both the image and mask. The original and augmented image and mask are displayed for comparison.**","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:80%;letter-spacing:0.5px;margin:0\"><b> </b>Image Outlier</p></div>\n","metadata":{}},{"cell_type":"code","source":"import cv2\n# Load the original image\nfile_path = \"/kaggle/input/UBC-OCEAN/train_images/51215.png\"\noriginal_image = cv2.imread(file_path)\n\n# Define the outlier parameters\noutlier_x = 100  # x-coordinate of the top-left corner of the rectangle\noutlier_y = 50   # y-coordinate of the top-left corner of the rectangle\noutlier_width = 50  # Width of the rectangle\noutlier_height = 30  # Height of the rectangle\n\n# Add the outlier (red rectangle) to the image\ncv2.rectangle(original_image, (outlier_x, outlier_y), (outlier_x + outlier_width, outlier_y + outlier_height), (0, 0, 255), -1)\n\n# Display the image with the outlier\nplt.imshow(cv2.cvtColor(original_image, cv2.COLOR_BGR2RGB))\nplt.title('Image with Outlier', fontsize = 15, fontweight = 'bold', color = 'darkgreen')\nplt.axis('off')\n\n# Save the plot\nplt.savefig('Image with Outlier.png')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-10-10T07:57:22.840197Z","iopub.execute_input":"2023-10-10T07:57:22.840583Z","iopub.status.idle":"2023-10-10T07:57:36.435007Z","shell.execute_reply.started":"2023-10-10T07:57:22.840556Z","shell.execute_reply":"2023-10-10T07:57:36.434269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:white;display:inline-block;border-radius:5px;background-color:#29a8ab;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:80%;letter-spacing:0.5px;margin:0\"><b> </b>Build a Model and Prediction</p></div>\n","metadata":{}},{"cell_type":"code","source":"# Load CSV \ndata = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntest_data = pd.read_csv('/kaggle/input/UBC-OCEAN/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:57:47.17497Z","iopub.execute_input":"2023-10-10T07:57:47.175391Z","iopub.status.idle":"2023-10-10T07:57:47.212792Z","shell.execute_reply.started":"2023-10-10T07:57:47.175359Z","shell.execute_reply":"2023-10-10T07:57:47.212039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define features and target variable\nfeatures = ['image_width', 'image_height', 'is_tma']\ntarget = 'label'\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(data[features], data[target], test_size=0.2, random_state=42)\n\n# Initialize and train the Random Forest Classifier\nclf = RandomForestClassifier(n_estimators=100, random_state=42)\nclf.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred = clf.predict(X_test)\n\n# Calculate the accuracy of the model\naccuracy = accuracy_score(y_test, y_pred)\nprint(f'Accuracy: {accuracy}')\n","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:57:50.728227Z","iopub.execute_input":"2023-10-10T07:57:50.728904Z","iopub.status.idle":"2023-10-10T07:57:50.950929Z","shell.execute_reply.started":"2023-10-10T07:57:50.728875Z","shell.execute_reply":"2023-10-10T07:57:50.949747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the Random Forest Classifier\nclf = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the model on the entire dataset\nclf.fit(data[features], data[target])\n\nnew_data = {\n    'image_width': [20000],\n    'image_height': [15000],\n    'is_tma': [False]\n}\n\nnew_df = pd.DataFrame(new_data)\n\n# Predict the label for the new data\npredicted_label = clf.predict(new_df)\nprint(predicted_label)\n","metadata":{"execution":{"iopub.status.busy":"2023-10-10T07:57:55.012758Z","iopub.execute_input":"2023-10-10T07:57:55.013099Z","iopub.status.idle":"2023-10-10T07:57:55.237175Z","shell.execute_reply.started":"2023-10-10T07:57:55.013074Z","shell.execute_reply":"2023-10-10T07:57:55.236019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\"> \"Your positive feedback and upvotes are incredibly appreciated! They inspire me to create more valuable content and help others in their learning journey. Your support fosters a vibrant community of knowledge-sharing. Thank you for considering an upvote, and best wishes on your learning journey!\" 😊📌</div>","metadata":{}}]}