{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":148683476,"sourceType":"kernelVersion"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:14.778358Z","iopub.execute_input":"2025-01-18T10:23:14.778742Z","iopub.status.idle":"2025-01-18T10:23:15.953245Z","shell.execute_reply.started":"2025-01-18T10:23:14.778701Z","shell.execute_reply":"2025-01-18T10:23:15.952273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA) Notebook\nThis notebook provides a basic framework for evaluating Exploratory Data Analysis (EDA) skills using Python. Students must execute each cell and pass the associated tests to proceed.","metadata":{}},{"cell_type":"code","source":"\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.offline as pyo\nfrom IPython.display import HTML","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:15.954827Z","iopub.execute_input":"2025-01-18T10:23:15.955238Z","iopub.status.idle":"2025-01-18T10:23:15.960245Z","shell.execute_reply.started":"2025-01-18T10:23:15.9552Z","shell.execute_reply":"2025-01-18T10:23:15.959259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_images\"\ntrain_thumbnails_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\ntest_thumbnails_path = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:15.962181Z","iopub.execute_input":"2025-01-18T10:23:15.962462Z","iopub.status.idle":"2025-01-18T10:23:15.987276Z","shell.execute_reply.started":"2025-01-18T10:23:15.962437Z","shell.execute_reply":"2025-01-18T10:23:15.9856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:15.988058Z","iopub.status.idle":"2025-01-18T10:23:15.988397Z","shell.execute_reply":"2025-01-18T10:23:15.988266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:15.989356Z","iopub.status.idle":"2025-01-18T10:23:15.989684Z","shell.execute_reply":"2025-01-18T10:23:15.989556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Size of the dataset: \", df.shape[0])\nassert df.shape[0] > 0, \"Dataset size must be greater than 0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:15.990874Z","iopub.status.idle":"2025-01-18T10:23:15.991247Z","shell.execute_reply":"2025-01-18T10:23:15.991091Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Visual Exploration**\n\n**Random Image Plotting**","metadata":{}},{"cell_type":"code","source":"def plot_images(folder_path: str, resize: bool = False):\n    # Get a list of image file names in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.endswith(('.jpg', '.jpeg', '.png', '.gif'))]\n    num_images_to_plot = 4\n    selected_images = random.sample(image_files, num_images_to_plot)\n    fig, axes = plt.subplots(2, 2, figsize=(12, 8))\n    for i, ax in enumerate(axes.flat):\n        if i < num_images_to_plot:\n            image_path = os.path.join(folder_path, selected_images[i])\n            img = Image.open(image_path)\n            if resize:\n                img = img.resize((512,512))\n            img = np.array(img)\n            ax.imshow(img)\n            ax.set_title(selected_images[i])\n            ax.axis('off')\n    plt.tight_layout()\n    plt.show()\n    \n\n# Test the function\nplot_images('/kaggle/input/UBC-OCEAN/train_thumbnails', resize=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns = ['Ovarian Cancer subtypes', 'count']  \nplt.bar(label_df['Ovarian Cancer subtypes'], label_df['count'], color='brown')\nplt.xlabel('Ovarian Cancer subtypes')\nplt.ylabel('Prevalence')\nplt.title('Ovarian Cancer subtypes Distribution')\nplt.show()\n\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:18.692481Z","iopub.execute_input":"2025-01-18T10:23:18.692928Z","iopub.status.idle":"2025-01-18T10:23:18.932007Z","shell.execute_reply.started":"2025-01-18T10:23:18.692892Z","shell.execute_reply":"2025-01-18T10:23:18.930993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns = ['Ovarian Cancer subtypes', 'count']\n\nlabel_df['color'] = pd.cut(label_df['count'], bins=len(['#FF7F7F', '#FF3030', '#DC143C', '#8B0000']), labels=['#FF7F7F', '#FF3030', '#DC143C', '#8B0000'])\n\nplt.bar(label_df['Ovarian Cancer subtypes'], label_df['count'], color=label_df['color'])\nplt.xlabel('Ovarian Cancer subtypes')\nplt.ylabel('Prevalence')\nplt.title('Ovarian Cancer subtypes Distribution')\nplt.show()\n\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:25.4653Z","iopub.execute_input":"2025-01-18T10:23:25.465645Z","iopub.status.idle":"2025-01-18T10:23:25.707597Z","shell.execute_reply.started":"2025-01-18T10:23:25.465616Z","shell.execute_reply":"2025-01-18T10:23:25.706502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=df, x='label', kde=True, color='#8B0000')\nplt.title('Subtype Distribution with Seaborn')\nplt.show()\nassert not df['label'].isnull().any(), \"Ensure there are no missing values in the label column\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:28.624989Z","iopub.execute_input":"2025-01-18T10:23:28.62536Z","iopub.status.idle":"2025-01-18T10:23:28.885479Z","shell.execute_reply.started":"2025-01-18T10:23:28.625332Z","shell.execute_reply":"2025-01-18T10:23:28.884355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_df = df.select_dtypes(include=['number'])\ncorrelation_matrix = numeric_df.corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='RdYlBu')\nplt.title('Correlation Matrix')\nplt.show()\nassert correlation_matrix.shape[0] > 0, \"Correlation matrix must be generated\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:31.998569Z","iopub.execute_input":"2025-01-18T10:23:31.998964Z","iopub.status.idle":"2025-01-18T10:23:32.22414Z","shell.execute_reply.started":"2025-01-18T10:23:31.998929Z","shell.execute_reply":"2025-01-18T10:23:32.223106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(data=df, x='image_id', y='is_tma', hue='label', palette='RdYlBu')\nplt.title('Scatter Plot of image_id vs is_tma')\nplt.show()\nassert 'image_id' in df.columns and 'is_tma' in df.columns, \"Ensure 'image_id' and 'is_tma' columns exist in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:38.704947Z","iopub.execute_input":"2025-01-18T10:23:38.705311Z","iopub.status.idle":"2025-01-18T10:23:39.088644Z","shell.execute_reply.started":"2025-01-18T10:23:38.705282Z","shell.execute_reply":"2025-01-18T10:23:39.087443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.pairplot(data=df, hue='label', palette='hsv')\nplt.suptitle('Pair Plot of Dataset Features', y=1.02)\nplt.show()\nassert 'label' in df.columns, \"Ensure the 'label' column is included for the pair plot\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:44.473254Z","iopub.execute_input":"2025-01-18T10:23:44.473707Z","iopub.status.idle":"2025-01-18T10:23:50.669156Z","shell.execute_reply.started":"2025-01-18T10:23:44.473667Z","shell.execute_reply":"2025-01-18T10:23:50.667834Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Box Plot**","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=df, x='label', y='image_height', palette='rainbow')\nplt.title('Box Plot of Image Height by Subtype')\nplt.show()\nassert 'image_height' in df.columns, \"Ensure 'image_height' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:54.47244Z","iopub.execute_input":"2025-01-18T10:23:54.472814Z","iopub.status.idle":"2025-01-18T10:23:54.863577Z","shell.execute_reply.started":"2025-01-18T10:23:54.472786Z","shell.execute_reply":"2025-01-18T10:23:54.862603Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count Plot","metadata":{"execution":{"iopub.status.busy":"2025-01-06T00:57:49.942255Z","iopub.execute_input":"2025-01-06T00:57:49.942633Z","iopub.status.idle":"2025-01-06T00:57:49.946847Z","shell.execute_reply.started":"2025-01-06T00:57:49.942586Z","shell.execute_reply":"2025-01-06T00:57:49.945718Z"}}},{"cell_type":"code","source":"sns.countplot(data=df, x='label', palette='twilight')\nplt.title('Count Plot of label')\nplt.show()\nassert 'label' in df.columns, \"Ensure 'label' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-18T10:23:58.521671Z","iopub.execute_input":"2025-01-18T10:23:58.522083Z","iopub.status.idle":"2025-01-18T10:23:58.744624Z","shell.execute_reply.started":"2025-01-18T10:23:58.522049Z","shell.execute_reply":"2025-01-18T10:23:58.743459Z"}},"outputs":[],"execution_count":null}]}