{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":148683476,"sourceType":"kernelVersion"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:19:57.904396Z","iopub.execute_input":"2025-01-06T18:19:57.904859Z","iopub.status.idle":"2025-01-06T18:20:03.15686Z","shell.execute_reply.started":"2025-01-06T18:19:57.904825Z","shell.execute_reply":"2025-01-06T18:20:03.155421Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA) Notebook\nThis notebook provides a basic framework for evaluating Exploratory Data Analysis (EDA) skills using Python. Students must execute each cell and pass the associated tests to proceed.","metadata":{}},{"cell_type":"code","source":"\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.offline as pyo\nfrom IPython.display import HTML","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:11.781624Z","iopub.execute_input":"2025-01-06T18:20:11.782052Z","iopub.status.idle":"2025-01-06T18:20:11.787823Z","shell.execute_reply.started":"2025-01-06T18:20:11.782018Z","shell.execute_reply":"2025-01-06T18:20:11.786596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_images\"\ntrain_thumbnails_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\ntest_thumbnails_path = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:15.957711Z","iopub.execute_input":"2025-01-06T18:20:15.958122Z","iopub.status.idle":"2025-01-06T18:20:15.963527Z","shell.execute_reply.started":"2025-01-06T18:20:15.95809Z","shell.execute_reply":"2025-01-06T18:20:15.962039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:17.952624Z","iopub.execute_input":"2025-01-06T18:20:17.953064Z","iopub.status.idle":"2025-01-06T18:20:17.963305Z","shell.execute_reply.started":"2025-01-06T18:20:17.953024Z","shell.execute_reply":"2025-01-06T18:20:17.962019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:19.471994Z","iopub.execute_input":"2025-01-06T18:20:19.472461Z","iopub.status.idle":"2025-01-06T18:20:19.483695Z","shell.execute_reply.started":"2025-01-06T18:20:19.472419Z","shell.execute_reply":"2025-01-06T18:20:19.482359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Size of the dataset: \", df.shape[0])\nassert df.shape[0] > 0, \"Dataset size must be greater than 0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:21.490881Z","iopub.execute_input":"2025-01-06T18:20:21.491313Z","iopub.status.idle":"2025-01-06T18:20:21.497609Z","shell.execute_reply.started":"2025-01-06T18:20:21.491277Z","shell.execute_reply":"2025-01-06T18:20:21.496334Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Visual Exploration**\n\n**Random Image Plotting**","metadata":{}},{"cell_type":"code","source":"def plot_images(folder_path: str, resize: bool = False):\n    # Get a list of image file names in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.endswith(('.jpg', '.jpeg', '.png', '.gif'))]\n    num_images_to_plot = 4\n    selected_images = random.sample(image_files, num_images_to_plot)\n    fig, axes = plt.subplots(2, 2, figsize=(12, 8))\n    for i, ax in enumerate(axes.flat):\n        if i < num_images_to_plot:\n            image_path = os.path.join(folder_path, selected_images[i])\n            img = Image.open(image_path)\n            if resize:\n                img = img.resize((512,512))\n            img = np.array(img)\n            ax.imshow(img)\n            ax.set_title(selected_images[i])\n            ax.axis('off')\n    plt.tight_layout()\n    plt.show()\n    \n\n# Test the function\nplot_images('/kaggle/input/UBC-OCEAN/train_thumbnails', resize=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:25.404738Z","iopub.execute_input":"2025-01-06T18:20:25.405097Z","iopub.status.idle":"2025-01-06T18:20:27.173385Z","shell.execute_reply.started":"2025-01-06T18:20:25.405063Z","shell.execute_reply":"2025-01-06T18:20:27.172161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns = ['Ovarian Cancer subtypes', 'count']  \nplt.bar(label_df['Ovarian Cancer subtypes'], label_df['count'], color='brown')\nplt.xlabel('Ovarian Cancer subtypes')\nplt.ylabel('Prevalence')\nplt.title('Ovarian Cancer subtypes Distribution')\nplt.show()\n\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:20:30.220264Z","iopub.execute_input":"2025-01-06T18:20:30.220694Z","iopub.status.idle":"2025-01-06T18:20:30.457064Z","shell.execute_reply.started":"2025-01-06T18:20:30.220663Z","shell.execute_reply":"2025-01-06T18:20:30.455891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns = ['Ovarian Cancer subtypes', 'count']\n\nlabel_df['color'] = pd.cut(label_df['count'], bins=len(['#FF7F7F', '#FF3030', '#DC143C', '#8B0000']), labels=['#FF7F7F', '#FF3030', '#DC143C', '#8B0000'])\n\nplt.bar(label_df['Ovarian Cancer subtypes'], label_df['count'], color=label_df['color'])\nplt.xlabel('Ovarian Cancer subtypes')\nplt.ylabel('Prevalence')\nplt.title('Ovarian Cancer subtypes Distribution')\nplt.show()\n\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:28:54.224062Z","iopub.execute_input":"2025-01-06T18:28:54.22457Z","iopub.status.idle":"2025-01-06T18:28:54.479269Z","shell.execute_reply.started":"2025-01-06T18:28:54.224527Z","shell.execute_reply":"2025-01-06T18:28:54.477678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=df, x='label', kde=True, color='#8B0000')\nplt.title('Subtype Distribution with Seaborn')\nplt.show()\nassert not df['label'].isnull().any(), \"Ensure there are no missing values in the label column\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:42:13.300169Z","iopub.execute_input":"2025-01-06T18:42:13.300619Z","iopub.status.idle":"2025-01-06T18:42:13.578738Z","shell.execute_reply.started":"2025-01-06T18:42:13.300587Z","shell.execute_reply":"2025-01-06T18:42:13.577573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_df = df.select_dtypes(include=['number'])\ncorrelation_matrix = numeric_df.corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='RdYlBu')\nplt.title('Correlation Matrix')\nplt.show()\nassert correlation_matrix.shape[0] > 0, \"Correlation matrix must be generated\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:37:01.740437Z","iopub.execute_input":"2025-01-06T18:37:01.740855Z","iopub.status.idle":"2025-01-06T18:37:02.053918Z","shell.execute_reply.started":"2025-01-06T18:37:01.740824Z","shell.execute_reply":"2025-01-06T18:37:02.05268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.scatterplot(data=df, x='image_id', y='is_tma', hue='label', palette='RdYlBu')\nplt.title('Scatter Plot of image_id vs is_tma')\nplt.show()\nassert 'image_id' in df.columns and 'is_tma' in df.columns, \"Ensure 'image_id' and 'is_tma' columns exist in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:37:15.379069Z","iopub.execute_input":"2025-01-06T18:37:15.379539Z","iopub.status.idle":"2025-01-06T18:37:15.746805Z","shell.execute_reply.started":"2025-01-06T18:37:15.379502Z","shell.execute_reply":"2025-01-06T18:37:15.745594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.pairplot(data=df, hue='label', palette='hsv')\nplt.suptitle('Pair Plot of Dataset Features', y=1.02)\nplt.show()\nassert 'label' in df.columns, \"Ensure the 'label' column is included for the pair plot\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:39:19.288332Z","iopub.execute_input":"2025-01-06T18:39:19.288773Z","iopub.status.idle":"2025-01-06T18:39:25.427716Z","shell.execute_reply.started":"2025-01-06T18:39:19.28874Z","shell.execute_reply":"2025-01-06T18:39:25.4262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Box Plot**","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=df, x='label', y='image_height', palette='rainbow')\nplt.title('Box Plot of Image Height by Subtype')\nplt.show()\nassert 'image_height' in df.columns, \"Ensure 'image_height' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:42:43.813703Z","iopub.execute_input":"2025-01-06T18:42:43.814178Z","iopub.status.idle":"2025-01-06T18:42:44.026039Z","shell.execute_reply.started":"2025-01-06T18:42:43.814138Z","shell.execute_reply":"2025-01-06T18:42:44.024886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count Plot","metadata":{"execution":{"iopub.status.busy":"2025-01-06T00:57:49.942255Z","iopub.execute_input":"2025-01-06T00:57:49.942633Z","iopub.status.idle":"2025-01-06T00:57:49.946847Z","shell.execute_reply.started":"2025-01-06T00:57:49.942586Z","shell.execute_reply":"2025-01-06T00:57:49.945718Z"}}},{"cell_type":"code","source":"sns.countplot(data=df, x='label', palette='twilight')\nplt.title('Count Plot of label')\nplt.show()\nassert 'label' in df.columns, \"Ensure 'label' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-06T18:44:21.410972Z","iopub.execute_input":"2025-01-06T18:44:21.411456Z","iopub.status.idle":"2025-01-06T18:44:21.589229Z","shell.execute_reply.started":"2025-01-06T18:44:21.411416Z","shell.execute_reply":"2025-01-06T18:44:21.587978Z"}},"outputs":[],"execution_count":null}]}