{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  Exploratory Data Analysis (EDA) Notebook \nThis notebook provides a basic framework for evaluating Exploratory Data Analysis (EDA) skills using Python. Students must execute each cell and pass the associated tests to proceed.","metadata":{}},{"cell_type":"code","source":"\n\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.offline as pyo\nfrom IPython.display import HTML\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:19:53.58762Z","iopub.execute_input":"2025-01-05T10:19:53.588021Z","iopub.status.idle":"2025-01-05T10:19:53.605611Z","shell.execute_reply.started":"2025-01-05T10:19:53.587984Z","shell.execute_reply":"2025-01-05T10:19:53.604281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_images\"\ntrain_thumbnails_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\ntest_thumbnails_path = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:19:53.581002Z","iopub.execute_input":"2025-01-05T10:19:53.581245Z","iopub.status.idle":"2025-01-05T10:19:53.586248Z","shell.execute_reply.started":"2025-01-05T10:19:53.581222Z","shell.execute_reply":"2025-01-05T10:19:53.584817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:27:23.608039Z","iopub.execute_input":"2025-01-05T10:27:23.608352Z","iopub.status.idle":"2025-01-05T10:27:23.616981Z","shell.execute_reply.started":"2025-01-05T10:27:23.60833Z","shell.execute_reply":"2025-01-05T10:27:23.615901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:27:30.901612Z","iopub.execute_input":"2025-01-05T10:27:30.902002Z","iopub.status.idle":"2025-01-05T10:27:30.920955Z","shell.execute_reply.started":"2025-01-05T10:27:30.901977Z","shell.execute_reply":"2025-01-05T10:27:30.91974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:54:03.246347Z","iopub.execute_input":"2025-01-05T10:54:03.246714Z","iopub.status.idle":"2025-01-05T10:54:03.253287Z","shell.execute_reply.started":"2025-01-05T10:54:03.246688Z","shell.execute_reply":"2025-01-05T10:54:03.252322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T10:55:11.739079Z","iopub.execute_input":"2025-01-05T10:55:11.739494Z","iopub.status.idle":"2025-01-05T10:55:11.766966Z","shell.execute_reply.started":"2025-01-05T10:55:11.739466Z","shell.execute_reply":"2025-01-05T10:55:11.765819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:01:18.632727Z","iopub.execute_input":"2025-01-05T11:01:18.633232Z","iopub.status.idle":"2025-01-05T11:01:18.643696Z","shell.execute_reply.started":"2025-01-05T11:01:18.633189Z","shell.execute_reply":"2025-01-05T11:01:18.642515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify dataset size\nprint(\"Size of the dataset: \", df.shape[0])\nassert df.shape[0] > 0, \"Dataset size must be greater than 0\"","metadata":{"nbgrader":{"grade":true,"grade_id":"dataset_size","locked":true,"points":1,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:04:13.801267Z","iopub.execute_input":"2025-01-05T11:04:13.801643Z","iopub.status.idle":"2025-01-05T11:04:13.808292Z","shell.execute_reply.started":"2025-01-05T11:04:13.801616Z","shell.execute_reply":"2025-01-05T11:04:13.80695Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visual Exploration\n### Random Image Plotting","metadata":{}},{"cell_type":"code","source":"Image.MAX_IMAGE_PIXELS = None #Increase the limit for image pixels to avoid DecompressionBombError\ndef plot_images(folder_path: str, resize: bool = False):\n   \n # Get a list of image file names in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.endswith(('.jpg', '.jpeg', '.png', '.gif'))]\n    num_images_to_plot = 6\n    selected_images = random.sample(image_files, num_images_to_plot)\n    fig, axes = plt.subplots(2, 3, figsize=(12, 8))\n    for i, ax in enumerate(axes.flat):\n        if i < num_images_to_plot:\n            image_path = os.path.join(folder_path, selected_images[i])\n            img = Image.open(image_path)\n            if resize:\n                img = img.resize((512,512))\n            img = np.array(img)\n            ax.imshow(img)\n            ax.set_title(selected_images[i])\n            ax.axis('off')\n    plt.tight_layout()\n    plt.show()\n\n# Test the function\nfolder_path = '/kaggle/input/UBC-OCEAN/train_images'  # Replace with your actual path\nplot_images(folder_path, resize=True)\n","metadata":{"nbgrader":{"grade":true,"grade_id":"plot_images","locked":true,"points":2,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:31:13.824246Z","iopub.execute_input":"2025-01-05T11:31:13.826336Z","iopub.status.idle":"2025-01-05T11:39:40.325971Z","shell.execute_reply.started":"2025-01-05T11:31:13.826227Z","shell.execute_reply":"2025-01-05T11:39:40.324127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Label Distribution","metadata":{}},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns = ['label', 'count']  # Rename the columns for clarity\n\nplt.bar(label_df['label'], label_df['count'], color='skyblue')  # Use the renamed columns\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.show()\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"\n","metadata":{"nbgrader":{"grade":true,"grade_id":"label_distribution","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:53:42.102483Z","iopub.execute_input":"2025-01-05T11:53:42.102889Z","iopub.status.idle":"2025-01-05T11:53:42.675502Z","shell.execute_reply.started":"2025-01-05T11:53:42.10286Z","shell.execute_reply":"2025-01-05T11:53:42.674318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class-wise Analysis","metadata":{}},{"cell_type":"code","source":"HGSC = df[df['label']==\"HGSC\"]\nEC = df[df['label']==\"EC\"]\nCC = df[df['label']==\"CC\"]\nLGSC = df[df['label']==\"LGSC\"]\nMC = df[df['label']==\"MC\"]\n\nplt.figure(figsize=(20, 6))\nplt.rcParams['font.size'] = 14\ncolors = ['red', 'lightblue', 'green','magenta', 'yellow']\nplt.pie([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)], \n        labels=['HGSC', 'EC', 'CC', 'LGSC', 'MC'], autopct='%1.1f%%', colors=colors)\nplt.title('Training Set')\nplt.show()\nassert sum([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)]) == len(df), \"Pie chart proportions must match dataset size\"","metadata":{"nbgrader":{"grade":true,"grade_id":"class_analysis","locked":true,"points":4,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:56:07.397882Z","iopub.execute_input":"2025-01-05T11:56:07.398428Z","iopub.status.idle":"2025-01-05T11:56:07.544099Z","shell.execute_reply.started":"2025-01-05T11:56:07.398394Z","shell.execute_reply":"2025-01-05T11:56:07.543004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter rows by label to calculate proportions\nHGSC = df[df['label'] == \"HGSC\"]\nEC = df[df['label'] == \"EC\"]\nCC = df[df['label'] == \"CC\"]\nLGSC = df[df['label'] == \"LGSC\"]\nMC = df[df['label'] == \"MC\"]\n\n# Correction: Adjusted figure size for better pie chart display\nplt.figure(figsize=(10, 10))  # Changed from (20, 6) to (10, 10) for a balanced layout\n\n# Set font size for readability\nplt.rcParams['font.size'] = 14\n\n# Create pie chart with appropriate colors and labels\ncolors = ['red', 'lightblue', 'green', 'magenta', 'yellow']\nplt.pie([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)], \n        labels=['HGSC', 'EC', 'CC', 'LGSC', 'MC'], autopct='%1.1f%%', colors=colors)\nplt.title('Training Set')  # Set chart title\nplt.show()\n\n# Assertion to ensure the pie chart proportions match the dataset size\nassert sum([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)]) == len(df), \"Pie chart proportions must match dataset size\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:57:50.030532Z","iopub.execute_input":"2025-01-05T11:57:50.031012Z","iopub.status.idle":"2025-01-05T11:57:50.245459Z","shell.execute_reply.started":"2025-01-05T11:57:50.030978Z","shell.execute_reply":"2025-01-05T11:57:50.244136Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Seaborn Visualization\n### Distribution Plot","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=df, x='label', kde=True, color='blue')\nplt.title('Label Distribution with Seaborn')\nplt.show()\nassert not df['label'].isnull().any(), \"Ensure there are no missing values in the label column\"","metadata":{"nbgrader":{"grade":true,"grade_id":"seaborn_distplot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T11:58:11.512873Z","iopub.execute_input":"2025-01-05T11:58:11.513336Z","iopub.status.idle":"2025-01-05T11:58:11.80315Z","shell.execute_reply.started":"2025-01-05T11:58:11.513305Z","shell.execute_reply":"2025-01-05T11:58:11.801941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#corrected version making sure there are noinf error\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Replace inf and -inf values in the 'label' column with NaN\ndf['label'] = df['label'].replace([np.inf, -np.inf], np.nan)  # Correction: Handle infinite values to avoid plotting issues\n\n# Drop rows with NaN values in the 'label' column\ndf = df.dropna(subset=['label'])  # Correction: Ensure no NaN values remain in the 'label' column\n\n# Use sns.countplot to plot the label distribution\nsns.countplot(data=df, x='label', palette='Blues')  # Visualization remains the same\n\n# Set the title and axis labels for clarity\nplt.title('Label Distribution with Seaborn')\nplt.xlabel('Labels')  # Added label for the x-axis\nplt.ylabel('Count')   # Added label for the y-axis\n\n# Display the plot\nplt.show()\n\n# Assertion to confirm the 'label' column is clean\nassert not df['label'].isnull().any(), \"Ensure there are no missing or inf values in the label column\"  # Final check to guarantee data integrity\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T12:02:00.398696Z","iopub.execute_input":"2025-01-05T12:02:00.399454Z","iopub.status.idle":"2025-01-05T12:02:00.646906Z","shell.execute_reply.started":"2025-01-05T12:02:00.399417Z","shell.execute_reply":"2025-01-05T12:02:00.645683Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlation Matrix","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load the dataset\ntrain_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n\n# Calculate the correlation matrix for numeric columns\nnumeric_df = train_df.select_dtypes(include=[np.number])\n\n# Plot the correlation matrix\nsns.heatmap(numeric_df.corr(), annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix for Training Data')\nplt.show()\n\n# Ensure the correlation matrix is valid\nassert numeric_df.corr().shape[0] > 0, \"Correlation matrix must be generated\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T13:38:01.349277Z","iopub.execute_input":"2025-01-05T13:38:01.349662Z","iopub.status.idle":"2025-01-05T13:38:01.642489Z","shell.execute_reply.started":"2025-01-05T13:38:01.349635Z","shell.execute_reply":"2025-01-05T13:38:01.641147Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Scatter Plot","metadata":{}},{"cell_type":"code","source":"sns.scatterplot(data=df=\"/kaggle/input/UBC-OCEAN/test.csv\", x='feature1', y='feature2', hue='label', palette='Set2')\nplt.title('Scatter Plot of Feature1 vs Feature2')\nplt.show()\nassert 'feature1' in df.columns and 'feature2' in df.columns, \"Ensure 'feature1' and 'feature2' columns exist in the dataset\"","metadata":{"nbgrader":{"grade":true,"grade_id":"scatter_plot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:13:55.38647Z","iopub.execute_input":"2025-01-05T15:13:55.38691Z","iopub.status.idle":"2025-01-05T15:13:55.394296Z","shell.execute_reply.started":"2025-01-05T15:13:55.386878Z","shell.execute_reply":"2025-01-05T15:13:55.392755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":" import pandas as pd\n\n# Load the dataset\ndf = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\n\n# Print the column names to confirm what is available\nprint(df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:50:40.014252Z","iopub.execute_input":"2025-01-05T15:50:40.014654Z","iopub.status.idle":"2025-01-05T15:50:40.031884Z","shell.execute_reply.started":"2025-01-05T15:50:40.014625Z","shell.execute_reply":"2025-01-05T15:50:40.030681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Make sure the columns 'image_width' and 'image_height' exist\nassert 'image_width' in df.columns and 'image_height' in df.columns, \"Ensure 'image_width' and 'image_height' columns exist in the dataset\"\n\n# Create a scatter plot\nsns.scatterplot(data=df, x='image_width', y='image_height', hue='label', palette='Set2')\n\n# Add a title and display the plot\nplt.title('Scatter Plot of Image Width vs Image Height')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:53:23.364968Z","iopub.execute_input":"2025-01-05T15:53:23.365338Z","iopub.status.idle":"2025-01-05T15:53:23.785098Z","shell.execute_reply.started":"2025-01-05T15:53:23.365309Z","shell.execute_reply":"2025-01-05T15:53:23.784029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Pair Plot","metadata":{}},{"cell_type":"code","source":"sns.pairplot(data=df, hue='label', palette='husl')\nplt.suptitle('Pair Plot of Dataset Features', y=1.02)\nplt.show()\nassert 'label' in df.columns, \"Ensure the 'label' column is included for the pair plot\"","metadata":{"nbgrader":{"grade":true,"grade_id":"pair_plot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:44:31.759173Z","iopub.execute_input":"2025-01-05T15:44:31.75953Z","iopub.status.idle":"2025-01-05T15:44:33.083015Z","shell.execute_reply.started":"2025-01-05T15:44:31.759503Z","shell.execute_reply":"2025-01-05T15:44:33.081314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Create the pair plot with 'label' as hue\nsns.pairplot(data=df, hue='label', palette='husl')\n\n# Set title for the plot\nplt.suptitle('Pair Plot of Dataset Features', y=1.02)\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T16:14:53.585518Z","iopub.execute_input":"2025-01-05T16:14:53.585938Z","iopub.status.idle":"2025-01-05T16:14:59.889306Z","shell.execute_reply.started":"2025-01-05T16:14:53.58591Z","shell.execute_reply":"2025-01-05T16:14:59.887992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Box Plot","metadata":{}},{"cell_type":"code","source":"sns.boxplot(data=df, x='label', y='numerical_feature', palette='cool')\nplt.title('Box Plot of Numerical Feature by Label')\nplt.show()\nassert 'numerical_feature' in df.columns, \"Ensure 'numerical_feature' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T16:22:46.242919Z","iopub.execute_input":"2025-01-05T16:22:46.243776Z","iopub.status.idle":"2025-01-05T16:22:46.273183Z","shell.execute_reply.started":"2025-01-05T16:22:46.243667Z","shell.execute_reply":"2025-01-05T16:22:46.271477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Ensure 'image_width' exists in the dataframe\nassert 'image_width' in df.columns, \"Ensure 'image_width' exists in the dataset\"\n\n# Create a boxplot using 'image_width' as the numerical feature\nsns.boxplot(data=df, x='label', y='image_width', palette='cool')\n\n# Set title for the plot\nplt.title('Box Plot of Image Width by Label')\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T16:25:09.173294Z","iopub.execute_input":"2025-01-05T16:25:09.173715Z","iopub.status.idle":"2025-01-05T16:25:09.480146Z","shell.execute_reply.started":"2025-01-05T16:25:09.173687Z","shell.execute_reply":"2025-01-05T16:25:09.478836Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count Plot","metadata":{}},{"cell_type":"code","source":"\ndf = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")  # Replace with your actual path to the dataset\n\nsns.countplot(data=df, x='categorical_feature', palette='pastel')\nplt.title('Count Plot of Categorical Feature')\nplt.show()\nassert 'categorical_feature' in df.columns, \"Ensure 'categorical_feature' exists in the dataset\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T16:12:02.741975Z","iopub.execute_input":"2025-01-05T16:12:02.742327Z","iopub.status.idle":"2025-01-05T16:12:02.766232Z","shell.execute_reply.started":"2025-01-05T16:12:02.742302Z","shell.execute_reply":"2025-01-05T16:12:02.764762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\n# Load the dataset\ndf = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")  # Adjust path as necessary\n\n# Ensure 'label' exists in the dataframe\nassert 'label' in df.columns, \"Ensure 'label' exists in the dataset\"\n\n# Create a count plot using 'label' as the categorical feature\nsns.countplot(data=df, x='label', palette='pastel')\n\n# Set the title for the plot\nplt.title('Count Plot of Categorical Feature')\n\n# Show the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T16:29:08.595622Z","iopub.execute_input":"2025-01-05T16:29:08.596033Z","iopub.status.idle":"2025-01-05T16:29:08.841472Z","shell.execute_reply.started":"2025-01-05T16:29:08.595999Z","shell.execute_reply":"2025-01-05T16:29:08.840466Z"}},"outputs":[],"execution_count":null}]}