{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  Exploratory Data Analysis (EDA) Notebook \nThis notebook provides a basic framework for evaluating Exploratory Data Analysis (EDA) skills using Python. Students must execute each cell and pass the associated tests to proceed.","metadata":{}},{"cell_type":"code","source":"\n\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport plotly.offline as pyo\nfrom IPython.display import HTML\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.566779Z","iopub.execute_input":"2025-01-05T15:03:35.567274Z","iopub.status.idle":"2025-01-05T15:03:35.573849Z","shell.execute_reply.started":"2025-01-05T15:03:35.567225Z","shell.execute_reply":"2025-01-05T15:03:35.572284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_images\"\ntrain_thumbnails_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\ntest_thumbnails_path = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.575984Z","iopub.execute_input":"2025-01-05T15:03:35.5764Z","iopub.status.idle":"2025-01-05T15:03:35.592163Z","shell.execute_reply.started":"2025-01-05T15:03:35.576368Z","shell.execute_reply":"2025-01-05T15:03:35.590959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.59488Z","iopub.execute_input":"2025-01-05T15:03:35.59524Z","iopub.status.idle":"2025-01-05T15:03:35.620118Z","shell.execute_reply.started":"2025-01-05T15:03:35.595211Z","shell.execute_reply":"2025-01-05T15:03:35.618576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.62224Z","iopub.execute_input":"2025-01-05T15:03:35.62261Z","iopub.status.idle":"2025-01-05T15:03:35.637412Z","shell.execute_reply.started":"2025-01-05T15:03:35.622581Z","shell.execute_reply":"2025-01-05T15:03:35.636007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Verify dataset size\nprint(\"Size of the dataset: \", df.shape[0])\nassert df.shape[0] > 0, \"Dataset size must be greater than 0\"","metadata":{"nbgrader":{"grade":true,"grade_id":"dataset_size","locked":true,"points":1,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.638595Z","iopub.execute_input":"2025-01-05T15:03:35.639518Z","iopub.status.idle":"2025-01-05T15:03:35.65866Z","shell.execute_reply.started":"2025-01-05T15:03:35.639462Z","shell.execute_reply":"2025-01-05T15:03:35.657509Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visual Exploration\n### Random Image Plotting","metadata":{}},{"cell_type":"code","source":"def plot_images(folder_path: str, resize: bool = False,num_images_to_plot=6):\n    # Get a list of image file names in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.endswith(('.jpg', '.jpeg', '.png', '.gif'))]\n    num_images_to_plot = 6\n    selected_images = random.sample(image_files, num_images_to_plot)\n    fig, axes = plt.subplots(2, 3, figsize=(12, 8))\n    for i, ax in enumerate(axes.flat):\n        if i < num_images_to_plot:\n            image_path = os.path.join(folder_path, selected_images[i])\n            img = Image.open(image_path)\n            if resize:\n                img = img.resize((512, 512))\n            img = np.array(img)\n            ax.imshow(img)\n            ax.set_title(selected_images[i])\n            ax.axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n","metadata":{"nbgrader":{"grade":true,"grade_id":"plot_images","locked":true,"points":2,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.660068Z","iopub.execute_input":"2025-01-05T15:03:35.660451Z","iopub.status.idle":"2025-01-05T15:03:35.682351Z","shell.execute_reply.started":"2025-01-05T15:03:35.66041Z","shell.execute_reply":"2025-01-05T15:03:35.680912Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Label Distribution","metadata":{}},{"cell_type":"code","source":"label_df = pd.DataFrame(df['label'].value_counts())\nlabel_df.reset_index(inplace=True)\nlabel_df.columns=['label','count']\nplt.bar(label_df['label'], label_df['count'], color='skyblue')\nplt.xlabel('Labels')\nplt.ylabel('Count')\nplt.title('Label Distribution')\nplt.show()\nassert len(label_df) > 0, \"Label distribution must be displayed correctly\"","metadata":{"nbgrader":{"grade":true,"grade_id":"label_distribution","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.683637Z","iopub.execute_input":"2025-01-05T15:03:35.68404Z","iopub.status.idle":"2025-01-05T15:03:35.881795Z","shell.execute_reply.started":"2025-01-05T15:03:35.68401Z","shell.execute_reply":"2025-01-05T15:03:35.88039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Class-wise Analysis","metadata":{}},{"cell_type":"code","source":"HGSC = df[df['label']==\"HGSC\"]\nEC = df[df['label']==\"EC\"]\nCC = df[df['label']==\"CC\"]\nLGSC = df[df['label']==\"LGSC\"]\nMC = df[df['label']==\"MC\"]\n\nplt.figure(figsize=(20, 6))\nplt.rcParams['font.size'] = 14\ncolors = ['red', 'lightblue', 'green','magenta', 'yellow']\nplt.pie([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)], \n        labels=['HGSC', 'EC', 'CC', 'LGSC', 'MC'], autopct='%1.1f%%', colors=colors)\nplt.title('Training Set')\nplt.show()\nassert sum([len(HGSC), len(EC), len(CC), len(LGSC), len(MC)]) == len(df), \"Pie chart proportions must match dataset size\"","metadata":{"nbgrader":{"grade":true,"grade_id":"class_analysis","locked":true,"points":4,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:35.883018Z","iopub.execute_input":"2025-01-05T15:03:35.88348Z","iopub.status.idle":"2025-01-05T15:03:36.060522Z","shell.execute_reply.started":"2025-01-05T15:03:35.883441Z","shell.execute_reply":"2025-01-05T15:03:36.058545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Seaborn Visualization\n### Distribution Plot","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nsns.histplot(data=df, x='label', kde=True, color='blue')\nplt.title('Label Distribution with Seaborn')\nplt.show()\nassert not df['label'].isnull().any(), \"Ensure there are no missing values in the label column\"","metadata":{"nbgrader":{"grade":true,"grade_id":"seaborn_distplot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:36.065945Z","iopub.execute_input":"2025-01-05T15:03:36.066664Z","iopub.status.idle":"2025-01-05T15:03:36.357591Z","shell.execute_reply.started":"2025-01-05T15:03:36.066608Z","shell.execute_reply":"2025-01-05T15:03:36.356275Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Correlation Matrix","metadata":{}},{"cell_type":"code","source":"numerical_df=df.select_dtypes(include=['number'])\ncorrelation_matrix = numerical_df.corr()\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()\nassert correlation_matrix.shape[0] > 0, \"Correlation matrix must be generated\"","metadata":{"nbgrader":{"grade":true,"grade_id":"correlation_matrix","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:36.358989Z","iopub.execute_input":"2025-01-05T15:03:36.359358Z","iopub.status.idle":"2025-01-05T15:03:36.909586Z","shell.execute_reply.started":"2025-01-05T15:03:36.359326Z","shell.execute_reply":"2025-01-05T15:03:36.908285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Scatter Plot","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\ndata = {\n    'feature1': [1, 2, 3, 4, 5],\n    'feature2': [5, 4, 3, 2, 1],\n    'label': ['A', 'B', 'A', 'B', 'A']\n}\ndf = pd.DataFrame(data)\n\nassert 'feature1' in df.columns and 'feature2' in df.columns, \"Ensure 'feature1' and 'feature2' columns exist in the dataset\"\nsns.scatterplot(data=df, x='feature1', y='feature2', hue='label', palette='Set2')\nplt.title('Scatter Plot of feature1 vs feature2')\nplt.show()\n","metadata":{"nbgrader":{"grade":true,"grade_id":"scatter_plot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:36.910897Z","iopub.execute_input":"2025-01-05T15:03:36.911339Z","iopub.status.idle":"2025-01-05T15:03:37.274295Z","shell.execute_reply.started":"2025-01-05T15:03:36.911289Z","shell.execute_reply":"2025-01-05T15:03:37.272916Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Pair Plot","metadata":{}},{"cell_type":"code","source":"sns.pairplot(data=df, hue='label', palette='husl')\nplt.suptitle('Pair Plot of Dataset Features', y=1.02)\nplt.show()\nassert 'label' in df.columns, \"Ensure the 'label' column is included for the pair plot\"","metadata":{"nbgrader":{"grade":true,"grade_id":"pair_plot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:37.275768Z","iopub.execute_input":"2025-01-05T15:03:37.276248Z","iopub.status.idle":"2025-01-05T15:03:38.816123Z","shell.execute_reply.started":"2025-01-05T15:03:37.276212Z","shell.execute_reply":"2025-01-05T15:03:38.814847Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Box Plot","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\ndata = {\n    'label': ['A', 'B', 'A', 'B', 'C'],\n    'numerical_feature': [1.5, 2.3, 3.7, 4.1, 2.9]\n}\ndf = pd.DataFrame(data)\nassert 'numerical_feature' in df.columns, \"Ensure 'numerical_feature' exists in the dataset\"\nsns.boxplot(data=df, x='label', y='numerical_feature', palette='cool')\nplt.title('Box Plot of Numerical Feature by Label')\nplt.show()\n","metadata":{"nbgrader":{"grade":true,"grade_id":"box_plot","locked":true,"points":3,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:38.817531Z","iopub.execute_input":"2025-01-05T15:03:38.81803Z","iopub.status.idle":"2025-01-05T15:03:39.086518Z","shell.execute_reply.started":"2025-01-05T15:03:38.817986Z","shell.execute_reply":"2025-01-05T15:03:39.085152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count Plot","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndata = {\n    'categorical_feature': ['A', 'B', 'A', 'C', 'B', 'C', 'A'],\n}\ndf = pd.DataFrame(data)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nassert 'categorical_feature' in df.columns, \"Ensure 'categorical_feature' exists in the dataset\"\nsns.countplot(data=df, x='categorical_feature', palette='pastel')\nplt.title('Count Plot of Categorical Feature')\nplt.show()\n","metadata":{"nbgrader":{"grade":true,"grade_id":"count_plot","locked":true,"points":2,"solution":false},"trusted":true,"execution":{"iopub.status.busy":"2025-01-05T15:03:39.087493Z","iopub.execute_input":"2025-01-05T15:03:39.087791Z","iopub.status.idle":"2025-01-05T15:03:39.285946Z","shell.execute_reply.started":"2025-01-05T15:03:39.087765Z","shell.execute_reply":"2025-01-05T15:03:39.284693Z"}},"outputs":[],"execution_count":null}]}