{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"DATA UNDERSTANDING","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:11.180112Z","iopub.execute_input":"2023-12-17T03:16:11.180571Z","iopub.status.idle":"2023-12-17T03:16:11.625422Z","shell.execute_reply.started":"2023-12-17T03:16:11.18053Z","shell.execute_reply":"2023-12-17T03:16:11.623626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ndf = pd.read_csv(train_csv_path)\n\n# Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:15.116651Z","iopub.execute_input":"2023-12-17T03:16:15.117278Z","iopub.status.idle":"2023-12-17T03:16:15.147009Z","shell.execute_reply.started":"2023-12-17T03:16:15.117241Z","shell.execute_reply":"2023-12-17T03:16:15.146122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:23.517387Z","iopub.execute_input":"2023-12-17T03:16:23.517777Z","iopub.status.idle":"2023-12-17T03:16:23.546647Z","shell.execute_reply.started":"2023-12-17T03:16:23.51774Z","shell.execute_reply":"2023-12-17T03:16:23.545445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:33.958772Z","iopub.execute_input":"2023-12-17T03:16:33.95925Z","iopub.status.idle":"2023-12-17T03:16:33.972395Z","shell.execute_reply.started":"2023-12-17T03:16:33.959206Z","shell.execute_reply":"2023-12-17T03:16:33.971474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:44.019035Z","iopub.execute_input":"2023-12-17T03:16:44.019505Z","iopub.status.idle":"2023-12-17T03:16:44.026015Z","shell.execute_reply.started":"2023-12-17T03:16:44.019461Z","shell.execute_reply":"2023-12-17T03:16:44.025244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:16:54.583506Z","iopub.execute_input":"2023-12-17T03:16:54.583898Z","iopub.status.idle":"2023-12-17T03:16:54.613887Z","shell.execute_reply.started":"2023-12-17T03:16:54.583867Z","shell.execute_reply":"2023-12-17T03:16:54.612362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(df.dtypes), len(df.dtypes),df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:03.341297Z","iopub.execute_input":"2023-12-17T03:17:03.341704Z","iopub.status.idle":"2023-12-17T03:17:03.350182Z","shell.execute_reply.started":"2023-12-17T03:17:03.34167Z","shell.execute_reply":"2023-12-17T03:17:03.348854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Get basic statistics about the dataset\nnum_rows = df.shape[0]\nnum_unique_images = df['image_id'].nunique()\nnum_unique_labels = df['label'].nunique()\nunique_labels = df['label'].unique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{num_unique_images=}\")\nprint(f\"{num_unique_labels=}\")\nprint(f\"{unique_labels=}\")\n\n# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:14.900289Z","iopub.execute_input":"2023-12-17T03:17:14.900682Z","iopub.status.idle":"2023-12-17T03:17:15.74182Z","shell.execute_reply.started":"2023-12-17T03:17:14.900649Z","shell.execute_reply":"2023-12-17T03:17:15.740595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist = df.select_dtypes(exclude=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:24.735581Z","iopub.execute_input":"2023-12-17T03:17:24.736072Z","iopub.status.idle":"2023-12-17T03:17:24.743625Z","shell.execute_reply.started":"2023-12-17T03:17:24.736031Z","shell.execute_reply":"2023-12-17T03:17:24.742094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist.hist(figsize=(40,20), alpha = 0.5, edgecolor='black',grid=False);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:31.652122Z","iopub.execute_input":"2023-12-17T03:17:31.652477Z","iopub.status.idle":"2023-12-17T03:17:32.627997Z","shell.execute_reply.started":"2023-12-17T03:17:31.652448Z","shell.execute_reply":"2023-12-17T03:17:32.627145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_id'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:40.734713Z","iopub.execute_input":"2023-12-17T03:17:40.735134Z","iopub.status.idle":"2023-12-17T03:17:40.747623Z","shell.execute_reply.started":"2023-12-17T03:17:40.735097Z","shell.execute_reply":"2023-12-17T03:17:40.746626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_id']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:49.262823Z","iopub.execute_input":"2023-12-17T03:17:49.263219Z","iopub.status.idle":"2023-12-17T03:17:49.40494Z","shell.execute_reply.started":"2023-12-17T03:17:49.263186Z","shell.execute_reply":"2023-12-17T03:17:49.404047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_width'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:17:58.621478Z","iopub.execute_input":"2023-12-17T03:17:58.621849Z","iopub.status.idle":"2023-12-17T03:17:58.632801Z","shell.execute_reply.started":"2023-12-17T03:17:58.621819Z","shell.execute_reply":"2023-12-17T03:17:58.631378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_width']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:18:06.463651Z","iopub.execute_input":"2023-12-17T03:18:06.465363Z","iopub.status.idle":"2023-12-17T03:18:06.61457Z","shell.execute_reply.started":"2023-12-17T03:18:06.465295Z","shell.execute_reply":"2023-12-17T03:18:06.613643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_height'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:18:14.005269Z","iopub.execute_input":"2023-12-17T03:18:14.005658Z","iopub.status.idle":"2023-12-17T03:18:14.019033Z","shell.execute_reply.started":"2023-12-17T03:18:14.005622Z","shell.execute_reply":"2023-12-17T03:18:14.017311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_height']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:18:21.057116Z","iopub.execute_input":"2023-12-17T03:18:21.057535Z","iopub.status.idle":"2023-12-17T03:18:21.210204Z","shell.execute_reply.started":"2023-12-17T03:18:21.057501Z","shell.execute_reply":"2023-12-17T03:18:21.209079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_numeric = df.select_dtypes(exclude=['object'])  # Exclude non-numeric columns\nm_corr = df_numeric.corr()\nfig = plt.figure(figsize=(10, 7.5))\nsns.heatmap(m_corr)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:18:29.539582Z","iopub.execute_input":"2023-12-17T03:18:29.540035Z","iopub.status.idle":"2023-12-17T03:18:29.865698Z","shell.execute_reply.started":"2023-12-17T03:18:29.540003Z","shell.execute_reply":"2023-12-17T03:18:29.86376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(40, 20))\nsns.heatmap(m_corr, cmap='seismic',\nlinewidths=0.75,\nlinecolor='black',\ncbar=True,\nvmin=-1,\nvmax=1,\nannot=True,\nannot_kws={'size':8, 'color':'black'})\nplt.tick_params(labelsize=10, rotation = 45)\nplt.title('Correlation Plot', size = 14);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:18:37.715054Z","iopub.execute_input":"2023-12-17T03:18:37.71544Z","iopub.status.idle":"2023-12-17T03:18:38.411114Z","shell.execute_reply.started":"2023-12-17T03:18:37.715411Z","shell.execute_reply":"2023-12-17T03:18:38.409971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA PREPARATION","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:08.198468Z","iopub.execute_input":"2023-12-17T03:19:08.198846Z","iopub.status.idle":"2023-12-17T03:19:08.5711Z","shell.execute_reply.started":"2023-12-17T03:19:08.198814Z","shell.execute_reply":"2023-12-17T03:19:08.569934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Penanganan Missing Value\n# Menggunakan SimpleImputer dengan strategi 'most_frequent' untuk mengisi nilai yang hilang dengan modus\nimputer = SimpleImputer(strategy='mean')\ndf['image_width'] = imputer.fit_transform(df[['image_width']])\ndf['image_height'] = imputer.fit_transform(df[['image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:16.81705Z","iopub.execute_input":"2023-12-17T03:19:16.817393Z","iopub.status.idle":"2023-12-17T03:19:16.838008Z","shell.execute_reply.started":"2023-12-17T03:19:16.817368Z","shell.execute_reply":"2023-12-17T03:19:16.83652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Data Cleaning\n#    Contoh: Mengganti jenis kelamin menjadi biner (0: False, 1: True)\ndf['is_tma'] = df['is_tma'].map({'False': 0, 'True': 1})","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:26.637553Z","iopub.execute_input":"2023-12-17T03:19:26.637954Z","iopub.status.idle":"2023-12-17T03:19:26.644725Z","shell.execute_reply.started":"2023-12-17T03:19:26.637925Z","shell.execute_reply":"2023-12-17T03:19:26.643462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Data Transformation\n#    Contoh: Standarisasi (z-score) variabel numerik menggunakan StandardScaler\nscaler = StandardScaler()\ndf[['image_width', 'image_height']] = scaler.fit_transform(df[['image_width', 'image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:34.375063Z","iopub.execute_input":"2023-12-17T03:19:34.375447Z","iopub.status.idle":"2023-12-17T03:19:34.386594Z","shell.execute_reply.started":"2023-12-17T03:19:34.375416Z","shell.execute_reply":"2023-12-17T03:19:34.384914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Data setelah Data Preparation:\")\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:41.910494Z","iopub.execute_input":"2023-12-17T03:19:41.910936Z","iopub.status.idle":"2023-12-17T03:19:41.923066Z","shell.execute_reply.started":"2023-12-17T03:19:41.910867Z","shell.execute_reply":"2023-12-17T03:19:41.921225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELLING","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:19:58.390999Z","iopub.execute_input":"2023-12-17T03:19:58.391414Z","iopub.status.idle":"2023-12-17T03:19:58.400461Z","shell.execute_reply.started":"2023-12-17T03:19:58.391381Z","shell.execute_reply":"2023-12-17T03:19:58.398575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.iloc[:, :-1].values\ny = df.iloc[:, 4].values","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:06.302504Z","iopub.execute_input":"2023-12-17T03:20:06.30296Z","iopub.status.idle":"2023-12-17T03:20:06.310734Z","shell.execute_reply.started":"2023-12-17T03:20:06.302923Z","shell.execute_reply":"2023-12-17T03:20:06.3088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['is_tma']","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:13.828008Z","iopub.execute_input":"2023-12-17T03:20:13.828531Z","iopub.status.idle":"2023-12-17T03:20:13.836748Z","shell.execute_reply.started":"2023-12-17T03:20:13.82849Z","shell.execute_reply":"2023-12-17T03:20:13.833707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding categorical data\n# Encoding the Independent Variable\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\ntransformer = ColumnTransformer(\n    [('encoder', OneHotEncoder(), [1])], \nremainder='passthrough')\nX = np.array(transformer.fit_transform(X), dtype=np.integer)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:21.356678Z","iopub.execute_input":"2023-12-17T03:20:21.357148Z","iopub.status.idle":"2023-12-17T03:20:21.384998Z","shell.execute_reply.started":"2023-12-17T03:20:21.357111Z","shell.execute_reply":"2023-12-17T03:20:21.383025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:23.698647Z","iopub.execute_input":"2023-12-17T03:20:23.699523Z","iopub.status.idle":"2023-12-17T03:20:23.705396Z","shell.execute_reply.started":"2023-12-17T03:20:23.699484Z","shell.execute_reply":"2023-12-17T03:20:23.704576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ini untuk HOLD OUT\n#Splitting dataset into training set and test set\nfrom sklearn.model_selection import train_test_split \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:35.282177Z","iopub.execute_input":"2023-12-17T03:20:35.282597Z","iopub.status.idle":"2023-12-17T03:20:35.288604Z","shell.execute_reply.started":"2023-12-17T03:20:35.282563Z","shell.execute_reply":"2023-12-17T03:20:35.287978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting classifier to the Training Set \nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.metrics import confusion_matrix, accuracy_score \nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:43.312849Z","iopub.execute_input":"2023-12-17T03:20:43.31332Z","iopub.status.idle":"2023-12-17T03:20:43.31974Z","shell.execute_reply.started":"2023-12-17T03:20:43.31328Z","shell.execute_reply":"2023-12-17T03:20:43.317633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instantiate learning model (k=3) \nclassifier = KNeighborsClassifier(n_neighbors=3)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:50.857326Z","iopub.execute_input":"2023-12-17T03:20:50.857694Z","iopub.status.idle":"2023-12-17T03:20:50.863295Z","shell.execute_reply.started":"2023-12-17T03:20:50.857651Z","shell.execute_reply":"2023-12-17T03:20:50.861945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model \nclassifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:20:58.822431Z","iopub.execute_input":"2023-12-17T03:20:58.822846Z","iopub.status.idle":"2023-12-17T03:20:58.835261Z","shell.execute_reply.started":"2023-12-17T03:20:58.822813Z","shell.execute_reply":"2023-12-17T03:20:58.834339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predicting the Test Set result \ny_pred = classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:21:06.27657Z","iopub.execute_input":"2023-12-17T03:21:06.27745Z","iopub.status.idle":"2023-12-17T03:21:06.302101Z","shell.execute_reply.started":"2023-12-17T03:21:06.277385Z","shell.execute_reply":"2023-12-17T03:21:06.300576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:21:12.552125Z","iopub.execute_input":"2023-12-17T03:21:12.552505Z","iopub.status.idle":"2023-12-17T03:21:12.566014Z","shell.execute_reply.started":"2023-12-17T03:21:12.552472Z","shell.execute_reply":"2023-12-17T03:21:12.564199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EVALUATION","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score (y_test, y_pred)*100 \nprint('Accuracy of our model is equal ' + str(round(accuracy, 2)) + '%.')","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:21:50.379852Z","iopub.execute_input":"2023-12-17T03:21:50.380262Z","iopub.status.idle":"2023-12-17T03:21:50.390023Z","shell.execute_reply.started":"2023-12-17T03:21:50.380232Z","shell.execute_reply":"2023-12-17T03:21:50.388373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, accuracy_score, recall_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:21:58.271994Z","iopub.execute_input":"2023-12-17T03:21:58.272407Z","iopub.status.idle":"2023-12-17T03:21:58.278381Z","shell.execute_reply.started":"2023-12-17T03:21:58.272376Z","shell.execute_reply":"2023-12-17T03:21:58.276888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Precision: %.3f' % precision_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:22:15.55346Z","iopub.execute_input":"2023-12-17T03:22:15.553846Z","iopub.status.idle":"2023-12-17T03:22:15.562011Z","shell.execute_reply.started":"2023-12-17T03:22:15.553815Z","shell.execute_reply":"2023-12-17T03:22:15.561263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Recall: %.3f' % recall_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:22:25.713953Z","iopub.execute_input":"2023-12-17T03:22:25.714373Z","iopub.status.idle":"2023-12-17T03:22:25.725698Z","shell.execute_reply.started":"2023-12-17T03:22:25.714341Z","shell.execute_reply":"2023-12-17T03:22:25.724631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nprint('F1 Score: %.3f' % f1_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T03:22:37.685602Z","iopub.execute_input":"2023-12-17T03:22:37.685954Z","iopub.status.idle":"2023-12-17T03:22:37.698433Z","shell.execute_reply.started":"2023-12-17T03:22:37.68593Z","shell.execute_reply":"2023-12-17T03:22:37.697016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}