{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"DATA UNDERSTANDING","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:01:02.267118Z","iopub.execute_input":"2023-12-17T02:01:02.267553Z","iopub.status.idle":"2023-12-17T02:01:02.713626Z","shell.execute_reply.started":"2023-12-17T02:01:02.267516Z","shell.execute_reply":"2023-12-17T02:01:02.712418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ndf = pd.read_csv(train_csv_path)\n\n# Inference\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:04:55.392058Z","iopub.execute_input":"2023-12-17T02:04:55.392503Z","iopub.status.idle":"2023-12-17T02:04:55.415234Z","shell.execute_reply.started":"2023-12-17T02:04:55.392468Z","shell.execute_reply":"2023-12-17T02:04:55.414301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:05:28.009538Z","iopub.execute_input":"2023-12-17T02:05:28.009989Z","iopub.status.idle":"2023-12-17T02:05:28.033647Z","shell.execute_reply.started":"2023-12-17T02:05:28.009941Z","shell.execute_reply":"2023-12-17T02:05:28.032825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:05:37.829295Z","iopub.execute_input":"2023-12-17T02:05:37.829672Z","iopub.status.idle":"2023-12-17T02:05:37.841057Z","shell.execute_reply.started":"2023-12-17T02:05:37.829643Z","shell.execute_reply":"2023-12-17T02:05:37.840053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:05:51.66424Z","iopub.execute_input":"2023-12-17T02:05:51.664624Z","iopub.status.idle":"2023-12-17T02:05:51.672047Z","shell.execute_reply.started":"2023-12-17T02:05:51.664596Z","shell.execute_reply":"2023-12-17T02:05:51.670778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:06:18.417047Z","iopub.execute_input":"2023-12-17T02:06:18.41744Z","iopub.status.idle":"2023-12-17T02:06:18.441088Z","shell.execute_reply.started":"2023-12-17T02:06:18.417399Z","shell.execute_reply":"2023-12-17T02:06:18.439882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(df.dtypes), len(df.dtypes),df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:06:38.622372Z","iopub.execute_input":"2023-12-17T02:06:38.622746Z","iopub.status.idle":"2023-12-17T02:06:38.631965Z","shell.execute_reply.started":"2023-12-17T02:06:38.622715Z","shell.execute_reply":"2023-12-17T02:06:38.630811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Get basic statistics about the dataset\nnum_rows = df.shape[0]\nnum_unique_images = df['image_id'].nunique()\nnum_unique_labels = df['label'].nunique()\nunique_labels = df['label'].unique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{num_unique_images=}\")\nprint(f\"{num_unique_labels=}\")\nprint(f\"{unique_labels=}\")\n\n# Plot the distribution of the target classes\nplt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:08:11.244078Z","iopub.execute_input":"2023-12-17T02:08:11.244483Z","iopub.status.idle":"2023-12-17T02:08:12.424585Z","shell.execute_reply.started":"2023-12-17T02:08:11.244452Z","shell.execute_reply":"2023-12-17T02:08:12.423494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist = df.select_dtypes(exclude=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:08:25.064535Z","iopub.execute_input":"2023-12-17T02:08:25.064924Z","iopub.status.idle":"2023-12-17T02:08:25.070714Z","shell.execute_reply.started":"2023-12-17T02:08:25.064893Z","shell.execute_reply":"2023-12-17T02:08:25.069499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist.hist(figsize=(40,20), alpha = 0.5, edgecolor='black',grid=False);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:08:35.959543Z","iopub.execute_input":"2023-12-17T02:08:35.959922Z","iopub.status.idle":"2023-12-17T02:08:37.154895Z","shell.execute_reply.started":"2023-12-17T02:08:35.959892Z","shell.execute_reply":"2023-12-17T02:08:37.153707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_id'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:08:45.853937Z","iopub.execute_input":"2023-12-17T02:08:45.854387Z","iopub.status.idle":"2023-12-17T02:08:45.865674Z","shell.execute_reply.started":"2023-12-17T02:08:45.85435Z","shell.execute_reply":"2023-12-17T02:08:45.864693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_id']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:08:53.663717Z","iopub.execute_input":"2023-12-17T02:08:53.664132Z","iopub.status.idle":"2023-12-17T02:08:53.828223Z","shell.execute_reply.started":"2023-12-17T02:08:53.664099Z","shell.execute_reply":"2023-12-17T02:08:53.827121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_width'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:01.091987Z","iopub.execute_input":"2023-12-17T02:09:01.092605Z","iopub.status.idle":"2023-12-17T02:09:01.104565Z","shell.execute_reply.started":"2023-12-17T02:09:01.092559Z","shell.execute_reply":"2023-12-17T02:09:01.103363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_width']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:10.653256Z","iopub.execute_input":"2023-12-17T02:09:10.653642Z","iopub.status.idle":"2023-12-17T02:09:10.818535Z","shell.execute_reply.started":"2023-12-17T02:09:10.653611Z","shell.execute_reply":"2023-12-17T02:09:10.817394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_height'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:17.331938Z","iopub.execute_input":"2023-12-17T02:09:17.332348Z","iopub.status.idle":"2023-12-17T02:09:17.346026Z","shell.execute_reply.started":"2023-12-17T02:09:17.332313Z","shell.execute_reply":"2023-12-17T02:09:17.344991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_height']);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:24.773792Z","iopub.execute_input":"2023-12-17T02:09:24.774158Z","iopub.status.idle":"2023-12-17T02:09:24.968098Z","shell.execute_reply.started":"2023-12-17T02:09:24.77413Z","shell.execute_reply":"2023-12-17T02:09:24.966567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_numeric = df.select_dtypes(exclude=['object'])  # Exclude non-numeric columns\nm_corr = df_numeric.corr()\nfig = plt.figure(figsize=(10, 7.5))\nsns.heatmap(m_corr)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:31.912931Z","iopub.execute_input":"2023-12-17T02:09:31.914245Z","iopub.status.idle":"2023-12-17T02:09:32.292443Z","shell.execute_reply.started":"2023-12-17T02:09:31.914184Z","shell.execute_reply":"2023-12-17T02:09:32.291313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(40, 20))\nsns.heatmap(m_corr, cmap='seismic',\nlinewidths=0.75,\nlinecolor='black',\ncbar=True,\nvmin=-1,\nvmax=1,\nannot=True,\nannot_kws={'size':8, 'color':'black'})\nplt.tick_params(labelsize=10, rotation = 45)\nplt.title('Correlation Plot', size = 14);","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:09:41.193146Z","iopub.execute_input":"2023-12-17T02:09:41.193628Z","iopub.status.idle":"2023-12-17T02:09:41.983588Z","shell.execute_reply.started":"2023-12-17T02:09:41.193593Z","shell.execute_reply":"2023-12-17T02:09:41.982263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA PREPARATION","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:10:04.611874Z","iopub.execute_input":"2023-12-17T02:10:04.612311Z","iopub.status.idle":"2023-12-17T02:10:04.997946Z","shell.execute_reply.started":"2023-12-17T02:10:04.612273Z","shell.execute_reply":"2023-12-17T02:10:04.99684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Penanganan Missing Value\n# Menggunakan SimpleImputer dengan strategi 'most_frequent' untuk mengisi nilai yang hilang dengan modus\nimputer = SimpleImputer(strategy='mean')\ndf['image_width'] = imputer.fit_transform(df[['image_width']])\ndf['image_height'] = imputer.fit_transform(df[['image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:10:12.182795Z","iopub.execute_input":"2023-12-17T02:10:12.183193Z","iopub.status.idle":"2023-12-17T02:10:12.203377Z","shell.execute_reply.started":"2023-12-17T02:10:12.183146Z","shell.execute_reply":"2023-12-17T02:10:12.202204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Data Cleaning\n#    Contoh: Mengganti jenis kelamin menjadi biner (0: False, 1: True)\ndf['is_tma'] = df['is_tma'].map({'False': 0, 'True': 1})","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:10:24.754279Z","iopub.execute_input":"2023-12-17T02:10:24.754785Z","iopub.status.idle":"2023-12-17T02:10:24.762529Z","shell.execute_reply.started":"2023-12-17T02:10:24.754744Z","shell.execute_reply":"2023-12-17T02:10:24.761124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Data Transformation\n#    Contoh: Standarisasi (z-score) variabel numerik menggunakan StandardScaler\nscaler = StandardScaler()\ndf[['image_width', 'image_height']] = scaler.fit_transform(df[['image_width', 'image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:10:34.069504Z","iopub.execute_input":"2023-12-17T02:10:34.069889Z","iopub.status.idle":"2023-12-17T02:10:34.082411Z","shell.execute_reply.started":"2023-12-17T02:10:34.06986Z","shell.execute_reply":"2023-12-17T02:10:34.081252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Data setelah Data Preparation:\")\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:10:42.027715Z","iopub.execute_input":"2023-12-17T02:10:42.028383Z","iopub.status.idle":"2023-12-17T02:10:42.040483Z","shell.execute_reply.started":"2023-12-17T02:10:42.028351Z","shell.execute_reply":"2023-12-17T02:10:42.03921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELLING","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_csv_path)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:11:27.134225Z","iopub.execute_input":"2023-12-17T02:11:27.134656Z","iopub.status.idle":"2023-12-17T02:11:27.146262Z","shell.execute_reply.started":"2023-12-17T02:11:27.13462Z","shell.execute_reply":"2023-12-17T02:11:27.144793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.iloc[:, :-1].values\ny = df.iloc[:, 4].values","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:11:37.592055Z","iopub.execute_input":"2023-12-17T02:11:37.592464Z","iopub.status.idle":"2023-12-17T02:11:37.598378Z","shell.execute_reply.started":"2023-12-17T02:11:37.592432Z","shell.execute_reply":"2023-12-17T02:11:37.597404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['is_tma']","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:11:46.26185Z","iopub.execute_input":"2023-12-17T02:11:46.262309Z","iopub.status.idle":"2023-12-17T02:11:46.267782Z","shell.execute_reply.started":"2023-12-17T02:11:46.262272Z","shell.execute_reply":"2023-12-17T02:11:46.266305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding categorical data\n# Encoding the Independent Variable\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\ntransformer = ColumnTransformer(\n    [('encoder', OneHotEncoder(), [1])], \nremainder='passthrough')\nX = np.array(transformer.fit_transform(X), dtype=np.integer)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:11:54.492041Z","iopub.execute_input":"2023-12-17T02:11:54.493117Z","iopub.status.idle":"2023-12-17T02:11:54.513325Z","shell.execute_reply.started":"2023-12-17T02:11:54.493069Z","shell.execute_reply":"2023-12-17T02:11:54.512225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:02.631965Z","iopub.execute_input":"2023-12-17T02:12:02.632378Z","iopub.status.idle":"2023-12-17T02:12:02.640517Z","shell.execute_reply.started":"2023-12-17T02:12:02.632344Z","shell.execute_reply":"2023-12-17T02:12:02.639255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ini untuk HOLD OUT\n#Splitting dataset into training set and test set\nfrom sklearn.model_selection import train_test_split \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.1, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:05.80183Z","iopub.execute_input":"2023-12-17T02:12:05.802303Z","iopub.status.idle":"2023-12-17T02:12:05.811177Z","shell.execute_reply.started":"2023-12-17T02:12:05.802266Z","shell.execute_reply":"2023-12-17T02:12:05.810031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting classifier to the Training Set \nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.metrics import confusion_matrix, accuracy_score \nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:14.710126Z","iopub.execute_input":"2023-12-17T02:12:14.710564Z","iopub.status.idle":"2023-12-17T02:12:14.715571Z","shell.execute_reply.started":"2023-12-17T02:12:14.710527Z","shell.execute_reply":"2023-12-17T02:12:14.714555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instantiate learning model (k=3) \nclassifier = KNeighborsClassifier(n_neighbors=3)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:23.828269Z","iopub.execute_input":"2023-12-17T02:12:23.828684Z","iopub.status.idle":"2023-12-17T02:12:23.833404Z","shell.execute_reply.started":"2023-12-17T02:12:23.828652Z","shell.execute_reply":"2023-12-17T02:12:23.832538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model \nclassifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:31.142033Z","iopub.execute_input":"2023-12-17T02:12:31.142649Z","iopub.status.idle":"2023-12-17T02:12:31.154722Z","shell.execute_reply.started":"2023-12-17T02:12:31.142613Z","shell.execute_reply":"2023-12-17T02:12:31.15362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predicting the Test Set result \ny_pred = classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:38.81206Z","iopub.execute_input":"2023-12-17T02:12:38.812508Z","iopub.status.idle":"2023-12-17T02:12:38.826172Z","shell.execute_reply.started":"2023-12-17T02:12:38.812472Z","shell.execute_reply":"2023-12-17T02:12:38.825352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:12:45.961989Z","iopub.execute_input":"2023-12-17T02:12:45.962409Z","iopub.status.idle":"2023-12-17T02:12:45.973451Z","shell.execute_reply.started":"2023-12-17T02:12:45.962373Z","shell.execute_reply":"2023-12-17T02:12:45.97224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EVALUATION","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score (y_test, y_pred)*100 \nprint('Accuracy of our model is equal ' + str(round(accuracy, 2)) + '%.')","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:14:13.135863Z","iopub.execute_input":"2023-12-17T02:14:13.136274Z","iopub.status.idle":"2023-12-17T02:14:13.14458Z","shell.execute_reply.started":"2023-12-17T02:14:13.136244Z","shell.execute_reply":"2023-12-17T02:14:13.143301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, accuracy_score, recall_score","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:14:21.536607Z","iopub.execute_input":"2023-12-17T02:14:21.537009Z","iopub.status.idle":"2023-12-17T02:14:21.541644Z","shell.execute_reply.started":"2023-12-17T02:14:21.536979Z","shell.execute_reply":"2023-12-17T02:14:21.540521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Precision: %.3f' % precision_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:14:28.903839Z","iopub.execute_input":"2023-12-17T02:14:28.904389Z","iopub.status.idle":"2023-12-17T02:14:28.918394Z","shell.execute_reply.started":"2023-12-17T02:14:28.904328Z","shell.execute_reply":"2023-12-17T02:14:28.917251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Recall: %.3f' % recall_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:14:36.239737Z","iopub.execute_input":"2023-12-17T02:14:36.24049Z","iopub.status.idle":"2023-12-17T02:14:36.249927Z","shell.execute_reply.started":"2023-12-17T02:14:36.240453Z","shell.execute_reply":"2023-12-17T02:14:36.248867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nprint('F1 Score: %.3f' % f1_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-17T02:14:42.594761Z","iopub.execute_input":"2023-12-17T02:14:42.595148Z","iopub.status.idle":"2023-12-17T02:14:42.606686Z","shell.execute_reply.started":"2023-12-17T02:14:42.595118Z","shell.execute_reply":"2023-12-17T02:14:42.60552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}