{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data understanding: berisi eksplorasi data mengenai komposisi dan sebaran data termasuk\nheatmap dan penjelasan keterkaitan antar data","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.neighbors import KNeighborsClassifier\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nclass KNNClassifier:\n    def __init__(self, n_neighbors=5):\n        self.knn_classifier = KNeighborsClassifier(n_neighbors=n_neighbors)\n\n    def train(self, X_train, y_train):\n        self.knn_classifier.fit(X_train, y_train)\n\n    def predict(self, X_test):\n        return self.knn_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:51:28.221209Z","iopub.execute_input":"2023-12-02T01:51:28.221607Z","iopub.status.idle":"2023-12-02T01:51:30.842902Z","shell.execute_reply.started":"2023-12-02T01:51:28.221572Z","shell.execute_reply":"2023-12-02T01:51:30.841792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 2\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:44:01.326646Z","iopub.execute_input":"2023-12-02T02:44:01.32707Z","iopub.status.idle":"2023-12-02T02:44:01.334166Z","shell.execute_reply.started":"2023-12-02T02:44:01.327039Z","shell.execute_reply":"2023-12-02T02:44:01.332751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:44:04.671729Z","iopub.execute_input":"2023-12-02T02:44:04.672113Z","iopub.status.idle":"2023-12-02T02:44:04.963629Z","shell.execute_reply.started":"2023-12-02T02:44:04.672085Z","shell.execute_reply":"2023-12-02T02:44:04.962299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:51:52.616555Z","iopub.execute_input":"2023-12-02T01:51:52.617742Z","iopub.status.idle":"2023-12-02T01:51:52.635413Z","shell.execute_reply.started":"2023-12-02T01:51:52.617707Z","shell.execute_reply":"2023-12-02T01:51:52.633741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:52:05.316318Z","iopub.execute_input":"2023-12-02T01:52:05.316747Z","iopub.status.idle":"2023-12-02T01:52:05.328995Z","shell.execute_reply.started":"2023-12-02T01:52:05.316715Z","shell.execute_reply":"2023-12-02T01:52:05.327835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:52:12.037052Z","iopub.execute_input":"2023-12-02T01:52:12.037512Z","iopub.status.idle":"2023-12-02T01:52:12.044705Z","shell.execute_reply.started":"2023-12-02T01:52:12.037477Z","shell.execute_reply":"2023-12-02T01:52:12.04332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe(exclude = np.number)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:52:16.471148Z","iopub.execute_input":"2023-12-02T01:52:16.471574Z","iopub.status.idle":"2023-12-02T01:52:16.490915Z","shell.execute_reply.started":"2023-12-02T01:52:16.471539Z","shell.execute_reply":"2023-12-02T01:52:16.489563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(df.dtypes), len(df.dtypes),df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:53:04.99656Z","iopub.execute_input":"2023-12-02T01:53:04.99695Z","iopub.status.idle":"2023-12-02T01:53:05.005768Z","shell.execute_reply.started":"2023-12-02T01:53:04.996923Z","shell.execute_reply":"2023-12-02T01:53:05.004538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ndf_hist = df.select_dtypes(exclude=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:56:58.586218Z","iopub.execute_input":"2023-12-02T01:56:58.586691Z","iopub.status.idle":"2023-12-02T01:56:58.593703Z","shell.execute_reply.started":"2023-12-02T01:56:58.586658Z","shell.execute_reply":"2023-12-02T01:56:58.59229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist.hist(figsize=(40,20), alpha = 0.5, edgecolor='black',grid=False);","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:57:14.472221Z","iopub.execute_input":"2023-12-02T01:57:14.472646Z","iopub.status.idle":"2023-12-02T01:57:15.920301Z","shell.execute_reply.started":"2023-12-02T01:57:14.472613Z","shell.execute_reply":"2023-12-02T01:57:15.919204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_id'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:59:04.247477Z","iopub.execute_input":"2023-12-02T01:59:04.24793Z","iopub.status.idle":"2023-12-02T01:59:04.262033Z","shell.execute_reply.started":"2023-12-02T01:59:04.247895Z","shell.execute_reply":"2023-12-02T01:59:04.260713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_id']);","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:59:25.586144Z","iopub.execute_input":"2023-12-02T01:59:25.586589Z","iopub.status.idle":"2023-12-02T01:59:25.890271Z","shell.execute_reply.started":"2023-12-02T01:59:25.586559Z","shell.execute_reply":"2023-12-02T01:59:25.888786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_width'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T01:59:51.926909Z","iopub.execute_input":"2023-12-02T01:59:51.9273Z","iopub.status.idle":"2023-12-02T01:59:51.938244Z","shell.execute_reply.started":"2023-12-02T01:59:51.927269Z","shell.execute_reply":"2023-12-02T01:59:51.937283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_width']);","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:00:08.246103Z","iopub.execute_input":"2023-12-02T02:00:08.246513Z","iopub.status.idle":"2023-12-02T02:00:08.527547Z","shell.execute_reply.started":"2023-12-02T02:00:08.246483Z","shell.execute_reply":"2023-12-02T02:00:08.526398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_height'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:00:32.346399Z","iopub.execute_input":"2023-12-02T02:00:32.346798Z","iopub.status.idle":"2023-12-02T02:00:32.359461Z","shell.execute_reply.started":"2023-12-02T02:00:32.346769Z","shell.execute_reply":"2023-12-02T02:00:32.358242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_height']);","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:00:49.291357Z","iopub.execute_input":"2023-12-02T02:00:49.291766Z","iopub.status.idle":"2023-12-02T02:00:49.576915Z","shell.execute_reply.started":"2023-12-02T02:00:49.291733Z","shell.execute_reply":"2023-12-02T02:00:49.575746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:02:37.706476Z","iopub.execute_input":"2023-12-02T02:02:37.707534Z","iopub.status.idle":"2023-12-02T02:02:37.716337Z","shell.execute_reply.started":"2023-12-02T02:02:37.707485Z","shell.execute_reply":"2023-12-02T02:02:37.71495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_numeric = df.select_dtypes(exclude=['object'])  # Exclude non-numeric columns\nm_corr = df_numeric.corr()\nfig = plt.figure(figsize=(10, 7.5))\nsns.heatmap(m_corr)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:04:26.502214Z","iopub.execute_input":"2023-12-02T02:04:26.502652Z","iopub.status.idle":"2023-12-02T02:04:26.906314Z","shell.execute_reply.started":"2023-12-02T02:04:26.502618Z","shell.execute_reply":"2023-12-02T02:04:26.904856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(40, 20))\nsns.heatmap(m_corr, cmap='seismic',\nlinewidths=0.75,\nlinecolor='black',\ncbar=True,\nvmin=-1,\nvmax=1,\nannot=True,\nannot_kws={'size':8, 'color':'black'})\nplt.tick_params(labelsize=10, rotation = 45)\nplt.title('Correlation Plot', size = 14);","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:05:09.941611Z","iopub.execute_input":"2023-12-02T02:05:09.942011Z","iopub.status.idle":"2023-12-02T02:05:10.799387Z","shell.execute_reply.started":"2023-12-02T02:05:09.941977Z","shell.execute_reply":"2023-12-02T02:05:10.798132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data preparation: berisi penanganan missing value jika ada, data cleaning dan data\ntransformation jika diperlukan","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:09:33.796965Z","iopub.execute_input":"2023-12-02T02:09:33.797417Z","iopub.status.idle":"2023-12-02T02:09:33.808258Z","shell.execute_reply.started":"2023-12-02T02:09:33.797381Z","shell.execute_reply":"2023-12-02T02:09:33.806862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Penanganan Missing Value\n# Menggunakan SimpleImputer dengan strategi 'most_frequent' untuk mengisi nilai yang hilang dengan modus\nimputer = SimpleImputer(strategy='mean')\ndf['image_width'] = imputer.fit_transform(df[['image_width']])\ndf['image_height'] = imputer.fit_transform(df[['image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:12:49.711267Z","iopub.execute_input":"2023-12-02T02:12:49.711722Z","iopub.status.idle":"2023-12-02T02:12:49.728015Z","shell.execute_reply.started":"2023-12-02T02:12:49.711689Z","shell.execute_reply":"2023-12-02T02:12:49.726465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Data Cleaning\n#    Contoh: Mengganti jenis kelamin menjadi biner (0: False, 1: True)\ndf['is_tma'] = df['is_tma'].map({'False': 0, 'True': 1})","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:13:51.390475Z","iopub.execute_input":"2023-12-02T02:13:51.390914Z","iopub.status.idle":"2023-12-02T02:13:51.398965Z","shell.execute_reply.started":"2023-12-02T02:13:51.390879Z","shell.execute_reply":"2023-12-02T02:13:51.397569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Data Transformation\n#    Contoh: Standarisasi (z-score) variabel numerik menggunakan StandardScaler\nscaler = StandardScaler()\ndf[['image_width', 'image_height']] = scaler.fit_transform(df[['image_width', 'image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:14:40.425729Z","iopub.execute_input":"2023-12-02T02:14:40.426156Z","iopub.status.idle":"2023-12-02T02:14:40.438549Z","shell.execute_reply.started":"2023-12-02T02:14:40.426123Z","shell.execute_reply":"2023-12-02T02:14:40.437432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Data setelah Data Preparation:\")\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:14:55.906141Z","iopub.execute_input":"2023-12-02T02:14:55.906947Z","iopub.status.idle":"2023-12-02T02:14:55.917701Z","shell.execute_reply.started":"2023-12-02T02:14:55.906908Z","shell.execute_reply":"2023-12-02T02:14:55.916394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Modeling: berupa pembagian data training dan testing termasuk salah satu evaluation\nexperiment seperti holdout/ cv/ k-fold cv, atau yang lainnya kemudian fitting dengan\nalgoritma beserta penjelasan model yang terbentuk","metadata":{}},{"cell_type":"code","source":"X = df.iloc[:, :-1].values\ny = df.iloc[:, 4].values","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:31:52.911536Z","iopub.execute_input":"2023-12-02T02:31:52.911917Z","iopub.status.idle":"2023-12-02T02:31:52.918495Z","shell.execute_reply.started":"2023-12-02T02:31:52.911889Z","shell.execute_reply":"2023-12-02T02:31:52.917082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df[\"is_tma\"] == False]","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:48:17.816987Z","iopub.execute_input":"2023-12-02T02:48:17.817488Z","iopub.status.idle":"2023-12-02T02:48:17.823659Z","shell.execute_reply.started":"2023-12-02T02:48:17.817451Z","shell.execute_reply":"2023-12-02T02:48:17.822752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['is_tma']","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:48:37.201735Z","iopub.execute_input":"2023-12-02T02:48:37.202104Z","iopub.status.idle":"2023-12-02T02:48:37.208099Z","shell.execute_reply.started":"2023-12-02T02:48:37.202077Z","shell.execute_reply":"2023-12-02T02:48:37.206695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding categorical data\n# Encoding the Independent Variable\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\ntransformer = ColumnTransformer(\n    [('encoder', OneHotEncoder(), [1])], \nremainder='passthrough')\nX = np.array(transformer.fit_transform(X), dtype=np.integer)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:36:37.03716Z","iopub.execute_input":"2023-12-02T02:36:37.037611Z","iopub.status.idle":"2023-12-02T02:36:37.04958Z","shell.execute_reply.started":"2023-12-02T02:36:37.03758Z","shell.execute_reply":"2023-12-02T02:36:37.048416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:37:34.671757Z","iopub.execute_input":"2023-12-02T02:37:34.672137Z","iopub.status.idle":"2023-12-02T02:37:34.679724Z","shell.execute_reply.started":"2023-12-02T02:37:34.672107Z","shell.execute_reply":"2023-12-02T02:37:34.678262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#jangan di run\nfrom sklearn.model_selection import train_test_split, cross_val_score, KFold\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:48:53.19882Z","iopub.execute_input":"2023-12-02T02:48:53.199196Z","iopub.status.idle":"2023-12-02T02:48:53.204793Z","shell.execute_reply.started":"2023-12-02T02:48:53.199168Z","shell.execute_reply":"2023-12-02T02:48:53.203674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assume X and y are your feature matrix and target variable\n# X and y should be already preprocessed and encoded if needed\n\n# Split the data into training and testing sets using holdout method\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:48:55.896912Z","iopub.execute_input":"2023-12-02T02:48:55.897273Z","iopub.status.idle":"2023-12-02T02:48:55.904367Z","shell.execute_reply.started":"2023-12-02T02:48:55.897244Z","shell.execute_reply":"2023-12-02T02:48:55.903072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize kNN classifier\nknn_classifier = KNeighborsClassifier(n_neighbors=3)  # You can adjust the value of 'n_neighbors'","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:49:00.431695Z","iopub.execute_input":"2023-12-02T02:49:00.432128Z","iopub.status.idle":"2023-12-02T02:49:00.437815Z","shell.execute_reply.started":"2023-12-02T02:49:00.43208Z","shell.execute_reply":"2023-12-02T02:49:00.436317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the model on the training data\nknn_classifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:49:03.425231Z","iopub.execute_input":"2023-12-02T02:49:03.425668Z","iopub.status.idle":"2023-12-02T02:49:03.440129Z","shell.execute_reply.started":"2023-12-02T02:49:03.425636Z","shell.execute_reply":"2023-12-02T02:49:03.438869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions on the testing set\ny_pred = knn_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:49:31.185691Z","iopub.execute_input":"2023-12-02T02:49:31.186068Z","iopub.status.idle":"2023-12-02T02:49:31.202226Z","shell.execute_reply.started":"2023-12-02T02:49:31.186039Z","shell.execute_reply":"2023-12-02T02:49:31.201273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model using holdout method\naccuracy_holdout = accuracy_score(y_test, y_pred)\nprint(\"Accuracy (Holdout): {:.2f}%\".format(accuracy_holdout * 100))\n\n# Print classification report and confusion matrix\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:49:58.409476Z","iopub.execute_input":"2023-12-02T02:49:58.409871Z","iopub.status.idle":"2023-12-02T02:49:58.432979Z","shell.execute_reply.started":"2023-12-02T02:49:58.409841Z","shell.execute_reply":"2023-12-02T02:49:58.431868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sampe sini\n# Evaluate the model using k-fold cross-validation\nkf = KFold(n_splits=5, shuffle=True, random_state=42)  # You can adjust the number of splits\ncv_scores = cross_val_score(knn_classifier, X, y, cv=kf)\n\n# Print average cross-validation accuracy\nprint(\"Cross-Validation Accuracy: {:.2f}%\".format(cv_scores.mean() * 100))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T02:50:45.354283Z","iopub.execute_input":"2023-12-02T02:50:45.354746Z","iopub.status.idle":"2023-12-02T02:50:45.412193Z","shell.execute_reply.started":"2023-12-02T02:50:45.354709Z","shell.execute_reply":"2023-12-02T02:50:45.411367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#ini untuk HOLD OUT\n#Splitting dataset into training set and test set\nfrom sklearn.model_selection import train_test_split \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:02:29.973373Z","iopub.execute_input":"2023-12-02T03:02:29.974053Z","iopub.status.idle":"2023-12-02T03:02:29.980226Z","shell.execute_reply.started":"2023-12-02T03:02:29.974021Z","shell.execute_reply":"2023-12-02T03:02:29.979406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting classifier to the Training Set \nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.metrics import confusion_matrix, accuracy_score \nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:02:38.926011Z","iopub.execute_input":"2023-12-02T03:02:38.926402Z","iopub.status.idle":"2023-12-02T03:02:38.931828Z","shell.execute_reply.started":"2023-12-02T03:02:38.926371Z","shell.execute_reply":"2023-12-02T03:02:38.930614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instantiate learning model (k=3) \nclassifier = KNeighborsClassifier(n_neighbors=3)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:02:46.679407Z","iopub.execute_input":"2023-12-02T03:02:46.679797Z","iopub.status.idle":"2023-12-02T03:02:46.684827Z","shell.execute_reply.started":"2023-12-02T03:02:46.679759Z","shell.execute_reply":"2023-12-02T03:02:46.683864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model \nclassifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:02:52.919436Z","iopub.execute_input":"2023-12-02T03:02:52.920218Z","iopub.status.idle":"2023-12-02T03:02:52.933474Z","shell.execute_reply.started":"2023-12-02T03:02:52.92015Z","shell.execute_reply":"2023-12-02T03:02:52.931627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predicting the Test Set result \ny_pred = classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:03:15.33288Z","iopub.execute_input":"2023-12-02T03:03:15.333727Z","iopub.status.idle":"2023-12-02T03:03:15.349019Z","shell.execute_reply.started":"2023-12-02T03:03:15.333682Z","shell.execute_reply":"2023-12-02T03:03:15.348032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:03:22.352314Z","iopub.execute_input":"2023-12-02T03:03:22.352707Z","iopub.status.idle":"2023-12-02T03:03:22.365345Z","shell.execute_reply.started":"2023-12-02T03:03:22.352677Z","shell.execute_reply":"2023-12-02T03:03:22.364011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluation: berisi metrikcs performansi yang sesuai untuk masing-masing komponen\nevaluation eksperiment yang digunakan (jika holdout harus dibandingkan beberapa\nkomposisi training: testing)\n","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score (y_test, y_pred)*100 \nprint('Accuracy of our model is equal ' + str(round(accuracy, 2)) + '%.')","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:03:31.426041Z","iopub.execute_input":"2023-12-02T03:03:31.426646Z","iopub.status.idle":"2023-12-02T03:03:31.435131Z","shell.execute_reply.started":"2023-12-02T03:03:31.426606Z","shell.execute_reply":"2023-12-02T03:03:31.434216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:04:25.349113Z","iopub.execute_input":"2023-12-02T03:04:25.349719Z","iopub.status.idle":"2023-12-02T03:04:25.356812Z","shell.execute_reply.started":"2023-12-02T03:04:25.349676Z","shell.execute_reply":"2023-12-02T03:04:25.355721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:04:33.895149Z","iopub.execute_input":"2023-12-02T03:04:33.895547Z","iopub.status.idle":"2023-12-02T03:04:33.901562Z","shell.execute_reply.started":"2023-12-02T03:04:33.895515Z","shell.execute_reply":"2023-12-02T03:04:33.900702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Precision: %.3f' % precision_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:08:36.53949Z","iopub.execute_input":"2023-12-02T03:08:36.539934Z","iopub.status.idle":"2023-12-02T03:08:36.549358Z","shell.execute_reply.started":"2023-12-02T03:08:36.539898Z","shell.execute_reply":"2023-12-02T03:08:36.548216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Recall: %.3f' % recall_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:08:43.308496Z","iopub.execute_input":"2023-12-02T03:08:43.308894Z","iopub.status.idle":"2023-12-02T03:08:43.318547Z","shell.execute_reply.started":"2023-12-02T03:08:43.308863Z","shell.execute_reply":"2023-12-02T03:08:43.317104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('F1 Score: %.3f' % f1_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-02T03:08:49.346927Z","iopub.execute_input":"2023-12-02T03:08:49.347306Z","iopub.status.idle":"2023-12-02T03:08:49.358901Z","shell.execute_reply.started":"2023-12-02T03:08:49.347278Z","shell.execute_reply":"2023-12-02T03:08:49.357172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}