{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-04T01:59:11.728229Z","iopub.execute_input":"2023-12-04T01:59:11.72874Z","iopub.status.idle":"2023-12-04T01:59:12.013477Z","shell.execute_reply.started":"2023-12-04T01:59:11.728705Z","shell.execute_reply":"2023-12-04T01:59:12.011671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA UNDERSTANDING","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.neighbors import KNeighborsClassifier\nimport matplotlib.pyplot as plt\n\n# Set the style for the plot\nsns.set(style=\"whitegrid\")\n\nclass KNNClassifier:\n    def __init__(self, n_neighbors=5):\n        self.knn_classifier = KNeighborsClassifier(n_neighbors=n_neighbors)\n\n    def train(self, X_train, y_train):\n        self.knn_classifier.fit(X_train, y_train)\n\n    def predict(self, X_test):\n        return self.knn_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:00:02.670708Z","iopub.execute_input":"2023-12-04T02:00:02.671177Z","iopub.status.idle":"2023-12-04T02:00:03.234329Z","shell.execute_reply.started":"2023-12-04T02:00:02.671142Z","shell.execute_reply":"2023-12-04T02:00:03.232418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    is_submission = False\n    \n    # Reproducibility\n    SEED = 42\n    \n    # Training\n    train_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\n    train_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    batch_size = 8\n    learning_rate = 1e-3\n    epochs = 2\n    \n    # Inference\n    test_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\n    test_thumbnail_paths = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:00:16.362Z","iopub.execute_input":"2023-12-04T02:00:16.36256Z","iopub.status.idle":"2023-12-04T02:00:16.370091Z","shell.execute_reply.started":"2023-12-04T02:00:16.362516Z","shell.execute_reply":"2023-12-04T02:00:16.368604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:00:29.221547Z","iopub.execute_input":"2023-12-04T02:00:29.22204Z","iopub.status.idle":"2023-12-04T02:00:29.552772Z","shell.execute_reply.started":"2023-12-04T02:00:29.221999Z","shell.execute_reply":"2023-12-04T02:00:29.551501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:00:45.523171Z","iopub.execute_input":"2023-12-04T02:00:45.523867Z","iopub.status.idle":"2023-12-04T02:00:45.544041Z","shell.execute_reply.started":"2023-12-04T02:00:45.523821Z","shell.execute_reply":"2023-12-04T02:00:45.542385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:00:55.015877Z","iopub.execute_input":"2023-12-04T02:00:55.016387Z","iopub.status.idle":"2023-12-04T02:00:55.031967Z","shell.execute_reply.started":"2023-12-04T02:00:55.016355Z","shell.execute_reply":"2023-12-04T02:00:55.030828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:01:06.793517Z","iopub.execute_input":"2023-12-04T02:01:06.794023Z","iopub.status.idle":"2023-12-04T02:01:06.804425Z","shell.execute_reply.started":"2023-12-04T02:01:06.793987Z","shell.execute_reply":"2023-12-04T02:01:06.80157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe(exclude = np.number)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:01:18.020617Z","iopub.execute_input":"2023-12-04T02:01:18.021446Z","iopub.status.idle":"2023-12-04T02:01:18.042274Z","shell.execute_reply.started":"2023-12-04T02:01:18.021406Z","shell.execute_reply":"2023-12-04T02:01:18.040277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(df.dtypes), len(df.dtypes),df.dtypes","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:01:27.803628Z","iopub.execute_input":"2023-12-04T02:01:27.804Z","iopub.status.idle":"2023-12-04T02:01:27.813818Z","shell.execute_reply.started":"2023-12-04T02:01:27.803968Z","shell.execute_reply":"2023-12-04T02:01:27.812616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#visualization\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ndf_hist = df.select_dtypes(exclude=['object'])","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:01:39.621762Z","iopub.execute_input":"2023-12-04T02:01:39.622275Z","iopub.status.idle":"2023-12-04T02:01:39.630498Z","shell.execute_reply.started":"2023-12-04T02:01:39.622236Z","shell.execute_reply":"2023-12-04T02:01:39.628914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_hist.hist(figsize=(40,20), alpha = 0.5, edgecolor='black',grid=False);","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:01:50.056318Z","iopub.execute_input":"2023-12-04T02:01:50.056801Z","iopub.status.idle":"2023-12-04T02:01:51.257171Z","shell.execute_reply.started":"2023-12-04T02:01:50.056768Z","shell.execute_reply":"2023-12-04T02:01:51.25518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_id'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:04.151094Z","iopub.execute_input":"2023-12-04T02:02:04.151472Z","iopub.status.idle":"2023-12-04T02:02:04.168988Z","shell.execute_reply.started":"2023-12-04T02:02:04.151442Z","shell.execute_reply":"2023-12-04T02:02:04.167828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_id']);","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:15.548209Z","iopub.execute_input":"2023-12-04T02:02:15.548557Z","iopub.status.idle":"2023-12-04T02:02:15.803146Z","shell.execute_reply.started":"2023-12-04T02:02:15.54853Z","shell.execute_reply":"2023-12-04T02:02:15.801666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_width'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:25.776409Z","iopub.execute_input":"2023-12-04T02:02:25.776961Z","iopub.status.idle":"2023-12-04T02:02:25.792414Z","shell.execute_reply.started":"2023-12-04T02:02:25.776926Z","shell.execute_reply":"2023-12-04T02:02:25.791138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_width']);","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:34.153546Z","iopub.execute_input":"2023-12-04T02:02:34.153933Z","iopub.status.idle":"2023-12-04T02:02:34.376721Z","shell.execute_reply.started":"2023-12-04T02:02:34.153903Z","shell.execute_reply":"2023-12-04T02:02:34.374829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['image_height'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:43.570857Z","iopub.execute_input":"2023-12-04T02:02:43.571298Z","iopub.status.idle":"2023-12-04T02:02:43.583873Z","shell.execute_reply.started":"2023-12-04T02:02:43.571269Z","shell.execute_reply":"2023-12-04T02:02:43.582261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,2))\nsns.boxplot(x=df['image_height']);","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:02:51.375288Z","iopub.execute_input":"2023-12-04T02:02:51.375796Z","iopub.status.idle":"2023-12-04T02:02:51.649216Z","shell.execute_reply.started":"2023-12-04T02:02:51.375758Z","shell.execute_reply":"2023-12-04T02:02:51.646867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_numeric = df.select_dtypes(exclude=['object'])  # Exclude non-numeric columns\nm_corr = df_numeric.corr()\nfig = plt.figure(figsize=(10, 7.5))\nsns.heatmap(m_corr)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:03:01.804742Z","iopub.execute_input":"2023-12-04T02:03:01.805178Z","iopub.status.idle":"2023-12-04T02:03:02.235549Z","shell.execute_reply.started":"2023-12-04T02:03:01.805147Z","shell.execute_reply":"2023-12-04T02:03:02.234654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(40, 20))\nsns.heatmap(m_corr, cmap='seismic',\nlinewidths=0.75,\nlinecolor='black',\ncbar=True,\nvmin=-1,\nvmax=1,\nannot=True,\nannot_kws={'size':8, 'color':'black'})\nplt.tick_params(labelsize=10, rotation = 45)\nplt.title('Correlation Plot', size = 14);","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:03:13.540355Z","iopub.execute_input":"2023-12-04T02:03:13.541025Z","iopub.status.idle":"2023-12-04T02:03:14.258708Z","shell.execute_reply.started":"2023-12-04T02:03:13.540991Z","shell.execute_reply":"2023-12-04T02:03:14.257484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA PREPARATION","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.impute import SimpleImputer","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:04:42.401121Z","iopub.execute_input":"2023-12-04T02:04:42.40154Z","iopub.status.idle":"2023-12-04T02:04:42.416767Z","shell.execute_reply.started":"2023-12-04T02:04:42.401507Z","shell.execute_reply":"2023-12-04T02:04:42.41487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. Penanganan Missing Value\n# Menggunakan SimpleImputer dengan strategi 'most_frequent' untuk mengisi nilai yang hilang dengan modus\nimputer = SimpleImputer(strategy='mean')\ndf['image_width'] = imputer.fit_transform(df[['image_width']])\ndf['image_height'] = imputer.fit_transform(df[['image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:04:50.993071Z","iopub.execute_input":"2023-12-04T02:04:50.993524Z","iopub.status.idle":"2023-12-04T02:04:51.019412Z","shell.execute_reply.started":"2023-12-04T02:04:50.993493Z","shell.execute_reply":"2023-12-04T02:04:51.017159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2. Data Cleaning\n#    Contoh: Mengganti jenis kelamin menjadi biner (0: False, 1: True)\ndf['is_tma'] = df['is_tma'].map({'False': 0, 'True': 1})","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:05:05.611118Z","iopub.execute_input":"2023-12-04T02:05:05.611491Z","iopub.status.idle":"2023-12-04T02:05:05.622869Z","shell.execute_reply.started":"2023-12-04T02:05:05.611466Z","shell.execute_reply":"2023-12-04T02:05:05.621232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3. Data Transformation\n#    Contoh: Standarisasi (z-score) variabel numerik menggunakan StandardScaler\nscaler = StandardScaler()\ndf[['image_width', 'image_height']] = scaler.fit_transform(df[['image_width', 'image_height']])","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:05:14.267785Z","iopub.execute_input":"2023-12-04T02:05:14.268434Z","iopub.status.idle":"2023-12-04T02:05:14.283236Z","shell.execute_reply.started":"2023-12-04T02:05:14.268399Z","shell.execute_reply":"2023-12-04T02:05:14.281927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Data setelah Data Preparation:\")\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:05:23.420045Z","iopub.execute_input":"2023-12-04T02:05:23.42056Z","iopub.status.idle":"2023-12-04T02:05:23.435066Z","shell.execute_reply.started":"2023-12-04T02:05:23.420526Z","shell.execute_reply":"2023-12-04T02:05:23.433937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODELLING","metadata":{}},{"cell_type":"code","source":"X = df.iloc[:, :-1].values\ny = df.iloc[:, 4].values","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:15.549474Z","iopub.execute_input":"2023-12-04T02:08:15.54983Z","iopub.status.idle":"2023-12-04T02:08:15.556364Z","shell.execute_reply.started":"2023-12-04T02:08:15.549803Z","shell.execute_reply":"2023-12-04T02:08:15.555237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[df[\"is_tma\"] == False]","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:05:53.474235Z","iopub.execute_input":"2023-12-04T02:05:53.474631Z","iopub.status.idle":"2023-12-04T02:05:53.482309Z","shell.execute_reply.started":"2023-12-04T02:05:53.474601Z","shell.execute_reply":"2023-12-04T02:05:53.480241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df['is_tma']","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:21.37068Z","iopub.execute_input":"2023-12-04T02:08:21.371033Z","iopub.status.idle":"2023-12-04T02:08:21.377168Z","shell.execute_reply.started":"2023-12-04T02:08:21.371006Z","shell.execute_reply":"2023-12-04T02:08:21.375644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not config.is_submission:\n    df = pd.read_csv(config.train_csv_path)\n\n    # Create the thumbnail df where is_tma == False\n    df = df[df[\"is_tma\"] == False]\n    \n    # Get basic statistics about the dataset\n    num_rows = df.shape[0]\n    num_unique_images = df['image_id'].nunique()\n    num_unique_labels = df['label'].nunique()\n    unique_labels = df['label'].unique()\n\n    print(f\"{num_rows=}\")\n    print(f\"{num_unique_images=}\")\n    print(f\"{num_unique_labels=}\")\n    print(f\"{unique_labels=}\")\n    \n    # Plot the distribution of the target classes\n    plt.figure(figsize=(10, 6))\n    sns.countplot(data=df, x='label', order=df['label'].value_counts().index)\n    plt.title('Distribution of Target Classes')\n    plt.xlabel('Label')\n    plt.ylabel('Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:09.09729Z","iopub.execute_input":"2023-12-04T02:08:09.097637Z","iopub.status.idle":"2023-12-04T02:08:09.341935Z","shell.execute_reply.started":"2023-12-04T02:08:09.09761Z","shell.execute_reply":"2023-12-04T02:08:09.340206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:26.974619Z","iopub.execute_input":"2023-12-04T02:08:26.975028Z","iopub.status.idle":"2023-12-04T02:08:26.983861Z","shell.execute_reply.started":"2023-12-04T02:08:26.974997Z","shell.execute_reply":"2023-12-04T02:08:26.983047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encoding categorical data\n# Encoding the Independent Variable\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\ntransformer = ColumnTransformer(\n    [('encoder', OneHotEncoder(), [1])], \nremainder='passthrough')\nX = np.array(transformer.fit_transform(X), dtype=np.integer)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:42.210669Z","iopub.execute_input":"2023-12-04T02:08:42.21123Z","iopub.status.idle":"2023-12-04T02:08:42.237588Z","shell.execute_reply.started":"2023-12-04T02:08:42.211187Z","shell.execute_reply":"2023-12-04T02:08:42.235343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:08:54.054024Z","iopub.execute_input":"2023-12-04T02:08:54.0544Z","iopub.status.idle":"2023-12-04T02:08:54.062795Z","shell.execute_reply.started":"2023-12-04T02:08:54.054369Z","shell.execute_reply":"2023-12-04T02:08:54.06138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ini untuk HOLD OUT\n#Splitting dataset into training set and test set\nfrom sklearn.model_selection import train_test_split \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.2, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:10:51.039689Z","iopub.execute_input":"2023-12-04T02:10:51.040377Z","iopub.status.idle":"2023-12-04T02:10:51.048954Z","shell.execute_reply.started":"2023-12-04T02:10:51.040347Z","shell.execute_reply":"2023-12-04T02:10:51.046855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting classifier to the Training Set \nfrom sklearn.neighbors import KNeighborsClassifier \nfrom sklearn.metrics import confusion_matrix, accuracy_score \nfrom sklearn.model_selection import cross_val_score","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:13:42.259333Z","iopub.execute_input":"2023-12-04T02:13:42.259664Z","iopub.status.idle":"2023-12-04T02:13:42.264821Z","shell.execute_reply.started":"2023-12-04T02:13:42.259638Z","shell.execute_reply":"2023-12-04T02:13:42.263637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Instantiate learning model (k=3) \nclassifier = KNeighborsClassifier(n_neighbors=3)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:13:52.241135Z","iopub.execute_input":"2023-12-04T02:13:52.241529Z","iopub.status.idle":"2023-12-04T02:13:52.247802Z","shell.execute_reply.started":"2023-12-04T02:13:52.241499Z","shell.execute_reply":"2023-12-04T02:13:52.246514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model \nclassifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:18:43.057286Z","iopub.execute_input":"2023-12-04T02:18:43.057687Z","iopub.status.idle":"2023-12-04T02:18:43.068507Z","shell.execute_reply.started":"2023-12-04T02:18:43.057655Z","shell.execute_reply":"2023-12-04T02:18:43.066471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Predicting the Test Set result \ny_pred = classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:19:13.793347Z","iopub.execute_input":"2023-12-04T02:19:13.794534Z","iopub.status.idle":"2023-12-04T02:19:13.808937Z","shell.execute_reply.started":"2023-12-04T02:19:13.79445Z","shell.execute_reply":"2023-12-04T02:19:13.807629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(y_test, y_pred)\ncm","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:19:56.747651Z","iopub.execute_input":"2023-12-04T02:19:56.749581Z","iopub.status.idle":"2023-12-04T02:19:56.759265Z","shell.execute_reply.started":"2023-12-04T02:19:56.749525Z","shell.execute_reply":"2023-12-04T02:19:56.75821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EVALUATION","metadata":{}},{"cell_type":"code","source":"accuracy = accuracy_score (y_test, y_pred)*100 \nprint('Accuracy of our model is equal ' + str(round(accuracy, 2)) + '%.')","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:20:23.393023Z","iopub.execute_input":"2023-12-04T02:20:23.393521Z","iopub.status.idle":"2023-12-04T02:20:23.407798Z","shell.execute_reply.started":"2023-12-04T02:20:23.393488Z","shell.execute_reply":"2023-12-04T02:20:23.405212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:20:32.424656Z","iopub.execute_input":"2023-12-04T02:20:32.425097Z","iopub.status.idle":"2023-12-04T02:20:32.431206Z","shell.execute_reply.started":"2023-12-04T02:20:32.425021Z","shell.execute_reply":"2023-12-04T02:20:32.430307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:20:41.170733Z","iopub.execute_input":"2023-12-04T02:20:41.171942Z","iopub.status.idle":"2023-12-04T02:20:41.180487Z","shell.execute_reply.started":"2023-12-04T02:20:41.171898Z","shell.execute_reply":"2023-12-04T02:20:41.178344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import precision_score, accuracy_score, recall_score","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:22:05.272518Z","iopub.execute_input":"2023-12-04T02:22:05.272922Z","iopub.status.idle":"2023-12-04T02:22:05.277331Z","shell.execute_reply.started":"2023-12-04T02:22:05.27289Z","shell.execute_reply":"2023-12-04T02:22:05.276591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Precision: %.3f' % precision_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:22:08.046432Z","iopub.execute_input":"2023-12-04T02:22:08.047072Z","iopub.status.idle":"2023-12-04T02:22:08.060809Z","shell.execute_reply.started":"2023-12-04T02:22:08.047019Z","shell.execute_reply":"2023-12-04T02:22:08.058954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Recall: %.3f' % recall_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:22:20.218661Z","iopub.execute_input":"2023-12-04T02:22:20.219243Z","iopub.status.idle":"2023-12-04T02:22:20.235532Z","shell.execute_reply.started":"2023-12-04T02:22:20.219199Z","shell.execute_reply":"2023-12-04T02:22:20.233293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nprint('F1 Score: %.3f' % f1_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2023-12-04T02:23:23.108098Z","iopub.execute_input":"2023-12-04T02:23:23.108519Z","iopub.status.idle":"2023-12-04T02:23:23.121738Z","shell.execute_reply.started":"2023-12-04T02:23:23.108488Z","shell.execute_reply":"2023-12-04T02:23:23.120647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}