{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.utils import shuffle\n\n# LOADING DATA & PREPARATION\n\ntrain_csv_path = \"/kaggle/input/UBC-OCEAN/train.csv\"\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_csv_path = \"/kaggle/input/UBC-OCEAN/test.csv\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\nsubmission_path = \"/kaggle/working/submission.csv\"\n\ntrain_df = pd.read_csv(train_csv_path)\ntest_df = pd.read_csv(test_csv_path)\n\n\n# Working on only thumbnails \nexisting_images = set(img.split('_')[0] for img in os.listdir(train_images_path))\nprint( len(existing_images))\ntrain_df['image_id'] = train_df['image_id'].astype(str)\nprint(train_df.count())\ntrain_df = train_df[train_df['image_id'].isin(existing_images)]\nprint(train_df.count())\n\nlabels = train_df['label']\nencoder = LabelEncoder()\nencoded_labels = encoder.fit_transform(labels)\ntrain_df['encoded_labels'] = encoded_labels\n\ndef preprocess_image(image_path, target_size=(128, 128)):\n    if os.path.exists(image_path):\n        image = cv2.imread(image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, target_size)\n        image = image / 255.0  # Normalize pixel values to [0, 1]\n        print(f\"LOG: prepricesing img {image_path}\")\n        return image\n    else:\n        return None\n\ntrain_df['image_id'] = train_df['image_id'].astype(str)\n\ntrain_images = []\ntrain_labels = []\nfor img_id in train_df['image_id']:\n    img_path = os.path.join(train_images_path, f\"{img_id}_thumbnail.png\")\n    image = preprocess_image(img_path)\n    if image is not None:\n        train_images.append(image)\n        train_labels.append(train_df[train_df['image_id'] == img_id]['encoded_labels'].values[0])\n\ntrain_images = np.array(train_images)\ntrain_labels = np.array(train_labels)\n\ntrain_images, train_labels = shuffle(train_images, train_labels, random_state=42)\n\ntrain_images, val_images, train_labels, val_labels = train_test_split(train_images, train_labels, test_size=0.2, random_state=42)\n\ntrain_images_flatten = train_images.reshape(train_images.shape[0], -1)\nval_images_flatten = val_images.reshape(val_images.shape[0], -1)\n\n# MODEL\n\nmodel = RandomForestClassifier(n_estimators=100, random_state=42)\n\nprint(\"FITTING THE MODEL\")\nmodel.fit(train_images_flatten, train_labels)\nval_predictions = model.predict(val_images_flatten)\n\n# Accuracy\naccuracy = accuracy_score(val_labels, val_predictions)\nprint(f\"Validation Accuracy: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:39:03.453022Z","iopub.execute_input":"2023-12-18T10:39:03.453782Z","iopub.status.idle":"2023-12-18T10:40:59.91021Z","shell.execute_reply.started":"2023-12-18T10:39:03.453746Z","shell.execute_reply":"2023-12-18T10:40:59.909062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ON TEST SET\n\ntest_df['image_id'] = test_df['image_id'].astype(str)\n\ntest_images = []\nfor img_id in test_df['image_id']:\n    print(test_images_path)\n    img_path = os.path.join(test_images_path, f\"{img_id}.png\")\n    print(img_path)\n    image = preprocess_image(img_path)\n    if image is not None:\n        test_images.append(image)\n\ntest_images = np.array(test_images)\ntest_images_flatten = test_images.reshape(test_images.shape[0], -1)\n\ntest_predictions = model.predict(test_images_flatten)\npredicted_labels = encoder.inverse_transform(test_predictions)\n\n# SUBMISSION\n\nsubmission_df = pd.DataFrame({'image_id': test_df['image_id'], 'label': predicted_labels})\nsubmission_df.to_csv(submission_path, index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:40:59.912173Z","iopub.execute_input":"2023-12-18T10:40:59.912538Z","iopub.status.idle":"2023-12-18T10:41:20.429119Z","shell.execute_reply.started":"2023-12-18T10:40:59.912507Z","shell.execute_reply":"2023-12-18T10:41:20.428005Z"},"trusted":true},"execution_count":null,"outputs":[]}]}