{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## Imports\nimport os\nimport random\nimport numpy as np\nimport pandas as pd\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = pow(2,40).__str__()\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn import svm\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, roc_curve, auc\nfrom sklearn.decomposition import PCA\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom PIL import Image\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom skimage.feature import hog\n\nrandom.seed(42)\nImage.MAX_IMAGE_PIXELS = 5_000_000_000","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-19T04:26:35.646343Z","iopub.execute_input":"2023-10-19T04:26:35.646848Z","iopub.status.idle":"2023-10-19T04:26:35.655846Z","shell.execute_reply.started":"2023-10-19T04:26:35.64681Z","shell.execute_reply":"2023-10-19T04:26:35.65428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Read data\ntrain_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\ntrain_images_path = \"/kaggle/input/UBC-OCEAN/train_images\"\ntest_images_path = \"/kaggle/input/UBC-OCEAN/test_images\"\ntrain_thumbnails_path = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\ntest_thumbnails_path = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n\nprint(\"Train dataframe\", train_df.shape)\nprint(\"Test dataframe\", test_df.shape)\ntrain_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:26:35.966103Z","iopub.execute_input":"2023-10-19T04:26:35.966621Z","iopub.status.idle":"2023-10-19T04:26:35.99863Z","shell.execute_reply.started":"2023-10-19T04:26:35.966585Z","shell.execute_reply":"2023-10-19T04:26:35.997658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Preprocessing\n\n## 1. Label encode categorical columns\nle = LabelEncoder()\nlabel_cols = [\"label\"]\nfor col in label_cols:\n    train_df[col] = le.fit_transform(train_df[col])","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:26:36.091237Z","iopub.execute_input":"2023-10-19T04:26:36.091717Z","iopub.status.idle":"2023-10-19T04:26:36.09923Z","shell.execute_reply.started":"2023-10-19T04:26:36.091685Z","shell.execute_reply":"2023-10-19T04:26:36.09768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Baseline model: As the label column is skewed, the largest values \"HGSC\" contributes towards 41% of the values, so the baseline score is 41% accuracy if all predictions are labelled as \"HGSC\". \n\nLB score: 0.14","metadata":{}},{"cell_type":"code","source":"## baseline\n# sample_submission = test_df.copy()\n# sample_submission[\"label\"] = \"HGSC\"\n# sample_submission.drop([\"image_height\", \"image_width\"], axis=1).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:26:38.958389Z","iopub.execute_input":"2023-10-19T04:26:38.958888Z","iopub.status.idle":"2023-10-19T04:26:38.963742Z","shell.execute_reply.started":"2023-10-19T04:26:38.958852Z","shell.execute_reply":"2023-10-19T04:26:38.962812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lets predict only bsaed on image height & image width - predicting only based on how big the image is for each class","metadata":{}},{"cell_type":"code","source":"## Split the data\nX, y = train_df.drop([\"image_id\", \"label\", \"is_tma\"], axis=1), train_df.label\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T01:42:27.655589Z","iopub.execute_input":"2023-10-19T01:42:27.656048Z","iopub.status.idle":"2023-10-19T01:42:27.669793Z","shell.execute_reply.started":"2023-10-19T01:42:27.65601Z","shell.execute_reply":"2023-10-19T01:42:27.668584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SVM Classifier\nsvm_clf = svm.SVC()\nsvm_clf.fit(X_train, y_train)\nprint(\"SVM Training Accuracy score:\",svm_clf.score(X_train, y_train))\nprint(\"SVM Validation Accuracy score:\",svm_clf.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T22:12:44.266956Z","iopub.execute_input":"2023-10-16T22:12:44.267348Z","iopub.status.idle":"2023-10-16T22:12:44.30262Z","shell.execute_reply.started":"2023-10-16T22:12:44.26732Z","shell.execute_reply":"2023-10-16T22:12:44.301246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LGBM\ngbm_clf = GradientBoostingClassifier(n_estimators=100, learning_rate=1.0, max_depth=1, random_state=0)\ngbm_clf.fit(X_train, y_train)\nprint(\"Gradient Boosting Classifier Training Accuracy score:\",gbm_clf.score(X_train, y_train))\nprint(\"Gradient Boosting Classifier Validation Accuracy score:\",gbm_clf.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-10-16T22:12:46.464668Z","iopub.execute_input":"2023-10-16T22:12:46.465011Z","iopub.status.idle":"2023-10-16T22:12:46.918166Z","shell.execute_reply.started":"2023-10-16T22:12:46.464987Z","shell.execute_reply":"2023-10-16T22:12:46.917003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # submission\n# test_predictions = gbm_clf.predict(test_df.drop(\"image_id\", axis=1))\n# test_df[\"label\"] = test_predictions\n# test_df.drop([\"image_height\", \"image_width\"], axis=1).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-16T22:12:48.275733Z","iopub.execute_input":"2023-10-16T22:12:48.276134Z","iopub.status.idle":"2023-10-16T22:12:48.29186Z","shell.execute_reply.started":"2023-10-16T22:12:48.276107Z","shell.execute_reply":"2023-10-16T22:12:48.290942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Predict by looking at the image thumbnail","metadata":{}},{"cell_type":"code","source":"# select random image\nrandom_image = Image.open(f\"{train_thumbnails_path}/{random.choice(train_df.image_id)}_thumbnail.png\")\nprint(\"Original Image size\", random_image.size)\n\n# crop image\nrandom_image = random_image.resize((500, 500))\nrandom_image","metadata":{"execution":{"iopub.status.busy":"2023-10-19T03:07:58.77648Z","iopub.execute_input":"2023-10-19T03:07:58.776986Z","iopub.status.idle":"2023-10-19T03:07:59.074121Z","shell.execute_reply.started":"2023-10-19T03:07:58.776951Z","shell.execute_reply":"2023-10-19T03:07:59.072482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load and preprocess images\ndef load_and_preprocess_images(directory, target_size=(500, 500)):\n    image_paths = [os.path.join(directory, filename) for filename in os.listdir(directory) if filename.endswith('.png')]\n    images = []\n    for path in image_paths:\n        img = cv2.imread(path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n        img = cv2.resize(img, target_size)\n        images.append(img.flatten())\n    return np.array(images)\n\n\nfeature_matrix = load_and_preprocess_images(train_thumbnails_path)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T03:55:17.556035Z","iopub.execute_input":"2023-10-19T03:55:17.556465Z","iopub.status.idle":"2023-10-19T03:56:51.664178Z","shell.execute_reply.started":"2023-10-19T03:55:17.556438Z","shell.execute_reply":"2023-10-19T03:56:51.662733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_matrix.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-19T03:56:51.666417Z","iopub.execute_input":"2023-10-19T03:56:51.66696Z","iopub.status.idle":"2023-10-19T03:56:51.674879Z","shell.execute_reply.started":"2023-10-19T03:56:51.666917Z","shell.execute_reply":"2023-10-19T03:56:51.673823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Feature scaling + PCA\nprint('Feature matrix shape is: ', feature_matrix.shape)\n\n# scaler = StandardScaler()\n# features_transformed = scaler.fit_transform(feature_matrix)\n\npca = PCA(n_components=500)\n# use fit_transform to run PCA on our standardized matrix\nfeatures_pca = pca.fit_transform(feature_matrix)\n\nprint('PCA matrix shape is: ', features_pca.shape)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T03:57:00.140963Z","iopub.execute_input":"2023-10-19T03:57:00.141377Z","iopub.status.idle":"2023-10-19T03:57:50.205191Z","shell.execute_reply.started":"2023-10-19T03:57:00.141331Z","shell.execute_reply":"2023-10-19T03:57:50.203673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exclude images who thumbnails are absent from the directory\ntrain_thumnails_image_ids = [int(filename.split(\"_\")[0]) for filename in os.listdir(train_thumbnails_path) if filename.endswith('.png')]\ntrain_thumbnails_df = train_df[train_df.image_id.isin(train_thumnails_image_ids)]","metadata":{"execution":{"iopub.status.busy":"2023-10-19T03:57:50.20757Z","iopub.execute_input":"2023-10-19T03:57:50.207897Z","iopub.status.idle":"2023-10-19T03:57:50.216338Z","shell.execute_reply.started":"2023-10-19T03:57:50.20787Z","shell.execute_reply":"2023-10-19T03:57:50.215122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Split data\nX_train, X_test, y_train, y_test = train_test_split(features_pca,\n                                                    train_thumbnails_df.label.values,\n                                                    test_size=.3,\n                                                    random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:33:07.025179Z","iopub.execute_input":"2023-10-19T04:33:07.025681Z","iopub.status.idle":"2023-10-19T04:33:07.364146Z","shell.execute_reply.started":"2023-10-19T04:33:07.025638Z","shell.execute_reply":"2023-10-19T04:33:07.362664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## SVM model\n# define support vector classifier\nsvm_clf = svm.SVC(kernel='linear', probability=True, random_state=42)\n\n# fit model\nsvm_clf.fit(X_train, y_train)\n\ny_pred = svm_clf.predict(X_test)\n\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"SVM model accuracy\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:33:08.981233Z","iopub.execute_input":"2023-10-19T04:33:08.981686Z","iopub.status.idle":"2023-10-19T04:35:51.760058Z","shell.execute_reply.started":"2023-10-19T04:33:08.981655Z","shell.execute_reply":"2023-10-19T04:35:51.758425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gbm_clf = GradientBoostingClassifier()\ngbm_clf.fit(X_train, y_train)\nprint(\"GBM accuracy\", gbm_clf.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:35:51.763062Z","iopub.execute_input":"2023-10-19T04:35:51.770812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## ROC AUC curve\nprobabilities = svm_clf.predict_proba(X_test)\n\n# select the probabilities for label 1.0\ny_proba = probabilities[:, 1]\n\nfalse_positive_rate, true_positive_rate, thresholds = roc_curve(y_test, y_proba, pos_label=1)\n\nroc_auc = auc(false_positive_rate, true_positive_rate)\n\nplt.title('Receiver Operating Characteristic')\nroc_plot = plt.plot(false_positive_rate,\n                    true_positive_rate,\n                    label='AUC = {:0.2f}'.format(roc_auc))\n\nplt.legend(loc=0)\nplt.plot([0,1], [0,1], ls='--')\nplt.ylabel('True Positive Rate')\nplt.xlabel('False Positive Rate');","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Submission\n\ntest_feature_matrix = load_and_preprocess_images(test_thumbnails_path)\n# test_features_pca = pca.transform(test_feature_matrix)\ntest_predictions = gbm_clf.predict(test_feature_matrix)\ntest_df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\ntest_df[\"label\"] = le.inverse_transform(test_predictions)\ntest_df.drop([\"image_height\", \"image_width\"], axis=1).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-19T04:32:14.71226Z","iopub.execute_input":"2023-10-19T04:32:14.712665Z","iopub.status.idle":"2023-10-19T04:32:15.369638Z","shell.execute_reply.started":"2023-10-19T04:32:14.712638Z","shell.execute_reply":"2023-10-19T04:32:15.367524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}