{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Simple Keras Training, only using the thumbnail images\n\nPlease consider upvoting this notebook\n\nAnd the inference notebook can be found here: https://www.kaggle.com/code/pjmathematician/ubco-keras-cnn-baseline-thumbnails-inference","metadata":{}},{"cell_type":"markdown","source":"## Import packages","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n\n\nfrom skimage import io\nfrom skimage.color import rgb2gray\nfrom skimage.transform import rescale, resize, downscale_local_mean\n\n\nfrom sklearn.preprocessing import LabelEncoder, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport matplotlib.pyplot as plt\n\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Flatten, Dropout\nfrom keras.utils import to_categorical","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-07T10:58:54.118117Z","iopub.execute_input":"2023-10-07T10:58:54.118468Z","iopub.status.idle":"2023-10-07T10:58:54.124594Z","shell.execute_reply.started":"2023-10-07T10:58:54.118439Z","shell.execute_reply":"2023-10-07T10:58:54.123326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    TRAIN_THUMBNAILS_PATH = \"/kaggle/input/UBC-OCEAN/train_thumbnails\"\n    TRAIN_FULL_PATH = \"/kaggle/input/UBC-OCEAN/train_images\"\n    TEST_THUMBNAILS_PATH = \"/kaggle/input/UBC-OCEAN/test_thumbnails\"\n    TEST_FULL_PATH = \"/kaggle/input/UBC-OCEAN/test_images\"","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:34:49.411712Z","iopub.execute_input":"2023-10-07T10:34:49.412343Z","iopub.status.idle":"2023-10-07T10:34:49.417017Z","shell.execute_reply.started":"2023-10-07T10:34:49.412308Z","shell.execute_reply":"2023-10-07T10:34:49.416157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:34:50.631463Z","iopub.execute_input":"2023-10-07T10:34:50.631816Z","iopub.status.idle":"2023-10-07T10:34:50.661115Z","shell.execute_reply.started":"2023-10-07T10:34:50.631789Z","shell.execute_reply":"2023-10-07T10:34:50.660156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = train['label'].unique().tolist()\nclasses.append(\"Other\")\nclasses","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:55:20.943103Z","iopub.execute_input":"2023-10-07T10:55:20.943458Z","iopub.status.idle":"2023-10-07T10:55:20.950998Z","shell.execute_reply.started":"2023-10-07T10:55:20.943432Z","shell.execute_reply":"2023-10-07T10:55:20.950099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Defining probabilities for later use, when predicting using random choice","metadata":{}},{"cell_type":"code","source":"proba = [0.4,0.2,0.1,0.075,0.075,0.15]","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:55:22.410891Z","iopub.execute_input":"2023-10-07T10:55:22.411533Z","iopub.status.idle":"2023-10-07T10:55:22.415868Z","shell.execute_reply.started":"2023-10-07T10:55:22.4115Z","shell.execute_reply":"2023-10-07T10:55:22.414926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only load thumbnails (is_tma == False), and resize them into (128,128)","metadata":{}},{"cell_type":"code","source":"images = []\nlabels = []\nfor (i, row) in tqdm(train.iterrows(), total = 538):\n    if row['is_tma']:\n        continue\n    img = io.imread(os.path.join(CFG.TRAIN_THUMBNAILS_PATH, str(row['image_id'])+\"_thumbnail.png\"))\n    img = rgb2gray(img)\n    img = resize(img, (128,128), anti_aliasing=False)\n    img = img.reshape(128,128,1)\n    images.append(img)\n    labels.append(row['label'])","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:34:52.873214Z","iopub.execute_input":"2023-10-07T10:34:52.873569Z","iopub.status.idle":"2023-10-07T10:37:25.453183Z","shell.execute_reply.started":"2023-10-07T10:34:52.873542Z","shell.execute_reply":"2023-10-07T10:37:25.452244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = np.array(images)\nlabels = np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T10:37:26.123897Z","iopub.execute_input":"2023-10-07T10:37:26.12452Z","iopub.status.idle":"2023-10-07T10:37:26.145356Z","shell.execute_reply.started":"2023-10-07T10:37:26.124475Z","shell.execute_reply":"2023-10-07T10:37:26.144431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = images.copy()\n\nencoder = OneHotEncoder(sparse_output = False)\nlabel_encoded = encoder.fit_transform(labels.reshape(-1,1), )\nY = label_encoded","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:00:23.707598Z","iopub.execute_input":"2023-10-07T11:00:23.707916Z","iopub.status.idle":"2023-10-07T11:00:23.740457Z","shell.execute_reply.started":"2023-10-07T11:00:23.707892Z","shell.execute_reply":"2023-10-07T11:00:23.739593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\n\nmodel.add(Conv2D(filters=64, kernel_size=3,strides=(2,1), padding='same', activation='relu', input_shape=(128,128,1)))\nmodel.add(Flatten())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dense(32, activation='relu'))\nmodel.add(Dense(5, activation='softmax'))\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:01:08.418485Z","iopub.execute_input":"2023-10-07T11:01:08.418831Z","iopub.status.idle":"2023-10-07T11:01:08.473835Z","shell.execute_reply.started":"2023-10-07T11:01:08.418805Z","shell.execute_reply":"2023-10-07T11:01:08.472934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X, Y, batch_size=8, epochs=6)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:01:19.081615Z","iopub.execute_input":"2023-10-07T11:01:19.081942Z","iopub.status.idle":"2023-10-07T11:01:30.012163Z","shell.execute_reply.started":"2023-10-07T11:01:19.081917Z","shell.execute_reply":"2023-10-07T11:01:30.011123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save(\"model.h5\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sample submission code provided, but will not work since the training data is not provided while submitting the notebook for testing. Please refer to this notebook for inference","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:01:30.51403Z","iopub.execute_input":"2023-10-07T11:01:30.514388Z","iopub.status.idle":"2023-10-07T11:01:30.526221Z","shell.execute_reply.started":"2023-10-07T11:01:30.514362Z","shell.execute_reply":"2023-10-07T11:01:30.525148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(69)\nthumbnails = os.listdir(CFG.TEST_THUMBNAILS_PATH)\npreds = []\nfor (i, row) in tqdm(test.iterrows()):\n    if str(row['image_id'])+\"_thumbnail.png\" not in thumbnails:\n        print(row['id'])\n        preds.append(np.random.choice(classes, p = proba))\n    img = io.imread(os.path.join(CFG.TEST_THUMBNAILS_PATH, str(row['image_id'])+\"_thumbnail.png\"))\n    img = rgb2gray(img)\n    img = resize(img, (128,128), anti_aliasing=False)\n    img = img.reshape(128,128,1)\n    prediction = encoder.inverse_transform(model.predict(np.array([img])))[0][0]\n    preds.append(prediction)","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:02:24.421171Z","iopub.execute_input":"2023-10-07T11:02:24.421813Z","iopub.status.idle":"2023-10-07T11:02:24.748321Z","shell.execute_reply.started":"2023-10-07T11:02:24.421782Z","shell.execute_reply":"2023-10-07T11:02:24.747317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\nsample_submission['label'] = preds\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2023-10-07T11:03:18.720735Z","iopub.execute_input":"2023-10-07T11:03:18.72106Z","iopub.status.idle":"2023-10-07T11:03:18.733689Z","shell.execute_reply.started":"2023-10-07T11:03:18.721035Z","shell.execute_reply":"2023-10-07T11:03:18.732423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.to_csv('submission.csv', index = False)","metadata":{},"execution_count":null,"outputs":[]}]}