{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"},{"sourceId":7073574,"sourceType":"datasetVersion","datasetId":4073952}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom shutil import copyfile\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-28T05:54:07.087112Z","iopub.execute_input":"2023-11-28T05:54:07.087493Z","iopub.status.idle":"2023-11-28T05:54:07.589145Z","shell.execute_reply.started":"2023-11-28T05:54:07.087462Z","shell.execute_reply":"2023-11-28T05:54:07.588279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checkout the random image that how it alooks like.","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimg = mpimg.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/52275_thumbnail.png')\nimgplot = plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:54:07.591227Z","iopub.execute_input":"2023-11-28T05:54:07.592264Z","iopub.status.idle":"2023-11-28T05:54:09.356852Z","shell.execute_reply.started":"2023-11-28T05:54:07.592226Z","shell.execute_reply":"2023-11-28T05:54:09.355876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**Creating the dataset**\n* Split the dataset in train and validation.\n* Put the data in own classes with the help of csv image id.","metadata":{}},{"cell_type":"code","source":"def building_data_classes(source_dir):\n    read_train_data = pd.read_csv(f'/kaggle/input/UBC-OCEAN/train.csv')    \n    read_train_dir = os.listdir(f'{source_dir}')\n    split_size = 0.9\n    train_length = int(len(read_train_dir) * 0.9)\n    valid_length = int(len(read_train_dir)) - train_length\n    \n    Train_dir = os.makedirs(f'train_data',exist_ok = True)\n    Valid_dir = os.makedirs(f'valid_data', exist_ok = True)\n\n    read_label = read_train_data['label'].value_counts()\n    for label_index in read_label.index:\n        os.makedirs(os.path.join('train_data' , label_index),exist_ok = True)\n        os.makedirs(os.path.join('valid_data' , label_index),exist_ok = True)\n        \n    for source_data in range(train_length):\n        image_name = read_train_dir[source_data].split('_')[0]\n        for label_filter in read_label.index:\n            filtered_data = read_train_data[read_train_data['label'] == label_filter].reset_index(drop = True)\n            for data_train in range(len(filtered_data)):\n                if str(image_name) == str(filtered_data.loc[data_train , 'image_id']):\n                    copyfile(f'{source_dir}{read_train_dir[source_data]}',f'train_data/{label_filter}/{read_train_dir[source_data]}')\n                    \n    for source_data in range(train_length , len(read_train_dir)):\n        image_name = read_train_dir[source_data].split('_')[0]\n        for label_filter in read_label.index:\n            filtered_data = read_train_data[read_train_data['label'] == label_filter].reset_index(drop = True)\n            for data_train in range(len(filtered_data)):\n                if str(image_name) == str(filtered_data.loc[data_train , 'image_id']):\n                    copyfile(f'{source_dir}{read_train_dir[source_data]}',f'valid_data/{label_filter}/{read_train_dir[source_data]}')\n\nbuilding_data_classes('/kaggle/input/UBC-OCEAN/train_thumbnails/')","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:54:09.357929Z","iopub.execute_input":"2023-11-28T05:54:09.358316Z","iopub.status.idle":"2023-11-28T05:54:57.853682Z","shell.execute_reply.started":"2023-11-28T05:54:09.358282Z","shell.execute_reply":"2023-11-28T05:54:57.852872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"Imagedatagenerator for the data augmentation and scaling the image","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ntraining_generator = ImageDataGenerator(rescale = 1./255,\n                                        width_shift_range = 0.2,\n                                        height_shift_range = 0.2,\n                                        zoom_range = 0.6,\n                                        rotation_range = 40,\n                                       )\n\nvalidation_generator = ImageDataGenerator(rescale = 1./255)\n\ntraining_work = training_generator.flow_from_directory('train_data',\n                                                       target_size=(300, 300),\n                                                       batch_size=8,\n                                                       class_mode='categorical'                                                       \n                                                        )\n\nvalid_work = validation_generator.flow_from_directory('valid_data',\n                                                       target_size=(300, 300),\n                                                       batch_size=8,\n                                                       class_mode='categorical'                                                       \n                                                        )","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:54:57.854978Z","iopub.execute_input":"2023-11-28T05:54:57.855332Z","iopub.status.idle":"2023-11-28T05:55:08.97312Z","shell.execute_reply.started":"2023-11-28T05:54:57.8553Z","shell.execute_reply":"2023-11-28T05:55:08.972246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating the Simple CNN Network to checkout how it performs on the dataset and what are the maximum validation accuracy,we can achieve on it.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.optimizers import RMSprop\n\n\nmodel = tf.keras.models.Sequential([\n                                    tf.keras.layers.Conv2D(32 , (3,3) , activation = 'relu', input_shape = (300,300,3)),\n                                    tf.keras.layers.MaxPooling2D(2,2),\n                                    tf.keras.layers.Conv2D(64 , (3,3) , activation = 'relu', input_shape = (300,300,3)),\n                                    tf.keras.layers.MaxPooling2D(2,2),\n                                    tf.keras.layers.Conv2D(128 , (3,3), activation = 'relu'),\n                                    tf.keras.layers.MaxPooling2D(2,2),\n                                    tf.keras.layers.Conv2D(264 , (3,3), activation = 'relu'),\n                                    tf.keras.layers.MaxPooling2D(2,2),\n                                    tf.keras.layers.Conv2D(512 , (3,3), activation = 'relu'),\n                                    tf.keras.layers.MaxPooling2D(2,2),\n                                    tf.keras.layers.Flatten(),\n                                    tf.keras.layers.Dense(512 , activation = 'relu'),\n                                    tf.keras.layers.Dense(256 , activation = 'relu'),\n                                    tf.keras.layers.Dense(128 , activation = 'relu'),\n                                    tf.keras.layers.Dense(5 , activation = 'softmax')\n                                  ])\n\nprint(model.summary())\n\n\nmodel.compile(\n              optimizer = RMSprop(lr=0.001),\n              loss = 'categorical_crossentropy',\n              metrics=['accuracy']\n            )\n    \nhistory = model.fit(training_work,\n                    epochs=15,\n                    verbose=1,\n                    validation_data=valid_work)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T05:55:08.975933Z","iopub.execute_input":"2023-11-28T05:55:08.976468Z","iopub.status.idle":"2023-11-28T06:18:35.26855Z","shell.execute_reply.started":"2023-11-28T05:55:08.97644Z","shell.execute_reply":"2023-11-28T06:18:35.267593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Download the inception v3 weights\n# !wget --no-check-certificate \\\n#     https://storage.googleapis.com/mledu-datasets/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5 \\\n#     -O /tmp/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5","metadata":{"execution":{"iopub.status.busy":"2023-11-28T06:18:35.270095Z","iopub.execute_input":"2023-11-28T06:18:35.270827Z","iopub.status.idle":"2023-11-28T06:18:39.364422Z","shell.execute_reply.started":"2023-11-28T06:18:35.270791Z","shell.execute_reply":"2023-11-28T06:18:39.363391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Inception Model**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.inception_v3 import InceptionV3\nfrom tensorflow.keras import layers\n\nlocal_weights_file = '/kaggle/input/inception-model-gaurav/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5'\n","metadata":{"execution":{"iopub.status.busy":"2023-11-28T12:19:40.843944Z","iopub.execute_input":"2023-11-28T12:19:40.844348Z","iopub.status.idle":"2023-11-28T12:19:40.849192Z","shell.execute_reply.started":"2023-11-28T12:19:40.844316Z","shell.execute_reply":"2023-11-28T12:19:40.848159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pre_trained_model = InceptionV3(input_shape = (300, 300, 3), \n                                include_top = False, \n                                weights = None)\n\npre_trained_model.load_weights(local_weights_file)\n\nfor layer in pre_trained_model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-11-28T12:19:41.743401Z","iopub.execute_input":"2023-11-28T12:19:41.744335Z","iopub.status.idle":"2023-11-28T12:19:45.3039Z","shell.execute_reply.started":"2023-11-28T12:19:41.744298Z","shell.execute_reply":"2023-11-28T12:19:45.303045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_layer = pre_trained_model.get_layer('mixed8')\nprint('last layer output shape: ', last_layer.output_shape)\nlast_output = last_layer.output\n\nfrom tensorflow.keras import Model\n\nx = layers.Flatten()(last_output)\nx = layers.Dense(512, activation='relu')(x)\nx = layers.Dropout(0.2)(x)       \nx = layers.Dense(256, activation='relu')(x)\nx = layers.Dense (5, activation='softmax')(x)      \n\nmodel_2 = Model(pre_trained_model.input, x) \nmodel_2.compile(optimizer = RMSprop(learning_rate=0.0001), \n              loss = 'categorical_crossentropy', \n              metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-28T06:18:41.811383Z","iopub.execute_input":"2023-11-28T06:18:41.811668Z","iopub.status.idle":"2023-11-28T06:18:41.89391Z","shell.execute_reply.started":"2023-11-28T06:18:41.811643Z","shell.execute_reply":"2023-11-28T06:18:41.892894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_2.fit(\n            training_work,\n            validation_data = valid_work,\n            steps_per_epoch = 50,\n            epochs = 50,\n            validation_steps = 25,\n            verbose = 1)\n           ","metadata":{"execution":{"iopub.status.busy":"2023-11-28T06:25:56.343527Z","iopub.execute_input":"2023-11-28T06:25:56.343925Z","iopub.status.idle":"2023-11-28T08:32:30.221757Z","shell.execute_reply.started":"2023-11-28T06:25:56.343897Z","shell.execute_reply":"2023-11-28T08:32:30.220707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Test the Model on testing dataset and create the submission file.**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import load_img, img_to_array\ndir_name = '/kaggle/input/UBC-OCEAN/test_thumbnails/'\ntest_images = os.listdir('/kaggle/input/UBC-OCEAN/test_thumbnails/')\n\nli_name = []\nfor image_name in test_images:\n    data_value = image_name.split('_')[0]\n    img = load_img(f'{dir_name}{image_name}', target_size=(300, 300))\n    x = img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    images = np.vstack([x])\n    classes = np.argmax(model_2.predict(images))\n    class_indices = training_work.class_indices\n    class_names = list(class_indices.keys())\n    get_name_id = class_names[classes]\n    li_name.append(data_value)\n    li_name.append(get_name_id)\n    \nsubmission_file = pd.DataFrame(li_name).T\nsubmission_file.columns = ['image_id','label']\nsubmission_file.to_csv('submission.csv',index = None)","metadata":{"execution":{"iopub.status.busy":"2023-11-28T09:08:41.139624Z","iopub.execute_input":"2023-11-28T09:08:41.140401Z","iopub.status.idle":"2023-11-28T09:08:41.366867Z","shell.execute_reply.started":"2023-11-28T09:08:41.140365Z","shell.execute_reply":"2023-11-28T09:08:41.366141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Note: We can do the experiment on the CNN network and the transfer learning**\n* Outliers part is missing, someone can try on this and add a new version.","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}