{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-20T17:38:50.598041Z","iopub.execute_input":"2023-11-20T17:38:50.598418Z","iopub.status.idle":"2023-11-20T17:38:50.619757Z","shell.execute_reply.started":"2023-11-20T17:38:50.59839Z","shell.execute_reply":"2023-11-20T17:38:50.618917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\nimport glob\nfrom matplotlib import pyplot as plt\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n\nimport torchmetrics\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict, Counter\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold, StratifiedGroupKFold\n\n# For Image Models\nimport timm\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n# For colored terminal text\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:50.621382Z","iopub.execute_input":"2023-11-20T17:38:50.62165Z","iopub.status.idle":"2023-11-20T17:38:50.630115Z","shell.execute_reply.started":"2023-11-20T17:38:50.621628Z","shell.execute_reply":"2023-11-20T17:38:50.629242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = {\n    \"seed\": 42,\n    \"epochs\": 10,\n    \"img_size\": 2048,\n    \"model_name\": \"tf_efficientnet_b0_ns\",\n    \"checkpoint_path\" : \"/kaggle/input/tf-efficientnet/pytorch/tf-efficientnet-b0/1/tf_efficientnet_b0_aa-827b6e33.pth\",\n    \"pretrained\" : \"/kaggle/input/ubc-efficienetnetb0-fold1of10-2048pix-thumbnails/Recall0.9178_Acc0.9437_Loss0.1685_epoch9.bin\",\n    \"num_classes\": 5,\n    \"train_batch_size\": 2,\n    \"valid_batch_size\": 4,\n    \"learning_rate\": 2e-5,\n    \"scheduler\": 'CosineAnnealingLR',\n    \"min_lr\": 1e-6,\n    \"T_max\": 500,\n    \"weight_decay\": 1e-6,\n    \"fold\" : 0,\n    \"n_fold\": 10,\n    \"n_accumulate\": 1,\n    \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n}","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:50.914261Z","iopub.execute_input":"2023-11-20T17:38:50.915059Z","iopub.status.idle":"2023-11-20T17:38:50.944399Z","shell.execute_reply.started":"2023-11-20T17:38:50.91503Z","shell.execute_reply":"2023-11-20T17:38:50.94344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG['seed'])","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:50.946064Z","iopub.execute_input":"2023-11-20T17:38:50.946363Z","iopub.status.idle":"2023-11-20T17:38:50.965683Z","shell.execute_reply.started":"2023-11-20T17:38:50.946338Z","shell.execute_reply":"2023-11-20T17:38:50.96471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/ubc-crop-training-raw-images'\nTEST_DIR = '/kaggle/input/UBC-OCEAN/test_images'","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:50.966765Z","iopub.execute_input":"2023-11-20T17:38:50.967085Z","iopub.status.idle":"2023-11-20T17:38:50.973104Z","shell.execute_reply.started":"2023-11-20T17:38:50.967049Z","shell.execute_reply":"2023-11-20T17:38:50.972068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ori = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ndf_ori","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:50.974136Z","iopub.execute_input":"2023-11-20T17:38:50.974438Z","iopub.status.idle":"2023-11-20T17:38:51.015948Z","shell.execute_reply.started":"2023-11-20T17:38:50.974414Z","shell.execute_reply":"2023-11-20T17:38:51.015105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, glob, gc\nimport cv2\nimport random\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom PIL import Image\nImage.MAX_IMAGE_PIXELS = None\n\nfrom collections import Counter\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.018895Z","iopub.execute_input":"2023-11-20T17:38:51.019224Z","iopub.status.idle":"2023-11-20T17:38:51.024072Z","shell.execute_reply.started":"2023-11-20T17:38:51.019194Z","shell.execute_reply":"2023-11-20T17:38:51.023212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = \"/kaggle/input/UBC-OCEAN\"\nTRAIN_DIR1 = '/kaggle/input/ubc-resize-images-2048pix/train_images'","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.025275Z","iopub.execute_input":"2023-11-20T17:38:51.025631Z","iopub.status.idle":"2023-11-20T17:38:51.03394Z","shell.execute_reply.started":"2023-11-20T17:38:51.0256Z","shell.execute_reply":"2023-11-20T17:38:51.033107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FILES = sorted(glob.glob(\"/kaggle/input/UBC-OCEAN/train_images*/*.png\"))\nID2FILE = {\n    int(os.path.basename(file).split(\".\")[0]) : file\n    for file in FILES\n}","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.034995Z","iopub.execute_input":"2023-11-20T17:38:51.03553Z","iopub.status.idle":"2023-11-20T17:38:51.047803Z","shell.execute_reply.started":"2023-11-20T17:38:51.035506Z","shell.execute_reply":"2023-11-20T17:38:51.047019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    return ID2FILE[image_id]","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.048791Z","iopub.execute_input":"2023-11-20T17:38:51.049101Z","iopub.status.idle":"2023-11-20T17:38:51.053145Z","shell.execute_reply.started":"2023-11-20T17:38:51.04907Z","shell.execute_reply":"2023-11-20T17:38:51.052273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\ndf['file_path'] = df['image_id'].apply(get_train_file_path)\ndf\ndf.to_csv(\"train_with_paths.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.054369Z","iopub.execute_input":"2023-11-20T17:38:51.054693Z","iopub.status.idle":"2023-11-20T17:38:51.070474Z","shell.execute_reply.started":"2023-11-20T17:38:51.054663Z","shell.execute_reply":"2023-11-20T17:38:51.069652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.07139Z","iopub.execute_input":"2023-11-20T17:38:51.071623Z","iopub.status.idle":"2023-11-20T17:38:51.086196Z","shell.execute_reply.started":"2023-11-20T17:38:51.071602Z","shell.execute_reply":"2023-11-20T17:38:51.085306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\n\n# Load the DataFrame with file paths\ndata = df\n\n# Split the data into training and validation sets\ntrain_data, val_data = train_test_split(data, test_size=0.2, random_state=42)\n\n# Create an image data generator object\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest',\n    validation_split=0.2  # set the validation split\n)\n\n# Set the paths to the images in the DataFrame\ntrain_data['file_path'] = train_data['file_path']\nval_data['file_path'] = val_data['file_path']\n\n# Create data generators\ntrain_generator = train_datagen.flow_from_dataframe(\n    dataframe=train_data,\n    x_col='file_path',\n    y_col='label',\n    target_size=(1080, 1080),\n    batch_size=32,\n    class_mode='categorical',\n    subset='training'\n)\n\nvalidation_generator = train_datagen.flow_from_dataframe(\n    dataframe=val_data,\n    x_col='file_path',\n    y_col='label',\n    target_size=(1080, 1080),\n    batch_size=32,\n    class_mode='categorical',\n    subset='validation'\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:38:51.087221Z","iopub.execute_input":"2023-11-20T17:38:51.087518Z","iopub.status.idle":"2023-11-20T17:39:00.857707Z","shell.execute_reply.started":"2023-11-20T17:38:51.087487Z","shell.execute_reply":"2023-11-20T17:39:00.856783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras.layers import Activation, Dropout, Flatten, Dense\n\n# As per your inputs\nIMG_WIDTH, IMG_HEIGHT = 1080, 1080\nNUM_CLASSES = 4  # adjust with your data, number of unique image labels\n\nmodel = Sequential()\nmodel.add(Conv2D(32, (3, 3), input_shape=(IMG_WIDTH, IMG_HEIGHT, 3)))  # We're working with images of 1080x1080 pixels, and they are colored (so, 3 channels)\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(Conv2D(32, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\n\nmodel.add(Conv2D(64, (3, 3)))\nmodel.add(Activation('relu'))\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\n\n# Now, we flatten the input for Dense layers\nmodel.add(Flatten())\nmodel.add(Dense(64))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(NUM_CLASSES))\nmodel.add(Activation('softmax'))","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:39:00.859058Z","iopub.execute_input":"2023-11-20T17:39:00.859872Z","iopub.status.idle":"2023-11-20T17:39:06.17894Z","shell.execute_reply.started":"2023-11-20T17:39:00.859836Z","shell.execute_reply":"2023-11-20T17:39:06.178167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:39:06.181662Z","iopub.execute_input":"2023-11-20T17:39:06.182459Z","iopub.status.idle":"2023-11-20T17:39:06.199975Z","shell.execute_reply.started":"2023-11-20T17:39:06.182424Z","shell.execute_reply":"2023-11-20T17:39:06.198931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n   train_generator,\n   steps_per_epoch=train_generator.samples // train_generator.batch_size,\n   epochs=64,  # Number of epochs\n   validation_data=validation_generator,\n   validation_steps=validation_generator.samples // validation_generator.batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-11-20T17:39:06.200991Z","iopub.execute_input":"2023-11-20T17:39:06.201308Z"},"trusted":true},"execution_count":null,"outputs":[]}]}