{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45867,"databundleVersionId":6924515,"sourceType":"competition"}],"dockerImageVersionId":30636,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":false,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-02-02T10:10:06.616253Z","iopub.execute_input":"2024-02-02T10:10:06.616661Z","iopub.status.idle":"2024-02-02T10:10:06.637943Z","shell.execute_reply.started":"2024-02-02T10:10:06.616627Z","shell.execute_reply":"2024-02-02T10:10:06.63677Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\nimport glob\nfrom matplotlib import pyplot as plt\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n# For Image Models\nimport timm\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n# For colored terminal text\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:11.821136Z","iopub.execute_input":"2024-02-02T10:10:11.821962Z","iopub.status.idle":"2024-02-02T10:10:11.830895Z","shell.execute_reply.started":"2024-02-02T10:10:11.821927Z","shell.execute_reply":"2024-02-02T10:10:11.829923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = {\n    \"seed\": 42,\n    \"epochs\": 5,\n    \"img_size\": 256,\n    \"model_name\": \"tf_efficientnet_b0_ns\",\n    \"checkpoint_path\" : \"/kaggle/input/tf-efficientnet/pytorch/tf-efficientnet-b0/1/tf_efficientnet_b0_aa-827b6e33.pth\",\n    \"pretrained\" : \"/kaggle/input/ubc-efficienetnetb0-fold1of10-2048pix-thumbnails/Recall0.9178_Acc0.9437_Loss0.1685_epoch9.bin\",\n    \"num_classes\": 5,\n    \"train_batch_size\": 8,\n    \"valid_batch_size\": 16,\n    \"learning_rate\": 0.0001, # <-4e-5\n    \"scheduler\": 'CosineAnnealingLR',\n    \"min_lr\": 2e-6,\n    \"T_max\": 500,\n    \"weight_decay\": 1e-6,\n    # Add loss setting\n    \"loss_type\": 'focal',\n    \"loss_params\": dict(alpha=1, gamma=2),\n    \"fold\" : 0,\n    \"n_fold\": 10,\n    \"n_accumulate\": 1,\n    \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n}","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-02-02T10:10:14.090815Z","iopub.execute_input":"2024-02-02T10:10:14.091638Z","iopub.status.idle":"2024-02-02T10:10:14.098264Z","shell.execute_reply.started":"2024-02-02T10:10:14.091606Z","shell.execute_reply":"2024-02-02T10:10:14.097271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=42):\n    '''Sets the seed of the entire notebook so results are the same every time we run.\n    This is for REPRODUCIBILITY.'''\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    # When running on the CuDNN backend, two further options must be set\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    # Set a fixed value for the hash seed\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    \nset_seed(CONFIG['seed'])","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:16.362438Z","iopub.execute_input":"2024-02-02T10:10:16.363128Z","iopub.status.idle":"2024-02-02T10:10:16.369743Z","shell.execute_reply.started":"2024-02-02T10:10:16.363096Z","shell.execute_reply":"2024-02-02T10:10:16.368811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/UBC-OCEAN'\nTRAIN_DIR = '/kaggle/input/UBC-OCEAN/train_thumbnails'\nTEST_DIR = '/kaggle/input/UBC-OCEAN/test_images'","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:06.765953Z","iopub.execute_input":"2024-02-02T10:22:06.766822Z","iopub.status.idle":"2024-02-02T10:22:06.770867Z","shell.execute_reply.started":"2024-02-02T10:22:06.766787Z","shell.execute_reply":"2024-02-02T10:22:06.769944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:07.78739Z","iopub.execute_input":"2024-02-02T10:22:07.788086Z","iopub.status.idle":"2024-02-02T10:22:07.804392Z","shell.execute_reply.started":"2024-02-02T10:22:07.788053Z","shell.execute_reply":"2024-02-02T10:22:07.803524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:08.70524Z","iopub.execute_input":"2024-02-02T10:22:08.705601Z","iopub.status.idle":"2024-02-02T10:22:08.716032Z","shell.execute_reply.started":"2024-02-02T10:22:08.705571Z","shell.execute_reply":"2024-02-02T10:22:08.715076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.duplicated()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:09.448408Z","iopub.execute_input":"2024-02-02T10:22:09.449085Z","iopub.status.idle":"2024-02-02T10:22:09.457839Z","shell.execute_reply.started":"2024-02-02T10:22:09.44905Z","shell.execute_reply":"2024-02-02T10:22:09.456967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:10.367972Z","iopub.execute_input":"2024-02-02T10:22:10.368356Z","iopub.status.idle":"2024-02-02T10:22:10.378959Z","shell.execute_reply.started":"2024-02-02T10:22:10.368324Z","shell.execute_reply":"2024-02-02T10:22:10.378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_file_path(image_id):\n    return f\"{TRAIN_DIR}/{image_id}_thumbnail.png\"\n#    return f\"{TRAIN_DIR}/{image_id}.png\"","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:11.859309Z","iopub.execute_input":"2024-02-02T10:22:11.860186Z","iopub.status.idle":"2024-02-02T10:22:11.864422Z","shell.execute_reply.started":"2024-02-02T10:22:11.86015Z","shell.execute_reply":"2024-02-02T10:22:11.863341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images = sorted(glob.glob(f\"{TRAIN_DIR}/*.png\"))","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:12.940211Z","iopub.execute_input":"2024-02-02T10:22:12.94076Z","iopub.status.idle":"2024-02-02T10:22:12.948583Z","shell.execute_reply.started":"2024-02-02T10:22:12.940711Z","shell.execute_reply":"2024-02-02T10:22:12.947599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/train.csv\")\ndf['file_path'] = df['image_id'].apply(get_train_file_path)\ndf = df[ df[\"file_path\"].isin(train_images) ].reset_index(drop=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:14.046761Z","iopub.execute_input":"2024-02-02T10:22:14.047124Z","iopub.status.idle":"2024-02-02T10:22:14.067071Z","shell.execute_reply.started":"2024-02-02T10:22:14.047093Z","shell.execute_reply":"2024-02-02T10:22:14.066206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\ncolors = ['#FFE4D6', '#FACBEA', '#D988B9', '#B0578D', '#EF9595']\n# Extracting label distribution\nlabels = df['label'].value_counts().index\nsizes = df['label'].value_counts().values\n\n# Extracting is_tma distribution\nis_tma_counts = df['is_tma'].value_counts()\n\n# Plotting side-by-side pie charts\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 10))\n\n# Plotting the pie chart for label distribution\nax1.pie(sizes, labels=labels, autopct='%1.1f%%', startangle=140, colors=colors)\nax1.set_title('Distribution of Labels')\n\n# Plotting the pie chart for is_tma distribution\nax2.pie(is_tma_counts, labels=is_tma_counts.index, autopct='%1.1f%%', startangle=140, colors=['#A0D8B3', '#E5F9DB'])\nax2.set_title('Distribution of is_tma')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:14.87787Z","iopub.execute_input":"2024-02-02T10:22:14.878855Z","iopub.status.idle":"2024-02-02T10:22:15.161928Z","shell.execute_reply.started":"2024-02-02T10:22:14.878809Z","shell.execute_reply":"2024-02-02T10:22:15.160927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(data=df, x='label', order=df['label'].value_counts().index)\nplt.title('Distribution of Target Classes')\nplt.xlabel('Label')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:17.840395Z","iopub.execute_input":"2024-02-02T10:22:17.840777Z","iopub.status.idle":"2024-02-02T10:22:18.141131Z","shell.execute_reply.started":"2024-02-02T10:22:17.840739Z","shell.execute_reply":"2024-02-02T10:22:18.14027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df, hue=\"label\")\nplt.suptitle(\"Pairplot by Class\", y=1.02)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:22:20.771706Z","iopub.execute_input":"2024-02-02T10:22:20.772668Z","iopub.status.idle":"2024-02-02T10:22:28.665813Z","shell.execute_reply.started":"2024-02-02T10:22:20.772627Z","shell.execute_reply":"2024-02-02T10:22:28.66483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(df[\"image_width\"], bins=30, kde=True)\nplt.title(\"Distribution of Image Width\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:23:03.512636Z","iopub.execute_input":"2024-02-02T10:23:03.513034Z","iopub.status.idle":"2024-02-02T10:23:03.950676Z","shell.execute_reply.started":"2024-02-02T10:23:03.513Z","shell.execute_reply":"2024-02-02T10:23:03.949624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(df[\"image_height\"], bins=30, kde=True)\nplt.title(\"Distribution of Image Height\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:23:21.403129Z","iopub.execute_input":"2024-02-02T10:23:21.403555Z","iopub.status.idle":"2024-02-02T10:23:21.802803Z","shell.execute_reply.started":"2024-02-02T10:23:21.403521Z","shell.execute_reply":"2024-02-02T10:23:21.80188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns_to_exclude = ['image_id']\nstyled_summaries = {}\n\nfor label in df['label'].unique():\n    filtered_data = df[df['label'] == label].drop(columns=columns_to_exclude)\n    sta_summary = filtered_data.describe(include=['float64', 'int64', 'float', 'int']).round(2)\n    styled_summary = sta_summary.T.style.background_gradient(cmap='magma', low=0.2, high=0.1).set_caption(f'<h2 style=\"text-align:center;font-size:15px\">{label} Summary Table')\n    styled_summaries[label] = styled_summary\n\n# Display the styled summaries for each label\nfor label, styled_summary in styled_summaries.items():\n    display(styled_summary)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:34.81626Z","iopub.execute_input":"2024-02-02T10:10:34.816653Z","iopub.status.idle":"2024-02-02T10:10:34.915782Z","shell.execute_reply.started":"2024-02-02T10:10:34.816623Z","shell.execute_reply":"2024-02-02T10:10:34.914783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Set the style of seaborn\nsns.set(style=\"whitegrid\")\n\n# Create subplots for 'image_width' and 'image_height'\nfig, axes = plt.subplots(nrows=1, ncols=2, figsize=(15, 6))\n\n# Boxplot for image_width\nsns.boxplot(x='label', y='image_width', data=df, palette='viridis', ax=axes[0])\naxes[0].set_title('Boxplot for image_width')\n\n# Boxplot for image_height\nsns.boxplot(x='label', y='image_height', data=df, palette='magma', ax=axes[1])\naxes[1].set_title('Boxplot for image_height')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:36.991784Z","iopub.execute_input":"2024-02-02T10:10:36.992542Z","iopub.status.idle":"2024-02-02T10:10:37.580706Z","shell.execute_reply.started":"2024-02-02T10:10:36.992509Z","shell.execute_reply":"2024-02-02T10:10:37.579764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for label in ['HGSC', 'CC', 'EC', 'LGSC', 'MC']:\n    df_tmp = df[df['label']==label]\n    image_id_list = list(df_tmp[df_tmp['is_tma']]['image_id'])\n    plt.figure(figsize=(20.0, 6.0))\n    \n    for i in range(len(image_id_list)):\n        image_id = image_id_list[i]\n        plt.subplot(1, 5, i+1)\n        if i == 0:\n            plt.title(f'image_id:{image_id} (TMA)', fontsize=14)\n            plt.ylabel(label, fontsize=14)\n        else:\n            plt.title(f'image_id:{image_id} (TMA)', fontsize=14)\n        io.imshow(f'/kaggle/input/UBC-OCEAN/train_images/{image_id}.png')\n        plt.tick_params(labelbottom=False, labelleft=False, labelright=False, labeltop=False, bottom=False, left=False, right=False, top=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:29:14.564309Z","iopub.execute_input":"2024-02-02T10:29:14.565166Z","iopub.status.idle":"2024-02-02T10:29:14.60225Z","shell.execute_reply.started":"2024-02-02T10:29:14.565131Z","shell.execute_reply":"2024-02-02T10:29:14.601245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\n\nlabel_column = 'label'\npath_column = 'file_path'\nimage_id_column = 'image_id'\n\nunique_labels = df[label_column].unique()\n\nimages_per_label = 5\n\n# Randomly select and plot 5 images for each label\nfor label in unique_labels:\n    label_data = df[df[label_column] == label]\n    sample_images = label_data.sample(min(images_per_label, len(label_data)))\n\n    plt.figure(figsize=(10, 3))\n    plt.suptitle(f'Images for Label {label}', y=1.1, fontsize=12)  # Adjusted font size\n\n    for i, (_, row) in enumerate(sample_images.iterrows()):\n        image_path = row[path_column]\n        image = Image.open(image_path)\n\n        plt.subplot(1, images_per_label, i + 1)\n        plt.imshow(image)\n        plt.title(f'Image ID: {row[image_id_column]}', fontsize=10)\n        plt.axis('off')\n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:39.518288Z","iopub.execute_input":"2024-02-02T10:10:39.51903Z","iopub.status.idle":"2024-02-02T10:10:59.726727Z","shell.execute_reply.started":"2024-02-02T10:10:39.518993Z","shell.execute_reply":"2024-02-02T10:10:59.725698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv('../input/UBC-OCEAN/test.csv')\ntest_data","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:10:59.728648Z","iopub.execute_input":"2024-02-02T10:10:59.729337Z","iopub.status.idle":"2024-02-02T10:10:59.741604Z","shell.execute_reply.started":"2024-02-02T10:10:59.7293Z","shell.execute_reply":"2024-02-02T10:10:59.740705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_thumbnails_folder_path = '../input/UBC-OCEAN/test_thumbnails'\ntest_data['full_path'] = test_data['image_id'].apply(lambda x: os.path.join(test_thumbnails_folder_path, f\"{x}_thumbnail.png\"))\n\ntest_data","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:01.610276Z","iopub.execute_input":"2024-02-02T10:11:01.610624Z","iopub.status.idle":"2024-02-02T10:11:01.621698Z","shell.execute_reply.started":"2024-02-02T10:11:01.610597Z","shell.execute_reply":"2024-02-02T10:11:01.620784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\n\nimage_id_column = 'image_id'\nwidth_column = 'image_width'\nheight_column = 'image_height'\npath_column = 'full_path'\n\nnum_images_to_plot = len(test_data)\n\nplt.figure(figsize=(10, 5 * num_images_to_plot))\nplt.suptitle(f'Images from Test Data', y=1.02, fontsize=16)\n\nfor i, (_, row) in enumerate(test_data.iterrows()):\n    image_path = row[path_column]\n    image = Image.open(image_path)\n\n    plt.subplot(num_images_to_plot, 1, i + 1)\n    plt.imshow(image)\n    plt.title(f'Image ID: {row[image_id_column]}')\n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:02.950412Z","iopub.execute_input":"2024-02-02T10:11:02.950795Z","iopub.status.idle":"2024-02-02T10:11:04.179034Z","shell.execute_reply.started":"2024-02-02T10:11:02.950766Z","shell.execute_reply":"2024-02-02T10:11:04.178094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = LabelEncoder()\ndf['label'] = encoder.fit_transform(df['label'])\n\nwith open(\"label_encoder.pkl\", \"wb\") as fp:\n    joblib.dump(encoder, fp)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:06.936642Z","iopub.execute_input":"2024-02-02T10:11:06.937008Z","iopub.status.idle":"2024-02-02T10:11:06.943947Z","shell.execute_reply.started":"2024-02-02T10:11:06.936977Z","shell.execute_reply":"2024-02-02T10:11:06.943031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedGroupKFold(n_splits=CONFIG['n_fold'], shuffle=True, random_state=CONFIG[\"seed\"])\n\nfor fold, ( _, val_) in enumerate(skf.split(X=df, y=df.label, groups=df.image_id)):\n      df.loc[val_ , \"kfold\"] = int(fold)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:08.83965Z","iopub.execute_input":"2024-02-02T10:11:08.840974Z","iopub.status.idle":"2024-02-02T10:11:09.403464Z","shell.execute_reply.started":"2024-02-02T10:11:08.840924Z","shell.execute_reply":"2024-02-02T10:11:09.402403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG['T_max'] = df[df[\"kfold\"]!=CONFIG[\"fold\"]].shape[0] * CONFIG['epochs'] // CONFIG['train_batch_size']\nCONFIG['T_max']","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:10.396641Z","iopub.execute_input":"2024-02-02T10:11:10.397393Z","iopub.status.idle":"2024-02-02T10:11:10.405239Z","shell.execute_reply.started":"2024-02-02T10:11:10.39736Z","shell.execute_reply":"2024-02-02T10:11:10.404252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\n \n# read image\nimg = cv2.imread('/kaggle/input/UBC-OCEAN/train_thumbnails/10143_thumbnail.png' , cv2.IMREAD_UNCHANGED)\n \n# get dimensions of image\ndimensions = img.shape\n \n# height, width, number of channels in image\nheight = img.shape[0]\nwidth = img.shape[1]\nchannels = img.shape[2]\n\nprint('Image Dimension    : ',dimensions)\nprint('Image Height       : ',height)\nprint('Image Width        : ',width)\nprint('Number of Channels : ',channels)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:11:13.774226Z","iopub.execute_input":"2024-02-02T10:11:13.774606Z","iopub.status.idle":"2024-02-02T10:11:14.070016Z","shell.execute_reply.started":"2024-02-02T10:11:13.774576Z","shell.execute_reply":"2024-02-02T10:11:14.069087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UBCDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n        self.labels = df['label'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        label = self.labels[index]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            'label': torch.tensor(label, dtype=torch.long)\n        }","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:12:13.434937Z","iopub.execute_input":"2024-02-02T10:12:13.43534Z","iopub.status.idle":"2024-02-02T10:12:13.443648Z","shell.execute_reply.started":"2024-02-02T10:12:13.435306Z","shell.execute_reply":"2024-02-02T10:12:13.442494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transforms = {\n    \"train\": A.Compose([\n        A.Resize(CONFIG['img_size'], CONFIG['img_size']),\n        A.Flip(p=0.5),\n        A.VerticalFlip(p=0.5),\n        A.RandomRotate90(p=1.0),\n        A.ShiftScaleRotate(shift_limit=0.1, \n                           scale_limit=0.15, \n                           rotate_limit=60, \n                           p=0.5),\n        A.HueSaturationValue(\n                hue_shift_limit=0.2,\n                sat_shift_limit=0.2,\n                val_shift_limit=0.2,\n                p=0.5\n            ),\n        A.RandomBrightnessContrast(\n                brightness_limit=(-0.1, 0.1), \n                contrast_limit=(-0.1, 0.1), \n                p=0.5\n            ),\n        A.Normalize(\n                mean=[0.468, 0.406, 0.465], \n                std=[0.384, 0.344, 0.383],\n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.),\n    \n    \"valid\": A.Compose([\n        A.Resize(CONFIG['img_size'], CONFIG['img_size']),\n        A.Normalize(\n                mean=[0.468, 0.406, 0.465], \n                std=[0.384, 0.344, 0.383],\n                max_pixel_value=255.0, \n                p=1.0\n            ),\n        ToTensorV2()], p=1.)\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:12:35.803885Z","iopub.execute_input":"2024-02-02T10:12:35.804507Z","iopub.status.idle":"2024-02-02T10:12:35.814565Z","shell.execute_reply.started":"2024-02-02T10:12:35.804472Z","shell.execute_reply":"2024-02-02T10:12:35.813684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if False:\n    df_train = df[df.kfold != fold].reset_index(drop=True)\n    train_dataset = UBCDataset(df_train,\n                               transforms=A.Compose([\n                                   A.Resize(CONFIG['img_size'], CONFIG['img_size']),\n                                   ToTensorV2()], p=1.)\n                              )\n    train_loader = DataLoader(train_dataset, batch_size=CONFIG['train_batch_size'], \n                              num_workers=2, pin_memory=True, drop_last=True)\n    mean = 0.0\n    std = 0.0\n    total_samples = 0\n\n    for batch_label in tqdm(train_loader):\n        batch = batch_label['image']\n        batch_size = batch.size(0)\n        data = batch.view(batch_size, batch.size(1), -1).float()\n        mean += data.mean(2).sum(0)\n        std += data.std(2).sum(0)\n        total_samples += batch_size\n\n    mean /= total_samples\n    std /= total_samples\n\n    print(\"Calculated Mean:\", mean/255)\n    print(\"Calculated Std:\", std/255)","metadata":{"execution":{"iopub.status.busy":"2024-02-02T10:12:48.857483Z","iopub.execute_input":"2024-02-02T10:12:48.858194Z","iopub.status.idle":"2024-02-02T10:12:48.866925Z","shell.execute_reply.started":"2024-02-02T10:12:48.858158Z","shell.execute_reply":"2024-02-02T10:12:48.865731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}