{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Reference: https://www.kaggle.com/code/shashwatraman/simple-unet-baseline-train-lb-0-580\n-----------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"# This is a notebook that records the research results in detail. (ongoing revision)\n# Please upvote if you find this useful. \n-----------------------------------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"#### Simple Unet Baseline (Infer) : Simple UNet Baseline (Infer) refers to a basic structure of the UNet, a deep learning model commonly used in the field of computer vision. UNet is primarily applied to tasks such as medical image analysis or image segmentation, where it excels at accurately segmenting specific parts of an image.\n\n#### Simple UNet Baseline (Infer) specifically pertains to the inference stage of the UNet model. The inference stage involves using the trained UNet model to make predictions on new input images. In the context of image segmentation tasks, these predictions entail predicting the labels for each pixel in a given image.\n\n#### Simple UNet Baseline (Infer) is a simplified version based on the basic structure of the UNet model, typically used for initial experiments or simpler problems. This model has lower complexity compared to other variations of UNet, offering faster training and inference speeds.\n\n----------------------------------------------------------------------------------------------------------------------------------------------------\n","metadata":{}},{"cell_type":"code","source":"from pathlib import Path  # Importing the Path module for working with file paths.\nimport os  # Importing the os module for interacting with the operating system.\nimport random  # Importing the random module for generating random numbers.\nimport math  # Importing the math module for mathematical operations.\nfrom collections import defaultdict  # Importing the defaultdict module for dictionaries with default values.\n\nimport numpy as np  # Importing the numpy module for working with multidimensional arrays and mathematical functions.\nimport pandas as pd  # Importing the pandas module for data manipulation and analysis.\nimport matplotlib.pyplot as plt  # Importing the pyplot module from matplotlib for data visualization.\n\nimport torch  # Importing the PyTorch deep learning framework.\nfrom torch import nn  # Importing the nn module for building neural network models.\nfrom torchvision import transforms  # Importing the transforms module from torchvision for image preprocessing.\nfrom torch.utils.data import Dataset, DataLoader  # Importing the Dataset and DataLoader modules for managing datasets and data loading.\n\nfrom PIL import Image  # Importing the Image module from PIL (Python Imaging Library) for image processing.\nfrom tqdm.notebook import tqdm  # Importing the tqdm module for visualizing progress.\nfrom transformers import get_cosine_schedule_with_warmup  # Importing the get_cosine_schedule_with_warmup function for learning rate scheduling in transformer models.\nfrom tqdm.auto import tqdm  # Importing tqdm module for automatically choosing the best tqdm module.\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path  # 파일 경로를 다루기 위한 Path 모듈을 가져옵니다.\nimport os  # 운영체제와 상호 작용하기 위한 os 모듈을 가져옵니다.\nimport random  # 난수 생성을 위한 random 모듈을 가져옵니다.\nimport math  # 수학 연산을 위한 math 모듈을 가져옵니다.\nfrom collections import defaultdict  # 기본값이 있는 딕셔너리를 다루기 위한 defaultdict 모듈을 가져옵니다.\n\nimport numpy as np  # 다차원 배열과 수학 함수를 다루기 위한 numpy 모듈을 가져옵니다.\nimport pandas as pd  # 데이터 조작과 분석을 위한 pandas 모듈을 가져옵니다.\nimport matplotlib.pyplot as plt  # 데이터 시각화를 위한 pyplot 모듈을 가져옵니다.\n\nimport torch  # 딥러닝 프레임워크인 PyTorch를 가져옵니다.\nfrom torch import nn  # 신경망 모델을 구성하기 위한 nn 모듈을 가져옵니다.\nfrom torchvision import transforms  # 이미지 전처리를 위한 transforms 모듈을 가져옵니다.\nfrom torch.utils.data import Dataset, DataLoader  # 데이터셋과 데이터로더를 관리하기 위한 Dataset과 DataLoader 모듈을 가져옵니다.\n\nfrom PIL import Image  # 이미지를 처리하기 위한 PIL(Python Imaging Library)의 Image 모듈을 가져옵니다.\nfrom tqdm.notebook import tqdm  # 진행 상황을 시각화하기 위한 tqdm 모듈을 가져옵니다.\nfrom transformers import get_cosine_schedule_with_warmup  # 트랜스포머 모델의 학습 스케줄링을 위한 get_cosine_schedule_with_warmup 함수를 가져옵니다.\nfrom tqdm.auto import tqdm  # 자동으로 최적의 tqdm 모듈을 선택하여 가져옵니다.\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys  # Import the sys module.\nsys.path.append(\"../input/pretrained-models-pytorch\")  # Add the parent directory path to sys.path.\nsys.path.append(\"../input/efficientnet-pytorch\")  # Add the parent directory path to sys.path.\nsys.path.append(\"/kaggle/input/smp-github/segmentation_models.pytorch-master\")  # Add the specified path to sys.path.\n\nimport segmentation_models_pytorch as smp  # Import the segmentation_models_pytorch module as smp.\n\nprint(f\"Segmentation Models version: {smp.__version__}\")  # Print the version of Segmentation Models.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys  # sys 모듈을 가져옵니다.\nsys.path.append(\"../input/pretrained-models-pytorch\")  # 상위 디렉토리 경로를 sys.path에 추가합니다.\nsys.path.append(\"../input/efficientnet-pytorch\")  # 상위 디렉토리 경로를 sys.path에 추가합니다.\nsys.path.append(\"/kaggle/input/smp-github/segmentation_models.pytorch-master\")  # 지정된 경로를 sys.path에 추가합니다.\n\nimport segmentation_models_pytorch as smp  # segmentation_models_pytorch 모듈을 smp로 가져옵니다.\n\nprint(f\"Segmentation Models version: {smp.__version__}\")  # Segmentation Models의 버전을 출력합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    batch_size = 32  # Set the batch size.\n    seed = 42  # Set the seed value.\n    thr = 0.02  # Set the threshold value.\n    \n    encoder = 'efficientnet-b0'  # Specify the encoder model to be used.\n    pretrained = False  # Determine whether to use pretrained weights.\n    weights = None  # Specify the path to the weight file.\n    classes = ['contrail']  # Set the list of classes.\n    activation = None  # Set the activation function.\n    in_chans = 3  # Set the number of input channels.\n    \n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')  # Set the device to be used. If CUDA is available, use GPU; otherwise, use CPU.\n    \n    image_size = 256  # Set the image size.\n    \n    model_ckpt = '/kaggle/input/unet-model/epoch-9.pth'  # Set the checkpoint file path for the model.\n\nclass Paths:\n    data = '/kaggle/input/google-research-identify-contrails-reduce-global-warming'  # Set the data path.\n    data_root = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/test/'  # Set the root path of the data.\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.212216Z","iopub.execute_input":"2023-06-14T12:59:34.212566Z","iopub.status.idle":"2023-06-14T12:59:34.245423Z","shell.execute_reply.started":"2023-06-14T12:59:34.212535Z","shell.execute_reply":"2023-06-14T12:59:34.242761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    batch_size = 32  # 배치 크기를 설정합니다.\n    seed = 42  # 시드 값(seed)을 설정합니다.\n    thr = 0.02  # 임계값(threshold)을 설정합니다.\n    \n    encoder = 'efficientnet-b0'  # 사용할 인코더 모델을 지정합니다.\n    pretrained = False  # 사전 훈련된 가중치를 사용할지 여부를 설정합니다.\n    weights = None  # 가중치 파일의 경로를 지정합니다.\n    classes = ['contrail']  # 클래스 목록을 설정합니다.\n    activation = None  # 활성화 함수를 설정합니다.\n    in_chans = 3  # 입력 채널 수를 설정합니다.\n    \n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')  # 사용할 디바이스를 설정합니다. CUDA가 사용 가능한 경우 GPU를 사용하고, 그렇지 않으면 CPU를 사용합니다.\n    \n    image_size = 256  # 이미지 크기를 설정합니다.\n    \n    model_ckpt = '/kaggle/input/unet-model/epoch-9.pth'  # 모델 체크포인트 파일의 경로를 설정합니다.\n\nclass Paths:\n    data = '/kaggle/input/google-research-identify-contrails-reduce-global-warming'  # 데이터 경로를 설정합니다.\n    data_root = '/kaggle/input/google-research-identify-contrails-reduce-global-warming/test/'  # 데이터 루트 경로를 설정합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=1234):\n    random.seed(seed)  # Set the seed for the random module.\n    np.random.seed(seed)  # Set the seed for the numpy module.\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)  # Set the seed for the Python hash function.\n    \n    torch.manual_seed(seed)  # Set the seed for the torch module.\n    torch.cuda.manual_seed(seed)  # Set the seed for CUDA operations.\n    torch.backends.cudnn.deterministic = False  # Set the cudnn deterministic flag to False.\n    torch.backends.cudnn.benchmark = True  # Enable the cudnn benchmark mode.","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.249121Z","iopub.execute_input":"2023-06-14T12:59:34.249809Z","iopub.status.idle":"2023-06-14T12:59:34.273997Z","shell.execute_reply.started":"2023-06-14T12:59:34.249778Z","shell.execute_reply":"2023-06-14T12:59:34.273037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(seed=1234):\n    random.seed(seed)  # random 모듈의 시드를 설정합니다.\n    np.random.seed(seed)  # numpy 모듈의 시드를 설정합니다.\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)  # Python 해시 함수의 시드를 설정합니다.\n    \n    torch.manual_seed(seed)  # torch 모듈의 시드를 설정합니다.\n    torch.cuda.manual_seed(seed)  # CUDA 작업의 시드를 설정합니다.\n    torch.backends.cudnn.deterministic = False  # cudnn의 deterministic 플래그를 False로 설정합니다.\n    torch.backends.cudnn.benchmark = True  # cudnn의 benchmark 모드를 활성화합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(Config.seed)  # Set the seed using the value specified in the Config class.","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.275561Z","iopub.execute_input":"2023-06-14T12:59:34.275918Z","iopub.status.idle":"2023-06-14T12:59:34.289633Z","shell.execute_reply.started":"2023-06-14T12:59:34.275887Z","shell.execute_reply":"2023-06-14T12:59:34.288408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set_seed(Config.seed)  # Config 클래스에서 지정된 시드 값을 사용하여 시드를 설정합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation","metadata":{}},{"cell_type":"code","source":"filenames = os.listdir(Paths.data_root)  # Get the list of filenames in the data root directory.\ntest_df = pd.DataFrame(filenames, columns=['record_id'])  # Create a DataFrame with the filenames, using 'record_id' as the column name.\n\ntest_df['path'] = Paths.data_root + test_df['record_id'].astype(str)  # Add a new column 'path' to the DataFrame, which combines the data root path with the filenames converted to strings.\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.291637Z","iopub.execute_input":"2023-06-14T12:59:34.292617Z","iopub.status.idle":"2023-06-14T12:59:34.309193Z","shell.execute_reply.started":"2023-06-14T12:59:34.292555Z","shell.execute_reply":"2023-06-14T12:59:34.308357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = os.listdir(Paths.data_root)  # 데이터 루트 디렉토리에서 파일 이름 목록을 가져옵니다.\ntest_df = pd.DataFrame(filenames, columns=['record_id'])  # 'record_id'를 열 이름으로 사용하여 파일 이름을 포함하는 DataFrame을 생성합니다.\n\ntest_df['path'] = Paths.data_root + test_df['record_id'].astype(str)  # DataFrame에 'path'라는 새로운 열을 추가하고, 데이터 루트 경로와 파일 이름을 문자열로 변환하여 결합합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.311616Z","iopub.execute_input":"2023-06-14T12:59:34.312159Z","iopub.status.idle":"2023-06-14T12:59:34.32399Z","shell.execute_reply.started":"2023-06-14T12:59:34.312129Z","shell.execute_reply":"2023-06-14T12:59:34.322875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ContrailsDataset(torch.utils.data.Dataset):\n    def __init__(self, df, train=True):\n        self.df = df  # Initialize the instance variable df to store the DataFrame.\n        self.trn = train  # Initialize the instance variable trn to indicate if it is a training dataset.\n\n    def read_record(self, directory):\n        record_data = {}  # Create a dictionary to store the record data.\n        for x in [\n            \"band_11\",\n            \"band_14\",\n            \"band_15\"\n        ]:\n            record_data[x] = np.load(os.path.join(directory, x + \".npy\"))  # Load data for each band and store it in the dictionary.\n\n        return record_data\n\n    def normalize_range(self, data, bounds):\n        \"\"\"Normalize data to the range [0, 1].\"\"\"\n        return (data - bounds[0]) / (bounds[1] - bounds[0])\n\n    def get_false_color(self, record_data):\n        _T11_BOUNDS = (243, 303)\n        _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n        _TDIFF_BOUNDS = (-4, 2)\n\n        N_TIMES_BEFORE = 4\n\n        r = self.normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n        g = self.normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n        b = self.normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n        false_color = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n        img = false_color[..., N_TIMES_BEFORE]\n\n        return img\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        con_path = row.path\n        data = self.read_record(con_path)\n\n        img = self.get_false_color(data)\n\n        img = torch.tensor(img)\n        img = img.permute(2, 0, 1)\n\n        return img.float()\n\n    def __len__(self):\n        return len(self.df)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.325531Z","iopub.execute_input":"2023-06-14T12:59:34.32592Z","iopub.status.idle":"2023-06-14T12:59:34.339682Z","shell.execute_reply.started":"2023-06-14T12:59:34.325888Z","shell.execute_reply":"2023-06-14T12:59:34.338743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ContrailsDataset(torch.utils.data.Dataset):\n    def __init__(self, df, train=True):\n        self.df = df  # DataFrame을 저장하기 위한 인스턴스 변수 df를 초기화합니다.\n        self.trn = train  # 훈련 데이터인지 여부를 나타내기 위한 인스턴스 변수 trn을 초기화합니다.\n\n    def read_record(self, directory):\n        record_data = {}  # 레코드 데이터를 저장하기 위한 딕셔너리를 생성합니다.\n        for x in [\n            \"band_11\",\n            \"band_14\",\n            \"band_15\"\n        ]:\n            record_data[x] = np.load(os.path.join(directory, x + \".npy\"))  # 각 밴드의 데이터를 불러와 딕셔너리에 저장합니다.\n\n        return record_data\n\n    def normalize_range(self, data, bounds):\n        \"\"\"데이터를 [0, 1] 범위로 정규화합니다.\"\"\"\n        return (data - bounds[0]) / (bounds[1] - bounds[0])\n\n    def get_false_color(self, record_data):\n        _T11_BOUNDS = (243, 303)\n        _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n        _TDIFF_BOUNDS = (-4, 2)\n\n        N_TIMES_BEFORE = 4\n\n        r = self.normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n        g = self.normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n        b = self.normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n        false_color = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n        img = false_color[..., N_TIMES_BEFORE]\n\n        return img\n\n    def __getitem__(self, index):\n        row = self.df.iloc[index]\n        con_path = row.path\n        data = self.read_record(con_path)\n\n        img = self.get_false_color(data)\n\n        img = torch.tensor(img)\n        img = img.permute(2, 0, 1)\n\n        return img.float()\n\n    def __len__(self):\n        return len(self.df)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = ContrailsDataset(\n        test_df,\n        train = False\n    )\n \ntest_dl = DataLoader(test_ds, batch_size=Config.batch_size, num_workers = 2)","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.341252Z","iopub.execute_input":"2023-06-14T12:59:34.341516Z","iopub.status.idle":"2023-06-14T12:59:34.351797Z","shell.execute_reply.started":"2023-06-14T12:59:34.341494Z","shell.execute_reply":"2023-06-14T12:59:34.35089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"class UNet(nn.Module):\n    def __init__(self, cfg):\n        super(UNet, self).__init__()\n\n        self.cfg = cfg  # Store the configuration object (cfg) as an instance variable.\n        self.training = True  # Set the instance variable to indicate if it is in training mode.\n\n        self.model = smp.Unet(\n            encoder_name=cfg.encoder,\n            encoder_weights=cfg.weights,\n            decoder_use_batchnorm=True,\n            classes=len(cfg.classes),\n            activation=cfg.activation,\n        )  # Create an instance of the smp.Unet model and store it as an instance variable.\n\n        self.loss_fn = smp.losses.DiceLoss(mode='binary')  # Store the Dice Loss as the loss function to be used.\n\n    def forward(self, imgs):\n        x = imgs  # Assign the input image to the variable x.\n        logits = self.model(x)  # Pass the input to the model to obtain logits.\n\n        return {\"logits\": logits.sigmoid()}  # Apply the sigmoid function to the logits and return the result.\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.35552Z","iopub.execute_input":"2023-06-14T12:59:34.35579Z","iopub.status.idle":"2023-06-14T12:59:34.364066Z","shell.execute_reply.started":"2023-06-14T12:59:34.355769Z","shell.execute_reply":"2023-06-14T12:59:34.363141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class UNet(nn.Module):\n    def __init__(self, cfg):\n        super(UNet, self).__init__()\n\n        self.cfg = cfg  # 설정 객체(cfg)를 인스턴스 변수로 저장합니다.\n        self.training = True  # 훈련 모드인지 여부를 나타내는 인스턴스 변수를 설정합니다.\n\n        self.model = smp.Unet(\n            encoder_name=cfg.encoder,\n            encoder_weights=cfg.weights,\n            decoder_use_batchnorm=True,\n            classes=len(cfg.classes),\n            activation=cfg.activation,\n        )  # smp.Unet 모델을 생성하여 인스턴스 변수로 저장합니다.\n\n        self.loss_fn = smp.losses.DiceLoss(mode='binary')  # Dice Loss를 사용하는 손실 함수를 인스턴스 변수로 저장합니다.\n\n    def forward(self, imgs):\n        x = imgs  # 입력 이미지를 변수 x에 할당합니다.\n        logits = self.model(x)  # 모델에 입력을 전달하여 로짓(logits)을 얻습니다.\n\n        return {\"logits\": logits.sigmoid()}  # 로짓에 시그모이드 함수를 적용하여 반환합니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = UNet(Config).to(Config.device)  # Create an instance of the UNet model and move it to the specified device.\n\nmodel.load_state_dict(torch.load(Config.model_ckpt, map_location=torch.device('cuda')))  # Load the model's state dictionary from the checkpoint file.","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:34.365698Z","iopub.execute_input":"2023-06-14T12:59:34.366033Z","iopub.status.idle":"2023-06-14T12:59:40.185238Z","shell.execute_reply.started":"2023-06-14T12:59:34.366002Z","shell.execute_reply":"2023-06-14T12:59:40.184282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = UNet(Config).to(Config.device)  # UNet 모델의 인스턴스를 생성하고 지정된 장치로 이동시킵니다.\n\nmodel.load_state_dict(torch.load(Config.model_ckpt, map_location=torch.device('cuda')))  # 모델의 상태 사전을 체크포인트 파일로부터 불러옵니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()  # Set the model to evaluation mode (no gradient computation and batch normalization in evaluation).\ntorch.set_grad_enabled(False)  # Disable gradient computation for efficient inference.\n\nval_data = defaultdict(list)  # Create a defaultdict to store validation data.\npbar = tqdm(enumerate(test_dl), total=len(test_dl), desc='Test')  # Create a progress bar for the test data.\nfor step, X in pbar:\n    X = X.to(Config.device)  # Move the input data to the specified device.\n\n    output = model(X)  # Perform forward pass to get the output.\n    for key, val in output.items():\n        val_data[key] += [output[key]]  # Append the output values to the corresponding key in val_data.\n\nfor key, val in output.items():\n    value = val_data[key]\n    if len(value[0].shape) == 0:\n        val_data[key] = torch.stack(value)  # Stack the output tensors along a new dimension.\n    else:\n        val_data[key] = torch.cat(value, dim=0).cpu().detach().numpy()  # Concatenate the output tensors along the specified dimension and move them to CPU for further processing.\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:40.186625Z","iopub.execute_input":"2023-06-14T12:59:40.187542Z","iopub.status.idle":"2023-06-14T12:59:45.789178Z","shell.execute_reply.started":"2023-06-14T12:59:40.187508Z","shell.execute_reply":"2023-06-14T12:59:45.788104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()  # 모델을 평가 모드로 설정합니다 (평가 시에는 기울기 계산과 배치 정규화 비활성화).\ntorch.set_grad_enabled(False)  # 효율적인 추론을 위해 기울기 계산을 비활성화합니다.\n\nval_data = defaultdict(list)  # 검증 데이터를 저장하기 위한 defaultdict를 생성합니다.\npbar = tqdm(enumerate(test_dl), total=len(test_dl), desc='Test')  # 테스트 데이터에 대한 진행 상황을 표시하는 프로그레스 바를 생성합니다.\nfor step, X in pbar:\n    X = X.to(Config.device)  # 입력 데이터를 지정된 장치로 이동합니다.\n\n    output = model(X)  # 순전파를 수행하여 출력을 얻습니다.\n    for key, val in output.items():\n        val_data[key] += [output[key]]  # 출력 값을 해당 키에 대응하는 val_data에 추가합니다.\n\nfor key, val in output.items():\n    value = val_data[key]\n    if len(value[0].shape) == 0:\n        val_data[key] = torch.stack(value)  # 출력 텐서들을 새로운 차원을 기준으로 쌓습니다.\n    else:\n        val_data[key] = torch.cat(value, dim=0).cpu().detach().numpy()  # 지정한 차원을 기준으로 출력 텐서들을 연결하고, 추가 처리를 위해 CPU로 이동시킵니다.","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rle_encode(x, fg_val=1):\n    \"\"\"\n    Args:\n        x:  numpy array of shape (height, width), 1 - mask, 0 - background\n    Returns: run length encoding as list\n    \"\"\"\n\n    dots = np.where(\n        x.T.flatten() == fg_val)[0]  # .T sets Fortran order down-then-right\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\ndef list_to_string(x):\n    \"\"\"\n    Converts list to a string representation\n    Empty list returns '-'\n    \"\"\"\n    if x: # non-empty list\n        s = str(x).replace(\"[\", \"\").replace(\"]\", \"\").replace(\",\", \"\")\n    else:\n        s = '-'\n    return s","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:45.790986Z","iopub.execute_input":"2023-06-14T12:59:45.791543Z","iopub.status.idle":"2023-06-14T12:59:45.799385Z","shell.execute_reply.started":"2023-06-14T12:59:45.791502Z","shell.execute_reply":"2023-06-14T12:59:45.798257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(Paths.data + '/sample_submission.csv', index_col='record_id')\n# Read the 'sample_submission.csv' file into submission, using 'record_id' as the index column.\n\nfor i, pred in enumerate(val_data['logits']):\n    rec = test_df['record_id'][i]\n    # Assign the i-th value from the 'record_id' column of test_df to rec.\n\n    mask = (pred[0] > Config.thr).astype(np.float32)\n    # Create a mask where values exceeding Config.thr in pred[0] are set to 1, and others are set to 0.\n\n    submission.loc[int(rec), 'encoded_pixels'] = list_to_string(rle_encode(mask))\n    # Assign the RLE-encoded string representation of the mask to the 'encoded_pixels' column of the row in submission corresponding to rec.\n\nsubmission.head()\n# Print the first five rows of submission.\n","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:45.801161Z","iopub.execute_input":"2023-06-14T12:59:45.801494Z","iopub.status.idle":"2023-06-14T12:59:45.830669Z","shell.execute_reply.started":"2023-06-14T12:59:45.801462Z","shell.execute_reply":"2023-06-14T12:59:45.829817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(Paths.data + '/sample_submission.csv', index_col='record_id')\n# 'sample_submission.csv' 파일을 읽어와 submission에 저장합니다. 인덱스 컬럼으로 'record_id'를 사용합니다.\n\nfor i, pred in enumerate(val_data['logits']):\n    rec = test_df['record_id'][i]\n    # test_df의 'record_id' 컬럼에서 i번째 값을 rec에 할당합니다.\n\n    mask = (pred[0] > Config.thr).astype(np.float32)\n    # pred[0]에서 Config.thr을 초과하는 값에 대해 1로, 그 외에는 0으로 이루어진 mask를 생성합니다.\n\n    submission.loc[int(rec), 'encoded_pixels'] = list_to_string(rle_encode(mask))\n    # submission에서 record_id가 rec에 해당하는 행의 'encoded_pixels' 컬럼에 mask의 RLE 인코딩 결과를 문자열로 변환하여 할당합니다.\n\nsubmission.head()\n# submission의 처음 다섯 개의 행을 출력합니다.\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-14T12:59:45.832099Z","iopub.execute_input":"2023-06-14T12:59:45.832944Z","iopub.status.idle":"2023-06-14T12:59:45.840012Z","shell.execute_reply.started":"2023-06-14T12:59:45.832911Z","shell.execute_reply":"2023-06-14T12:59:45.838972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}}]}