{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport json\nimport glob\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom path import Path\nfrom tqdm import tqdm\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:45:59.309389Z","iopub.execute_input":"2023-05-22T03:45:59.309786Z","iopub.status.idle":"2023-05-22T03:45:59.347068Z","shell.execute_reply.started":"2023-05-22T03:45:59.309757Z","shell.execute_reply":"2023-05-22T03:45:59.345934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"# Functions\ndef load_ground_truth(train_paths):\n    \"\"\"\n    Load all groud truth data\n    \n    parameters\n    ----------\n    train_paths: list[str]\n        the list of paths of training data\n    \n    returns\n    ----------\n    ground_truths: numpy.array\n        the np array of ground truth\n    \"\"\"\n    ground_truths = []\n    for path in tqdm(train_paths):\n        ground_truths.append(np.load(f'{path}/human_pixel_masks.npy'))\n    \n    return np.array(ground_truths)\n\n\ndef plot_bands(path):\n    \"\"\"\n    plot band images \n    \n    parameters\n    ----------\n    path: str\n        the path containing the plot target file\n    \"\"\"\n    band_numbers = ['08', '09', '10', '11', '12', '13', '14', '15', '16']\n    \n    num_cols = 8\n    num_rows = len(band_numbers)\n    \n    fig, axes = plt.subplots(num_rows, num_cols, figsize=(16, 16))\n    \n    for i, band_number in enumerate(band_numbers):\n        img = np.load(f'{path}/band_{band_number}.npy')\n        for j in range(num_cols):\n            axes[i, j].imshow(img[:, :, j])\n            axes[i, j].set_title(f'Band {band_number}\\nTime step {j - 4}')\n    plt.tight_layout()\n    plt.show()\n\n\ndef check_diff_wavelength(path):\n    \"\"\"\n    Check the wavelength of each bands\n    \n    parameters\n    ----------\n    path: str\n        the path containing the plot target file\n    \"\"\"\n    band_numbers = ['08', '09', '10', '11', '12', '13', '14', '15', '16']\n    \n    for i, band_number in enumerate(band_numbers):\n        img = np.load(f'{path}/band_{band_number}.npy')\n        x = [i for i in range(img.shape[0])]\n        plt.plot(x, img[0, :, 4], label=band_number)\n    plt.legend()\n    plt.show()\n\n\ndef plot_masks(path):\n    \"\"\"\n    plot mask images \n    \n    parameters\n    ----------\n    path: str\n        the path containing the plot target file\n    \"\"\"\n    masks = np.load(f'{path}/human_individual_masks.npy')\n    ground_truth = np.load(f'{path}/human_pixel_masks.npy')\n    \n    num_cols = masks.shape[3] + 1\n    \n    fig, axes = plt.subplots(1, num_cols, figsize=(16, 16))\n    \n    for i in range(num_cols):\n        \n        if i == (num_cols - 1):\n            axes[i].imshow(ground_truth)\n            axes[i].set_title(f'Ground truth')\n        else:\n            axes[i].imshow(masks[:, :, :, i])\n            axes[i].set_title(f'Label {i}')\n\n            \ndef plot_hist_ground_truth(ground_truths):\n    \"\"\"\n    Plot histgram of ground truth's the number of sum of labels\n    \n    parameters\n    ----------\n    ground_truths: numpy.array\n        the np array of ground truth\n    \"\"\"\n    sum_labels = []\n    \n    for ground_truth in tqdm(ground_truths):\n        sum_labels.append(ground_truth.sum())\n    \n    plt.hist(sum_labels, bins=100)\n    plt.xlabel('sum of labels')\n    plt.ylabel('frequency')\n    plt.show()\n#     print(Counter(sum_labels))","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:45:59.349337Z","iopub.execute_input":"2023-05-22T03:45:59.349811Z","iopub.status.idle":"2023-05-22T03:45:59.368982Z","shell.execute_reply.started":"2023-05-22T03:45:59.349779Z","shell.execute_reply":"2023-05-22T03:45:59.367726Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load data","metadata":{}},{"cell_type":"code","source":"BASE_DIR = Path('/kaggle/input/google-research-identify-contrails-reduce-global-warming')\n\ntrain_paths = glob.glob(f'{BASE_DIR}/train/*')\nval_paths = glob.glob(f'{BASE_DIR}/validation/*')\ntest_paths = glob.glob(f'{BASE_DIR}/test/*')\n\n# Check the number \nprint(f'trains:\\t{len(train_paths)}')\nprint(f'vals:\\t{len(val_paths)}')\nprint(f'tests:\\t{len(test_paths)}')","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:45:59.370542Z","iopub.execute_input":"2023-05-22T03:45:59.371336Z","iopub.status.idle":"2023-05-22T03:45:59.775997Z","shell.execute_reply.started":"2023-05-22T03:45:59.3713Z","shell.execute_reply":"2023-05-22T03:45:59.7748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ground_truths = load_ground_truth(train_paths)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:45:59.779048Z","iopub.execute_input":"2023-05-22T03:45:59.780223Z","iopub.status.idle":"2023-05-22T03:48:58.990935Z","shell.execute_reply.started":"2023-05-22T03:45:59.78018Z","shell.execute_reply":"2023-05-22T03:48:58.989912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Metadata train/validation","metadata":{}},{"cell_type":"code","source":"train_metadata = pd.read_json(f'{BASE_DIR}/train_metadata.json')\nval_metadata = pd.read_json(f'{BASE_DIR}/validation_metadata.json')\n\ntrain_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:48:58.992269Z","iopub.execute_input":"2023-05-22T03:48:58.992597Z","iopub.status.idle":"2023-05-22T03:48:59.440701Z","shell.execute_reply.started":"2023-05-22T03:48:58.992552Z","shell.execute_reply":"2023-05-22T03:48:59.439644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:48:59.442282Z","iopub.execute_input":"2023-05-22T03:48:59.443264Z","iopub.status.idle":"2023-05-22T03:48:59.457416Z","shell.execute_reply.started":"2023-05-22T03:48:59.443232Z","shell.execute_reply":"2023-05-22T03:48:59.456296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Understanding the band images\nEach band files contain the 4 times before and 3 times after images. So, band file's shape is (height, width, 8).","metadata":{}},{"cell_type":"code","source":"train_files = os.listdir(train_paths[0])\ntrain_files.sort()\ntrain_files","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:48:59.458865Z","iopub.execute_input":"2023-05-22T03:48:59.459197Z","iopub.status.idle":"2023-05-22T03:48:59.47479Z","shell.execute_reply.started":"2023-05-22T03:48:59.459168Z","shell.execute_reply":"2023-05-22T03:48:59.473229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_bands(train_paths[100])","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:56:37.607641Z","iopub.execute_input":"2023-05-22T03:56:37.608083Z","iopub.status.idle":"2023-05-22T03:56:47.529743Z","shell.execute_reply.started":"2023-05-22T03:56:37.608048Z","shell.execute_reply":"2023-05-22T03:56:47.52864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_diff_wavelength(train_paths[100])","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:56:28.500958Z","iopub.execute_input":"2023-05-22T03:56:28.50143Z","iopub.status.idle":"2023-05-22T03:56:28.887119Z","shell.execute_reply.started":"2023-05-22T03:56:28.501389Z","shell.execute_reply":"2023-05-22T03:56:28.88625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Undestanding the masks and ground truth","metadata":{}},{"cell_type":"code","source":"plot_masks(train_paths[100])","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:56:28.888455Z","iopub.execute_input":"2023-05-22T03:56:28.888887Z","iopub.status.idle":"2023-05-22T03:56:29.873795Z","shell.execute_reply.started":"2023-05-22T03:56:28.888859Z","shell.execute_reply":"2023-05-22T03:56:29.872601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_hist_ground_truth(ground_truths)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T03:56:29.876025Z","iopub.execute_input":"2023-05-22T03:56:29.876465Z","iopub.status.idle":"2023-05-22T03:56:31.464829Z","shell.execute_reply.started":"2023-05-22T03:56:29.876436Z","shell.execute_reply":"2023-05-22T03:56:31.463664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the Ground Truth data has no labels at all. Therefore, during training, the dominant data becomes the 0 data, so it is necessary to adjust the data.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}