{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport math\nfrom IPython import display\nimport matplotlib.pyplot as plt\nfrom matplotlib import animation\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-18T03:29:00.493484Z","iopub.execute_input":"2023-06-18T03:29:00.493945Z","iopub.status.idle":"2023-06-18T03:29:00.500889Z","shell.execute_reply.started":"2023-06-18T03:29:00.493907Z","shell.execute_reply":"2023-06-18T03:29:00.499232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    VER = 1\n    AUTHOR = 'takaito'\n    COMPETITION = 'google-research-identify-contrails-reduce-global-warming'\n    DATA_PATH = '/kaggle/input/google-research-identify-contrails-reduce-global-warming'\n    N_TIMES_BEFORE = 4","metadata":{"execution":{"iopub.status.busy":"2023-06-18T03:09:02.915683Z","iopub.execute_input":"2023-06-18T03:09:02.916074Z","iopub.status.idle":"2023-06-18T03:09:02.924637Z","shell.execute_reply.started":"2023-06-18T03:09:02.916043Z","shell.execute_reply":"2023-06-18T03:09:02.923229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class fastnumpyio:\n    def load(file):\n        file=open(file,\"rb\")\n        header = file.read(128)\n        descr = str(header[19:25], 'utf-8').replace(\"'\",\"\").replace(\" \",\"\")\n        shape = tuple(int(num) for num in str(header[60:120], 'utf-8').replace(', }', '').replace('(', '').replace(')', '').split(','))\n        datasize = np.lib.format.descr_to_dtype(descr).itemsize\n        for dimension in shape:\n            datasize *= dimension\n        return np.ndarray(shape, dtype=descr, buffer=file.read(datasize))\n\ndef read_record(record_id, directory):\n    record_data = {}\n    for target in [\"band_11\", \"band_14\", \"band_15\", \"human_pixel_masks\", \"human_individual_masks\"]:\n        try:\n            record_data[target] = fastnumpyio.load(f'{directory}/{record_id}/{target}.npy')\n            # with open(f'{directory}/{record_id}/{target}.npy', 'rb') as f:\n            #     record_data[target] = np.load(f)\n            if target[:5] == 'band_':\n                record_data[target] = record_data[target][..., 3:6]\n        except Exception as e:\n            pass\n    return record_data\n\ndef normalize_range(data, bounds):\n    \"\"\"Maps data to the range [0, 1].\"\"\"\n    return (data - bounds[0]) / (bounds[1] - bounds[0])\n\ndef get_false_color(record_data):\n    _T11_BOUNDS = (243, 303)\n    _CLOUD_TOP_TDIFF_BOUNDS = (-4, 5)\n    _TDIFF_BOUNDS = (-4, 2)\n    r = normalize_range(record_data[\"band_15\"] - record_data[\"band_14\"], _TDIFF_BOUNDS)\n    g = normalize_range(record_data[\"band_14\"] - record_data[\"band_11\"], _CLOUD_TOP_TDIFF_BOUNDS)\n    b = normalize_range(record_data[\"band_14\"], _T11_BOUNDS)\n    false_color = np.clip(np.stack([r, g, b], axis=2), 0, 1)\n    return false_color\n\ndef read_meta_data(flag):\n    df = pd.read_json(f'{CFG.DATA_PATH}/{flag}_metadata.json')\n    df['month'] = df['timestamp'].dt.month\n    df['hour'] = df['timestamp'].dt.hour\n    df['flag'] = flag\n    df = df.drop(['projection_wkt'], axis=1)\n    return df\n\ndef read_record_id(flag):\n    df = pd.DataFrame({'record_id': os.listdir(f'{CFG.DATA_PATH}/{flag}')})\n    df['record_id'] = df['record_id'].astype(int)\n    df = df.sort_values('record_id').reset_index(drop=True)\n    labeler_count_list = []\n    shape_0_list = []\n    shape_1_list = []\n    for record_id in tqdm(df['record_id']):\n        record_data = read_record(record_id, f'{CFG.DATA_PATH}/{flag}')\n        false_color = get_false_color(record_data)\n        img_shape = false_color.shape\n        shape_0_list.append(img_shape[0])\n        shape_1_list.append(img_shape[1])\n        if flag == 'train':\n            labeler_count_list.append(record_data['human_individual_masks'].shape[-1])\n    df['shape_0'] = shape_0_list\n    df['shape_1'] = shape_1_list\n    if flag == 'train':\n        df['labeler_count'] = labeler_count_list\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-06-18T03:29:11.649681Z","iopub.execute_input":"2023-06-18T03:29:11.650143Z","iopub.status.idle":"2023-06-18T03:29:11.672304Z","shell.execute_reply.started":"2023-06-18T03:29:11.650104Z","shell.execute_reply":"2023-06-18T03:29:11.671174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train_df = read_meta_data('train')\nmeta_valid_df = read_meta_data('validation')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T03:29:12.201976Z","iopub.execute_input":"2023-06-18T03:29:12.202369Z","iopub.status.idle":"2023-06-18T03:29:12.491084Z","shell.execute_reply.started":"2023-06-18T03:29:12.202338Z","shell.execute_reply":"2023-06-18T03:29:12.489761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = read_record_id('train')\nvalid_df = read_record_id('validation')\ntest_df = read_record_id('test')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T03:29:12.494134Z","iopub.execute_input":"2023-06-18T03:29:12.495171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train_df.to_csv('meta_train.csv', index=False)\nmeta_valid_df.to_csv('meta_validation.csv', index=False)\ntrain_df.to_csv('train.csv', index=False)\nvalid_df.to_csv('validation.csv', index=False)\ntest_df.to_csv('test.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}