{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Resources\n- training:  \n    - step1: training with thumbnails  \n        - https://www.kaggle.com/datasets/motono0223/ubc-efficienetnetb0-fold1of10-2048pix-thumbnails  \n            - train@512pix --> train@1024pix --> train@2048pix\n    - step2: fine tuning without thumbnails  \n        - https://www.kaggle.com/code/motono0223/ubc-finetune-cnn-without-thumbnails\n\n- training data:  \n    - step1: https://www.kaggle.com/code/motono0223/ubc-crop-training-thumbnails  \n    - step2: https://www.kaggle.com/code/motono0223/ubc-crop-training-raw-images  \n\n- inference:  \n    - this notebook : https://www.kaggle.com/code/motono0223/ubc-infer-efficientnetb0-crop-resize-2048pix\n    - resizing script : https://www.kaggle.com/code/motono0223/script-resize  \n    - inference script : https://www.kaggle.com/code/motono0223/script-ubc-inference-crop","metadata":{}},{"cell_type":"markdown","source":"# Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nfrom pathos.multiprocessing import ProcessingPool","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:13:59.083167Z","iopub.execute_input":"2023-11-04T03:13:59.083524Z","iopub.status.idle":"2023-11-04T03:13:59.489833Z","shell.execute_reply.started":"2023-11-04T03:13:59.083496Z","shell.execute_reply":"2023-11-04T03:13:59.48899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters","metadata":{}},{"cell_type":"code","source":"TH_TMA_FILE_SIZE   = 1.5\nBATCHSIZE          = 8\nIMGSIZE            = 2048\nMODELNAME          = \"tf_efficientnet_b0_ns\"\nWEIGHT_COURSETUNE  = \"/kaggle/input/ubc-efficienetnetb0-fold1of10-2048pix-thumbnails/Recall0.9178_Acc0.9437_Loss0.1685_epoch9.bin\"\nWEIGHT_FINETUNE    = \"/kaggle/input/ubc-finetune-cnn-without-thumbnails/Recall0.8726_Acc0.9067_Loss0.2489_epoch3.bin\"\nLABELPKL           = \"/kaggle/input/ubc-finetune-cnn-without-thumbnails/label_encoder.pkl\"","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:13:59.491348Z","iopub.execute_input":"2023-11-04T03:13:59.49172Z","iopub.status.idle":"2023-11-04T03:13:59.496445Z","shell.execute_reply.started":"2023-11-04T03:13:59.491695Z","shell.execute_reply":"2023-11-04T03:13:59.495616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract TMA images from test data by file size","metadata":{}},{"cell_type":"code","source":"!du -s -m /kaggle/input/UBC-OCEAN/test_images/* > amount.csv\ndf_size = pd.read_csv(\"amount.csv\", delimiter='\\t', header=None).rename(columns={0:\"size_mb\", 1:\"file\"})\ndf_size[\"image_id\"] = df_size[\"file\"].apply(lambda x: int(os.path.basename(x).split(\".\")[0] ) )\ndf_size[\"size_mb_log10\"] = np.log10(df_size[\"size_mb\"].values)\ndf_size[\"is_tma\"] = df_size[\"size_mb_log10\"] < TH_TMA_FILE_SIZE\ndf_size","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:13:59.497752Z","iopub.execute_input":"2023-11-04T03:13:59.498066Z","iopub.status.idle":"2023-11-04T03:14:00.512956Z","shell.execute_reply.started":"2023-11-04T03:13:59.498037Z","shell.execute_reply":"2023-11-04T03:14:00.51191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create thumbnail images of WSL data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/UBC-OCEAN/test.csv\")\ndf = pd.merge(df, df_size[[\"image_id\", \"is_tma\"]], on=\"image_id\", how=\"left\")\ndf[\"raw_file_path\"] = df[\"image_id\"].apply(lambda x: f\"/kaggle/input/UBC-OCEAN/test_images/{x}.png\")\ndf[\"thumbnail_file_path\"] = df[\"image_id\"].apply(lambda x: f\"/kaggle/input/UBC-OCEAN/test_thumbnails/{x}_thumbnail.png\")\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:14:00.515365Z","iopub.execute_input":"2023-11-04T03:14:00.5157Z","iopub.status.idle":"2023-11-04T03:14:00.547823Z","shell.execute_reply.started":"2023-11-04T03:14:00.515672Z","shell.execute_reply":"2023-11-04T03:14:00.546838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_image(data):\n    #print(data)\n    image_id, raw_file_path, thumbnail_file_path, is_tma = data\n    # WSL data\n    if not is_tma:\n        file_path = f\"/tmp/test_images/{image_id}.png\"\n        !python /kaggle/input/script-resize/resize.py --imgsize {IMGSIZE} --input {raw_file_path} --output {file_path} > /dev/null\n        return file_path\n    # TMA data\n    else:\n        return thumbnail_file_path","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:14:00.549434Z","iopub.execute_input":"2023-11-04T03:14:00.5501Z","iopub.status.idle":"2023-11-04T03:14:00.556381Z","shell.execute_reply.started":"2023-11-04T03:14:00.550073Z","shell.execute_reply":"2023-11-04T03:14:00.555281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Resize images\n#  - WSL data : resize image of \"test_images/{image_id}.png\" into \"/tmp/test_images/{image_id}.jpg\"\n#  - TMA data : use thumbnail images, as is.\n!mkdir -p /tmp/test_images/\nfile_paths = []\ndata_iter = zip(df[\"image_id\"], df[\"raw_file_path\"], df[\"thumbnail_file_path\"], df[\"is_tma\"])\npool = ProcessingPool(nodes=2)\nfile_paths = pool.map(lambda x: convert_image(x), data_iter)\ndf[\"file_path\"] = file_paths\ndisplay(df)","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:14:00.557685Z","iopub.execute_input":"2023-11-04T03:14:00.558009Z","iopub.status.idle":"2023-11-04T03:14:29.335993Z","shell.execute_reply.started":"2023-11-04T03:14:00.557971Z","shell.execute_reply":"2023-11-04T03:14:29.334603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save test.csv for inference","metadata":{}},{"cell_type":"code","source":"df = df[[\"image_id\", \"file_path\"]]\ndf.to_csv(\"/tmp/test.csv\", index=False)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-11-04T03:14:29.338365Z","iopub.execute_input":"2023-11-04T03:14:29.338856Z","iopub.status.idle":"2023-11-04T03:14:29.356638Z","shell.execute_reply.started":"2023-11-04T03:14:29.338813Z","shell.execute_reply":"2023-11-04T03:14:29.355656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference (Finetuned model)","metadata":{}},{"cell_type":"code","source":"%%time\nTESTCSV   = \"/tmp/test.csv\"\n\n!python /kaggle/input/script-ubc-inference-crop/infer.py \\\n    --imgsize   {IMGSIZE}            \\\n    --batchsize {BATCHSIZE}          \\\n    --weight    {WEIGHT_FINETUNE}    \\\n    --labelpkl  {LABELPKL}           \\\n    --modelname {MODELNAME}          \\\n    --testcsv   {TESTCSV}\n\n!cp log.csv log_ft.csv\n!cp submission.csv submission_ft.csv\n!cp preds.npy preds_ft.npy\n\ndf_log_ft = pd.read_csv(\"log_ft.csv\")\ndisplay(df_log_ft)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-04T03:14:29.357951Z","iopub.execute_input":"2023-11-04T03:14:29.358268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference (Course tuned model)","metadata":{}},{"cell_type":"code","source":"%%time\n# The script runs inference with thumbnail images when the option --testcsv is not set.\n!python /kaggle/input/script-ubc-inference-crop/infer.py \\\n    --imgsize   {IMGSIZE}           \\\n    --batchsize {BATCHSIZE}         \\\n    --weight    {WEIGHT_COURSETUNE} \\\n    --labelpkl  {LABELPKL}\n\n!cp log.csv log_ct.csv\n!cp submission.csv submission_ct.csv\n!cp preds.npy preds_ct.npy\n\ndf_log_ct = pd.read_csv(\"log_ct.csv\")\ndisplay(df_log_ct)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensemble","metadata":{}},{"cell_type":"code","source":"df_log = pd.concat([df_log_ft, df_log_ct])\ndisplay(df_log)\n\ndfs = []\nfor image_id, gdf in df_log.groupby(\"image_id\"):\n    gdf = gdf.sort_values(\"conf\", ascending=False).reset_index()\n    dfs.append( gdf.iloc[:1, :] )\ndf_log = pd.concat(dfs).reset_index(drop=True)\ndisplay(df_log)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub = pd.read_csv(\"/kaggle/input/UBC-OCEAN/sample_submission.csv\")\ndf_sub = pd.merge(df_sub[[\"image_id\"]], df_log[[\"image_id\", \"label\"]], on=\"image_id\", how=\"left\")\ndf_sub.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}