{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Change image size limit before import opencv\nimport os\nos.environ[\"OPENCV_IO_MAX_IMAGE_PIXELS\"] = pow(2,40).__str__()\n\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport os\nfrom skimage import io\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-10T13:24:29.229079Z","iopub.execute_input":"2023-10-10T13:24:29.229747Z","iopub.status.idle":"2023-10-10T13:24:30.736315Z","shell.execute_reply.started":"2023-10-10T13:24:29.229704Z","shell.execute_reply":"2023-10-10T13:24:30.735467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/UBC-OCEAN/train.csv')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-10-10T13:24:30.73794Z","iopub.execute_input":"2023-10-10T13:24:30.739238Z","iopub.status.idle":"2023-10-10T13:24:30.786052Z","shell.execute_reply.started":"2023-10-10T13:24:30.739196Z","shell.execute_reply":"2023-10-10T13:24:30.784864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check PNG Size","metadata":{}},{"cell_type":"code","source":"df_train.plot.scatter(x='image_width', y='image_height')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T01:39:46.548534Z","iopub.execute_input":"2023-10-09T01:39:46.548942Z","iopub.status.idle":"2023-10-09T01:39:46.865933Z","shell.execute_reply.started":"2023-10-09T01:39:46.548908Z","shell.execute_reply":"2023-10-09T01:39:46.865078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[['image_width', 'image_height']].min()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T01:39:46.869883Z","iopub.execute_input":"2023-10-09T01:39:46.870206Z","iopub.status.idle":"2023-10-09T01:39:46.882818Z","shell.execute_reply.started":"2023-10-09T01:39:46.870178Z","shell.execute_reply":"2023-10-09T01:39:46.88148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[df_train['image_width']==2964]","metadata":{"execution":{"iopub.status.busy":"2023-10-09T01:40:14.006425Z","iopub.execute_input":"2023-10-09T01:40:14.006778Z","iopub.status.idle":"2023-10-09T01:40:14.021277Z","shell.execute_reply.started":"2023-10-09T01:40:14.006751Z","shell.execute_reply":"2023-10-09T01:40:14.019996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Resize & Save","metadata":{}},{"cell_type":"code","source":"# Config\nsize = (2964, 2964)","metadata":{"execution":{"iopub.status.busy":"2023-10-10T13:24:30.787516Z","iopub.execute_input":"2023-10-10T13:24:30.787992Z","iopub.status.idle":"2023-10-10T13:24:30.7918Z","shell.execute_reply.started":"2023-10-10T13:24:30.787964Z","shell.execute_reply":"2023-10-10T13:24:30.790764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save folder\nos.mkdir('/kaggle/working/reduced/')\n\nfor image_id in tqdm(df_train['image_id'].to_list()):\n    # Background\n    base_pic=np.zeros((size[1],size[0],3),np.uint8)\n    \n    # Read the original image if there is no thumbnail.\n    pic = f'/kaggle/input/UBC-OCEAN/train_thumbnails/{image_id}_thumbnail.png'\n    if ~os.path.isfile(pic):\n        pic = f'/kaggle/input/UBC-OCEAN/train_images/{image_id}.png'\n    pic1=cv2.imread(pic,cv2.IMREAD_COLOR)\n    h,w=pic1.shape[:2]\n    ash=size[1]/h\n    asw=size[0]/w\n    if asw<ash:\n        sizeas=(int(w*asw),int(h*asw))\n    else:\n        sizeas=(int(w*ash),int(h*ash))\n    pic1 = cv2.resize(pic1,dsize=sizeas)\n    base_pic[int(size[1]/2-sizeas[1]/2):int(size[1]/2+sizeas[1]/2),\n    int(size[0]/2-sizeas[0]/2):int(size[0]/2+sizeas[0]/2),:]=pic1\n    cv2.imwrite(f'reduced/{image_id}.png', base_pic)","metadata":{"execution":{"iopub.status.busy":"2023-10-10T13:24:42.514769Z","iopub.execute_input":"2023-10-10T13:24:42.515134Z","iopub.status.idle":"2023-10-10T13:25:02.033987Z","shell.execute_reply.started":"2023-10-10T13:24:42.515103Z","shell.execute_reply":"2023-10-10T13:25:02.032755Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample Visualization\nfor label in ['HGSC', 'CC', 'EC', 'LGSC', 'MC']:\n    df_tmp = df_train[df_train['label']==label]\n    image_id_list = list(df_tmp[~df_tmp['is_tma']]['image_id'].sample(5))\n    plt.figure(figsize=(20.0, 6.0))\n    for i in range(len(image_id_list)):\n        image_id = image_id_list[i]\n        plt.subplot(1, 5, i+1)\n        if i == 0:\n            plt.title(f'image_id:{image_id} (WSI)', fontsize=14)\n            plt.ylabel(label, fontsize=14)\n        else:\n            plt.title(f'image_id:{image_id} (WSI)', fontsize=14)\n            \n        io.imshow(f'reduced/{image_id}.png')\n        #plt.tick_params(labelbottom=False, labelleft=False, labelright=False, labeltop=False, bottom=False, left=False, right=False, top=False)\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}