{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## general thoughts on data\nAs we can see, the size of an image comparitevely huge, hence reading it might be problematic due to memory limits set by Kaggle. \n\nSo far, I have tried to load each image using *Pillow, cv2, tifffile*. However, read function of these libraries crashed while reading large images with more than 1.2 Gb. \n\nThank to this [notebook](https://www.kaggle.com/code/dschettler8845/mcsai-how-to-interact-with-large-tif-files), I've learnt about **pyvips**, and turns out it can read large files without any problem. ","metadata":{}},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get -y install libvips-dev\n!pip install pyvips","metadata":{"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-22T10:44:44.736847Z","iopub.execute_input":"2022-08-22T10:44:44.737422Z","iopub.status.idle":"2022-08-22T10:46:32.951582Z","shell.execute_reply.started":"2022-08-22T10:44:44.737374Z","shell.execute_reply":"2022-08-22T10:46:32.950022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport torch\nimport pyvips\nimport openslide\n\nMAIN_DIR = \"/kaggle/input/mayo-clinic-strip-ai/\"\n\ntrain_df = pd.read_csv(\"/kaggle/input/mayo-clinic-strip-ai/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/mayo-clinic-strip-ai/test.csv\")\nother_df = pd.read_csv(\"/kaggle/input/mayo-clinic-strip-ai/other.csv\")\n\ntrain_df['img_path'] = MAIN_DIR + \"train/\" + train_df['image_id'] + \".tif\"\nother_df['img_path'] = MAIN_DIR + \"other/\" + other_df['image_id'] + \".tif\"\ntest_df['img_path'] = MAIN_DIR + \"test/\" + test_df['image_id'] + \".tif\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-22T10:58:40.111409Z","iopub.execute_input":"2022-08-22T10:58:40.11206Z","iopub.status.idle":"2022-08-22T10:58:42.875269Z","shell.execute_reply.started":"2022-08-22T10:58:40.112003Z","shell.execute_reply":"2022-08-22T10:58:42.873843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_data = [\n    other_df.drop(columns = [\"label\", \"other_specified\",]),\n    train_df.drop(columns = [\"center_id\", \"label\"]),\n    test_df.drop(columns = [\"center_id\"])\n]\n\nALL_DF = pd.concat(all_data, ignore_index = True)\nALL_DF.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-22T10:58:45.468172Z","iopub.execute_input":"2022-08-22T10:58:45.468717Z","iopub.status.idle":"2022-08-22T10:58:45.50537Z","shell.execute_reply.started":"2022-08-22T10:58:45.468676Z","shell.execute_reply":"2022-08-22T10:58:45.504045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Memory sizes distribution","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nimg_sizes = []\nlargest_file, max_size = None, 0\n\nfor idx, row in ALL_DF.iterrows():\n    img_path, img_id = row['img_path'], row['image_id']\n    st_size = os.stat(img_path).st_size * 9.314e-10\n    \n    if st_size > max_size:\n        largest_file, max_size = row['img_path'], st_size\n    \n    img_sizes.append(st_size)\n\nsns.histplot(img_sizes)\nplt.xlabel(\"Image size in Gigabytes\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-22T10:58:46.338765Z","iopub.execute_input":"2022-08-22T10:58:46.339372Z","iopub.status.idle":"2022-08-22T10:58:48.690116Z","shell.execute_reply.started":"2022-08-22T10:58:46.339314Z","shell.execute_reply":"2022-08-22T10:58:48.688774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading the largest file","metadata":{}},{"cell_type":"code","source":"largest_file, max_size","metadata":{"execution":{"iopub.status.busy":"2022-08-22T10:58:48.692331Z","iopub.execute_input":"2022-08-22T10:58:48.692701Z","iopub.status.idle":"2022-08-22T10:58:48.700173Z","shell.execute_reply.started":"2022-08-22T10:58:48.692666Z","shell.execute_reply":"2022-08-22T10:58:48.698861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time \n\ndef load_scaled_down_slide(image_path, downsample_by, to_numpy=True):\n    return pyvips.Image.new_from_file(image_path).resize(1/downsample_by,kernel='nearest').numpy()\n\n\nstart = time.time()\n\ndownscaled_img = load_scaled_down_slide(largest_file, downsample_by = 20)\n\nprint(\"\\ntime: \", time.time() - start)\ndownscaled_img.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:07:12.336154Z","iopub.execute_input":"2022-08-22T11:07:12.336706Z","iopub.status.idle":"2022-08-22T11:07:55.015529Z","shell.execute_reply.started":"2022-08-22T11:07:12.336664Z","shell.execute_reply":"2022-08-22T11:07:55.014086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"519/60","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:09:44.346783Z","iopub.execute_input":"2022-08-22T11:09:44.347258Z","iopub.status.idle":"2022-08-22T11:09:44.354986Z","shell.execute_reply.started":"2022-08-22T11:09:44.347221Z","shell.execute_reply":"2022-08-22T11:09:44.35386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(downscaled_img)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:08:13.679869Z","iopub.execute_input":"2022-08-22T11:08:13.680328Z","iopub.status.idle":"2022-08-22T11:08:15.153125Z","shell.execute_reply.started":"2022-08-22T11:08:13.68029Z","shell.execute_reply":"2022-08-22T11:08:15.151486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Further steps\n\nAs a Kaggle newbie, I tried to make an end-to-end notebook that cleans data, trains the model, and submit predictions. However, after [learning](https://www.kaggle.com/c/severstal-steel-defect-detection/discussion/113195) how to create own dataset and load it to the notebook, I think best approach would be as follows: \n\n> 1. Convert all tiff images into Low Resolution images using pyvips *(and seam-carving)*, and download the Low Resolution images.\n> 2. Then, create own private dataset from downloaded Low Resolution models. \n> 3. Train you model. Downloaded the best weights and write a submission pipeline.  ","metadata":{}},{"cell_type":"code","source":"pyvips.Image","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:04:03.533924Z","iopub.execute_input":"2022-08-22T11:04:03.534308Z","iopub.status.idle":"2022-08-22T11:04:03.543825Z","shell.execute_reply.started":"2022-08-22T11:04:03.534274Z","shell.execute_reply":"2022-08-22T11:04:03.542466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tifffile as tifi\nimport time","metadata":{"execution":{"iopub.status.busy":"2022-08-22T10:59:01.008335Z","iopub.execute_input":"2022-08-22T10:59:01.00884Z","iopub.status.idle":"2022-08-22T10:59:01.014665Z","shell.execute_reply.started":"2022-08-22T10:59:01.008802Z","shell.execute_reply":"2022-08-22T10:59:01.013319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ns_1 = time.time()\nimg = tifi.imread(largest_file)\nt_1 = time.time()-s_1\nprint(f\"image reading time: {t1}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T10:59:01.631279Z","iopub.execute_input":"2022-08-22T10:59:01.631887Z","iopub.status.idle":"2022-08-22T11:00:30.674766Z","shell.execute_reply.started":"2022-08-22T10:59:01.631832Z","shell.execute_reply":"2022-08-22T11:00:30.672405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s_2 = time.time()\nresized = cv2.resize(img, (2073, 4886, 3), interpolation = cv2.INTER_NEAREST)\nt_2 = time.time()-s_2\nprint(f\"image resizing time: {t1}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:00:30.676049Z","iopub.status.idle":"2022-08-22T11:00:30.676541Z","shell.execute_reply.started":"2022-08-22T11:00:30.676316Z","shell.execute_reply":"2022-08-22T11:00:30.67634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"total time\")\nprint(t2+t1)","metadata":{"execution":{"iopub.status.busy":"2022-08-22T11:00:30.678023Z","iopub.status.idle":"2022-08-22T11:00:30.67847Z","shell.execute_reply.started":"2022-08-22T11:00:30.67826Z","shell.execute_reply":"2022-08-22T11:00:30.678281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}