{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# INSTALLING THE PACKAGES","metadata":{}},{"cell_type":"code","source":"!sudo apt-get update\n!sudo apt-get -y install libvips-dev\n!pip install pyvips","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# IMPORTING PACKAGES","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport pyvips\n\nimport tensorflow as tf\nimport tifffile as tiff \nimport os\nimport cv2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LOAD DATA","metadata":{}},{"cell_type":"code","source":"trainImgDir = \"../input/mayo-clinic-strip-ai/train\"\ncsvPath = \"../input/mayo-clinic-strip-ai/train.csv\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(csvPath)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = []\nlabels = []","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The function to load high high memory file by downsampling.\ndef load_scaled_down_slide(image_path, downsample_by=10, to_numpy=True):\n    return pyvips.Image.new_from_file(image_path).resize(1/downsample_by).numpy()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(\"./512x512_Images\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# SAVE NEW DATASET","metadata":{}},{"cell_type":"markdown","source":"The image is split into two based on their size and resized.","metadata":{}},{"cell_type":"code","source":"for i, (Id, label) in enumerate(zip(df.image_id.values, df.label.values)):\n    fullImage = load_scaled_down_slide(f\"{trainImgDir}/{str(Id)}.tif\")\n    h, w, c = fullImage.shape\n    for i in range(0, 2):\n        fileName = f\"{str(Id)}_{i}.png\"\n        labels.append(label)\n        ids.append(fileName)\n        \n        if round(h/w) == 2:\n            cv2.imwrite(\"./512x512_Images/\"+fileName, tf.image.resize(fullImage[(h//2)*i:(h//2)*(i+1), :, :], (512,512)).numpy().astype('uint8'))\n        elif round(h/w) == 0:\n            cv2.imwrite(\"./512x512_Images/\"+fileName, tf.image.resize(fullImage[:, (h//2)*i:(h//2)*(i+1), :], (512,512)).numpy().astype('uint8'))\n        else:\n            cv2.imwrite(\"./512x512_Images/\"+fileName, tf.image.resize(fullImage, (512,512)).numpy().astype('uint8'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save new csv file.\n\nnewDf = pd.DataFrame({\"image_id\":ids, \"label\":labels})\nnewDf.to_csv(\"resizedTrain.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The dataset link is [here](http://)**","metadata":{}}]}