{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-08T16:54:11.736833Z","iopub.execute_input":"2022-09-08T16:54:11.737292Z","iopub.status.idle":"2022-09-08T16:54:12.369512Z","shell.execute_reply.started":"2022-09-08T16:54:11.737201Z","shell.execute_reply":"2022-09-08T16:54:12.368493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport tensorflow as tf\nimport tensorflow.keras.preprocessing as image\nimport tensorflow_io as tfio\nfrom sklearn.model_selection import train_test_split\nimport cv2\nimport tifffile as tifi\nimport gc\nimport os\nimport openslide\nfrom openslide import OpenSlide\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:12.370834Z","iopub.execute_input":"2022-09-08T16:54:12.371149Z","iopub.status.idle":"2022-09-08T16:54:21.141791Z","shell.execute_reply.started":"2022-09-08T16:54:12.37112Z","shell.execute_reply":"2022-09-08T16:54:21.140697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_train_file(x):\n    return \"../input/mayo-clinic-strip-ai/train/\" + x + \".tif\"\n\ndef make_test_file(x):\n    return x + \".tif\"","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:21.143127Z","iopub.execute_input":"2022-09-08T16:54:21.14382Z","iopub.status.idle":"2022-09-08T16:54:21.151451Z","shell.execute_reply.started":"2022-09-08T16:54:21.143786Z","shell.execute_reply":"2022-09-08T16:54:21.150415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#This is where we will look for the csv data, because it's important to know what patients match with which images\n#Patients can have more than 1 image\ntrain = pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')\ntrain.head()\ntrain_data = pd.DataFrame({'image_id': train.image_id.apply(make_train_file), 'label': train.label})\nImage.MAX_IMAGE_PIXELS = 5_000_000_000","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:21.15377Z","iopub.execute_input":"2022-09-08T16:54:21.154843Z","iopub.status.idle":"2022-09-08T16:54:21.202374Z","shell.execute_reply.started":"2022-09-08T16:54:21.154808Z","shell.execute_reply":"2022-09-08T16:54:21.201197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:21.204053Z","iopub.execute_input":"2022-09-08T16:54:21.204814Z","iopub.status.idle":"2022-09-08T16:54:21.220592Z","shell.execute_reply.started":"2022-09-08T16:54:21.204777Z","shell.execute_reply":"2022-09-08T16:54:21.219788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '../input/mayo-clinic-strip-ai/train/'","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:21.316066Z","iopub.execute_input":"2022-09-08T16:54:21.31709Z","iopub.status.idle":"2022-09-08T16:54:21.32524Z","shell.execute_reply.started":"2022-09-08T16:54:21.317045Z","shell.execute_reply":"2022-09-08T16:54:21.324405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\n\nfrom tensorflow.keras.callbacks import EarlyStopping\nearly_stopping = EarlyStopping(\n    min_delta = 0.1,\n    patience = 5,\n    restore_best_weights = True\n)\n\nmodel = keras.Sequential([\n    layers.Conv2D(filters = 16, kernel_size = (3,3),input_shape = (1,128,128,4), padding = 'same', activation = 'relu'),\n    layers.Conv2D(filters = 32, kernel_size = (3,3), padding = 'same', activation = 'relu'),\n    layers.Conv2D(filters = 64, kernel_size = (3,3), padding = 'same', activation = 'relu'),\n    layers.Flatten(),\n    layers.Dense(128, activation='relu'),\n    layers.Dropout(0.25),\n    layers.Dense(2),\n])\n\n\n\nmodel.compile(\n    optimizer='adam',\n    loss='mean_squared_error',\n    metrics=['accuracy'],\n)\n\nfor x in range(int(train_data.size / 2)):\n    img_path = train_data.image_id[x]\n    img_tmp = OpenSlide(img_path)\n    region= (1000,1000)    \n    size  = (5000, 5000)\n    img_tmp = img_tmp.read_region(region, 0, size)\n    img_tmp = tf.image.resize(img_tmp, (128, 128))\n    img_tmp = img_tmp/255\n    img_tmp = np.reshape(img_tmp, [1, 128, 128, 4])\n    label = train_data.label[x]\n    if(label == \"CE\"):\n        label = 0\n    elif(label == \"LAA\"):\n        label = 1\n    \n    history = model.fit(x = img_tmp,y = np.array([label]), callbacks = [early_stopping])\n    \n    print(x)\n    del img_tmp  # to free memory\n    gc.collect() # to free memory\n    print(gc.collect())\n","metadata":{"execution":{"iopub.status.busy":"2022-09-08T16:54:21.457285Z","iopub.execute_input":"2022-09-08T16:54:21.457723Z","iopub.status.idle":"2022-09-08T17:49:22.068091Z","shell.execute_reply.started":"2022-09-08T16:54:21.457683Z","shell.execute_reply":"2022-09-08T17:49:22.065702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n\n# history_frame = pd.DataFrame(history.history)\n# history_frame.loc[:, ['loss', 'val_loss']].plot()\n# history_frame.loc[:, ['accuracy', 'val_accuracy']].plot();","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:22.076381Z","iopub.execute_input":"2022-09-08T17:49:22.076808Z","iopub.status.idle":"2022-09-08T17:49:22.082503Z","shell.execute_reply.started":"2022-09-08T17:49:22.076773Z","shell.execute_reply":"2022-09-08T17:49:22.081001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\nimport os\nimport cv2\nimport tifffile as tifi\n\npath = '../input/mayo-clinic-strip-ai/test/'\ntest = pd.read_csv('../input/mayo-clinic-strip-ai/test.csv')\ntest_data = pd.DataFrame({'image_id': test.image_id.apply(make_test_file)})\n\nimage_size_height = 128\nimage_size_width  = 128\n\n\npreds = []\nfor img_path in test_data.image_id:\n    img_tmp = OpenSlide(path + img_path)\n    region= (1000,1000)    \n    size  = (5000, 5000)\n    img_tmp = img_tmp.read_region(region, 0, size)\n    img_tmp = tf.image.resize(img_tmp, (128, 128))\n    img_tmp = img_tmp/255\n    img_tmp = np.reshape(img_tmp, [1, 128, 128, 4])\n    \n    pred_tmp = model.predict(img_tmp)\n    \n    preds.append(pred_tmp)\n    \n    del img_tmp  # to free memory\n    del pred_tmp # to free memory\n    gc.collect() # to free memory","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:22.084326Z","iopub.execute_input":"2022-09-08T17:49:22.084849Z","iopub.status.idle":"2022-09-08T17:49:43.939448Z","shell.execute_reply.started":"2022-09-08T17:49:22.0848Z","shell.execute_reply":"2022-09-08T17:49:43.938372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:43.941507Z","iopub.execute_input":"2022-09-08T17:49:43.941942Z","iopub.status.idle":"2022-09-08T17:49:43.959583Z","shell.execute_reply.started":"2022-09-08T17:49:43.941906Z","shell.execute_reply":"2022-09-08T17:49:43.958267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(preds)","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:43.973662Z","iopub.execute_input":"2022-09-08T17:49:43.974404Z","iopub.status.idle":"2022-09-08T17:49:43.992093Z","shell.execute_reply.started":"2022-09-08T17:49:43.974362Z","shell.execute_reply":"2022-09-08T17:49:43.990594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = pd.DataFrame(np.concatenate(preds))\n\nsubmission = pd.read_csv('../input/mayo-clinic-strip-ai/sample_submission.csv')\nsubmission.CE = preds.iloc[ : , : 1]\nsubmission.LAA = preds.iloc[ : , 1: 2]\n\nsubmission = submission.groupby(\"patient_id\").mean()\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:43.993965Z","iopub.execute_input":"2022-09-08T17:49:43.994518Z","iopub.status.idle":"2022-09-08T17:49:44.057545Z","shell.execute_reply.started":"2022-09-08T17:49:43.994467Z","shell.execute_reply":"2022-09-08T17:49:44.05613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[[\"patient_id\", \"CE\", \"LAA\"]].to_csv(\"submission.csv\", index=False)\n\n!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-09-08T17:49:44.059353Z","iopub.execute_input":"2022-09-08T17:49:44.059758Z","iopub.status.idle":"2022-09-08T17:49:45.36946Z","shell.execute_reply.started":"2022-09-08T17:49:44.059724Z","shell.execute_reply":"2022-09-08T17:49:45.367763Z"},"trusted":true},"execution_count":null,"outputs":[]}]}