{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.250073,"end_time":"2022-08-08T16:15:37.006923","exception":false,"start_time":"2022-08-08T16:15:36.75685","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-11T15:01:25.776596Z","iopub.execute_input":"2022-08-11T15:01:25.777367Z","iopub.status.idle":"2022-08-11T15:01:25.965539Z","shell.execute_reply.started":"2022-08-11T15:01:25.777268Z","shell.execute_reply":"2022-08-11T15:01:25.964153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset=pd.read_csv('../input/mayo-clinic-strip-ai/train.csv')","metadata":{"papermill":{"duration":0.029138,"end_time":"2022-08-08T16:15:37.044281","exception":false,"start_time":"2022-08-08T16:15:37.015143","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:25.967686Z","iopub.execute_input":"2022-08-11T15:01:25.968887Z","iopub.status.idle":"2022-08-11T15:01:25.985153Z","shell.execute_reply.started":"2022-08-11T15:01:25.968837Z","shell.execute_reply":"2022-08-11T15:01:25.984183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset.describe()","metadata":{"papermill":{"duration":0.058175,"end_time":"2022-08-08T16:15:37.109091","exception":false,"start_time":"2022-08-08T16:15:37.050916","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:25.987677Z","iopub.execute_input":"2022-08-11T15:01:25.988165Z","iopub.status.idle":"2022-08-11T15:01:26.027943Z","shell.execute_reply.started":"2022-08-11T15:01:25.988117Z","shell.execute_reply":"2022-08-11T15:01:26.027046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_dataset.head()","metadata":{"papermill":{"duration":0.024386,"end_time":"2022-08-08T16:15:37.14082","exception":false,"start_time":"2022-08-08T16:15:37.116434","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:26.030518Z","iopub.execute_input":"2022-08-11T15:01:26.030881Z","iopub.status.idle":"2022-08-11T15:01:26.043513Z","shell.execute_reply.started":"2022-08-11T15:01:26.03085Z","shell.execute_reply":"2022-08-11T15:01:26.042534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset=pd.concat([text_dataset,pd.get_dummies(text_dataset['label'])], axis=1)\nfrom sklearn.preprocessing import LabelEncoder\nbi=LabelEncoder()\npoint=bi.fit_transform(text_dataset['label'])\ndataset=dataset.drop(['label', 'CE', 'LAA'], axis=1)","metadata":{"papermill":{"duration":1.279725,"end_time":"2022-08-08T16:15:38.427405","exception":false,"start_time":"2022-08-08T16:15:37.14768","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:26.044739Z","iopub.execute_input":"2022-08-11T15:01:26.045686Z","iopub.status.idle":"2022-08-11T15:01:27.174989Z","shell.execute_reply.started":"2022-08-11T15:01:26.045635Z","shell.execute_reply":"2022-08-11T15:01:27.173421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset=dataset.assign(label=point)\ndataset","metadata":{"papermill":{"duration":0.029107,"end_time":"2022-08-08T16:15:38.46316","exception":false,"start_time":"2022-08-08T16:15:38.434053","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:27.176635Z","iopub.execute_input":"2022-08-11T15:01:27.177755Z","iopub.status.idle":"2022-08-11T15:01:27.199403Z","shell.execute_reply.started":"2022-08-11T15:01:27.177687Z","shell.execute_reply":"2022-08-11T15:01:27.197985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nplt.title('Labels')\nplt.pie(text_dataset['label'].value_counts(), labels=text_dataset['label'].unique())\nplt.show()","metadata":{"papermill":{"duration":0.360384,"end_time":"2022-08-08T16:15:38.830639","exception":false,"start_time":"2022-08-08T16:15:38.470255","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:27.201594Z","iopub.execute_input":"2022-08-11T15:01:27.202104Z","iopub.status.idle":"2022-08-11T15:01:27.552542Z","shell.execute_reply.started":"2022-08-11T15:01:27.202029Z","shell.execute_reply":"2022-08-11T15:01:27.549715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(text_dataset['center_id'], kind='kde')\nsns.pairplot(text_dataset)","metadata":{"papermill":{"duration":1.295245,"end_time":"2022-08-08T16:15:40.143358","exception":false,"start_time":"2022-08-08T16:15:38.848113","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:27.55464Z","iopub.execute_input":"2022-08-11T15:01:27.555801Z","iopub.status.idle":"2022-08-11T15:01:28.823399Z","shell.execute_reply.started":"2022-08-11T15:01:27.555734Z","shell.execute_reply":"2022-08-11T15:01:28.822202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax=[]\nfor i in range(1, 12):\n    ax.append('ax'+str(i))\nfig, ax=plt.subplots(1,5, sharey=True)\nfor i in range(0, 5):\n    ax[i].set_title('center_id'+str(i+1))\n    ax[i].bar(text_dataset[text_dataset['center_id']==i+1]['label'].unique(), text_dataset[text_dataset['center_id']==i+1]['label'].value_counts())\nplt.show()","metadata":{"papermill":{"duration":0.44922,"end_time":"2022-08-08T16:15:40.600961","exception":false,"start_time":"2022-08-08T16:15:40.151741","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:28.824873Z","iopub.execute_input":"2022-08-11T15:01:28.825951Z","iopub.status.idle":"2022-08-11T15:01:29.274553Z","shell.execute_reply.started":"2022-08-11T15:01:28.825911Z","shell.execute_reply":"2022-08-11T15:01:29.273695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(text_dataset['center_id'])","metadata":{"papermill":{"duration":0.339811,"end_time":"2022-08-08T16:15:40.949291","exception":false,"start_time":"2022-08-08T16:15:40.60948","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:29.27751Z","iopub.execute_input":"2022-08-11T15:01:29.278592Z","iopub.status.idle":"2022-08-11T15:01:29.563415Z","shell.execute_reply.started":"2022-08-11T15:01:29.278554Z","shell.execute_reply":"2022-08-11T15:01:29.562115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(text_dataset['label'])","metadata":{"papermill":{"duration":0.204686,"end_time":"2022-08-08T16:15:41.162487","exception":false,"start_time":"2022-08-08T16:15:40.957801","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:29.565008Z","iopub.execute_input":"2022-08-11T15:01:29.566304Z","iopub.status.idle":"2022-08-11T15:01:29.770408Z","shell.execute_reply.started":"2022-08-11T15:01:29.566252Z","shell.execute_reply":"2022-08-11T15:01:29.769275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create folder for files path**","metadata":{}},{"cell_type":"code","source":"train_folder=[]\nfor i in text_dataset['image_id']:\n    train_folder.append(i)","metadata":{"papermill":{"duration":0.025221,"end_time":"2022-08-08T16:15:41.943751","exception":false,"start_time":"2022-08-08T16:15:41.91853","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:29.771633Z","iopub.execute_input":"2022-08-11T15:01:29.774847Z","iopub.status.idle":"2022-08-11T15:01:29.78098Z","shell.execute_reply.started":"2022-08-11T15:01:29.774809Z","shell.execute_reply":"2022-08-11T15:01:29.77965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Images are to heavy to open with cv2 or PIL, so we will use OpenSlide**","metadata":{}},{"cell_type":"code","source":"import openslide\nfrom openslide import OpenSlide\nslide = OpenSlide('../input/mayo-clinic-strip-ai/train/006388_0.tif') # opening a full slide\n\nregion = (0, 0) # location of the top left pixel\nlevel = 0 # level of the picture (we have only 0)\nsize = (10000, 10000) # region size in pixels\n\nregion = slide.read_region(region, level, size)\n\nplt.figure(figsize=(8, 8))\nplt.imshow(region)\nplt.show()","metadata":{"papermill":{"duration":30.146154,"end_time":"2022-08-08T16:16:12.098858","exception":false,"start_time":"2022-08-08T16:15:41.952704","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:01:29.783045Z","iopub.execute_input":"2022-08-11T15:01:29.783838Z","iopub.status.idle":"2022-08-11T15:02:00.680592Z","shell.execute_reply.started":"2022-08-11T15:01:29.78379Z","shell.execute_reply":"2022-08-11T15:02:00.679357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create target matrix**","metadata":{}},{"cell_type":"code","source":"target = pd.get_dummies(text_dataset['label'])[:7:]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:02:00.681917Z","iopub.execute_input":"2022-08-11T15:02:00.682907Z","iopub.status.idle":"2022-08-11T15:02:00.689445Z","shell.execute_reply.started":"2022-08-11T15:02:00.682866Z","shell.execute_reply":"2022-08-11T15:02:00.688347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Than we will create our tensor files as follows**","metadata":{}},{"cell_type":"code","source":"train_path='../input/mayo-clinic-strip-ai/train/'\ntrain_dataset_images=[]\n#target = dataset['label'][:7:].to_list()\ndef return_train_images(train_dataset_images, train_path):\n    for i in train_folder[:7:]:\n        slide=OpenSlide(train_path+i+'.tif')\n        region=(0,0)\n        level=0\n        size=(10000, 10000)\n        train_dataset_images.append(np.array(slide.read_region(region, level, size)))\n    train_dataset_images=np.array(train_dataset_images)\n    return train_dataset_images","metadata":{"papermill":{"duration":119.520918,"end_time":"2022-08-08T16:18:11.635107","exception":false,"start_time":"2022-08-08T16:16:12.114189","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:02:00.69111Z","iopub.execute_input":"2022-08-11T15:02:00.692029Z","iopub.status.idle":"2022-08-11T15:02:00.703794Z","shell.execute_reply.started":"2022-08-11T15:02:00.691992Z","shell.execute_reply":"2022-08-11T15:02:00.702487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset_images=return_train_images(train_dataset_images, train_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:02:00.705222Z","iopub.execute_input":"2022-08-11T15:02:00.705765Z","iopub.status.idle":"2022-08-11T15:03:28.224744Z","shell.execute_reply.started":"2022-08-11T15:02:00.70573Z","shell.execute_reply":"2022-08-11T15:03:28.223576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Our model architechture**\n\n1. **Resize layer (to fit ResNet50 input)**\n2. **Convolutional layer for dimensional reduction**\n3. **ResNet50 pre-trained model**\n4. **Batchnorm layer to avoid overfitting**\n5. **Double points dense layer**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport keras.backend as K\nimport keras\nfrom keras.models import Model\nfrom keras.layers import Dense, BatchNormalization, Conv2D, Resizing\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom keras.callbacks import EarlyStopping\ninput_image = keras.Input(shape=(10000, 10000, 4))\nx = Resizing(224 ,224, interpolation=\"bilinear\", crop_to_aspect_ratio=False)(input_image)\nx = Conv2D(3, (1, 1))(x)\nx = ResNet50()(x)\nx = BatchNormalization()(x)\noutput_score = Dense(2, activation = 'sigmoid')(x)\nmodel = Model(input_image, output_score)","metadata":{"papermill":{"duration":16.002335,"end_time":"2022-08-08T16:18:27.653332","exception":false,"start_time":"2022-08-08T16:18:11.650997","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-08-11T15:03:28.226626Z","iopub.execute_input":"2022-08-11T15:03:28.227311Z","iopub.status.idle":"2022-08-11T15:03:43.495148Z","shell.execute_reply.started":"2022-08-11T15:03:28.227261Z","shell.execute_reply":"2022-08-11T15:03:43.493976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Validation and trainable data**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(train_dataset_images, target, shuffle=False, test_size=0.4)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:43.496874Z","iopub.execute_input":"2022-08-11T15:03:43.497536Z","iopub.status.idle":"2022-08-11T15:03:45.429983Z","shell.execute_reply.started":"2022-08-11T15:03:43.497504Z","shell.execute_reply":"2022-08-11T15:03:45.428795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**To integrate data to our model we firstly need to convert it to tf.tensor type**","metadata":{}},{"cell_type":"code","source":"def tf_tensor(X_train, X_test, y_train, y_test):\n    X_train=tf.convert_to_tensor(X_train)\n    y_train=tf.convert_to_tensor(y_train)\n    X_test=tf.convert_to_tensor(X_test)\n    y_test=tf.convert_to_tensor(y_test)\n    return X_train, X_test, y_train, y_test\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:45.431857Z","iopub.execute_input":"2022-08-11T15:03:45.432271Z","iopub.status.idle":"2022-08-11T15:03:45.439824Z","shell.execute_reply.started":"2022-08-11T15:03:45.432234Z","shell.execute_reply":"2022-08-11T15:03:45.43849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now we can compile & fit our model**","metadata":{}},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy', optimizer='adam')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:45.441518Z","iopub.execute_input":"2022-08-11T15:03:45.441866Z","iopub.status.idle":"2022-08-11T15:03:45.471125Z","shell.execute_reply.started":"2022-08-11T15:03:45.441833Z","shell.execute_reply":"2022-08-11T15:03:45.470153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test=tf_tensor(X_train, X_test, y_train, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:45.472661Z","iopub.execute_input":"2022-08-11T15:03:45.47373Z","iopub.status.idle":"2022-08-11T15:03:50.033235Z","shell.execute_reply.started":"2022-08-11T15:03:45.47368Z","shell.execute_reply":"2022-08-11T15:03:50.031321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#history=model.fit(X_train, y_train, epochs=3, batch_size=10, validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T15:03:50.035152Z","iopub.execute_input":"2022-08-11T15:03:50.035951Z","iopub.status.idle":"2022-08-11T15:03:50.07524Z","shell.execute_reply.started":"2022-08-11T15:03:50.035893Z","shell.execute_reply":"2022-08-11T15:03:50.074189Z"},"trusted":true},"execution_count":null,"outputs":[]}]}