{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- Verilerin okunması\n- Preprocessing\n    . Windowing\n    . ....\n- Data generatorların oluşturulması\n- Model oluşturma\n- Eğitim\n- Tahmin\n    . .....\n","metadata":{}},{"cell_type":"code","source":"pip install pynrrd\npip install keras_applications","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:32:40.230104Z","iopub.execute_input":"2023-09-05T10:32:40.230453Z","iopub.status.idle":"2023-09-05T10:32:51.468331Z","shell.execute_reply.started":"2023-09-05T10:32:40.230427Z","shell.execute_reply":"2023-09-05T10:32:51.467118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pydicom\nimport os\nimport matplotlib.pyplot as plt\nimport collections\nfrom tqdm import tqdm_notebook as tqdm\nfrom datetime import datetime\nimport seaborn as sns\nimport nrrd\nfrom math import ceil, floor, log\nimport cv2\nimport tensorflow as tf\nimport keras\nimport sys\n# from keras_applications.resnet import ResNet50\nfrom keras_applications.inception_v3 import InceptionV3\nfrom sklearn.model_selection import ShuffleSplit\nfrom sklearn.metrics import confusion_matrix, classification_report, multilabel_confusion_matrix\nfrom keras import backend as K","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:17:05.165093Z","iopub.execute_input":"2023-09-05T11:17:05.165509Z","iopub.status.idle":"2023-09-05T11:17:05.17322Z","shell.execute_reply.started":"2023-09-05T11:17:05.16548Z","shell.execute_reply":"2023-09-05T11:17:05.171464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Verilerin Okunması","metadata":{}},{"cell_type":"code","source":"input_path = \"/kaggle/input/rsna-intracranial-hemorrhage-detection/rsna-intracranial-hemorrhage-detection/\"\ntest_images_dir = input_path + 'stage_2_test/'\ntrain_images_dir = input_path + 'stage_2_train/'","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:36:12.302941Z","iopub.execute_input":"2023-09-05T10:36:12.303279Z","iopub.status.idle":"2023-09-05T10:36:12.308088Z","shell.execute_reply.started":"2023-09-05T10:36:12.303253Z","shell.execute_reply":"2023-09-05T10:36:12.306976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Ön işleme","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/code/reppic/gradient-sigmoid-windowing\ndef correct_dcm(dcm):\n    x = dcm.pixel_array + 1000\n    px_mode = 4096\n    x[x>=px_mode] = x[x>=px_mode] - px_mode\n    dcm.PixelData = x.tobytes()\n    dcm.RescaleIntercept = -1000\n\ndef window_image(dcm, window_center, window_width):\n    if (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100):\n        correct_dcm(dcm)\n    img = dcm.pixel_array * dcm.RescaleSlope + dcm.RescaleIntercept\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    img = np.clip(img, img_min, img_max)\n\n    return img\n\ndef bsb_window(dcm):\n    brain_img = window_image(dcm, 40, 80)\n    subdural_img = window_image(dcm, 80, 200)\n    soft_img = window_image(dcm, 40, 380)\n    \n    brain_img = (brain_img - 0) / 80\n    subdural_img = (subdural_img - (-20)) / 200\n    soft_img = (soft_img - (-150)) / 380\n    bsb_img = np.array([brain_img, subdural_img, soft_img]).transpose(1,2,0)\n\n    return bsb_img","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:48:19.972664Z","iopub.execute_input":"2023-09-05T10:48:19.973019Z","iopub.status.idle":"2023-09-05T10:48:19.981283Z","shell.execute_reply.started":"2023-09-05T10:48:19.972991Z","shell.execute_reply":"2023-09-05T10:48:19.980239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom = pydicom.dcmread(train_images_dir + 'ID_5c8b5d701' + '.dcm')\n#                                     ID  Label\n# 4045566          ID_5c8b5d701_epidural      0\n# 4045567  ID_5c8b5d701_intraparenchymal      1\n# 4045568  ID_5c8b5d701_intraventricular      0\n# 4045569      ID_5c8b5d701_subarachnoid      1\n# 4045570          ID_5c8b5d701_subdural      1\n# 4045571               ID_5c8b5d701_any      1\nfig, ax = plt.subplots(1,2)\nax[0].imshow(dicom.pixel_array)\nax[0].set_title(\"orjinal\")\nax[1].imshow(bsb_window(dicom), cmap=plt.cm.bone)\nax[1].set_title(\"windowing\")","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:48:25.708812Z","iopub.execute_input":"2023-09-05T10:48:25.709205Z","iopub.status.idle":"2023-09-05T10:48:26.130902Z","shell.execute_reply.started":"2023-09-05T10:48:25.709175Z","shell.execute_reply":"2023-09-05T10:48:26.130063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Corrections çalıştığı kontrol ediliyor","metadata":{}},{"cell_type":"code","source":"def window_without_correction(dcm, window_center, window_width):\n    img = dcm.pixel_array * dcm.RescaleSlope + dcm.RescaleIntercept\n    img_min = window_center - window_width // 2\n    img_max = window_center + window_width // 2\n    img = np.clip(img, img_min, img_max)\n    return img\n\ndef window_testing(img, window):\n    brain_img = window(img, 40, 80)\n    subdural_img = window(img, 80, 200)\n    soft_img = window(img, 40, 380)\n    \n    brain_img = (brain_img - 0) / 80\n    subdural_img = (subdural_img - (-20)) / 200\n    soft_img = (soft_img - (-150)) / 380\n    bsb_img = np.array([brain_img, subdural_img, soft_img]).transpose(1,2,0)\n\n    return bsb_img\n","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:55:49.931362Z","iopub.execute_input":"2023-09-05T10:55:49.931764Z","iopub.status.idle":"2023-09-05T10:55:49.939374Z","shell.execute_reply.started":"2023-09-05T10:55:49.931733Z","shell.execute_reply":"2023-09-05T10:55:49.938077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example of a \"bad data point\" (i.e. (dcm.BitsStored == 12) and (dcm.PixelRepresentation == 0) and (int(dcm.RescaleIntercept) > -100) == True)\n\ndicom = pydicom.dcmread(train_images_dir + \"ID_036db39b7\" + \".dcm\")\n\nfig, ax = plt.subplots(1, 2)\n\nax[0].imshow(window_testing(dicom, window_without_correction), cmap=plt.cm.bone);\nax[0].set_title(\"original\")\nax[1].imshow(window_testing(dicom, window_image), cmap=plt.cm.bone);\nax[1].set_title(\"corrected\");","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:55:52.402572Z","iopub.execute_input":"2023-09-05T10:55:52.402903Z","iopub.status.idle":"2023-09-05T10:55:52.84238Z","shell.execute_reply.started":"2023-09-05T10:55:52.402878Z","shell.execute_reply":"2023-09-05T10:55:52.841162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### inception v3 için bsb_window kullanılarak görüntüleri 3 kanallı dicom'lara dönüştürüyor, resize yapıyor ve okuyor","metadata":{}},{"cell_type":"code","source":"def _read(path, desired_size):\n    \"\"\"Will be used in DataGenerator\"\"\"\n    \n    dcm = pydicom.dcmread(path)\n    \n    try:\n        img = bsb_window(dcm)\n    except:\n        img = np.zeros(desired_size)\n    img = cv2.resize(img, desired_size[:2], interpolation=cv2.INTER_LINEAR)\n    return img\n\n# Another sanity check \nplt.imshow(\n    _read(train_images_dir+'ID_5c8b5d701'+'.dcm', (128, 128)), cmap=plt.cm.bone)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T10:58:40.347033Z","iopub.execute_input":"2023-09-05T10:58:40.347407Z","iopub.status.idle":"2023-09-05T10:58:40.61429Z","shell.execute_reply.started":"2023-09-05T10:58:40.347378Z","shell.execute_reply":"2023-09-05T10:58:40.613203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data generators","metadata":{}},{"cell_type":"code","source":"class DataGenerator(keras.utils.Sequence):\n\n    def __init__(self, list_IDs, labels=None, batch_size=1, img_size=(512, 512, 1), \n                 img_dir=train_images_dir, *args, **kwargs):\n\n        self.list_IDs = list_IDs\n        self.labels = labels\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.img_dir = img_dir\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(ceil(len(self.indices) / self.batch_size))\n\n    def __getitem__(self, index):\n        indices = self.indices[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indices]\n        \n        if self.labels is not None:\n            X, Y = self.__data_generation(list_IDs_temp)\n            return X, Y\n        else:\n            X = self.__data_generation(list_IDs_temp)\n            return X\n        \n    def on_epoch_end(self):\n        if self.labels is not None: # for training phase we undersample and shuffle\n            # keep probability of any=0 and any=1\n            keep_prob = self.labels.iloc[:, 0].map({0: 0.35, 1: 0.5})\n            keep = (keep_prob > np.random.rand(len(keep_prob)))\n            self.indices = np.arange(len(self.list_IDs))[keep]\n            np.random.shuffle(self.indices)\n        else:\n            self.indices = np.arange(len(self.list_IDs))\n\n    def __data_generation(self, list_IDs_temp):\n        X = np.empty((self.batch_size, *self.img_size))\n        \n        if self.labels is not None: # training phase\n            Y = np.empty((self.batch_size, 6), dtype=np.float32)\n        \n            for i, ID in enumerate(list_IDs_temp):\n                X[i,] = _read(self.img_dir + ID + \".dcm\", self.img_size)\n                Y[i,] = self.labels.loc[ID].values\n        \n            return X, Y\n        \n        else: # test phase\n            for i, ID in enumerate(list_IDs_temp):\n                X[i,] = _read(self.img_dir+ID+\".dcm\", self.img_size)\n            \n            return X","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:14:26.09573Z","iopub.execute_input":"2023-09-05T11:14:26.096081Z","iopub.status.idle":"2023-09-05T11:14:26.108572Z","shell.execute_reply.started":"2023-09-05T11:14:26.096053Z","shell.execute_reply":"2023-09-05T11:14:26.107342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Loss function ve metrik","metadata":{}},{"cell_type":"code","source":"def weighted_log_loss(y_true, y_pred):\n    \"\"\"\n    Can be used as the loss function in model.compile()\n    ---------------------------------------------------\n    \"\"\"\n    \n    class_weights = np.array([2., 1., 1., 1., 1., 1.])\n    eps = K.epsilon()\n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n\n    out = -(         y_true  * K.log(      y_pred) * class_weights\n            + (1.0 - y_true) * K.log(1.0 - y_pred) * class_weights)\n    \n    return K.mean(out, axis=-1)\n\n\ndef _normalized_weighted_average(arr, weights=None):\n    \"\"\"\n    A simple Keras implementation that mimics that of \n    numpy.average(), specifically for this competition\n    \"\"\"\n    \n    if weights is not None:\n        scl = K.sum(weights)\n        weights = K.expand_dims(weights, axis=1)\n        return K.sum(K.dot(arr, weights), axis=1) / scl\n    return K.mean(arr, axis=1)\n\n\ndef weighted_loss(y_true, y_pred):\n    \"\"\"\n    Will be used as the metric in model.compile()\n    ---------------------------------------------\n    \n    Similar to the custom loss function 'weighted_log_loss()' above\n    but with normalized weights, which should be very similar \n    to the official competition metric:\n        https://www.kaggle.com/kambarakun/lb-probe-weights-n-of-positives-scoring\n    and hence:\n        sklearn.metrics.log_loss with sample weights\n    \"\"\"\n    \n    class_weights = K.variable([2., 1., 1., 1., 1., 1.])\n    \n    eps = K.epsilon()\n    \n    y_pred = K.clip(y_pred, eps, 1.0-eps)\n\n    loss = -(        y_true  * K.log(      y_pred)\n            + (1.0 - y_true) * K.log(1.0 - y_pred))\n    \n    loss_samples = _normalized_weighted_average(loss, class_weights)\n    \n    return K.mean(loss_samples)\n\n\ndef weighted_log_loss_metric(trues, preds):\n    \"\"\"\n    Will be used to calculate the log loss \n    of the validation set in PredictionCheckpoint()\n    ------------------------------------------\n    \"\"\"\n    class_weights = [2., 1., 1., 1., 1., 1.]\n    \n    epsilon = 1e-7\n    \n    preds = np.clip(preds, epsilon, 1-epsilon)\n    loss = trues * np.log(preds) + (1 - trues) * np.log(1 - preds)\n    loss_samples = np.average(loss, axis=1, weights=class_weights)\n\n    return - loss_samples.mean()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:38:15.425705Z","iopub.execute_input":"2023-09-05T11:38:15.426055Z","iopub.status.idle":"2023-09-05T11:38:15.43762Z","shell.execute_reply.started":"2023-09-05T11:38:15.426032Z","shell.execute_reply":"2023-09-05T11:38:15.43609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model oluşturma","metadata":{}},{"cell_type":"code","source":"class PredictionCheckpoint(keras.callbacks.Callback):\n    \n    def __init__(self, test_df, valid_df, \n                 test_images_dir=test_images_dir, \n                 valid_images_dir=train_images_dir, \n                 batch_size=32, input_size=(224, 224, 3)):\n        \n        self.test_df = test_df\n        self.valid_df = valid_df\n        self.test_images_dir = test_images_dir\n        self.valid_images_dir = valid_images_dir\n        self.batch_size = batch_size\n        self.input_size = input_size\n        \n    def on_train_begin(self, logs={}):\n        self.test_predictions = []\n        self.valid_predictions = []\n        \n    def on_epoch_end(self,batch, logs={}):\n        self.test_predictions.append(\n            self.model.predict_generator(\n                DataGenerator(self.test_df.index, None, self.batch_size, self.input_size, self.test_images_dir), verbose=2)[:len(self.test_df)])","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:27:06.336876Z","iopub.execute_input":"2023-09-05T11:27:06.337263Z","iopub.status.idle":"2023-09-05T11:27:06.345695Z","shell.execute_reply.started":"2023-09-05T11:27:06.337235Z","shell.execute_reply":"2023-09-05T11:27:06.344689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class MyDeepModel:\n    def __init__(self, engine, input_dims, batch_size=5, num_epochs=4, learning_rate=1e-3, \n                 decay_rate=1.0, decay_steps=1, weights=\"imagenet\", verbose=1):\n        \n        self.engine = engine\n        self.input_dims = input_dims\n        self.batch_size = batch_size\n        self.num_epochs = num_epochs\n        self.learning_rate = learning_rate\n        self.decay_rate = decay_rate\n        self.decay_steps = decay_steps\n        self.weights = weights\n        self.verbose = verbose\n        self._build()\n\n    def _build(self):\n        \n        \n        engine = self.engine(include_top=False, weights=self.weights, input_shape=self.input_dims,\n                             backend = keras.backend, layers = keras.layers,\n                             models = keras.models, utils = keras.utils)\n        \n        x = keras.layers.GlobalAveragePooling2D(name='avg_pool')(engine.output)\n#         x = keras.layers.Dropout(0.2)(x)\n#         x = keras.layers.Dense(keras.backend.int_shape(x)[1], activation=\"relu\", name=\"dense_hidden_1\")(x)\n#         x = keras.layers.Dropout(0.1)(x)\n        out = keras.layers.Dense(6, activation=\"sigmoid\", name='dense_output')(x)\n\n        self.model = keras.models.Model(inputs=engine.input, outputs=out)\n\n        self.model.compile(loss=\"binary_crossentropy\", optimizer=keras.optimizers.Adam(), metrics=[weighted_loss])\n    \n\n    def fit_and_predict(self, train_df, valid_df, test_df):\n        \n        # callbacks\n        pred_history = PredictionCheckpoint(test_df, valid_df, input_size=self.input_dims)\n        #checkpointer = keras.callbacks.ModelCheckpoint(filepath='%s-{epoch:02d}.hdf5' % self.engine.__name__, verbose=1, save_weights_only=True, save_best_only=False)\n        scheduler = keras.callbacks.LearningRateScheduler(lambda epoch: self.learning_rate * pow(self.decay_rate, floor(epoch / self.decay_steps)))\n        \n        self.model.fit_generator(\n            DataGenerator(\n                train_df.index, \n                train_df, \n                self.batch_size, \n                self.input_dims, \n                train_images_dir\n            ),\n            epochs=self.num_epochs,\n            verbose=self.verbose,\n            use_multiprocessing=True,\n            workers=4,\n            callbacks=[pred_history, scheduler]\n        )\n        \n        return pred_history\n    \n    def save(self, path):\n        self.model.save_weights(path)\n    \n    def load(self, path):\n        self.model.load_weights(path)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:28:41.457263Z","iopub.execute_input":"2023-09-05T11:28:41.457648Z","iopub.status.idle":"2023-09-05T11:28:41.469413Z","shell.execute_reply.started":"2023-09-05T11:28:41.457622Z","shell.execute_reply":"2023-09-05T11:28:41.468386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### .csv dosyalarını okuma","metadata":{}},{"cell_type":"code","source":"def read_testset(filename=input_path+\"stage_2_sample_submission.csv\"):\n    df = pd.read_csv(filename)\n    df[\"Image\"] = df[\"ID\"].str.slice(stop=12)\n    df[\"Diagnosis\"] = df[\"ID\"].str.slice(start=13)\n    \n    df = df.loc[:, [\"Label\", \"Diagnosis\", \"Image\"]]\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    \n    return df\n\ndef read_trainset(filename=input_path+\"stage_2_train.csv\"):\n    df = pd.read_csv(filename)\n    df[\"Image\"] = df[\"ID\"].str.slice(stop=12)\n    df[\"Diagnosis\"] = df[\"ID\"].str.slice(start=13)\n    \n    duplicates_to_remove = [\n        56346, 56347, 56348, 56349,\n        56350, 56351, 1171830, 1171831,\n        1171832, 1171833, 1171834, 1171835,\n        3705312, 3705313, 3705314, 3705315,\n        3705316, 3705317, 3842478, 3842479,\n        3842480, 3842481, 3842482, 3842483\n    ]\n    \n    df = df.drop(index=duplicates_to_remove)\n    df = df.reset_index(drop=True)\n    \n    df = df.loc[:, [\"Label\", \"Diagnosis\", \"Image\"]]\n    df = df.set_index(['Image', 'Diagnosis']).unstack(level=-1)\n    \n    return df\n\n    \ntest_df = read_testset()\ndf = read_trainset()","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:37:41.466189Z","iopub.execute_input":"2023-09-05T11:37:41.466572Z","iopub.status.idle":"2023-09-05T11:37:53.729028Z","shell.execute_reply.started":"2023-09-05T11:37:41.466544Z","shell.execute_reply":"2023-09-05T11:37:53.72808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train model and predict","metadata":{}},{"cell_type":"code","source":"# train set (00%) and validation set (10%)\nss = ShuffleSplit(n_splits=10, test_size=0.1, random_state=42).split(df.index)\n\n# lets go for the first fold only\ntrain_idx, valid_idx = next(ss)\n\n# obtain model\nmodel = MyDeepModel(engine=InceptionV3, input_dims=(256, 256, 3), batch_size=32, learning_rate=5e-4,\n                    num_epochs=5, decay_rate=0.8, decay_steps=1, weights=\"imagenet\", verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:38:27.895981Z","iopub.execute_input":"2023-09-05T11:38:27.896331Z","iopub.status.idle":"2023-09-05T11:38:30.32381Z","shell.execute_reply.started":"2023-09-05T11:38:27.896285Z","shell.execute_reply":"2023-09-05T11:38:30.322669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_and_predict(df.iloc[train_idx], df.iloc[valid_idx], test_df)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:40:11.688793Z","iopub.execute_input":"2023-09-05T11:40:11.689147Z","iopub.status.idle":"2023-09-05T11:40:18.564087Z","shell.execute_reply.started":"2023-09-05T11:40:11.689121Z","shell.execute_reply":"2023-09-05T11:40:18.562385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load(\"/kaggle/input/test-weight/weights\")","metadata":{"execution":{"iopub.status.busy":"2023-09-05T11:41:44.494395Z","iopub.execute_input":"2023-09-05T11:41:44.494713Z","iopub.status.idle":"2023-09-05T11:41:44.572115Z","shell.execute_reply.started":"2023-09-05T11:41:44.49469Z","shell.execute_reply":"2023-09-05T11:41:44.570963Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}