{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%matplotlib inline\nimport numpy as np \nimport pandas as pd\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import Xception\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nimport tensorflow as tf\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:21:45.710721Z","iopub.execute_input":"2023-01-23T19:21:45.711102Z","iopub.status.idle":"2023-01-23T19:21:55.972368Z","shell.execute_reply.started":"2023-01-23T19:21:45.711073Z","shell.execute_reply":"2023-01-23T19:21:55.971273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -qU python-gdcm pydicom pylibjpeg","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:21:55.974445Z","iopub.execute_input":"2023-01-23T19:21:55.97511Z","iopub.status.idle":"2023-01-23T19:22:09.429069Z","shell.execute_reply.started":"2023-01-23T19:21:55.975074Z","shell.execute_reply":"2023-01-23T19:22:09.427878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport glob\nimport cv2\nimport seaborn as sns\n\n# To work with DICOM images\nimport gdcm\nimport pydicom\n\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:09.430746Z","iopub.execute_input":"2023-01-23T19:22:09.431135Z","iopub.status.idle":"2023-01-23T19:22:10.568699Z","shell.execute_reply.started":"2023-01-23T19:22:09.431104Z","shell.execute_reply":"2023-01-23T19:22:10.56776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rescale_img_to_hu(dcm_ds):\n    \"\"\"\n    Rescales the image to Hounsfield unit.\n    Thank you https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\n    \"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    \"\"\"\n    Thank you\n    https://www.kaggle.com/code/radek1/eda-training-a-fast-ai-model-submission\n    although with some changes.\n    \"\"\"\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    print(patient_dir)\n    num_images = len([name for name in os.listdir(patient_dir)])\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(2, 2, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_file in enumerate(os.listdir(patient_dir)):\n        img_path = os.path.join(patient_dir, img_file)\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")\n        # Break if there are more than 4 images for visualization purposes\n        if i==3: break","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:10.571392Z","iopub.execute_input":"2023-01-23T19:22:10.571782Z","iopub.status.idle":"2023-01-23T19:22:10.581038Z","shell.execute_reply.started":"2023-01-23T19:22:10.571745Z","shell.execute_reply":"2023-01-23T19:22:10.579522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pd = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_pd.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:10.5827Z","iopub.execute_input":"2023-01-23T19:22:10.583204Z","iopub.status.idle":"2023-01-23T19:22:10.750665Z","shell.execute_reply.started":"2023-01-23T19:22:10.583139Z","shell.execute_reply":"2023-01-23T19:22:10.749713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add path column for further data generator\ntrain_pd['path'] = '/kaggle/input/rsna-breast-cancer-256-pngs/' + train_pd['patient_id'].astype(str) + '_' + train_pd['image_id'].astype(str) + '.png'\n\n# Convert target to string\ntrain_pd['cancer'] = train_pd['cancer'].astype(str)\n\n# Convert target to onehot\nfrom sklearn.preprocessing import OneHotEncoder\nohe = OneHotEncoder()\ntrain_pd['cancer_one_hot'] = ohe.fit_transform(train_pd[['cancer']]).toarray().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:10.752176Z","iopub.execute_input":"2023-01-23T19:22:10.752576Z","iopub.status.idle":"2023-01-23T19:22:10.989583Z","shell.execute_reply.started":"2023-01-23T19:22:10.752538Z","shell.execute_reply":"2023-01-23T19:22:10.988574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imbalanced target\ntrain_pd['cancer'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:10.990993Z","iopub.execute_input":"2023-01-23T19:22:10.991344Z","iopub.status.idle":"2023-01-23T19:22:11.004318Z","shell.execute_reply.started":"2023-01-23T19:22:10.991312Z","shell.execute_reply":"2023-01-23T19:22:11.003207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Relationship cancer with implant\n# train_pd['implant'].value_counts()\ntrain_pd.groupby('implant')['cancer'].mean()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:11.006064Z","iopub.execute_input":"2023-01-23T19:22:11.006771Z","iopub.status.idle":"2023-01-23T19:22:11.07638Z","shell.execute_reply.started":"2023-01-23T19:22:11.006735Z","shell.execute_reply":"2023-01-23T19:22:11.075258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Target wrt difficult case --> More than 10% were difficult to diagnose negative\ntrain_pd[['cancer', 'difficult_negative_case']].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:11.078574Z","iopub.execute_input":"2023-01-23T19:22:11.079388Z","iopub.status.idle":"2023-01-23T19:22:11.108812Z","shell.execute_reply.started":"2023-01-23T19:22:11.079352Z","shell.execute_reply":"2023-01-23T19:22:11.107679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show me one image with cancer\npatient_id_with_cancer = train_pd[train_pd.cancer=='1']['patient_id'].iloc[0]\nshow_images_for_patient(10011)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:11.112931Z","iopub.execute_input":"2023-01-23T19:22:11.113258Z","iopub.status.idle":"2023-01-23T19:22:16.848714Z","shell.execute_reply.started":"2023-01-23T19:22:11.113229Z","shell.execute_reply":"2023-01-23T19:22:16.847719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Should we care about implants?\npatient_id_with_implants = train_pd[train_pd.implant==1]['patient_id'].iloc[0]\nshow_images_for_patient(patient_id_with_implants)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:16.849788Z","iopub.execute_input":"2023-01-23T19:22:16.850195Z","iopub.status.idle":"2023-01-23T19:22:23.19956Z","shell.execute_reply.started":"2023-01-23T19:22:16.85015Z","shell.execute_reply":"2023-01-23T19:22:23.198699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Knee code model \nxception = Xception(weights=\"imagenet\")#,input_shape=(256, 256, 3),\n   # include_top=False)\nx=  xception.layers[-3].output\n\nx = tf.keras.layers.Conv2D(filters= 1024, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 256, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 64, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 5, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nGAP = tf.keras.layers.GlobalAveragePooling2D()(x)\npred = tf.keras.activations.softmax(GAP)\n\nxception_model = Model(inputs=xception.input,outputs=pred)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:43:10.866467Z","iopub.execute_input":"2023-01-23T18:43:10.866967Z","iopub.status.idle":"2023-01-23T18:43:17.268492Z","shell.execute_reply.started":"2023-01-23T18:43:10.866928Z","shell.execute_reply":"2023-01-23T18:43:17.26752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = tf.keras.applications.Xception(\n    weights='imagenet',  # Load weights pre-trained on ImageNet.\n    input_shape=(256, 256, 3),\n    include_top=False\n)\n# Freeze the layers of the base model\nbase_model.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:23.201194Z","iopub.execute_input":"2023-01-23T19:22:23.201835Z","iopub.status.idle":"2023-01-23T19:22:29.936625Z","shell.execute_reply.started":"2023-01-23T19:22:23.201795Z","shell.execute_reply":"2023-01-23T19:22:29.935627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Input layer\ninputs = tf.keras.Input(shape=(256, 256, 3))\n\n# Base model layer\nx = base_model(inputs, training=False)\n\n#x=  base_model.layers[-3].output\n\nx = tf.keras.layers.Conv2D(filters= 1024, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 256, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 64, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n\nx = tf.keras.layers.Conv2D(filters= 5, kernel_size= 3, padding= \"same\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Activation(\"relu\")(x)\n# Pooling to reduce the number of dimensions\nx = tf.keras.layers.GlobalAveragePooling2D()(x)\n\n# Dense layer to learn new stuff\nx = tf.keras.layers.Dense(16, activation='relu')(x)\n\n# Output layer for loss\noutputs = tf.keras.layers.Dense(2, activation='softmax')(x)\n\n# All togeteher now... all togeeeeether\nmodel = tf.keras.Model(inputs, outputs)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:29.938259Z","iopub.execute_input":"2023-01-23T19:22:29.938665Z","iopub.status.idle":"2023-01-23T19:22:30.349935Z","shell.execute_reply.started":"2023-01-23T19:22:29.938614Z","shell.execute_reply":"2023-01-23T19:22:30.348968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:07:08.774949Z","iopub.execute_input":"2023-01-23T19:07:08.775335Z","iopub.status.idle":"2023-01-23T19:07:08.793706Z","shell.execute_reply.started":"2023-01-23T19:07:08.775303Z","shell.execute_reply":"2023-01-23T19:07:08.792543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#knee code compile\nxception_model.compile(optimizer = tf.keras.optimizers.Adam(learning_rate=0.00001,decay=0.0001),\n                 metrics=[\"acc\"],\n                 loss= tf.keras.losses.sparse_categorical_crossentropy)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T18:43:17.269684Z","iopub.execute_input":"2023-01-23T18:43:17.269988Z","iopub.status.idle":"2023-01-23T18:43:17.288545Z","shell.execute_reply.started":"2023-01-23T18:43:17.269961Z","shell.execute_reply":"2023-01-23T18:43:17.28732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    optimizer=tf.keras.optimizers.Adam(),\n    loss='categorical_crossentropy',\n    metrics=[tf.keras.metrics.BinaryAccuracy()]\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:30.35145Z","iopub.execute_input":"2023-01-23T19:22:30.351832Z","iopub.status.idle":"2023-01-23T19:22:30.370956Z","shell.execute_reply.started":"2023-01-23T19:22:30.351795Z","shell.execute_reply":"2023-01-23T19:22:30.370078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create data generator\nfrom keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    horizontal_flip=False,\n    vertical_flip=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:30.372429Z","iopub.execute_input":"2023-01-23T19:22:30.372828Z","iopub.status.idle":"2023-01-23T19:22:30.377984Z","shell.execute_reply.started":"2023-01-23T19:22:30.372795Z","shell.execute_reply":"2023-01-23T19:22:30.376898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ndf_train, df_val = train_test_split(train_pd, test_size=0.1, stratify=train_pd['cancer'])","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:30.379615Z","iopub.execute_input":"2023-01-23T19:22:30.379975Z","iopub.status.idle":"2023-01-23T19:22:30.538703Z","shell.execute_reply.started":"2023-01-23T19:22:30.379943Z","shell.execute_reply":"2023-01-23T19:22:30.537609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create training flow\ntrain_flow = datagen.flow_from_dataframe(\n    df_train,\n    x_col='path',\n    y_col='cancer_one_hot',\n    target_size=(256, 256),\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=32,\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:22:30.542092Z","iopub.execute_input":"2023-01-23T19:22:30.542826Z","iopub.status.idle":"2023-01-23T19:24:58.760385Z","shell.execute_reply.started":"2023-01-23T19:22:30.542774Z","shell.execute_reply":"2023-01-23T19:24:58.758423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create validation flow\nval_flow = datagen.flow_from_dataframe(\n    df_val,\n    x_col='path',\n    y_col='cancer_one_hot',\n    target_size=(256, 256),\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=32,\n    shuffle=True\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:24:58.762558Z","iopub.execute_input":"2023-01-23T19:24:58.763695Z","iopub.status.idle":"2023-01-23T19:25:15.59053Z","shell.execute_reply.started":"2023-01-23T19:24:58.763653Z","shell.execute_reply":"2023-01-23T19:25:15.58952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import class_weight\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                 classes= (0.0,1.0),\n                                                 y= df_train['cancer_one_hot'][0])\nclass_weights = dict(enumerate(class_weights))","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:25:15.593529Z","iopub.execute_input":"2023-01-23T19:25:15.594439Z","iopub.status.idle":"2023-01-23T19:25:15.602329Z","shell.execute_reply.started":"2023-01-23T19:25:15.5944Z","shell.execute_reply":"2023-01-23T19:25:15.601212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(\n    x=train_flow,\n    epochs=3,\n    validation_data=val_flow,\n   class_weight={0: 1, 1:10}    # Since the dataset is very imbalanced I gave some weights to hopefully help the NN a bit\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-23T19:25:15.603765Z","iopub.execute_input":"2023-01-23T19:25:15.604229Z","iopub.status.idle":"2023-01-23T19:46:00.972749Z","shell.execute_reply.started":"2023-01-23T19:25:15.60419Z","shell.execute_reply":"2023-01-23T19:46:00.971586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}