{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport cv2\ntarget_size = (224, 224)\nimport matplotlib.pyplot as plt\n# Define a function that takes a grayscale image and replicates the single channel into three channels\ndef extrect_rgb_image(x):\n    grayscale_image = tf.io.read_file(x)\n    grayscale_image = tf.image.decode_png(grayscale_image, channels=1)  # channels=1 to load as grayscale\n   # Convert the grayscale image to RGB\n    rgb_image = tf.image.grayscale_to_rgb(grayscale_image)\n    return  rgb_image.numpy()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:14:26.576719Z","iopub.execute_input":"2023-02-21T09:14:26.577115Z","iopub.status.idle":"2023-02-21T09:14:33.496876Z","shell.execute_reply.started":"2023-02-21T09:14:26.577027Z","shell.execute_reply":"2023-02-21T09:14:33.495922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.mkdir(\"/kaggle/working/data\")\nos.mkdir(\"/kaggle/working/data/cencer\")\nos.mkdir(\"/kaggle/working/data/noncencer\")\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:15:51.174233Z","iopub.execute_input":"2023-02-21T09:15:51.174929Z","iopub.status.idle":"2023-02-21T09:15:51.180865Z","shell.execute_reply.started":"2023-02-21T09:15:51.174891Z","shell.execute_reply":"2023-02-21T09:15:51.179883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.image as mpimg\n\nfrom PIL import Image\nimport numpy as np\ndef copy_images_convert_rgb(source_path,dest_path,image_number=1000):\n      # load images into list for creation\n        i=0\n        for img in os.listdir(source_path):\n            path_ = os.path.join(source_path,img)\n            img = extrect_rgb_image(path_)\n            cv2.imwrite(dest_path+str(i)+\".png\",img)\n            if i==image_number:\n                break\n            i=i+1","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:15:54.998119Z","iopub.execute_input":"2023-02-21T09:15:54.998498Z","iopub.status.idle":"2023-02-21T09:15:55.005299Z","shell.execute_reply.started":"2023-02-21T09:15:54.998444Z","shell.execute_reply":"2023-02-21T09:15:55.004137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"copy_images_convert_rgb(source_path=\"/kaggle/input/rsna-crpped-images/cancer\",dest_path=\"/kaggle/working/data/cencer/\",image_number=1100)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:16:00.749972Z","iopub.execute_input":"2023-02-21T09:16:00.750332Z","iopub.status.idle":"2023-02-21T09:16:12.278315Z","shell.execute_reply.started":"2023-02-21T09:16:00.750301Z","shell.execute_reply":"2023-02-21T09:16:12.277345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"copy_images_convert_rgb(source_path=\"/kaggle/input/rsna-crpped-images/nocancer\",dest_path=\"/kaggle/working/data/noncencer/\",image_number=11000)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:16:12.282224Z","iopub.execute_input":"2023-02-21T09:16:12.282535Z","iopub.status.idle":"2023-02-21T09:17:47.199763Z","shell.execute_reply.started":"2023-02-21T09:16:12.282508Z","shell.execute_reply":"2023-02-21T09:17:47.198795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(os.listdir(\"/kaggle/working/data/cencer\")),len(os.listdir(\"/kaggle/working/data/noncencer\"))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:47.201065Z","iopub.execute_input":"2023-02-21T09:17:47.201422Z","iopub.status.idle":"2023-02-21T09:17:47.218279Z","shell.execute_reply.started":"2023-02-21T09:17:47.201388Z","shell.execute_reply":"2023-02-21T09:17:47.217261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Set the directory where the images are located\nimage_dir = '/kaggle/working/data/cencer'\n\n# Get a list of all image file names in the directory\nimage_files = [f for f in os.listdir(image_dir) if f.endswith('.png')]\n\n# Set the number of rows and columns for the subplot grid\nnum_rows = 10\nnum_cols = 5\n\n# Create a new figure and set its size\nfig = plt.figure(figsize=(25, 50))\n\n# Loop through the image files and display them in a subplot\nfor i, image_file in enumerate(image_files):\n    # Load the image using PIL\n    image = Image.open(os.path.join(image_dir, image_file))\n    \n    # Add the image to the subplot\n    subplot = fig.add_subplot(num_rows, num_cols, i+1)\n    subplot.imshow(image)\n    subplot.axis('off')\n    \n    # Break out of the loop if we've displayed all the images\n    if i >= (num_rows * num_cols) - 1:\n        break\n\n# Show the final subplot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:47.220982Z","iopub.execute_input":"2023-02-21T09:17:47.221337Z","iopub.status.idle":"2023-02-21T09:17:50.284816Z","shell.execute_reply.started":"2023-02-21T09:17:47.221303Z","shell.execute_reply":"2023-02-21T09:17:50.283495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Define the directories for the training and validation sets\ntrain_dir = '/kaggle/working/data'\n#val_dir = 'path/to/validation/directory'\n\n# Define the target image size and batch size\ntarget_size = (224, 224)\nbatch_size = 32\n\n# Define the data generator for the training set\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    validation_split=0.2) # 20% of the data will be used for validation\n\n# Create a generator for the training set\ntrain_generator = train_datagen.flow_from_directory(\n    train_dir,\n    target_size=target_size,\n    batch_size=batch_size,\n    class_mode='binary',\n    subset='training') # specify the subset to use for training\n\n# Create a generator for the validation set\nval_generator = train_datagen.flow_from_directory(\n    train_dir, # use the same directory as for the training set\n    target_size=target_size,\n    batch_size=batch_size,\n    class_mode='binary',\n    subset='validation') # specify the subset to use for validation\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:50.286183Z","iopub.execute_input":"2023-02-21T09:17:50.286636Z","iopub.status.idle":"2023-02-21T09:17:50.632198Z","shell.execute_reply.started":"2023-02-21T09:17:50.286588Z","shell.execute_reply":"2023-02-21T09:17:50.631169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator.class_indices","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:50.633496Z","iopub.execute_input":"2023-02-21T09:17:50.636605Z","iopub.status.idle":"2023-02-21T09:17:50.644405Z","shell.execute_reply.started":"2023-02-21T09:17:50.636572Z","shell.execute_reply":"2023-02-21T09:17:50.643435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator.class_indices = {'noncancer': 0, 'cancer': 1}\nval_generator.class_indices = {'noncancer': 0, 'cancer': 1}\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:50.645956Z","iopub.execute_input":"2023-02-21T09:17:50.646333Z","iopub.status.idle":"2023-02-21T09:17:50.651915Z","shell.execute_reply.started":"2023-02-21T09:17:50.646228Z","shell.execute_reply":"2023-02-21T09:17:50.65085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n\n# Load the model and weights from the saved file\nmodel = load_model('/kaggle/input/best-model/best.h5')","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:17:50.653718Z","iopub.execute_input":"2023-02-21T09:17:50.654131Z","iopub.status.idle":"2023-02-21T09:17:53.494511Z","shell.execute_reply.started":"2023-02-21T09:17:50.654096Z","shell.execute_reply":"2023-02-21T09:17:53.493571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow import keras\n# from tensorflow.keras import layers\n# from tensorflow.keras.applications import ResNet50\n# from tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout\n# from tensorflow.keras.models import Model\n\n\n# # Load the pre-trained ResNet50 model\n# base_model = ResNet50(include_top=False, weights='imagenet',input_shape=(224,224,3))\n\n# # Add a new global average pooling layer and a dense layer for classification\n# x = base_model.output\n# x = GlobalAveragePooling2D()(x)\n# x = Dense(1024, activation='relu')(x)\n\n# # Add a dropout layer\n# x = Dropout(0.5)(x)\n\n# # Add the final output layer for binary classification\n# predictions = Dense(1, activation='sigmoid')(x)\n\n# # Create the final model by combining the base model with the new layers\n# model = Model(inputs=base_model.input, outputs=predictions)\n\n# # Freeze the weights of the pre-trained layers to prevent overfitting\n# for layer in base_model.layers:\n#     layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2023-02-20T23:17:41.750178Z","iopub.execute_input":"2023-02-20T23:17:41.750549Z","iopub.status.idle":"2023-02-20T23:17:41.755123Z","shell.execute_reply.started":"2023-02-20T23:17:41.750508Z","shell.execute_reply":"2023-02-20T23:17:41.75392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compile the model with an optimizer and loss function\noptimizer = tf.keras.optimizers.Adam(learning_rate=0.0001)\nloss_fn = 'binary_crossentropy'\nmetrics = ['accuracy',tf.keras.metrics.Precision(),tf.keras.metrics.Recall()]\nmodel.compile(optimizer=optimizer, loss=loss_fn, metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:18:14.762814Z","iopub.execute_input":"2023-02-21T09:18:14.763682Z","iopub.status.idle":"2023-02-21T09:18:14.789653Z","shell.execute_reply.started":"2023-02-21T09:18:14.763631Z","shell.execute_reply":"2023-02-21T09:18:14.788658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model on the medical image dataset\nhistory = model.fit(train_generator,\n                epochs=1,\n                validation_data=val_generator)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:18:16.770337Z","iopub.execute_input":"2023-02-21T09:18:16.77071Z","iopub.status.idle":"2023-02-21T09:20:48.169385Z","shell.execute_reply.started":"2023-02-21T09:18:16.770677Z","shell.execute_reply":"2023-02-21T09:20:48.168485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nshutil.rmtree(\"/kaggle/working/data\", ignore_errors=True)\nshutil.rmtree(\"/kaggle/working/data/cencer\", ignore_errors=True)\nshutil.rmtree(\"/kaggle/working/data/noncencer\", ignore_errors=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:17.739165Z","iopub.execute_input":"2023-02-21T09:21:17.73957Z","iopub.status.idle":"2023-02-21T09:21:18.191834Z","shell.execute_reply.started":"2023-02-21T09:21:17.739537Z","shell.execute_reply":"2023-02-21T09:21:18.190561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:25.360676Z","iopub.execute_input":"2023-02-21T09:21:25.361052Z","iopub.status.idle":"2023-02-21T09:21:25.3798Z","shell.execute_reply.started":"2023-02-21T09:21:25.36102Z","shell.execute_reply":"2023-02-21T09:21:25.378913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_path = \"/kaggle/input/rsna-breast-cancer-detection/test_images/10008\"","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:32.571209Z","iopub.execute_input":"2023-02-21T09:21:32.571603Z","iopub.status.idle":"2023-02-21T09:21:32.576393Z","shell.execute_reply.started":"2023-02-21T09:21:32.57157Z","shell.execute_reply":"2023-02-21T09:21:32.575303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = []\nfor index, row in test_df.iterrows():\n    \n    path = base_path + '/' + str(row.image_id) + '.dcm' \n    paths.append(path)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:35.0771Z","iopub.execute_input":"2023-02-21T09:21:35.07777Z","iopub.status.idle":"2023-02-21T09:21:35.087781Z","shell.execute_reply.started":"2023-02-21T09:21:35.077731Z","shell.execute_reply":"2023-02-21T09:21:35.0867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['image_path'] = paths","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:39.623989Z","iopub.execute_input":"2023-02-21T09:21:39.624512Z","iopub.status.idle":"2023-02-21T09:21:39.630377Z","shell.execute_reply.started":"2023-02-21T09:21:39.624411Z","shell.execute_reply":"2023-02-21T09:21:39.629471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(test_df['image_path'])","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:49.674149Z","iopub.execute_input":"2023-02-21T09:21:49.674535Z","iopub.status.idle":"2023-02-21T09:21:49.682219Z","shell.execute_reply.started":"2023-02-21T09:21:49.674502Z","shell.execute_reply":"2023-02-21T09:21:49.681115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\ndef read_xray(file_path, img_size=None):\n    \"\"\"\n    Read the dicom data and get the image\n    Args:\n        file_path: The path of the dicom file\n        img_size: Size of the output image\n        \n    \"\"\"\n    test_data = np.zeros((len(file_path),img_size[0],img_size[1],3))\n    i = 1\n    for image in file_path:\n        \n        dicom = pydicom.read_file(image)\n        img = dicom.pixel_array\n\n        if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n            img = np.max(img) - img\n\n        if img_size:\n            img = cv2.resize(img, img_size)\n\n        # Add channel dim at First\n        img = img[np.newaxis]\n\n        # Converting img to float32\n        img = img / np.max(img)\n        img = img.astype(\"float32\")\n        rgb_image = tf.tile(tf.expand_dims(img[0], axis=-1), multiples=[1, 1, 3])\n        test_data[i-1,...] = rgb_image.numpy()\n        i = i+1\n    return test_data","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:53.773897Z","iopub.execute_input":"2023-02-21T09:21:53.774247Z","iopub.status.idle":"2023-02-21T09:21:53.880973Z","shell.execute_reply.started":"2023-02-21T09:21:53.774215Z","shell.execute_reply":"2023-02-21T09:21:53.880025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data  = read_xray(test_df['image_path'], img_size=(224,224))","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:21:56.192617Z","iopub.execute_input":"2023-02-21T09:21:56.192974Z","iopub.status.idle":"2023-02-21T09:21:58.960989Z","shell.execute_reply.started":"2023-02-21T09:21:56.192945Z","shell.execute_reply":"2023-02-21T09:21:58.960019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:22:00.135917Z","iopub.execute_input":"2023-02-21T09:22:00.136269Z","iopub.status.idle":"2023-02-21T09:22:01.179886Z","shell.execute_reply.started":"2023-02-21T09:22:00.136238Z","shell.execute_reply":"2023-02-21T09:22:01.178849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the original test data with the prediction_id column\ntest_data_orig = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\n# Copy the prediction_id column from the original test data\nprediction_ids = test_data_orig['prediction_id'].copy()\n\n# Flatten the predictions array\npredictions = y_pred.ravel()\n\n# Create a dataframe with the required format\nsubmission = pd.DataFrame({'prediction_id': prediction_ids, 'cancer':1- predictions}).groupby('prediction_id').mean().reset_index()\n\n# Save the dataframe to a CSV file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:22:47.122069Z","iopub.execute_input":"2023-02-21T09:22:47.122429Z","iopub.status.idle":"2023-02-21T09:22:47.136943Z","shell.execute_reply.started":"2023-02-21T09:22:47.122399Z","shell.execute_reply":"2023-02-21T09:22:47.135669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/working/submission.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2023-02-21T09:22:49.387018Z","iopub.execute_input":"2023-02-21T09:22:49.38738Z","iopub.status.idle":"2023-02-21T09:22:49.400815Z","shell.execute_reply.started":"2023-02-21T09:22:49.387347Z","shell.execute_reply":"2023-02-21T09:22:49.399492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}