{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# !pip install python-gdcm -q\n# !pip install pylibjpeg -q\n# !pip install wandb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pydicom\nimport numpy as np\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom pathlib import Path\nimport glob\nimport pandas as pd\n# import pylibjpeg\n\nimport numpy as np\nfrom tensorflow import keras\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nimport random\nfrom scipy.ndimage import gaussian_filter\nfrom scipy import ndimage\nimport cv2\nimport matplotlib.pyplot as plt\nfrom skimage.transform import rescale, resize, downscale_local_mean\nfrom tqdm import tqdm\n\n# !wandb login\n","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:37:37.245507Z","iopub.execute_input":"2023-01-03T20:37:37.245856Z","iopub.status.idle":"2023-01-03T20:37:39.832555Z","shell.execute_reply.started":"2023-01-03T20:37:37.245785Z","shell.execute_reply":"2023-01-03T20:37:39.831492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# source: https://www.kaggle.com/code/allunia/rsna-csf-cervical-spine-fracture-eda/notebook\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef show_images_for_patient(patient_id):\n    patient_dir = os.path.join('../input/rsna-breast-cancer-detection/train_images', str(patient_id))\n    num_images = len(glob.glob(f\"{patient_dir}/*\"))\n    print(f\"Number of images for patient: {num_images}\")\n    fig, axs = plt.subplots(5,3, figsize=(24,15))\n    axs = axs.flatten()\n    for i, img_path in enumerate(list(Path(patient_dir).iterdir())):\n        ds = pydicom.dcmread(img_path)\n        axs[i].imshow(rescale_img_to_hu(ds), cmap=\"bone\")\n        \ndef multi_class_labels(data, labels=[1]):\n    if data==1:\n        return [0,1]\n    return [1,0]\n\ndef image_resize(image, width = None, height = None, inter = cv2.INTER_LINEAR):\n\n    dim = None\n    (h, w) = image.shape[:2]\n\n    if width is None and height is None:\n        return image\n\n    if width is None:\n        r = height / float(h)\n        dim = (int(w * r), height)\n    else:\n        r = width / float(w)\n        dim = (width, int(h * r))\n    resized = cv2.resize(image, dim, interpolation = inter)\n\n    return resized\n\n\n        \n# show_images_for_patient(55706)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:37:53.908908Z","iopub.execute_input":"2023-01-03T20:37:53.909638Z","iopub.status.idle":"2023-01-03T20:37:53.924183Z","shell.execute_reply.started":"2023-01-03T20:37:53.909599Z","shell.execute_reply":"2023-01-03T20:37:53.923001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_csv = train_csv.sample(frac=1).reset_index(drop=True)\ntrain_csv=train_csv[train_csv['cancer']==1]\ntrain_csv.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:37:55.168259Z","iopub.execute_input":"2023-01-03T20:37:55.168976Z","iopub.status.idle":"2023-01-03T20:37:55.287833Z","shell.execute_reply.started":"2023-01-03T20:37:55.168913Z","shell.execute_reply":"2023-01-03T20:37:55.286846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model(INPUT_SHAPE, OUTPUT_SIZE, output_bias=None):\n    model = tf.keras.applications.resnet.ResNet152 (weights='imagenet', input_shape=(256,256,3), include_top=False, classes=1000)\n    # model = tf.keras.applications.resnet50.ResNet50(weights='imagenet', input_shape=(200,200,3), include_top=False, classes=1000)\n    x = model.output\n    x = tf.keras.layers.Conv2D(32, (3,3), strides=(1, 1), padding='valid')(x)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Flatten()(x)\n    x = tf.keras.layers.Dense(64, activation='sigmoid')(x)\n    # x = tf.keras.layers.BatchNormalization()(x)\n    # x = tf.keras.layers.Dropout(0.5)(x)\n    if OUTPUT_SIZE==1:\n        activation='sigmoid'\n    else:\n        activation='softmax'\n    \n    # print(activation)\n    x = tf.keras.layers.Dense(OUTPUT_SIZE, activation=activation)(x)\n    # x = tf.keras.layers.Dropout(0.2)(x)\n\n    model = tf.keras.models.Model(inputs=model.input, outputs=x)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:37:56.403825Z","iopub.execute_input":"2023-01-03T20:37:56.404556Z","iopub.status.idle":"2023-01-03T20:37:56.412759Z","shell.execute_reply.started":"2023-01-03T20:37:56.404517Z","shell.execute_reply":"2023-01-03T20:37:56.411797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class Generator:\n#     def __init__(self, image_list):\n#         np.random.seed(10)\n#         train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n# #         train_csv = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n#         base_path='/kaggle/input/rsna-mammography-images-as-pngs/images_as_pngs_cv2_256/'\n#         # saving image path into train dataframe\n#         train_csv['img_path']= f'{base_path}/train_images_processed_cv2_256'\\\n#                             + '/' + train_csv.patient_id.astype(str)\\\n#                             + '/' + train_csv.image_id.astype(str)\\\n#                             + '.png'\n#         train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n#         case = train_csv[train_csv['cancer']==1]\n#         control = train_csv[train_csv['cancer']==0]\n\n#         frames = [case, control.loc[:len(case)]]\n#         train_csv = pd.concat(frames)\n        \n#         train_csv = train_csv.sample(frac=1).reset_index(drop=True)\n#         _image=[]\n#         _label=[]\n#         for encounter in tqdm(train_csv.iterrows()):\n#             img_path = os.path.join(f'{base_path}/train_images_processed_cv2_256', str(encounter[1]['patient_id']), str(encounter[1]['image_id'])+ '.png')\n#             image=tf.io.read_file(img_path)\n#             image = tf.io.decode_png(image)\n\n#             _image.append(resize(image[:,:,0], (256,256)))\n#             _label.append(encounter[1]['cancer'])\n        \n#         self.n_image=len(_image)\n#         self.image=_image\n#         self.label=_label\n#     def get_item(self):\n#         while True:\n#             randomindex = random.randint(0, self.n_image-1)\n#             X_placeholder = np.zeros((256, 256, 3, 1), dtype=np.float32)\n#             y_placeholder = np.zeros((2), dtype=np.int16)\n#             image = self.image[randomindex]\n#             label = self.label[randomindex]\n                    \n#             image=(image-np.min(image))/(np.max(image)-np.min(image))\n#             X = np.array(np.stack((np.array(image),)*3, axis=2))\n#             y = multi_class_labels(label, np.array([0,1]))\n#             X_placeholder[:, :, :, 0] = X\n#             y_placeholder = y\n#             yield X_placeholder, y_placeholder\n            \n\n# training_generator=Generator('/kaggle/input/rsna-breast-cancer-detection/train.csv')\n# train_tf_gen = tf.data.Dataset.from_generator(training_generator.get_item, (tf.float32, tf.int16), (tf.TensorShape([256,256,3,1]), tf.TensorShape([2])))\n# # train_tf_gen = train_tf_gen.cache()\n# train_batches = train_tf_gen.batch(32)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:37:57.502002Z","iopub.execute_input":"2023-01-03T20:37:57.502733Z","iopub.status.idle":"2023-01-03T20:38:12.654247Z","shell.execute_reply.started":"2023-01-03T20:37:57.502693Z","shell.execute_reply":"2023-01-03T20:38:12.653361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BATCH_SIZE=300\n\n# METRICS = [\n#     keras.metrics.TruePositives(name='tp'),\n#     keras.metrics.FalsePositives(name='fp'),\n#     keras.metrics.TrueNegatives(name='tn'),\n#     keras.metrics.FalseNegatives(name='fn'), \n#     keras.metrics.BinaryAccuracy(name='accuracy'),\n#     keras.metrics.Precision(name='precision'),\n#     keras.metrics.Recall(name='recall'),\n#     keras.metrics.AUC(name='auc'),\n#     keras.metrics.AUC(name='prc', curve='PR'), # precision-recall curve\n# ]\n\nmodel=make_model((256,256,3), 2)\n# model.compile(\n#     optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n#     loss=keras.losses.BinaryCrossentropy(),\n#     metrics=METRICS)\n# # model.load_weights('iqa.h5')\n# early_stopping = tf.keras.callbacks.EarlyStopping(\n#     monitor='val_prc', \n#     verbose=1,\n#     patience=20,\n#     mode='max',\n#     restore_best_weights=True)\n\n# !mkdir weights\n\n# check_pointer = tf.keras.callbacks.ModelCheckpoint(filepath=\"./weights/iqa_seg.h5\", verbose=1, save_best_only=True, save_weights_only=True)\n\n# history = model.fit(\n#     train_batches,\n#     steps_per_epoch=25,\n#     epochs=100,\n#     callbacks=[early_stopping, check_pointer],\n#     validation_data=train_batches,\n#     validation_steps=10,\n#     )\n\nmodel.load_weights('/kaggle/working/weights/iqa_seg.h5')","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:38:20.855945Z","iopub.execute_input":"2023-01-03T20:38:20.857132Z","iopub.status.idle":"2023-01-03T20:38:26.033771Z","shell.execute_reply.started":"2023-01-03T20:38:20.857086Z","shell.execute_reply":"2023-01-03T20:38:26.032533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\n\nDATASET_PATH='/kaggle/input/rsna-breast-cancer-detection/'\ntest_df = pd.read_csv(os.path.join(DATASET_PATH, \"test.csv\"))\ndisplay(test_df.head())\nprint(f'cases: {len(test_df)}')\n\n# Show sample submission example\n\ndf_sub = pd.read_csv(os.path.join(DATASET_PATH, \"sample_submission.csv\"))\ndisplay(df_sub.head())\nprint(f'cases: {len(df_sub)}')","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:38:38.255721Z","iopub.execute_input":"2023-01-03T20:38:38.256122Z","iopub.status.idle":"2023-01-03T20:38:38.287459Z","shell.execute_reply.started":"2023-01-03T20:38:38.256084Z","shell.execute_reply":"2023-01-03T20:38:38.286382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path):\n    print(image_path)\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels = 3)\n    img = tf.image.resize(img, [256, 256])\n    img = tf.cast(img, dtype = tf.float32)\n    img = img/255.0\n    return img\n\n# test_paths=[]\n# test_dir='/kaggle/input/rsnatest/test_images_256/10008/'\n# img_path= os.listdir (test_dir)\n\n\n\nDF_PATH = '/kaggle/input/rsna-breast-cancer-detection'\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\ntest_path='/kaggle/input/rsnatest/test_images_256/'\n\ntest_dir = f'{test_path}'\n\n\n\n\ntest_df['img_path']= f'{test_dir}/'\\\n                    + '/' + test_df.patient_id.astype(str)\\\n                    + '/' + test_df.image_id.astype(str)\\\n                    + '.png'\n\n\ntest_df['img_path'][1]\nimage = tf.keras.preprocessing.image.load_img(test_df['img_path'][1])\n        \nprint(image.getdata)\nplt.imshow(image)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:38:39.299397Z","iopub.execute_input":"2023-01-03T20:38:39.299785Z","iopub.status.idle":"2023-01-03T20:38:39.518905Z","shell.execute_reply.started":"2023-01-03T20:38:39.299753Z","shell.execute_reply":"2023-01-03T20:38:39.517866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds=[]\nfor i in range (len(test_df['img_path'])):\n    image = tf.keras.preprocessing.image.load_img(test_df['img_path'][i])\n    image=tf.expand_dims(np.array(image), 0)\n    pred=np.argmax(model.predict(np.asarray(image)))\n    preds.append(pred)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:38:40.288988Z","iopub.execute_input":"2023-01-03T20:38:40.289356Z","iopub.status.idle":"2023-01-03T20:38:44.300336Z","shell.execute_reply.started":"2023-01-03T20:38:40.289325Z","shell.execute_reply":"2023-01-03T20:38:44.299259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'prediction_id':test_df.prediction_id,\n                        'cancer':preds})\n\npred_df['cancer']=(np.argmax(pred_df.cancer)).astype(int)\n\n\npred_df.to_csv('sample_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T20:38:44.30241Z","iopub.execute_input":"2023-01-03T20:38:44.302749Z","iopub.status.idle":"2023-01-03T20:38:44.312068Z","shell.execute_reply.started":"2023-01-03T20:38:44.302721Z","shell.execute_reply":"2023-01-03T20:38:44.310666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}