{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%capture\n# Source: https://www.kaggle.com/code/remekkinas/fast-dicom-processing-1-6-2x-faster?scriptVersionId=113360473\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n   !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:37:46.27457Z","iopub.execute_input":"2023-01-21T12:37:46.27496Z","iopub.status.idle":"2023-01-21T12:38:16.114008Z","shell.execute_reply.started":"2023-01-21T12:37:46.274926Z","shell.execute_reply":"2023-01-21T12:38:16.11236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Test Images\nimport numpy as np\nimport pandas as pd\nimport os\nimport random\nimport pydicom\nimport cv2\nfrom PIL import Image\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.117002Z","iopub.execute_input":"2023-01-21T12:38:16.117678Z","iopub.status.idle":"2023-01-21T12:38:16.124554Z","shell.execute_reply.started":"2023-01-21T12:38:16.117633Z","shell.execute_reply":"2023-01-21T12:38:16.12361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() # TPU detection\nexcept ValueError:\n    tpu = None\n    gpus = tf.config.experimental.list_logical_devices(\"GPU\")\n    \nif tpu:\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    STRATEGY = tf.distribute.experimental.TPUStrategy(tpu,) \n    print('Running on TPU ', tpu.cluster_spec().as_dict()['worker'])\nelif len(gpus) > 1:\n    STRATEGY = tf.distribute.MirroredStrategy([gpu.name for gpu in gpus])\n    print('Running on multiple GPUs ', [gpu.name for gpu in gpus])\nelif len(gpus) == 1:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on single GPU ', gpus[0].name)\nelse:\n    STRATEGY = tf.distribute.get_strategy() \n    print('Running on CPU')\nprint(\"Number of accelerators: \", STRATEGY.num_replicas_in_sync)\n\nAUTOTUNE = tf.data.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.125928Z","iopub.execute_input":"2023-01-21T12:38:16.126936Z","iopub.status.idle":"2023-01-21T12:38:16.15414Z","shell.execute_reply.started":"2023-01-21T12:38:16.126897Z","shell.execute_reply":"2023-01-21T12:38:16.153157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed all random number generators\nSEED = 43\ndef seed_everything(seed=SEED):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.155828Z","iopub.execute_input":"2023-01-21T12:38:16.156664Z","iopub.status.idle":"2023-01-21T12:38:16.173466Z","shell.execute_reply.started":"2023-01-21T12:38:16.156627Z","shell.execute_reply":"2023-01-21T12:38:16.172086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGET_HEIGHT = 512\nTARGET_WIDTH = 512\nN_CHANNELS = 1\nINPUT_SHAPE = (TARGET_HEIGHT, TARGET_WIDTH, N_CHANNELS)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.177341Z","iopub.execute_input":"2023-01-21T12:38:16.177624Z","iopub.status.idle":"2023-01-21T12:38:16.185365Z","shell.execute_reply.started":"2023-01-21T12:38:16.177596Z","shell.execute_reply":"2023-01-21T12:38:16.184299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(file_path, size=None, apply_clahe=False, debug=False, save=False):\n    # Read Dicom File\n    dicom = pydicom.dcmread(file_path)\n    image = dicom.pixel_array\n\n    # Normalize [0,1] range\n    image = (image - image.min()) / (image.max() - image.min())\n\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":  \n        image = 1 - image\n\n    # Convert to uint8 image in range [0, 255]\n    image = (image * 255).astype(np.uint8)\n    \n    # Flip T0 Left/Right Orientation\n    #h0, w0 = image.shape\n    #if image[:,int(-w0 * 0.10):].sum() > image[:,:int(w0 * 0.10)].sum():\n    #    image = np.flip(image, axis=1)\n    \n    # Save original image\n    if debug:\n        image0 = np.copy(image)\n    \n    # Always crop 10 pixels for weird border noise/lines\n    #image = image[int(h0 * 2e-2):-int(h0 * 2e-2),int(w0 * 2e-2):-int(w0 * 2e-2)]\n    \n    #if crop_image:\n    #    image = crop(image, debug=debug)\n        \n    # Resize\n    if size is not None:\n        # Pad black pixels to make square image\n        #h, w = image.shape\n        #if (h / w) > TARGET_HEIGHT_WIDTH_RATIO:\n        #    pad = int(h / TARGET_HEIGHT_WIDTH_RATIO - w)\n        #    image = np.pad(image, [[0,0], [0, pad]])\n        #    h, w = image.shape\n        #else:\n        #    pad = int(0.50 * (w * TARGET_HEIGHT_WIDTH_RATIO - h))\n        #    image = np.pad(image, [[pad, pad], [0,0]])\n        #    h, w = image.shape\n        # Resize\n        image = cv2.resize(image, size, interpolation=cv2.INTER_AREA)\n        \n    # Apply CLAHE contrast enhancement\n    if apply_clahe:\n        image = CLAHE.apply(image)\n        \n    # Save Only\n    if save:\n        image_id = file_path.split('/')[-1].split('.')[0]\n        cv2.imwrite(f'{image_id}.png', image)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.186951Z","iopub.execute_input":"2023-01-21T12:38:16.187642Z","iopub.status.idle":"2023-01-21T12:38:16.19852Z","shell.execute_reply.started":"2023-01-21T12:38:16.187607Z","shell.execute_reply":"2023-01-21T12:38:16.197626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # To decode datatype\n# def preprocess_image(image):\n#     image = tf.image.decode_png(image, channels=3)\n\n#     image = tf.cast(image, tf.float32)\n#     image = tf.reshape(image, [*[512]*2, 3])\n#     print(image.shape)\n#     image /= 255.0  # normalize to [0,1] range         # IMAGE NORMALIZE\n\n#     return image\n\n# # Read image, and process\n# def load_and_preprocess_image(path):\n#     image = tf.io.read_file(path)\n#     return preprocess_image(image)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.200074Z","iopub.execute_input":"2023-01-21T12:38:16.200794Z","iopub.status.idle":"2023-01-21T12:38:16.21288Z","shell.execute_reply.started":"2023-01-21T12:38:16.200738Z","shell.execute_reply":"2023-01-21T12:38:16.211748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_test_dataset(df,batch_size,with_labels = False, shuffle= False):\n    \n    \n#     # Create Dataset\n#     if with_labels:\n#         dataset = tf.data.Dataset.from_tensor_slices((df['img_path'].values, df['cancer'].values))\n#     else:\n#         dataset = tf.data.Dataset.from_tensor_slices(\n#             (df['img_path'].values)\n#         )\n\n#     # Image preprocessing\n#     dataset = dataset.map(load_and_preprocess_image) \n    \n\n#     # Get one batch\n#     dataset = dataset.shuffle(1024, reshuffle_each_iteration = True) if shuffle else dataset\n#     dataset = dataset.batch(batch_size,drop_remainder=False)\n#     dataset = dataset.prefetch(AUTOTUNE)\n#     #dataset = dataset.shape\n\n#     return dataset","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.214398Z","iopub.execute_input":"2023-01-21T12:38:16.21523Z","iopub.status.idle":"2023-01-21T12:38:16.226929Z","shell.execute_reply.started":"2023-01-21T12:38:16.215196Z","shell.execute_reply":"2023-01-21T12:38:16.225874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_image_paths = []\n# test_dir = os.listdir('/kaggle/tmp/output/test/')\n# for i in range(len(test_dir)):\n#     img_path = '/kaggle/tmp/output/test' + '/' + test_dir[i]\n#     test_image_paths.append(img_path)\n\n\n# test_df['img_path'] = test_image_paths\n# test_dataset = create_test_dataset(\n#     test_df,\n#     batch_size  = 8, \n#     with_labels = False, \n#     shuffle = False\n# )","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.228483Z","iopub.execute_input":"2023-01-21T12:38:16.229256Z","iopub.status.idle":"2023-01-21T12:38:16.238971Z","shell.execute_reply.started":"2023-01-21T12:38:16.229217Z","shell.execute_reply":"2023-01-21T12:38:16.237978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_pfbeta(from_logits=True, beta=1.0, epsilon=1e-07):\n    \n    def pfbeta(y_true, y_pred):\n        y_pred = tf.cond(\n            tf.cast(from_logits, dtype=tf.bool),\n            lambda: tf.nn.sigmoid(y_pred),\n            lambda: y_pred,\n        )\n        y_true = tf.reshape(y_true, [-1])\n        y_pred = tf.reshape(y_pred, [-1])\n\n        ctp = tf.reduce_sum(y_true * y_pred, axis=-1)\n        cfp = tf.reduce_sum(y_pred, axis=-1) - ctp\n\n        c_precision = ctp / (ctp + cfp)\n        c_recall = ctp / tf.reduce_sum(y_true)\n        \n        def compute_fractions():\n            numerator = c_precision * c_recall\n            denominator = beta**2 * c_precision + c_recall\n            return (1 + beta**2) * tf.math.divide_no_nan(numerator, denominator)\n        \n        return tf.cond(\n            tf.logical_and(\n                tf.greater(c_precision, 0.), tf.greater(c_recall, 0.)\n            ),\n            compute_fractions,\n            lambda: tf.constant(0, dtype=tf.float32)\n        )\n    \n    return pfbeta","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.24066Z","iopub.execute_input":"2023-01-21T12:38:16.241449Z","iopub.status.idle":"2023-01-21T12:38:16.252079Z","shell.execute_reply.started":"2023-01-21T12:38:16.241411Z","shell.execute_reply":"2023-01-21T12:38:16.250784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def binary_focal_loss(gamma=2., alpha=.25, class_weights=None):\n    \"\"\"\n    Binary form of focal loss.\n      FL(p_t) = -alpha * (1 - p_t)**gamma * log(p_t)\n      where p = sigmoid(x), p_t = p or 1 - p depending on if the label is 1 or 0, respectively.\n    References:\n        https://arxiv.org/pdf/1708.02002.pdf\n    Usage:\n     model.compile(loss=[binary_focal_loss(alpha=.25, gamma=2)], metrics=[\"accuracy\"], optimizer=adam)\n    \"\"\"\n    def binary_focal_loss_fixed(y_true, y_pred):\n        \"\"\"\n        y_true shape need be (None,1)\n        y_pred need be compute after sigmoid\n        \"\"\"\n        y_true = tf.cast(y_true, tf.float32)\n        y_pred = tf.clip_by_value(y_pred, tf.keras.backend.epsilon(), 1-tf.keras.backend.epsilon())\n        pt_1 = tf.where(tf.equal(y_true, 1), y_pred, tf.ones_like(y_pred))\n        pt_0 = tf.where(tf.equal(y_true, 0), y_pred, tf.zeros_like(y_pred))\n\n        epsilon = tf.keras.backend.epsilon()\n        # clip to prevent NaN's and Inf's\n        pt_1 = tf.keras.backend.clip(pt_1, epsilon, 1. - epsilon)\n        pt_0 = tf.keras.backend.clip(pt_0, epsilon, 1. - epsilon)\n        \n        if class_weights is not None:\n            weight_0 = class_weights[0]\n            weight_1 = class_weights[1]\n            \n            loss = -weight_0*tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -weight_1*tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n            \n        else:\n            \n            loss = -tf.keras.backend.mean(alpha * tf.keras.backend.pow(1. - pt_1, gamma) * tf.keras.backend.log(pt_1)) \\\n               -tf.keras.backend.mean((1 - alpha) * tf.keras.backend.pow(pt_0, gamma) * tf.keras.backend.log(1. - pt_0))\n        \n        \n        return loss\n\n    return binary_focal_loss_fixed","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.253682Z","iopub.execute_input":"2023-01-21T12:38:16.254397Z","iopub.status.idle":"2023-01-21T12:38:16.267357Z","shell.execute_reply.started":"2023-01-21T12:38:16.254359Z","shell.execute_reply":"2023-01-21T12:38:16.266446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    # Create the model\n    # Verify Mixed Policy Settings\n    print(f'Compute dtype: {tf.keras.mixed_precision.global_policy().compute_dtype}')\n    print(f'Variable dtype: {tf.keras.mixed_precision.global_policy().variable_dtype}')\n    \n    with STRATEGY.scope():\n        \n        # Set seed for deterministic weights initialization\n        seed_everything()\n        \n        model = tf.keras.Sequential()\n        \n        # Add the first convolutional layer\n        model.add(keras.layers.Conv2D(32, kernel_size=(5, 5), activation='relu', input_shape=(TARGET_HEIGHT,TARGET_WIDTH, 1)))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Add the second convolutional layer\n        model.add(keras.layers.Conv2D(32, kernel_size=(5, 5), activation='relu'))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Add the third convolutional layer\n        model.add(keras.layers.Conv2D(64, kernel_size=(3, 3), activation='relu'))\n        model.add(keras.layers.MaxPooling2D(pool_size=(2, 2)))\n\n        # Flatten the output from the convolutional layers\n        model.add(keras.layers.Flatten())\n\n        # Add a dense layer\n        model.add(keras.layers.Dense(256, activation='relu'))\n\n        # Add the final dense layer for binary classification\n        model.add(keras.layers.Dense(1, activation='sigmoid'))\n        \n        # We will use the famous Adam optimizer for fast learning\n        optimizer = tf.optimizers.Adam(learning_rate=1e-3, epsilon=1e-7, clipnorm=10.0)\n        \n        # Loss\n        loss = binary_focal_loss(gamma=2., alpha=.25, class_weights = {0: 1., 1: 10.})\n\n        \n        # Metrics\n        metrics = [\n            #tfa.metrics.F1Score(num_classes=1, threshold=0.50),\n            tf_pfbeta(beta=1.0, from_logits=True),\n            tf.keras.metrics.Precision(),\n            tf.keras.metrics.Recall(),\n            tf.keras.metrics.AUC(),\n            tf.keras.metrics.BinaryAccuracy(),\n        ]\n        \n        model.compile(optimizer=optimizer, loss=loss, metrics=metrics)\n\n\n        return model","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.26921Z","iopub.execute_input":"2023-01-21T12:38:16.26992Z","iopub.status.idle":"2023-01-21T12:38:16.284023Z","shell.execute_reply.started":"2023-01-21T12:38:16.269883Z","shell.execute_reply":"2023-01-21T12:38:16.283205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    \n    # Inputs, note the names are equal to the dictionary keys in the dataset\n    image = tf.keras.layers.Input(INPUT_SHAPE, name='image', dtype=tf.uint8)\n\n    model = create_model() \n\n    #model.load_weights('/kaggle/input/rsna-efficientnetv2-training-tensorflow-tpu-ds/model.h5')\n    \n    #model.load_weights('/kaggle/input/my-rsna-effnetv2t-model/model.h5')\n    \n    #model.load_weights('/kaggle/input/my-tuned-rsna-effnetv2t/model.h5')\n    \n    model.load_weights('/kaggle/input/inferencersna/best-model.h5')\n\n    model.trainable = False\n\n    model.compile()\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.285588Z","iopub.execute_input":"2023-01-21T12:38:16.286251Z","iopub.status.idle":"2023-01-21T12:38:16.299568Z","shell.execute_reply.started":"2023-01-21T12:38:16.286217Z","shell.execute_reply":"2023-01-21T12:38:16.298728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pretrained File Path: '/kaggle/input/sartorius-training-dataset/model.h5'\ntf.keras.backend.clear_session()\n# enable XLA optmizations\ntf.config.optimizer.set_jit(True)\nmodel = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:16.304431Z","iopub.execute_input":"2023-01-21T12:38:16.304697Z","iopub.status.idle":"2023-01-21T12:38:19.664858Z","shell.execute_reply.started":"2023-01-21T12:38:16.304673Z","shell.execute_reply":"2023-01-21T12:38:19.663818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:19.666274Z","iopub.execute_input":"2023-01-21T12:38:19.666648Z","iopub.status.idle":"2023-01-21T12:38:19.67559Z","shell.execute_reply.started":"2023-01-21T12:38:19.666611Z","shell.execute_reply":"2023-01-21T12:38:19.673343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\ndef get_file_path(args):\n    patient_id, image_id = args\n    return f'/kaggle/input/rsna-breast-cancer-detection/test_images/{patient_id}/{image_id}.dcm'\n    \ntest['file_path'] = test[['patient_id', 'image_id']].apply(get_file_path, axis=1)\n\ndisplay(test.info())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:19.677337Z","iopub.execute_input":"2023-01-21T12:38:19.678601Z","iopub.status.idle":"2023-01-21T12:38:19.713136Z","shell.execute_reply.started":"2023-01-21T12:38:19.678564Z","shell.execute_reply":"2023-01-21T12:38:19.711917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess a single image and saves it\ndef preprocess_and_save_image(args):\n    (patient_id, laterality), g = args\n    cancer = 0.0\n    for row_idx, row in g.iterrows():\n        process(row['file_path'], size=(TARGET_WIDTH, TARGET_HEIGHT), save=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:19.714839Z","iopub.execute_input":"2023-01-21T12:38:19.715583Z","iopub.status.idle":"2023-01-21T12:38:19.721782Z","shell.execute_reply.started":"2023-01-21T12:38:19.715547Z","shell.execute_reply":"2023-01-21T12:38:19.720481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess all images in parallel using Joblib\nimport joblib\nfrom joblib import Parallel, delayed\nfrom multiprocessing import cpu_count\njobs = [joblib.delayed(preprocess_and_save_image)(args) for args in test.groupby(['patient_id', 'laterality'])]\nSUBMISSION_ROWS = joblib.Parallel(\n    n_jobs=cpu_count(),\n    verbose=1,\n    backend='multiprocessing',\n    prefer='threads',\n)(jobs)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:19.723789Z","iopub.execute_input":"2023-01-21T12:38:19.724241Z","iopub.status.idle":"2023-01-21T12:38:22.868439Z","shell.execute_reply.started":"2023-01-21T12:38:19.724206Z","shell.execute_reply":"2023-01-21T12:38:22.867041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SUBMISSION_ROWS = []\n\nfor (patient_id, laterality), g in tqdm(test.groupby(['patient_id', 'laterality'])):\n    cancer = 0\n    for row_idx, row in g.iterrows():\n        # Load Image\n        image_id = row['image_id']\n        \n        image = cv2.imread(f'{image_id}.png', -1)\n        # Expand to Batch HxW -> 1xHxWx1\n        \n        \n        image = np.expand_dims(image, axis=[0,3])\n        # Make Prediction\n        cancer += model.predict_on_batch(image).squeeze() / len(g)\n        # Remove Image PNG\n        os.remove(f'{image_id}.png')\n        \n    # Add Submission Row\n    SUBMISSION_ROWS.append({\n        'prediction_id': f'{patient_id}_{laterality}',\n        #'cancer': np.int8(cancer > THRESHOLD_BEST),\n        'cancer': cancer,\n    })","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:22.870793Z","iopub.execute_input":"2023-01-21T12:38:22.87159Z","iopub.status.idle":"2023-01-21T12:38:25.424317Z","shell.execute_reply.started":"2023-01-21T12:38:22.871541Z","shell.execute_reply":"2023-01-21T12:38:25.423317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create DataFrame from submission rows\nsubmission_df = pd.DataFrame(SUBMISSION_ROWS)\n\ndisplay(submission_df.info())\ndisplay(submission_df.head())","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:25.425831Z","iopub.execute_input":"2023-01-21T12:38:25.426519Z","iopub.status.idle":"2023-01-21T12:38:25.452759Z","shell.execute_reply.started":"2023-01-21T12:38:25.426478Z","shell.execute_reply":"2023-01-21T12:38:25.451927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', columns=['prediction_id','cancer'],index=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:25.453989Z","iopub.execute_input":"2023-01-21T12:38:25.455034Z","iopub.status.idle":"2023-01-21T12:38:25.462156Z","shell.execute_reply.started":"2023-01-21T12:38:25.454995Z","shell.execute_reply":"2023-01-21T12:38:25.461184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity Check\ndisplay(pd.read_csv('submission.csv').head())","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:25.463539Z","iopub.execute_input":"2023-01-21T12:38:25.464908Z","iopub.status.idle":"2023-01-21T12:38:25.480044Z","shell.execute_reply.started":"2023-01-21T12:38:25.464847Z","shell.execute_reply":"2023-01-21T12:38:25.478944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# THRESHOLD = 0.10\n\n# #preds = np.mean([prediction], 0)\n# preds = (preds > THRESHOLD).astype(int)\n# test_df[\"cancer\"] = preds","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:25.483449Z","iopub.execute_input":"2023-01-21T12:38:25.484297Z","iopub.status.idle":"2023-01-21T12:38:25.490491Z","shell.execute_reply.started":"2023-01-21T12:38:25.484269Z","shell.execute_reply":"2023-01-21T12:38:25.489437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_df['prediction_id'] = test_df['patient_id'].astype(str) + \"_\" + test_df['laterality']\n\n# sub = test_df[['prediction_id', 'cancer']].groupby(\"prediction_id\").mean().reset_index()\n\n# sub.to_csv('/kaggle/working/submission.csv', index=False)\n\n# sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-21T12:38:25.492219Z","iopub.execute_input":"2023-01-21T12:38:25.492878Z","iopub.status.idle":"2023-01-21T12:38:25.502687Z","shell.execute_reply.started":"2023-01-21T12:38:25.492826Z","shell.execute_reply":"2023-01-21T12:38:25.501341Z"},"trusted":true},"execution_count":null,"outputs":[]}]}