{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"! pip install -q /kaggle/input/keras-cv-core-namex/namex-0.0.7-py3-none-any.whl\n! pip install -q /kaggle/input/keras-cv-core-namex/keras_core-0.1.4-py3-none-any.whl\n! pip install -q /kaggle/input/keras-cv-core-namex/keras_cv-0.6.1-py3-none-any.whl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-03T05:30:39.698035Z","iopub.execute_input":"2023-09-03T05:30:39.699042Z","iopub.status.idle":"2023-09-03T05:31:13.25462Z","shell.execute_reply.started":"2023-09-03T05:30:39.698996Z","shell.execute_reply":"2023-09-03T05:31:13.252788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"\nimport keras_core as keras\nimport keras_cv\n\nimport gc\nimport cv2\nimport pydicom\nfrom joblib import Parallel, delayed\n\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nfrom glob import glob","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:32:26.71966Z","iopub.execute_input":"2023-09-03T06:32:26.720214Z","iopub.status.idle":"2023-09-03T06:32:48.480528Z","shell.execute_reply.started":"2023-09-03T06:32:26.720169Z","shell.execute_reply":"2023-09-03T06:32:48.479141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\nIMAGE_DIR = \"/tmp/dataset/rsna-atd\"\nINPUT_MODEL_PATH = \"/input/rsna-train-keras-yolov8/yolov8_xl_atd.keras\"\nMODEL_PATH = \"/kaggle/working/yolov8_xl_atd.keras\"\n\nSTRIDE = 10","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:35:23.922602Z","iopub.execute_input":"2023-09-03T06:35:23.923019Z","iopub.status.idle":"2023-09-03T06:35:23.929257Z","shell.execute_reply.started":"2023-09-03T06:35:23.922989Z","shell.execute_reply":"2023-09-03T06:35:23.928075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    IMAGE_SIZE = [256, 256]\n    RESIZE_DIM = 256\n    BATCH_SIZE = 32\n    AUTOTUNE = tf.data.AUTOTUNE\n    TARGET_COLS  = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\"]\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:35:24.220551Z","iopub.execute_input":"2023-09-03T06:35:24.2218Z","iopub.status.idle":"2023-09-03T06:35:24.23072Z","shell.execute_reply.started":"2023-09-03T06:35:24.221744Z","shell.execute_reply":"2023-09-03T06:35:24.228992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp /kaggle/input/rsna-train-keras-yolov8/yolov8_xl_atd.keras /kaggle/working/\n","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:32:25.472278Z","iopub.execute_input":"2023-09-03T06:32:25.472704Z","iopub.status.idle":"2023-09-03T06:32:26.715924Z","shell.execute_reply.started":"2023-09-03T06:32:25.472667Z","shell.execute_reply":"2023-09-03T06:32:26.713527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = keras.models.load_model(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:35:27.857974Z","iopub.execute_input":"2023-09-03T06:35:27.858434Z","iopub.status.idle":"2023-09-03T06:35:28.011798Z","shell.execute_reply.started":"2023-09-03T06:35:27.8584Z","shell.execute_reply":"2023-09-03T06:35:28.007113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:35:38.152442Z","iopub.execute_input":"2023-09-03T06:35:38.152922Z","iopub.status.idle":"2023-09-03T06:35:38.207639Z","shell.execute_reply.started":"2023-09-03T06:35:38.152887Z","shell.execute_reply":"2023-09-03T06:35:38.2064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DATA PIPELINE","metadata":{}},{"cell_type":"code","source":"meta_df = pd.read_csv(f'{BASE_PATH}/test_series_meta.csv')\n\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:37:59.107127Z","iopub.execute_input":"2023-09-03T06:37:59.107828Z","iopub.status.idle":"2023-09-03T06:37:59.14809Z","shell.execute_reply.started":"2023-09-03T06:37:59.10779Z","shell.execute_reply":"2023-09-03T06:37:59.14704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#looking for duplicate values\nnum_rows = meta_df.shape[0]\nunique_patients = meta_df['patient_id'].value_counts()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:38:41.955811Z","iopub.execute_input":"2023-09-03T06:38:41.956456Z","iopub.status.idle":"2023-09-03T06:38:41.976465Z","shell.execute_reply.started":"2023-09-03T06:38:41.956412Z","shell.execute_reply":"2023-09-03T06:38:41.974849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:39:21.442909Z","iopub.execute_input":"2023-09-03T06:39:21.443875Z","iopub.status.idle":"2023-09-03T06:39:21.453625Z","shell.execute_reply.started":"2023-09-03T06:39:21.443822Z","shell.execute_reply":"2023-09-03T06:39:21.452097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PREPROCESSING \n\nmeta_df[\"dicom_folder\"] = BASE_PATH + \"/\" + \"test_images\"\\\n                                    + \"/\" + meta_df.patient_id.astype(str)\\\n                                    + \"/\" + meta_df.series_id.astype(str)\n\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:40:42.817075Z","iopub.execute_input":"2023-09-03T06:40:42.817776Z","iopub.status.idle":"2023-09-03T06:40:42.841651Z","shell.execute_reply.started":"2023-09-03T06:40:42.817721Z","shell.execute_reply":"2023-09-03T06:40:42.840211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_folders = meta_df.dicom_folder.to_list()\ntest_paths = []\nfor folder in tqdm(test_folders):\n    test_paths += sorted(glob(os.path.join(folder, '*dcm')))[::STRIDE]","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:43:17.941654Z","iopub.execute_input":"2023-09-03T06:43:17.942186Z","iopub.status.idle":"2023-09-03T06:43:17.993368Z","shell.execute_reply.started":"2023-09-03T06:43:17.94215Z","shell.execute_reply":"2023-09-03T06:43:17.991964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame(test_paths, columns=[\"dicom_path\"])\ntest_df[\"patient_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-3]).astype(int)\ntest_df[\"series_id\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-2]).astype(int)\ntest_df[\"instance_number\"] = test_df.dicom_path.map(lambda x: x.split(\"/\")[-1].replace(\".dcm\",\"\")).astype(int)\n\ntest_df[\"image_path\"] = f\"{IMAGE_DIR}/test_images\"\\\n                    + \"/\" + test_df.patient_id.astype(str)\\\n                    + \"/\" + test_df.series_id.astype(str)\\\n                    + \"/\" + test_df.instance_number.astype(str) +\".png\"\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:44:22.646273Z","iopub.execute_input":"2023-09-03T06:44:22.64687Z","iopub.status.idle":"2023-09-03T06:44:22.677059Z","shell.execute_reply.started":"2023-09-03T06:44:22.646829Z","shell.execute_reply":"2023-09-03T06:44:22.675528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if patients are repeated by finding the number of unique patient IDs\nnum_rows = test_df.shape[0]\nunique_patients = test_df[\"patient_id\"].nunique()\n\nprint(f\"{num_rows=}\")\nprint(f\"{unique_patients=}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:46:08.039402Z","iopub.execute_input":"2023-09-03T06:46:08.039848Z","iopub.status.idle":"2023-09-03T06:46:08.051414Z","shell.execute_reply.started":"2023-09-03T06:46:08.039819Z","shell.execute_reply":"2023-09-03T06:46:08.049691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"DICOM TO PNG PIPELINE","metadata":{}},{"cell_type":"code","source":"!rm -r {IMAGE_DIR}\nos.makedirs(f\"{IMAGE_DIR}/train_images\", exist_ok=True)\nos.makedirs(f\"{IMAGE_DIR}/test_images\", exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:46:58.613517Z","iopub.execute_input":"2023-09-03T06:46:58.614048Z","iopub.status.idle":"2023-09-03T06:46:59.810619Z","shell.execute_reply.started":"2023-09-03T06:46:58.614013Z","shell.execute_reply":"2023-09-03T06:46:59.808678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def standardize_pixel_array(dcm):\n    # Correct DICOM pixel_array if PixelRepresentation == 1.\n    pixel_array = dcm.pixel_array\n    if dcm.PixelRepresentation == 1:\n        bit_shift = dcm.BitsAllocated - dcm.BitsStored\n        dtype = pixel_array.dtype \n        new_array = (pixel_array << bit_shift).astype(dtype) >>  bit_shift\n        pixel_array = pydicom.pixel_data_handlers.util.apply_modality_lut(new_array, dcm)\n    return pixel_array\n\ndef read_xray(path, fix_monochrome=True):\n    dicom = pydicom.dcmread(path)\n    data = standardize_pixel_array(dicom)\n    data = data - np.min(data)\n    data = data / (np.max(data) + 1e-5)\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = 1.0 - data\n    return data\n\ndef resize_and_save(file_path):\n    img = read_xray(file_path)\n    h, w = img.shape[:2]  # orig hw\n    img = cv2.resize(img, (config.RESIZE_DIM, config.RESIZE_DIM), cv2.INTER_LINEAR)\n    img = (img * 255).astype(np.uint8)\n    \n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    sub_path = file_path.split(\"/\",4)[-1].split(\".dcm\")[0] + \".png\"\n    infos = sub_path.split(\"/\")\n    pid = infos[-3]\n    sid = infos[-2]\n    iid = infos[-1]; iid = iid.replace(\".png\",\"\")\n    new_path = os.path.join(IMAGE_DIR, sub_path)\n    os.makedirs(new_path.rsplit(\"/\",1)[0], exist_ok=True)\n    cv2.imwrite(new_path, img)\n    return","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:47:42.570154Z","iopub.execute_input":"2023-09-03T06:47:42.57074Z","iopub.status.idle":"2023-09-03T06:47:42.59407Z","shell.execute_reply.started":"2023-09-03T06:47:42.570696Z","shell.execute_reply":"2023-09-03T06:47:42.591265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfile_paths = test_df.dicom_path.tolist()\n_ = Parallel(n_jobs=2, backend=\"threading\")(\n    delayed(resize_and_save)(file_path) for file_path in tqdm(file_paths, leave=True, position=0)\n)\n\ndel _; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:47:59.273034Z","iopub.execute_input":"2023-09-03T06:47:59.273912Z","iopub.status.idle":"2023-09-03T06:48:00.726087Z","shell.execute_reply.started":"2023-09-03T06:47:59.273866Z","shell.execute_reply":"2023-09-03T06:48:00.724668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"tf.data pipeline","metadata":{}},{"cell_type":"code","source":"def decode_image(image_path):\n    file_bytes = tf.io.read_file(image_path)\n    image = tf.io.decode_png(file_bytes, channels=3, dtype=tf.uint8)\n    image = tf.image.resize(image, config.IMAGE_SIZE, method=\"bilinear\")\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\ndef build_dataset(image_paths):\n    ds = (\n        tf.data.Dataset.from_tensor_slices(image_paths)\n        .map(decode_image, num_parallel_calls=config.AUTOTUNE)\n        .shuffle(config.BATCH_SIZE * 10)\n        .batch(config.BATCH_SIZE)\n        .prefetch(config.AUTOTUNE)\n    )\n    return ds","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:49:34.29101Z","iopub.execute_input":"2023-09-03T06:49:34.291602Z","iopub.status.idle":"2023-09-03T06:49:34.302176Z","shell.execute_reply.started":"2023-09-03T06:49:34.29156Z","shell.execute_reply":"2023-09-03T06:49:34.30065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = test_df.image_path.to_list()\n\nds = build_dataset(paths)\n\nimages = next(iter(ds))\n\nimages.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:50:37.59352Z","iopub.execute_input":"2023-09-03T06:50:37.593981Z","iopub.status.idle":"2023-09-03T06:50:37.882241Z","shell.execute_reply.started":"2023-09-03T06:50:37.59395Z","shell.execute_reply":"2023-09-03T06:50:37.880876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_cv.visualization.plot_image_gallery(\n    images=images,\n    value_range=(0, 1),\n    rows=1,\n    cols=3,\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:50:52.723193Z","iopub.execute_input":"2023-09-03T06:50:52.72399Z","iopub.status.idle":"2023-09-03T06:50:53.108644Z","shell.execute_reply.started":"2023-09-03T06:50:52.723934Z","shell.execute_reply":"2023-09-03T06:50:53.106874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def post_proc(pred):\n    proc_pred = np.empty((pred.shape[0], 2*2 + 3*3), dtype=\"float32\")\n\n    # bowel, extravasation\n    proc_pred[:, 0] = pred[:, 0]\n    proc_pred[:, 1] = 1 - proc_pred[:, 0]\n    proc_pred[:, 2] = pred[:, 1]\n    proc_pred[:, 3] = 1 - proc_pred[:, 2]\n    \n    # liver, kidney, sneel\n    proc_pred[:, 4:7] = pred[:, 2:5]\n    proc_pred[:, 7:10] = pred[:, 5:8]\n    proc_pred[:, 10:13] = pred[:, 8:11]\n\n    return proc_pred","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:51:54.760058Z","iopub.execute_input":"2023-09-03T06:51:54.760639Z","iopub.status.idle":"2023-09-03T06:51:54.772575Z","shell.execute_reply.started":"2023-09-03T06:51:54.760596Z","shell.execute_reply":"2023-09-03T06:51:54.770095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting unique patient IDs from test dataset\npatient_ids = test_df[\"patient_id\"].unique()\n\n# Initializing array to store predictions\npatient_preds = np.zeros(\n    shape=(len(patient_ids), 2*2 + 3*3),\n    dtype=\"float32\"\n)\n\n# Iterating over each patient\nfor pidx, patient_id in tqdm(enumerate(patient_ids), total=len(patient_ids), desc=\"Patients \"):\n    print(f\"Patient ID: {patient_id}\")\n    \n    # Query the dataframe for a particular patient\n    \n    patient_df = test_df.query(\"patient_id == patient_id\")\n    \n    # Getting image paths for a patient\n    patient_paths = patient_df.image_path.tolist()\n\n    # Building dataset for prediction\n    dtest = build_dataset(patient_paths)\n    \n    # Predicting with the model\n    pred = model.predict(dtest)\n    pred = np.concatenate(pred, axis=-1).astype(\"float32\")\n    pred = pred[:len(patient_paths), :]\n    pred = np.mean(pred.reshape(1, len(patient_paths), 11), axis=0)\n    pred = np.max(pred, axis=0, keepdims=True)\n    \n    patient_preds[pidx, :] += post_proc(pred)[0]\n    \n\n    # Deleting variables to free up memory \n    del patient_df, patient_paths, dtest, pred; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:56:38.380219Z","iopub.execute_input":"2023-09-03T06:56:38.38091Z","iopub.status.idle":"2023-09-03T06:56:43.486619Z","shell.execute_reply.started":"2023-09-03T06:56:38.380865Z","shell.execute_reply":"2023-09-03T06:56:43.485736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submission","metadata":{}},{"cell_type":"code","source":"!rm -rf {MODEL_PATH}\n","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:57:08.721741Z","iopub.execute_input":"2023-09-03T06:57:08.722239Z","iopub.status.idle":"2023-09-03T06:57:09.914763Z","shell.execute_reply.started":"2023-09-03T06:57:08.722202Z","shell.execute_reply":"2023-09-03T06:57:09.9121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Submission\npred_df = pd.DataFrame({\"patient_id\":patient_ids,})\npred_df[config.TARGET_COLS] = patient_preds.astype(\"float32\")\n\n# Align with sample submission\nsub_df = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\nsub_df = sub_df[[\"patient_id\"]]\nsub_df = sub_df.merge(pred_df, on=\"patient_id\", how=\"left\")\n\n# Store submission\nsub_df.to_csv(\"submission.csv\",index=False)\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:57:19.598165Z","iopub.execute_input":"2023-09-03T06:57:19.598785Z","iopub.status.idle":"2023-09-03T06:57:19.673034Z","shell.execute_reply.started":"2023-09-03T06:57:19.598741Z","shell.execute_reply":"2023-09-03T06:57:19.671427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-03T06:58:23.859088Z","iopub.execute_input":"2023-09-03T06:58:23.859664Z","iopub.status.idle":"2023-09-03T06:58:23.886931Z","shell.execute_reply.started":"2023-09-03T06:58:23.859625Z","shell.execute_reply":"2023-09-03T06:58:23.88497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}