{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://i.imgur.com/3iLsS6i.png\">\n\n<center><h1> -Fase de exploracion- </h1></center>\n\n> 🦴 **Goal**: Detect and localize cervical spine fractures within CT scans.\n\n### Fracturas cervicales (cuello roto)\nHay **7 huesos que forman las vértebras cervicales** (cuello). Sostienen la cabeza y la conectan con los hombros y el cuerpo. Una fractura, o ruptura, en una de las vértebras cervicales se denomina comúnmente cuello roto.([src here](https://orthoinfo.aaos.org/en/diseases--conditions/cervical-fracture-broken-neck/#:~:text=A%20fracture%2C%20or%20break%2C%20in,Athletes%20are%20also%20at%20risk.)).\n\n<center><img src=\"https://i.imgur.com/hynMD8Z.png\" width=700></center>\n\n### ⬇ Libraries","metadata":{}},{"cell_type":"code","source":"!pip install -qU \"python-gdcm\" pydicom pylibjpeg \"opencv-python-headless\"","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-28T14:03:14.990062Z","iopub.execute_input":"2022-10-28T14:03:14.9913Z","iopub.status.idle":"2022-10-28T14:03:26.819282Z","shell.execute_reply.started":"2022-10-28T14:03:14.991238Z","shell.execute_reply":"2022-10-28T14:03:26.81775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Libraries\nimport os\nimport re\nimport gc\nimport cv2\nimport wandb\nfrom PIL import Image\nimport random\nimport math\nimport shutil\nfrom glob import glob\nfrom tqdm import tqdm\nfrom pprint import pprint\nfrom time import time\nimport warnings\nimport itertools\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib as mpl\nfrom matplotlib import cm\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.patches import Rectangle\nfrom IPython.display import display_html\nplt.rcParams.update({'font.size': 16})\n\n# .dcm handling\nimport pydicom\nimport nibabel as nib\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\n\n# Environment check\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"WANDB_SILENT\"] = \"true\"\nCONFIG = {'competition': 'RSNA_SpineFructure', '_wandb_kernel': 'aot'}\n\n# Custom colors\nclass clr:\n    S = '\\033[1m' + '\\033[94m'\n    E = '\\033[0m'\n    \nmy_colors = [\"#5EAFD9\", \"#449DD1\", \"#3977BB\", \n             \"#2D51A5\", \"#5C4C8F\", \"#8B4679\",\n             \"#C53D4C\", \"#E23836\", \"#FF4633\", \"#FF5746\"]\nCMAP1 = ListedColormap(my_colors)\n\nprint(clr.S+\"Notebook Color Schemes:\"+clr.E)\nsns.palplot(sns.color_palette(my_colors))","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:26.823286Z","iopub.execute_input":"2022-10-28T14:03:26.823845Z","iopub.status.idle":"2022-10-28T14:03:26.933408Z","shell.execute_reply.started":"2022-10-28T14:03:26.823789Z","shell.execute_reply":"2022-10-28T14:03:26.931988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Secrets\nfrom kaggle_secrets import UserSecretsClient\nimport wandb\nfrom wandb.keras import WandbCallback\n\nwandb.login()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:26.935592Z","iopub.execute_input":"2022-10-28T14:03:26.936502Z","iopub.status.idle":"2022-10-28T14:03:26.956912Z","shell.execute_reply.started":"2022-10-28T14:03:26.936446Z","shell.execute_reply":"2022-10-28T14:03:26.955407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_values_on_bars(axs, h_v=\"v\", space=0.4):\n    '''Plots the value at the end of the a seaborn barplot.\n    axs: the ax of the plot\n    h_v: weather or not the barplot is vertical/ horizontal'''\n    \n    def _show_on_single_plot(ax):\n        if h_v == \"v\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() / 2\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_height())\n                ax.text(_x, _y, format(value, ','), ha=\"center\") \n        elif h_v == \"h\":\n            for p in ax.patches:\n                _x = p.get_x() + p.get_width() + float(space)\n                _y = p.get_y() + p.get_height()\n                value = int(p.get_width())\n                ax.text(_x, _y, format(value, ','), ha=\"left\")\n\n    if isinstance(axs, np.ndarray):\n        for idx, ax in np.ndenumerate(axs):\n            _show_on_single_plot(ax)\n    else:\n        _show_on_single_plot(axs)\n        \n        \ndef atoi(text):\n    return int(text) if text.isdigit() else text\n\ndef natural_keys(text):\n    '''\n    alist.sort(key=natural_keys) sorts in human order\n    http://nedbatchelder.com/blog/200712/human_sorting.html\n    (See Toothy's implementation in the comments)\n    '''\n    return [ atoi(c) for c in re.split(r'(\\d+)', text) ]\n        \n        \n# === 🐝 W&B ===\ndef save_dataset_artifact(run_name, artifact_name, path):\n    '''Saves dataset to W&B Artifactory.\n    run_name: name of the experiment\n    artifact_name: under what name should the dataset be stored\n    path: path to the dataset'''\n    \n    run = wandb.init(project='RSNA_SpineFructure', \n                     name=run_name, \n                     config=CONFIG)\n    artifact = wandb.Artifact(name=artifact_name, \n                              type='dataset')\n    artifact.add_file(path)\n\n    wandb.log_artifact(artifact)\n    wandb.finish()\n    print(\"Artifact has been saved successfully.\")\n    \n    \ndef create_wandb_plot(x_data=None, y_data=None, x_name=None, y_name=None, title=None, log=None, plot=\"line\"):\n    '''Create and save lineplot/barplot in W&B Environment.\n    x_data & y_data: Pandas Series containing x & y data\n    x_name & y_name: strings containing axis names\n    title: title of the graph\n    log: string containing name of log'''\n    \n    data = [[label, val] for (label, val) in zip(x_data, y_data)]\n    table = wandb.Table(data=data, columns = [x_name, y_name])\n    \n    if plot == \"line\":\n        wandb.log({log : wandb.plot.line(table, x_name, y_name, title=title)})\n    elif plot == \"bar\":\n        wandb.log({log : wandb.plot.bar(table, x_name, y_name, title=title)})\n    elif plot == \"scatter\":\n        wandb.log({log : wandb.plot.scatter(table, x_name, y_name, title=title)})\n        \n        \ndef create_wandb_hist(x_data=None, x_name=None, title=None, log=None):\n    '''Create and save histogram in W&B Environment.\n    x_data: Pandas Series containing x values\n    x_name: strings containing axis name\n    title: title of the graph\n    log: string containing name of log'''\n    \n    data = [[x] for x in x_data]\n    table = wandb.Table(data=data, columns=[x_name])\n    wandb.log({log : wandb.plot.histogram(table, x_name, title=title)})\n    \n    \n# 🐝 Log Cover Photo\nrun = wandb.init(project='RSNA_SpineFructure', name='CoverPhoto', config=CONFIG)\ncover = plt.imread(\"../input/rsna-fracture-detection/Kaggle Covers (6).png\")\nwandb.log({\"example\": wandb.Image(cover)})\nwandb.finish()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-10-28T14:03:26.960494Z","iopub.execute_input":"2022-10-28T14:03:26.961395Z","iopub.status.idle":"2022-10-28T14:03:36.141708Z","shell.execute_reply.started":"2022-10-28T14:03:26.961356Z","shell.execute_reply":"2022-10-28T14:03:36.140542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Tabular Data [.csv]\n\n#### 🦴 **To Note**:\n* **Train Dataframe**:\n    * `patient_overall` - whether or not there is at least 1 vertebrae that is fractured\n    * `C1 - C7` - which of the vertebraes (if any) are fractured\n    * 2019 unique IDs in total\n* **Train BBox Dataframe**:\n    * contains bbox coordinates with the exact location of the fractures\n* **Train Images folder**:\n    * 2019 folders containing the `.dcm` CT scans of the patient","metadata":{}},{"cell_type":"code","source":"run = wandb.init(project='RSNA_SpineFructure', name='tabular_explore', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:36.143394Z","iopub.execute_input":"2022-10-28T14:03:36.144608Z","iopub.status.idle":"2022-10-28T14:03:40.322935Z","shell.execute_reply.started":"2022-10-28T14:03:36.144568Z","shell.execute_reply":"2022-10-28T14:03:40.321542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE = \"../input/rsna-2022-cervical-spine-fracture-detection\"","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.326654Z","iopub.execute_input":"2022-10-28T14:03:40.32708Z","iopub.status.idle":"2022-10-28T14:03:40.334431Z","shell.execute_reply.started":"2022-10-28T14:03:40.327024Z","shell.execute_reply":"2022-10-28T14:03:40.333144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_data():\n    '''Reads in all .csv files.'''\n    \n    train = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/train.csv\")\n    train_bbox = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/train_bounding_boxes.csv\")\n    test = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/test.csv\")\n    ss = pd.read_csv(\"../input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv\")\n    \n    return train, train_bbox, test, ss\n\n\ndef get_csv_info(csv, name=\"Default\"):\n    '''Prints main information for the speciffied .csv file.'''\n    \n    print(clr.S+f\"=== {name} ===\"+clr.E)\n    print(clr.S+f\"Shape:\"+clr.E, csv.shape)\n    print(clr.S+f\"Missing Values:\"+clr.E, csv.isna().sum().sum(), \"total missing datapoints.\")\n    print(clr.S+\"Columns:\"+clr.E, list(csv.columns), \"\\n\")\n    \n    display_html(csv.head())\n    print(\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.336291Z","iopub.execute_input":"2022-10-28T14:03:40.336648Z","iopub.status.idle":"2022-10-28T14:03:40.349578Z","shell.execute_reply.started":"2022-10-28T14:03:40.336616Z","shell.execute_reply":"2022-10-28T14:03:40.348492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in the data\ntrain, train_bbox, test, ss = read_data()\n\n# Print useful information on it\nfor csv, name in zip([train, train_bbox, test, ss],\n                     [\"Train\", \"Train Bbox\", \"Test\", \"Sample Submission\"]):\n    get_csv_info(csv, name)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.351232Z","iopub.execute_input":"2022-10-28T14:03:40.35192Z","iopub.status.idle":"2022-10-28T14:03:40.46395Z","shell.execute_reply.started":"2022-10-28T14:03:40.351886Z","shell.execute_reply":"2022-10-28T14:03:40.46282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I. Análisis de Fracturas\n\n🦴 **Main Takeaways**:\n* *la clase esta balanceada* - hay una proporción de ~ 50%-50% entre pacientes con y sin fracturas\n\n* *C7* -este es el hueso con las lesiones más frecuentes\n* *C3* - este es el hueso con menos lesiones","metadata":{}},{"cell_type":"code","source":"dt = pd.melt(train, \n             id_vars=['StudyInstanceUID', 'patient_overall'],\n             var_name=\"Vertebra\",\n             value_name=\"Flag\")\n\n# Plot\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(24, 12))\nfig.suptitle('Fractures - Main Analysis', \n             weight=\"bold\", size=25)\n\nsns.countplot(data=train, x=\"patient_overall\", ax=ax1, palette=[my_colors[0], my_colors[4]])\nshow_values_on_bars(ax1, h_v=\"v\", space=0.4)\nax1.set_title(\"Indicador de fractura general del paciente [frecuencia]\", weight=\"bold\", size=19)\nax1.set_xlabel(\"Bandera de fractura\", size = 18, weight=\"bold\")\nax1.set_ylabel(\"\")\nax1.set_yticks([])\n\nhatches = itertools.cycle(['', '//'])\nfor i, bar in enumerate(ax1.patches):\n    hatch = next(hatches)\n    bar.set_hatch(hatch)\n\nsns.countplot(data=dt, x=\"Vertebra\", hue=\"Flag\", ax=ax2, palette=[my_colors[0], my_colors[4]])\nshow_values_on_bars(ax2, h_v=\"v\", space=0.4)\nax2.set_title(\"Marca de fractura en el tipo de vértebra [frecuencia]\", weight=\"bold\", size=19)\nax2.set_xlabel(\"Vertebra\", size = 18, weight=\"bold\")\nax2.set_ylabel(\"\")\nax2.set_yticks([])\n\nfor i, bar in enumerate(ax2.patches):\n    hatch = ''\n    if i in [7, 8, 9, 10, 11, 12, 13]:\n        hatch = '//'\n    bar.set_hatch(hatch)\n\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:06:02.279907Z","iopub.execute_input":"2022-10-28T14:06:02.280376Z","iopub.status.idle":"2022-10-28T14:06:02.689619Z","shell.execute_reply.started":"2022-10-28T14:06:02.28034Z","shell.execute_reply":"2022-10-28T14:06:02.688282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log in barplot to Dashboard\ncreate_wandb_plot(x_data=train[\"patient_overall\"].value_counts().index,\n                  y_data=train[\"patient_overall\"].value_counts().values,\n                  x_name=\"Fracture Flag\", y_name=\"Frequency\",\n                  title=\"Patient Overall Fracture Flag\",\n                  log=\"bar_overall\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:23.907854Z","iopub.execute_input":"2022-10-28T14:18:23.908394Z","iopub.status.idle":"2022-10-28T14:18:24.194442Z","shell.execute_reply.started":"2022-10-28T14:18:23.908349Z","shell.execute_reply":"2022-10-28T14:18:24.193341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### II.Número de Lesiones / Paciente\n\n🦴 **Main Takeaways**:\n* *lesiones en 1 hueso* - de los 961 casos con al menos 1 fractura, más de la mitad tienen una lesión en solo 1 hueso\n* *lesiones en múltiples huesos* - muy pocos casos tienen lesiones en 4 huesos o más al mismo tiempo","metadata":{}},{"cell_type":"code","source":"train[\"total_fractures\"] = train.iloc[:, 2:].sum(axis=1)\n\n# Plot\nplt.figure(figsize=(24, 12))\naxs = sns.countplot(data=train, x=\"total_fractures\", palette=my_colors)\nshow_values_on_bars(axs, h_v=\"v\", space=0.4)\nplt.title(\"Número de huesos fracturados por paciente\", weight=\"bold\", size=19)\nplt.xlabel(\"Huesos Totales Fracturados\", size = 18, weight=\"bold\")\nplt.ylabel(\"\")\nplt.yticks([])\n\n# Hatch\nfor i, bar in enumerate(axs.patches):\n    hatch = ''\n    if i==0:\n        hatch = '-'\n    bar.set_hatch(hatch)\n\n# Arrow\nstyle = \"Simple, tail_width=2, head_width=14, head_length=16\"\nkw = dict(arrowstyle=style, color=my_colors[2])\narrow = patches.FancyArrowPatch((1.8, 900), (1, 660),\n                             connectionstyle=\"arc3,rad=.10\", **kw)\nplt.gca().add_patch(arrow)\nplt.text(x=2, y=900, s=f\"{round(623/961*100,2)}% los casos tienen\",\n         color=my_colors[2], size=17, weight=\"bold\")\nplt.text(x=2, y=860, s=f\"solo 1 hueso fracturado\",\n         color=my_colors[2], size=17, weight=\"bold\")\nplt.text(x=2, y=820, s=f\"al mismo tiempo.\",\n         color=my_colors[2], size=17, weight=\"bold\")\n    \n# Line    \nplt.axvline(x=3.5, linestyle = '--', color=my_colors[5], lw=2)\nplt.text(x=3.6, y=250, s=f\"{round(35/961*100,2)}% los casos tienen más de 4 huesos fracturados al mismo tiempo.\",\n         color=my_colors[5], size=17, weight=\"bold\")\n\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:25.408871Z","iopub.execute_input":"2022-10-28T14:18:25.409336Z","iopub.status.idle":"2022-10-28T14:18:25.76827Z","shell.execute_reply.started":"2022-10-28T14:18:25.40929Z","shell.execute_reply":"2022-10-28T14:18:25.766875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log in barplot to Dashboard\ncreate_wandb_plot(x_data=train[\"total_fractures\"].value_counts().index,\n                  y_data=train[\"total_fractures\"].value_counts().values,\n                  x_name=\"Fracture Bones Fractured\", y_name=\"Frequency\",\n                  title=\"Number of Bones Fractured per Patient\",\n                  log=\"bar_bones\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:29.074076Z","iopub.execute_input":"2022-10-28T14:18:29.074752Z","iopub.status.idle":"2022-10-28T14:18:29.351149Z","shell.execute_reply.started":"2022-10-28T14:18:29.074711Z","shell.execute_reply":"2022-10-28T14:18:29.349952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Finish this Experiment\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:30.204222Z","iopub.execute_input":"2022-10-28T14:18:30.204671Z","iopub.status.idle":"2022-10-28T14:18:35.362453Z","shell.execute_reply.started":"2022-10-28T14:18:30.204627Z","shell.execute_reply":"2022-10-28T14:18:35.361365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Image Data [.dcm]\n\n## 2.1 Images Understanding\n\n🦴 **Pasos**: primero veamos algunos ejemplos para entender cómo se ven estas imágenes. Luego, podemos extraer más detalles de los conjuntos de datos `.dcm` y explorar las características generales (como el tamaño de la imagen, la cantidad de cortes por paciente, etc.).","metadata":{}},{"cell_type":"code","source":"run = wandb.init(project='RSNA_SpineFructure', name='image_explore', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:37.655989Z","iopub.execute_input":"2022-10-28T14:18:37.656441Z","iopub.status.idle":"2022-10-28T14:18:40.723764Z","shell.execute_reply.started":"2022-10-28T14:18:37.656407Z","shell.execute_reply":"2022-10-28T14:18:40.72254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_dcm_images(patient_id, rows=4, cols=8, random=0):\n    '''Show .dcm images based on id.'''\n    \n    N = rows*cols\n    wandb_logs = []\n\n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(24,15))\n    fig.suptitle(f'ID: {patient_id}', weight=\"bold\", size=20)\n\n    # Get .dcm paths\n    dcm_paths = glob(f\"{BASE}/train_images/{patient_id}/*\")\n    print(clr.S+\"Number of TOTAL Slices:\"+clr.E, len(dcm_paths))\n    dcm_paths.sort(key=natural_keys)\n    dcm_paths = dcm_paths[random:(random+N)]\n    # Get corresponging datasets and images\n    datasets = [pydicom.dcmread(path) for path in dcm_paths]\n    images = [apply_voi_lut(dataset.pixel_array, dataset) for dataset in datasets]\n\n    # Loop through the information\n    for data, img, i in zip(datasets, images, range(N)):\n        slice_no = data.SOPInstanceUID.split(\".\")[-1]\n\n        # Plot the image\n        x = i // cols\n        y = i % cols\n\n        axes[x, y].imshow(img, cmap=\"bone\")\n        axes[x, y].set_title(f\"Slice: {slice_no}\", \n                  fontsize=14, weight='bold')\n        axes[x, y].axis('off');\n        \n        # 🐝 Create Image\n        wandb_logs.append(wandb.Image(img, \n                                      caption=f\"Slice: {slice_no}\"))\n        \n    # 🐝 Log images into Dashboard\n    wandb.log({f\"{data.SOPInstanceUID}\": wandb_logs})","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:41.009262Z","iopub.execute_input":"2022-10-28T14:18:41.010429Z","iopub.status.idle":"2022-10-28T14:18:41.024933Z","shell.execute_reply.started":"2022-10-28T14:18:41.010382Z","shell.execute_reply":"2022-10-28T14:18:41.023153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example 1","metadata":{}},{"cell_type":"code","source":"show_dcm_images(\"1.2.826.0.1.3680043.10001\",\n                rows=4, cols=8)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:45.629838Z","iopub.execute_input":"2022-10-28T14:18:45.630279Z","iopub.status.idle":"2022-10-28T14:18:49.532978Z","shell.execute_reply.started":"2022-10-28T14:18:45.630243Z","shell.execute_reply":"2022-10-28T14:18:49.532031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example 2","metadata":{}},{"cell_type":"code","source":"show_dcm_images(\"1.2.826.0.1.3680043.14994\", \n                rows=4, cols=8, random=23)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:18:50.238388Z","iopub.execute_input":"2022-10-28T14:18:50.239212Z","iopub.status.idle":"2022-10-28T14:18:54.295589Z","shell.execute_reply.started":"2022-10-28T14:18:50.239172Z","shell.execute_reply":"2022-10-28T14:18:54.294131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Finish Experiment\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:30.200792Z","iopub.execute_input":"2022-10-28T14:23:30.201284Z","iopub.status.idle":"2022-10-28T14:23:34.292602Z","shell.execute_reply.started":"2022-10-28T14:23:30.201246Z","shell.execute_reply":"2022-10-28T14:23:34.291667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 Saving the images\n\nThe `get_images()` function reads all the `.dcm` files for each Study Instance and saves the corresponding images as `.png` files. \n\n🦴 **File Saving**: Each saved file is named as follows: `StudyInstanceUID/InstanceNumber.png`\n* e.g.: `1.2.826.0.1.3680043.53/4.png`\n    * `StudyInstanceUID` -> 1.2.826.0.1.3680043.53\n    * `InstanceNumber` -> 4 (slice number) in `.png` format\n    \n*! **Update**: I've made a mistake and kept only the images that contain a bbox and excluded all the .dcm files that didn't have any fractures - I've rectified this mistake in V11 of this notebook.*","metadata":{}},{"cell_type":"code","source":"def get_images():\n    '''\n    Retrieves and saves all .dcm images into a zip file.\n    '''\n\n    for k in tqdm(range(len(train))):\n        \n        dt = train.iloc[k, :]\n        os.mkdir(\"../working/png_images/\" + str(dt.StudyInstanceUID))\n\n        # Get all .dcm paths for this Instance\n        dcm_paths = glob(f\"{BASE}/train_images/{dt.StudyInstanceUID}/*\")\n\n        for path in dcm_paths:\n            # Get datasets\n            dataset = pydicom.dcmread(path)\n\n            # Get images\n            image = apply_voi_lut(dataset.pixel_array, dataset)\n            cv2.imwrite(f\"../working/png_images/{dataset.StudyInstanceUID}/{dataset.InstanceNumber}.png\", image)\n            \n    print(\"PNG Images Saved.\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:34.294135Z","iopub.execute_input":"2022-10-28T14:23:34.294601Z","iopub.status.idle":"2022-10-28T14:23:34.302829Z","shell.execute_reply.started":"2022-10-28T14:23:34.294558Z","shell.execute_reply":"2022-10-28T14:23:34.301453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 🦴**Note**: For the purpose of this notebook running faster, I am commenting the cell below (that creates, saves and archives the `.png` images). You can find the full dataset [here](https://www.kaggle.com/datasets/andradaolteanu/rsna-fracture-detection).","metadata":{}},{"cell_type":"code","source":"# # === Uncomment cell to run ===\n\n# # Create file to store the images\n# ! rm -rf png_images\n# ! mkdir png_images\n\n# # Save images\n# get_images()\n\n# # Zip the png folder and remove original\n# zip_name = './zip_png_images/'\n# directory_name = './png_images/'\n\n# shutil.make_archive(zip_name, 'zip', directory_name)\n# ! rm -rf png_images","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:39.333082Z","iopub.execute_input":"2022-10-28T14:23:39.333959Z","iopub.status.idle":"2022-10-28T14:23:39.339694Z","shell.execute_reply.started":"2022-10-28T14:23:39.333914Z","shell.execute_reply":"2022-10-28T14:23:39.338395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_files = len(glob(\"../input/rsna-fracture-detection/zip_png_images/*\"))\nprint(clr.S+\"Total .png files downloaded and saved:\"+clr.E, total_files)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:49.430657Z","iopub.execute_input":"2022-10-28T14:23:49.431116Z","iopub.status.idle":"2022-10-28T14:23:49.445602Z","shell.execute_reply.started":"2022-10-28T14:23:49.431082Z","shell.execute_reply":"2022-10-28T14:23:49.444626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. DICOM Metadata\n\nslice## 3.1 Recuperación de metadatos para 1 archivo\n\n🦴 Por el momento, la única información que me gustaría recuperar es la siguiente:\n* Filas -> la altura de la tomografía computarizada/imagen\n* Columnas -> el ancho de la tomografía computarizada/imagen\n* SOPInstanceUID -> Identificador único que contiene `StudyInstanceUID` + número de segmento\n* ContentDate -> la fecha en que comenzó la creación de datos de píxeles de imagen\n* SliceThickness -> da el grosor de la rebanada de la imagen (*POR HACER: tal vez parear con `Spacing Between Slices` - da la distancia entre dos rebanadas adyacentes*)\n* Número de instancia -> número de segmento\n* ImagePositionPatient -> las coordenadas x, y y z de la esquina superior izquierda (centro del primer vóxel transmitido) de la imagen, en mm\n* ImageOrientationPatient -> los cosenos directores de la primera fila y la primera columna con respecto al paciente\n> 🦴 **Note**: all attribute explanations [are here](https://dicom.innolitics.com/ciods/rt-dose/image-plane/00200037).","metadata":{}},{"cell_type":"code","source":"def get_observation_data(path):\n    '''\n    Get information from the .dcm files\n    '''\n\n    dataset = pydicom.read_file(path)\n    \n    # Dictionary to store the information from the image\n    observation_data = {\n        \"Rows\" : dataset.get(\"Rows\"),\n        \"Columns\" : dataset.get(\"Columns\"),\n        \"SOPInstanceUID\" : dataset.get(\"SOPInstanceUID\"),\n        \"ContentDate\" : dataset.get(\"ContentDate\"),\n        \"SliceThickness\" : dataset.get(\"SliceThickness\"),\n        \"InstanceNumber\" : dataset.get(\"InstanceNumber\"),\n        \"ImagePositionPatient\" : dataset.get(\"ImagePositionPatient\"),\n        \"ImageOrientationPatient\" : dataset.get(\"ImageOrientationPatient\"),\n    }\n\n    # String columns\n    str_columns = [\"SOPInstanceUID\", \"ContentDate\", \n                   \"SliceThickness\", \"InstanceNumber\"]\n    for k in str_columns:\n        observation_data[k] = str(dataset.get(k)) if k in dataset else None\n\n    \n    return observation_data","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:53.755645Z","iopub.execute_input":"2022-10-28T14:23:53.756144Z","iopub.status.idle":"2022-10-28T14:23:53.765181Z","shell.execute_reply.started":"2022-10-28T14:23:53.756105Z","shell.execute_reply":"2022-10-28T14:23:53.763877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# An example\npath = \"../input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/109.dcm\"\nexample = get_observation_data(path)\npprint(example)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:23:58.994429Z","iopub.execute_input":"2022-10-28T14:23:58.995474Z","iopub.status.idle":"2022-10-28T14:23:59.006458Z","shell.execute_reply.started":"2022-10-28T14:23:58.995429Z","shell.execute_reply":"2022-10-28T14:23:59.005402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.2 Create Metadata file","metadata":{}},{"cell_type":"code","source":"def get_metadata():\n    '''\n    Retrieves the desired metadata from the .dcm files and saves it into dataframe.\n    '''\n    \n    exceptions = 0\n    dicts = []\n\n    for k in tqdm(range(len(train))):\n        dt = train.iloc[k, :]\n\n        if dt.patient_overall == 1:\n            # Get all .dcm paths for this Instance\n            dcm_paths = glob(f\"{BASE}/train_images/{dt.StudyInstanceUID}/*\")\n\n            for path in dcm_paths:\n                try:\n                    # Get datasets\n                    dataset = get_observation_data(path)\n                    dicts.append(dataset)\n                except Exception as e:\n                    exceptions += 1\n                    continue\n                    \n    # Convert into df\n    meta_train_data = pd.DataFrame(data=dicts, columns=example.keys())\n    # Export information\n    meta_train_data.to_csv(\"meta_train.csv\", index=False)\n            \n    print(f\"Metadata created. Number of total fails: {exceptions}.\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:02.02669Z","iopub.execute_input":"2022-10-28T14:24:02.027146Z","iopub.status.idle":"2022-10-28T14:24:02.037711Z","shell.execute_reply.started":"2022-10-28T14:24:02.027107Z","shell.execute_reply":"2022-10-28T14:24:02.035885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 🦴 **Note**: All data is available to download [here](https://www.kaggle.com/datasets/andradaolteanu/rsna-fracture-detection).","metadata":{}},{"cell_type":"code","source":"# Create and save the metadata\n# This cell takes ~ 1 hour to run\n# get_metadata()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:08.35044Z","iopub.execute_input":"2022-10-28T14:24:08.350891Z","iopub.status.idle":"2022-10-28T14:24:08.356424Z","shell.execute_reply.started":"2022-10-28T14:24:08.350854Z","shell.execute_reply":"2022-10-28T14:24:08.354964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.3 Metadata Explore\n\nLet's play a bit with the data we have just created 😁","metadata":{}},{"cell_type":"code","source":"run = wandb.init(project='RSNA_SpineFructure', name='meta_explore', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:13.244115Z","iopub.execute_input":"2022-10-28T14:24:13.244528Z","iopub.status.idle":"2022-10-28T14:24:16.544942Z","shell.execute_reply.started":"2022-10-28T14:24:13.244496Z","shell.execute_reply":"2022-10-28T14:24:16.541334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in saved metadata\nmeta_train = pd.read_csv(\"../input/rsna-fracture-detection/meta_train.csv\")\nmeta_train[\"StudyInstanceUID\"] = meta_train[\"SOPInstanceUID\"].apply(lambda x: \".\".join(x.split(\".\")[:-2]))\n\n# Information\nprint(clr.S+\"Total number of .dcm files: (CT scans)\"+clr.E, len(meta_train))\nprint(clr.S+\"[sanity check] Total Number of Study Instances\"+clr.E, meta_train[\"StudyInstanceUID\"].nunique())\n\n# 🐝 Log info into Dashboard\nwandb.log({\"total_dcm_files\": len(meta_train),\n           \"total_study_instances\": meta_train[\"StudyInstanceUID\"].nunique()})","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:18.769498Z","iopub.execute_input":"2022-10-28T14:24:18.769945Z","iopub.status.idle":"2022-10-28T14:24:19.939548Z","shell.execute_reply.started":"2022-10-28T14:24:18.769907Z","shell.execute_reply":"2022-10-28T14:24:19.938164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I. Distribución de archivos .dcm en la instancia de estudio\n\n🦴 **A tener en cuenta**:\n* La distribución está sesgada a la derecha\n* 78% de todas las observaciones 200 a 400 cortes\n* Solo el 6% de las instancias de estudio tienen menos de 200 cortes\n* Hay 50 instancias de estudio con más de 600 cortes","metadata":{}},{"cell_type":"code","source":"# Get the data\ndata = meta_train[\"StudyInstanceUID\"].value_counts().reset_index()\ndata.columns = [\"StudyInstanceUID\", \"count\"]","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:24.27079Z","iopub.execute_input":"2022-10-28T14:24:24.271263Z","iopub.status.idle":"2022-10-28T14:24:24.313909Z","shell.execute_reply.started":"2022-10-28T14:24:24.271226Z","shell.execute_reply":"2022-10-28T14:24:24.312685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot 1\nplt.figure(figsize=(24, 12))\nsns.distplot(data[\"count\"], rug=True, bins=10,\n             rug_kws={\"color\": my_colors[1]},\n             kde_kws={\"color\": my_colors[8], \"lw\": 5, \"alpha\": 0.7},\n             hist_kws={\"histtype\": \"step\", \"linewidth\": 3, \"alpha\": 1, \"color\": my_colors[1]}\n            )\n\nplt.title(\"Distribución de archivos .dcm en la instancia de estudio\", weight=\"bold\", size=25)\nplt.xlabel(\"Número de rebanadas\", size = 18, weight=\"bold\")\nplt.ylabel(\"Frecuencia\")\n\n# Arrow\nstyle = \"Simple, tail_width=1, head_width=12, head_length=14\"\nkw = dict(arrowstyle=style, color=\"black\")\narrow = patches.FancyArrowPatch((600, 0.0033), (300, 0.0020),\n                             connectionstyle=\"arc3,rad=-.10\", **kw)\nplt.gca().add_patch(arrow)\n\nplt.text(x=400, y=0.0035, s=f\"El 78% de las observaciones tienen de 200 a 400 segmentos.\", \n         color=\"black\", size=17)\n\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:27.648826Z","iopub.execute_input":"2022-10-28T14:24:27.649288Z","iopub.status.idle":"2022-10-28T14:24:28.023367Z","shell.execute_reply.started":"2022-10-28T14:24:27.649246Z","shell.execute_reply":"2022-10-28T14:24:28.022131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log in histogram to Dashboard\ncreate_wandb_hist(x_data=data[\"count\"], \n                  x_name=\"Number of Slices\",\n                  title=\"Distribution of .dcm files on Study Instance\",\n                  log=\"hist_slices\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:31.657417Z","iopub.execute_input":"2022-10-28T14:24:31.657905Z","iopub.status.idle":"2022-10-28T14:24:31.972813Z","shell.execute_reply.started":"2022-10-28T14:24:31.657867Z","shell.execute_reply":"2022-10-28T14:24:31.971559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random.seed(25)\n\n# Add a fictive y\n# this \"y\" doesn't mean anything, it's just for\n# showcasing purposes\ndata[\"y\"] = [random.randint(0, 100) for i in range(len(data))]\nperc = round(data[data[\"count\"]<=10].shape[0]/len(data), 3)*100\n\nplt.figure(figsize=(24, 12))\nsns.scatterplot(data=data, x=\"count\", y=\"y\", size=\"count\", alpha=0.65, sizes=(500, 2000),\n               hue=\"count\", palette=CMAP1)\n\nplt.title(\"Distribución de archivos .dcm en la instancia de estudio\", weight=\"bold\", size=25)\nplt.xlabel(\"Numero de segmentos\", size = 18, weight=\"bold\")\nplt.ylabel(\"\")\nplt.yticks([])\n\nplt.axvline(x=200, linestyle = '--', color=\"black\", lw=4)\nplt.axvline(x=400, linestyle = '--', color=\"black\", lw=4)\nplt.text(x=225, y=50, s=f\"~El 80% de los datos está aquí.\", color=\"black\", size=20, weight=\"bold\")\n\nplt.axvline(x=600, linestyle = '--', color=\"black\", lw=4)\nplt.text(x=610, y=7, s=f\"solo 50 instancias con segmentos >=600\", color=\"black\", size=14, weight=\"bold\")\nplt.arrow(x=600, y=5, dx=200, dy=0, color=\"black\", lw=4, \n          head_width=2, head_length=8)\n\nplt.legend('',frameon=False)\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:24:43.392265Z","iopub.execute_input":"2022-10-28T14:24:43.392703Z","iopub.status.idle":"2022-10-28T14:24:43.934213Z","shell.execute_reply.started":"2022-10-28T14:24:43.392669Z","shell.execute_reply":"2022-10-28T14:24:43.931135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###II. Tamaños de imagen\n\n🦴 **Nota**:\n* El 99,6% de las imágenes tienen un tamaño fijo de 512 por 512.\n* El resto de las porciones (el otro 0,3 %) también debe cambiarse de tamaño a 512x512.","metadata":{}},{"cell_type":"code","source":"# Create column with full image size (height x width)\nmeta_train[\"ImageSize\"] = meta_train[\"Rows\"].astype(str) + \" x \" + meta_train[\"Columns\"].astype(str)\n\n\n# Plot\nplt.figure(figsize=(24, 10))\naxs = sns.countplot(data=meta_train, x=\"ImageSize\", palette=my_colors[5:])\nshow_values_on_bars(axs, h_v=\"v\", space=0.4)\nplt.title(\"Frequency of Image Sizes in .dcm data\", weight=\"bold\", size=19)\nplt.xlabel(\"Image Size (height x width)\", size = 18, weight=\"bold\")\nplt.ylabel(\"\")\nplt.yticks([])\n\n# Hatch\nfor i, bar in enumerate(axs.patches):\n    hatch = ''\n    if i==0:\n        hatch = '/'\n    bar.set_hatch(hatch)\n\n# Arrow\nstyle = \"Simple, tail_width=2, head_width=14, head_length=16\"\nkw = dict(arrowstyle=style, color=my_colors[4])\narrow = patches.FancyArrowPatch((0.8, 300000), (0.2, 200000),\n                             connectionstyle=\"arc3,rad=.10\", **kw)\nplt.gca().add_patch(arrow)\nplt.text(x=0.8, y=300000, s=f\"99.6% of images have size 512x512\",\n         color=my_colors[4], size=17, weight=\"bold\")\n\nplt.text(x=1, y=70000, s=f\"The rest 0.32% should be resized too as 512x512\",\n         color=my_colors[6], size=17, weight=\"bold\")\nplt.axhline(xmin=0.4, xmax=0.95, y=60000, linestyle = '--', color=my_colors[6], lw=2)\n\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:17.669308Z","iopub.execute_input":"2022-10-28T14:25:17.669712Z","iopub.status.idle":"2022-10-28T14:25:18.57536Z","shell.execute_reply.started":"2022-10-28T14:25:17.66968Z","shell.execute_reply":"2022-10-28T14:25:18.574101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log in barplot to Dashboard\ncreate_wandb_plot(x_data=meta_train[\"ImageSize\"].value_counts().index,\n                  y_data=meta_train[\"ImageSize\"].value_counts().values,\n                  x_name=\"Image Size (height x width)\", y_name=\"Frequency\",\n                  title=\"Frequency of Image Sizes in .dcm data\",\n                  log=\"bar_imgsize\", plot=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.913169Z","iopub.status.idle":"2022-10-28T14:03:40.91362Z","shell.execute_reply.started":"2022-10-28T14:03:40.913392Z","shell.execute_reply":"2022-10-28T14:03:40.913411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### III. Content Date\n\n> 🦴 **Note**: All .dcm files have the same creation date (2022-07-27), so this column will be dropped.","metadata":{}},{"cell_type":"code","source":"print(clr.S+\"The *only* ContentDate is:\"+clr.E, meta_train[\"ContentDate\"].value_counts().index)\n\nmeta_train.drop(columns=\"ContentDate\", axis=1, inplace=True)\nprint(clr.S+\"Column ContentDate was dropped.\"+clr.E)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:36.359651Z","iopub.execute_input":"2022-10-28T14:25:36.360107Z","iopub.status.idle":"2022-10-28T14:25:36.475379Z","shell.execute_reply.started":"2022-10-28T14:25:36.360069Z","shell.execute_reply":"2022-10-28T14:25:36.474081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:47.670624Z","iopub.execute_input":"2022-10-28T14:25:47.671051Z","iopub.status.idle":"2022-10-28T14:25:52.199623Z","shell.execute_reply.started":"2022-10-28T14:25:47.671003Z","shell.execute_reply":"2022-10-28T14:25:52.198607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Bounding Boxes\n\nLet's now analyse **only the slices/images with bounding boxes** in them (aka the slices that contain 1 or more ruptures.","metadata":{}},{"cell_type":"code","source":"run = wandb.init(project='RSNA_SpineFructure', name='bbox_explore', config=CONFIG)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:52.201353Z","iopub.execute_input":"2022-10-28T14:25:52.201985Z","iopub.status.idle":"2022-10-28T14:25:55.508873Z","shell.execute_reply.started":"2022-10-28T14:25:52.201948Z","shell.execute_reply":"2022-10-28T14:25:55.507363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(clr.S+\"Number of Study Instances that contain bounding boxes:\"+clr.E,\n      train_bbox[\"StudyInstanceUID\"].nunique())","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:56.753605Z","iopub.execute_input":"2022-10-28T14:25:56.754078Z","iopub.status.idle":"2022-10-28T14:25:56.765692Z","shell.execute_reply.started":"2022-10-28T14:25:56.754024Z","shell.execute_reply":"2022-10-28T14:25:56.764848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.1 Number of Bounding Boxes per Study Instance\n\n🦴 **Note**:\n* ~50% of all slices have about 25 or less bboxes per study.\n    * ! ~16% of slices have less than 10 bboxes per study instance.\n* there are 6 studies that are outliers - with more than 100 bboxes per study.\n\n*! Keep in mind: number of bboxes == number of fractures*.","metadata":{}},{"cell_type":"code","source":"# Plot 1\nplt.figure(figsize=(24, 12))\nsns.distplot(train_bbox[\"StudyInstanceUID\"].value_counts().values, rug=True, bins=20,\n             rug_kws={\"color\": my_colors[1]},\n             kde_kws={\"color\": my_colors[8], \"lw\": 5, \"alpha\": 0.7},\n             hist_kws={\"histtype\": \"step\", \"linewidth\": 3, \"alpha\": 1, \"color\": my_colors[1]}\n            )\n\nplt.title(\"Number of Bounding Boxes per Study Instance\", weight=\"bold\", size=25)\nplt.xlabel(\"Number of BBoxes / Study Instance\", size = 18, weight=\"bold\")\nplt.ylabel(\"Frequency\")\n\n# Arrow\nstyle = \"Simple, tail_width=1, head_width=12, head_length=14\"\nkw = dict(arrowstyle=style, color=\"black\")\narrow = patches.FancyArrowPatch((70, 0.022), (20, 0.018),\n                             connectionstyle=\"arc3,rad=.10\", **kw)\nplt.gca().add_patch(arrow)\nplt.text(x=70, y=0.022, s=f\"~50% of slices have less than 25 bboxes/study.\", \n         color=\"black\", size=17)\n\nstyle = \"Simple, tail_width=1, head_width=12, head_length=14\"\nkw = dict(arrowstyle=style, color=\"black\")\narrow = patches.FancyArrowPatch((160, 0.008), (150, 0.001),\n                             connectionstyle=\"arc3,rad=.10\", **kw)\nplt.gca().add_patch(arrow)\nplt.text(x=130, y=0.0085, s=f\"Only 6 slices have 100+ bboxes/study.\", \n         color=\"black\", size=17)\n\nsns.despine(right=True, top=True, left=True);","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:25:59.764209Z","iopub.execute_input":"2022-10-28T14:25:59.764999Z","iopub.status.idle":"2022-10-28T14:26:00.129206Z","shell.execute_reply.started":"2022-10-28T14:25:59.764957Z","shell.execute_reply":"2022-10-28T14:26:00.128195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Log in histogram to Dashboard\ncreate_wandb_hist(x_data=train_bbox[\"StudyInstanceUID\"].value_counts().values, \n                  x_name=\"Number of BBoxes / Study Instance\",\n                  title=\"Number of Bounding Boxes per Study Instance\",\n                  log=\"hist_bboxes\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:07.797687Z","iopub.execute_input":"2022-10-28T14:26:07.79813Z","iopub.status.idle":"2022-10-28T14:26:08.08457Z","shell.execute_reply.started":"2022-10-28T14:26:07.79809Z","shell.execute_reply":"2022-10-28T14:26:08.083639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4.2 BBoxes on Images\n\nLet's now plot some bounding boxes on out CT scans.","metadata":{}},{"cell_type":"code","source":"# Compute x2 and y2\ntrain_bbox[\"x2\"] = train_bbox[\"x\"] + train_bbox[\"width\"]\ntrain_bbox[\"y2\"] = train_bbox[\"y\"] + train_bbox[\"height\"]\n\n# Rename columns\ntrain_bbox.rename(columns={\"x\": \"x1\", \"y\": \"y1\"}, inplace=True)\n\n# Change to int\ntrain_bbox[\"x1\"] = train_bbox[\"x1\"].apply(lambda x: int(x))\ntrain_bbox[\"x2\"] = train_bbox[\"x2\"].apply(lambda x: int(x))\ntrain_bbox[\"y1\"] = train_bbox[\"y1\"].apply(lambda x: int(x))\ntrain_bbox[\"y2\"] = train_bbox[\"y2\"].apply(lambda x: int(x))\n\ntrain_bbox.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:09.321457Z","iopub.execute_input":"2022-10-28T14:26:09.321956Z","iopub.status.idle":"2022-10-28T14:26:09.366958Z","shell.execute_reply.started":"2022-10-28T14:26:09.321915Z","shell.execute_reply":"2022-10-28T14:26:09.365539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_dcm_bboxes(studyInstanceUID, rows, cols):\n    \n    # Set defaults\n    BASE_PATH = \"../input/rsna-2022-cervical-spine-fracture-detection/train_images\"\n    N = rows*cols\n    \n    fig, axes = plt.subplots(nrows=rows, ncols=cols, figsize=(24,25))\n\n    # Get dataframe & .dcm paths\n    bbox_df = train_bbox[train_bbox[\"StudyInstanceUID\"]==studyInstanceUID].reset_index(drop=True).head(N)\n    png_paths = [f\"{BASE_PATH}/{studyInstanceUID}/{slice_no}.dcm\" \n                 for slice_no in bbox_df[\"slice_number\"]]\n\n    for path, k in zip(png_paths, range(N)):\n        dataset = pydicom.dcmread(path)\n        img = apply_voi_lut(dataset.pixel_array, dataset)\n        slice_no = png_paths[k].split(\"/\")[-1].split(\".\")[0]\n\n        # BBoxes\n        bbox_list = (bbox_df.loc[k, \"x1\"], bbox_df.loc[k, \"y1\"],\n                     bbox_df.loc[k, \"x2\"], bbox_df.loc[k, \"y2\"])\n        x1, y1, x2, y2 = [ int(x) for x in bbox_list ]\n\n        # Plot the image\n        x_plot = k // cols\n        y_plot = k % cols\n        \n        cv2.rectangle(img, (x1, y1), (x2, y2), (226, 56, 54), 5)\n        axes[x_plot, y_plot].imshow(img, cmap=\"bone\")\n        axes[x_plot, y_plot].set_title(f\"Slice: {slice_no}\", \n                                       fontsize=14, weight='bold')\n        axes[x_plot, y_plot].axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:18.039625Z","iopub.execute_input":"2022-10-28T14:26:18.040079Z","iopub.status.idle":"2022-10-28T14:26:18.055253Z","shell.execute_reply.started":"2022-10-28T14:26:18.040026Z","shell.execute_reply":"2022-10-28T14:26:18.05418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example 1","metadata":{}},{"cell_type":"code","source":"show_dcm_bboxes(studyInstanceUID=\"1.2.826.0.1.3680043.5783\",\n                rows=6, cols=6)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:22.313663Z","iopub.execute_input":"2022-10-28T14:26:22.31546Z","iopub.status.idle":"2022-10-28T14:26:25.985391Z","shell.execute_reply.started":"2022-10-28T14:26:22.315402Z","shell.execute_reply":"2022-10-28T14:26:25.98406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example 2","metadata":{}},{"cell_type":"code","source":"show_dcm_bboxes(studyInstanceUID=\"1.2.826.0.1.3680043.19778\",\n                rows=6, cols=6)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:37.352276Z","iopub.execute_input":"2022-10-28T14:26:37.352684Z","iopub.status.idle":"2022-10-28T14:26:41.141588Z","shell.execute_reply.started":"2022-10-28T14:26:37.352652Z","shell.execute_reply":"2022-10-28T14:26:41.140313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Example 3","metadata":{}},{"cell_type":"code","source":"show_dcm_bboxes(studyInstanceUID=\"1.2.826.0.1.3680043.31077\",\n                rows=6, cols=6)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:49.067238Z","iopub.execute_input":"2022-10-28T14:26:49.067677Z","iopub.status.idle":"2022-10-28T14:26:53.324607Z","shell.execute_reply.started":"2022-10-28T14:26:49.067643Z","shell.execute_reply":"2022-10-28T14:26:53.322902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Finish Experiment\nwandb.finish()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:53.326903Z","iopub.execute_input":"2022-10-28T14:26:53.327483Z","iopub.status.idle":"2022-10-28T14:26:58.203506Z","shell.execute_reply.started":"2022-10-28T14:26:53.327439Z","shell.execute_reply":"2022-10-28T14:26:58.202434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. 3D CT Scans\n\n> 🙏 This part is completely **inspired** by [Seungwon Song](https://www.kaggle.com/songseungwon)'s notebook: [Break Time☕️ Just Enjoy drawing 3D cervical spine](https://www.kaggle.com/code/songseungwon/break-time-just-enjoy-drawing-3d-cervical-spine/notebook) which I LOVED so I had to create something similar and learn ♥\n\n🦴 **nifti file format** -> NIfTI (`.nii`) is a type of file format for neuroimaging.","metadata":{}},{"cell_type":"code","source":"# Get .nii paths\nnii_paths = glob(\"../input/rsna-2022-cervical-spine-fracture-detection/segmentations/*\")\nprint(clr.S+\"Total paths in [segmentations] folder:\"+clr.E, len(nii_paths))","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:26:59.689213Z","iopub.execute_input":"2022-10-28T14:26:59.68996Z","iopub.status.idle":"2022-10-28T14:26:59.697168Z","shell.execute_reply.started":"2022-10-28T14:26:59.689914Z","shell.execute_reply":"2022-10-28T14:26:59.695897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DrawMaskSample():\n    \n    def __init__(self, nii_paths, i, ax):\n        self.i = i\n        self.ax = ax\n        self.nii_sample = nib.load(nii_paths[self.i]).get_fdata()\n        self.get_xyz()\n        self.draw_sample_3d()\n        \n    def get_xyz(self):\n        self.xyz_li = []\n        cnt = 0\n        max_cnt = self.nii_sample.shape[-1]\n        for iter_z, (iter_img) in enumerate(self.nii_sample.transpose(2,0,1)):\n            for iter_x, iter_arr_y in enumerate(iter_img):\n                iter_arr_y = np.where(iter_arr_y)[0]\n                if len(iter_arr_y) >= 1:\n                    iter_arr_y = list(set([iter_arr_y.max(), iter_arr_y.min()]))\n                    xyz = [(iter_x, iter_y, iter_z) for iter_y in iter_arr_y if np.any(iter_y)]\n                    self.xyz_li.append(xyz)\n            cnt += 1\n\n    def draw_sample_3d(self):\n        xyz_matrix = np.array(list(itertools.chain.from_iterable(self.xyz_li)))\n        X = xyz_matrix[:,0]\n        Y = xyz_matrix[:,1]\n        Z = xyz_matrix[:,2]\n\n        self.ax.scatter(X, Y, Z, s=1, alpha=0.05, color='#e3dac9')\n        xlim, ylim, zlim = self.nii_sample.shape\n        self.ax.set_xlim(0, xlim)\n        self.ax.set_ylim(0, ylim)\n        self.ax.set_zlim(0, zlim)\n        self.ax.set_title(f'sample - ({self.i+1})')\n        self.ax.set_yticklabels([])\n        self.ax.set_xticklabels([])\n        self.ax.set_zticklabels([])\n        self.ax.set_facecolor('#081921')","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:27:11.583921Z","iopub.execute_input":"2022-10-28T14:27:11.585076Z","iopub.status.idle":"2022-10-28T14:27:11.598617Z","shell.execute_reply.started":"2022-10-28T14:27:11.585017Z","shell.execute_reply":"2022-10-28T14:27:11.5976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(24,24))\nfor i in range(9):\n    ax = fig.add_subplot(int(f'33{i+1}'), projection='3d')\n    DrawMaskSample(nii_paths, i, ax)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:27:21.353897Z","iopub.execute_input":"2022-10-28T14:27:21.35438Z","iopub.status.idle":"2022-10-28T14:27:55.295457Z","shell.execute_reply.started":"2022-10-28T14:27:21.354332Z","shell.execute_reply":"2022-10-28T14:27:55.294235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 💽 Save Data","metadata":{}},{"cell_type":"code","source":"# Save datasets\ntrain.to_csv(\"train.csv\", index=False)\nmeta_train.to_csv(\"meta_train_clean.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:28:57.616106Z","iopub.execute_input":"2022-10-28T14:28:57.616665Z","iopub.status.idle":"2022-10-28T14:28:59.462348Z","shell.execute_reply.started":"2022-10-28T14:28:57.616615Z","shell.execute_reply":"2022-10-28T14:28:59.461082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 🐝 Make them artifacts and store in Dashboard\nsave_dataset_artifact(run_name=\"save_train\", artifact_name=\"train\",\n                      path=\"../input/rsna-fracture-detection/train.csv\")\nsave_dataset_artifact(run_name=\"save_meta\", artifact_name=\"train_meta\",\n                      path=\"../input/rsna-fracture-detection/meta_train_clean.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:28:59.464423Z","iopub.execute_input":"2022-10-28T14:28:59.46527Z","iopub.status.idle":"2022-10-28T14:29:17.192758Z","shell.execute_reply.started":"2022-10-28T14:28:59.465221Z","shell.execute_reply":"2022-10-28T14:29:17.191529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sns.palplot(sns.color_palette(my_colors))","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.948905Z","iopub.status.idle":"2022-10-28T14:03:40.949325Z","shell.execute_reply.started":"2022-10-28T14:03:40.949125Z","shell.execute_reply":"2022-10-28T14:03:40.949145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# === WIP ===","metadata":{"execution":{"iopub.status.busy":"2022-10-28T14:03:40.950353Z","iopub.status.idle":"2022-10-28T14:03:40.95074Z","shell.execute_reply.started":"2022-10-28T14:03:40.950543Z","shell.execute_reply":"2022-10-28T14:03:40.950561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<center><img src=\"https://i.imgur.com/0cx4xXI.png\"></center>\n\n### 🐝 W&B Dashboard\n\n> My [W&B Dashboard](https://wandb.ai/andrada/RSNA_SpineFructure?workspace=user-andrada).\n\n<center><video src=\"https://i.imgur.com/FXnUE0O.mp4\" width=800 controls></center>\n\n<center><img src=\"https://i.imgur.com/knxTRkO.png\"></center>\n\n### My Specs\n\n* 🖥 Z8 G4 Workstation\n* 💾 2 CPUs & 96GB Memory\n* 🎮 2x NVIDIA A6000\n* 💻 Zbook Studio G7 on the go","metadata":{}}]}