{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pylibjpeg python-gdcm pylibjpeg-libjpeg","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:03:34.542189Z","iopub.execute_input":"2023-01-11T13:03:34.542651Z","iopub.status.idle":"2023-01-11T13:03:51.856828Z","shell.execute_reply.started":"2023-01-11T13:03:34.542618Z","shell.execute_reply":"2023-01-11T13:03:51.855357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\n\nimport pydicom\nimport pylibjpeg\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom itertools import islice\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-11T13:04:37.196168Z","iopub.execute_input":"2023-01-11T13:04:37.197766Z","iopub.status.idle":"2023-01-11T13:04:38.261988Z","shell.execute_reply.started":"2023-01-11T13:04:37.197707Z","shell.execute_reply":"2023-01-11T13:04:38.260317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:05:09.273787Z","iopub.execute_input":"2023-01-11T13:05:09.275431Z","iopub.status.idle":"2023-01-11T13:05:09.44426Z","shell.execute_reply.started":"2023-01-11T13:05:09.275347Z","shell.execute_reply":"2023-01-11T13:05:09.442436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"site_id\", df[\"site_id\"].unique(), df[\"site_id\"].isna().sum())\nprint(\"patient_id\", df[\"patient_id\"].nunique(), df[\"patient_id\"].isna().sum())\nprint(\"image_id\", df[\"image_id\"].nunique(), df[\"image_id\"].isna().sum())\nprint(\"laterality\", df[\"laterality\"].unique(), df[\"laterality\"].isna().sum())\nprint(\"view\", df[\"view\"].unique(), df[\"view\"].isna().sum())\nprint(\"age\", df[\"age\"].nunique(), df[\"age\"].isna().sum(), df[\"age\"].mean(), df[\"age\"].std())\nprint(\"cancer\", df[\"cancer\"].unique(), df[\"cancer\"].isna().sum())\nprint(\"biopsy\", df[\"biopsy\"].unique(), df[\"biopsy\"].isna().sum())\nprint(\"invasive\", df[\"invasive\"].unique(), df[\"invasive\"].isna().sum())\nprint(\"BIRADS\", df[\"BIRADS\"].unique(), df[\"BIRADS\"].isna().sum())\nprint(\"density\", df[\"density\"].unique(), df[\"density\"].isna().sum())\nprint(\"machine_id\", df[\"machine_id\"].unique(), df[\"machine_id\"].isna().sum())\nprint(\"difficult_negative_case\", df[\"difficult_negative_case\"].unique(), df[\"difficult_negative_case\"].isna().sum())","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:02.863429Z","iopub.execute_input":"2023-01-10T12:16:02.86382Z","iopub.status.idle":"2023-01-10T12:16:02.924473Z","shell.execute_reply.started":"2023-01-10T12:16:02.863783Z","shell.execute_reply":"2023-01-10T12:16:02.923328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def take(n, iterable):\n    \"\"\"Return the first n items of the iterable as a list.\"\"\"\n    return list(islice(iterable, n))\n\ndef rescale_img_to_hu(dcm_ds):\n    \"\"\"Rescales the image to Hounsfield unit.\"\"\"\n    dcm_ds.PhotometricInterpretation = 'YBR_FULL'\n    data = dcm_ds.pixel_array\n    if dcm_ds.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data\n    return data * dcm_ds.RescaleSlope + dcm_ds.RescaleIntercept\n\ndef GetDataByPatient(patient_id, subset=[\"L_CC\", \"L_MLO\", \"R_CC\", \"R_MLO\"]):\n    if not subset or len(subset) == 0:\n        print(\"Input Params Error\")\n        return 0\n    if len(subset) == 4:\n        fig, axs = plt.subplots(1, 4, figsize=(30,10), squeeze=False)\n    else:\n        fig, axs = plt.subplots(1, len(subset), figsize=(9,12), squeeze=False)\n    axs = axs.flatten()\n    df_select = df[df[\"patient_id\"] == patient_id]\n    showarr = []\n    for type_str in subset:\n        if \"L\" in type_str.split(\"_\")[0]:\n            if \"CC\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"L\") & (df_select[\"view\"] == \"CC\")][\"image_id\"].values[0])\n                continue\n            if \"MLO\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"L\") & (df_select[\"view\"] == \"MLO\")][\"image_id\"].values[0])\n                continue\n        if \"R\" in type_str.split(\"_\")[0]:\n            if \"CC\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"R\") & (df_select[\"view\"] == \"CC\")][\"image_id\"].values[0])\n                continue\n            if \"MLO\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"R\") & (df_select[\"view\"] == \"MLO\")][\"image_id\"].values[0])\n                continue\n                \n    for i, imageid in enumerate(showarr):\n        img_path = \"../input/rsna-breast-cancer-detection/train_images/\" + str(patient_id) + \"/\" + str(imageid) + \".dcm\"\n        ds = pydicom.dcmread(img_path)\n        axs[i].grid(False)\n        axs[i].title.set_text(str(df_select[df_select[\"image_id\"] == imageid][\"patient_id\"].iloc[0]) + \" \" + \n                              str(df_select[df_select[\"image_id\"] == imageid][\"laterality\"].iloc[0]) + \" \" + \n                              str(df_select[df_select[\"image_id\"] == imageid][\"view\"].iloc[0]) + \" \" + \n                              str(df_select[df_select[\"image_id\"] == imageid][\"cancer\"].iloc[0]))\n        rescaleimg = rescale_img_to_hu(ds)\n        rescaleimg = rescaleimg * (255.0 / rescaleimg.max())\n        if df_select[df_select[\"image_id\"] == imageid][\"machine_id\"].iloc[0] in [29, 210]:\n             rescaleimg = rescaleimg * -1 + rescaleimg.max()\n        shape = rescaleimg.shape\n        axs[i].imshow(rescaleimg, cmap=\"bone\")\n        \n    plt.show()\n    return shape\n\ndef GetParamsByPatient(patient_id, subset=[\"L_CC\", \"L_MLO\", \"R_CC\", \"R_MLO\"]):\n    if not subset or len(subset) == 0:\n        print(\"Input Params Error\")\n        return 0\n    df_select = df[df[\"patient_id\"] == patient_id]\n    showarr = []\n    for type_str in subset:\n        if \"L\" in type_str.split(\"_\")[0]:\n            if \"CC\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"L\") & (df_select[\"view\"] == \"CC\")][\"image_id\"].values[0])\n                continue\n            if \"MLO\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"L\") & (df_select[\"view\"] == \"MLO\")][\"image_id\"].values[0])\n                continue\n        if \"R\" in type_str.split(\"_\")[0]:\n            if \"CC\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"R\") & (df_select[\"view\"] == \"CC\")][\"image_id\"].values[0])\n                continue\n            if \"MLO\" in type_str.split(\"_\")[1]:\n                showarr.append(df_select[(df_select[\"laterality\"] == \"R\") & (df_select[\"view\"] == \"MLO\")][\"image_id\"].values[0])\n                continue\n     \n    result_params_arr = {}\n    for i, imageid in enumerate(showarr):\n        img_path = \"../input/rsna-breast-cancer-detection/train_images/\" + str(patient_id) + \"/\" + str(imageid) + \".dcm\"\n        ds = pydicom.dcmread(img_path)\n        result_params_arr[imageid] = []\n        result_params_arr[imageid].append(ds)\n    \n    return result_params_arr","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:04:43.795206Z","iopub.execute_input":"2023-01-11T13:04:43.79569Z","iopub.status.idle":"2023-01-11T13:04:43.828773Z","shell.execute_reply.started":"2023-01-11T13:04:43.795649Z","shell.execute_reply":"2023-01-11T13:04:43.827104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:02.955614Z","iopub.execute_input":"2023-01-10T12:16:02.956053Z","iopub.status.idle":"2023-01-10T12:16:02.975012Z","shell.execute_reply.started":"2023-01-10T12:16:02.956016Z","shell.execute_reply":"2023-01-10T12:16:02.973728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=df[\"view\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:02.976839Z","iopub.execute_input":"2023-01-10T12:16:02.977563Z","iopub.status.idle":"2023-01-10T12:16:03.26429Z","shell.execute_reply.started":"2023-01-10T12:16:02.977515Z","shell.execute_reply":"2023-01-10T12:16:03.263423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"errordict = {}\nfor patient_id in tqdm(df[\"patient_id\"].unique(), leave=False, position=0):\n    dfshape = df[df[\"patient_id\"] == patient_id].shape[0]\n    if dfshape != 4:\n        errordict[patient_id] = dfshape\n        \nlen(errordict)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:03.265441Z","iopub.execute_input":"2023-01-10T12:16:03.266203Z","iopub.status.idle":"2023-01-10T12:16:05.626006Z","shell.execute_reply.started":"2023-01-10T12:16:03.266167Z","shell.execute_reply":"2023-01-10T12:16:05.622147Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"errordictnp = np.array(list(errordict.values()))\nprint(\"Count:\", errordictnp.shape)\nprint(\"Max:\", errordictnp.max())\nprint(\"Min:\", errordictnp.min())\nprint(\"Mean:\", errordictnp.mean())\nprint(\"Std:\", errordictnp.std())","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:26.062086Z","iopub.execute_input":"2023-01-10T12:16:26.062942Z","iopub.status.idle":"2023-01-10T12:16:26.072703Z","shell.execute_reply.started":"2023-01-10T12:16:26.062862Z","shell.execute_reply":"2023-01-10T12:16:26.071312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr_bad = []\nfor row in tqdm(errordict.items(), leave=False, position=0):\n    current_df = df[df[\"patient_id\"] == row[0]]\n    current_df_L = current_df[current_df[\"laterality\"] == \"L\"]\n    current_df_R = current_df[current_df[\"laterality\"] == \"R\"]\n    if \"MLO\" not in current_df_L[\"view\"].values:\n        arr_bad.append(current_df_L[\"patient_id\"].values[0])\n        continue\n    if \"CC\" not in current_df_L[\"view\"].values:\n        arr_bad.append(current_df_L[\"patient_id\"].values[0])\n        continue\n    if \"MLO\" not in current_df_R[\"view\"].values:\n        arr_bad.append(current_df_R[\"patient_id\"].values[0])\n        continue\n    if \"CC\" not in current_df_R[\"view\"].values:\n        arr_bad.append(current_df_R[\"patient_id\"].values[0])\n        continue\n\nprint(len(arr_bad))","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.630243Z","iopub.status.idle":"2023-01-10T12:16:05.630677Z","shell.execute_reply.started":"2023-01-10T12:16:05.630481Z","shell.execute_reply":"2023-01-10T12:16:05.630502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indexes = [] \nfor row in take(10, errordict.items()):\n    indexes.append(row[0])\n\nindexes","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.631838Z","iopub.status.idle":"2023-01-10T12:16:05.632454Z","shell.execute_reply.started":"2023-01-10T12:16:05.632161Z","shell.execute_reply":"2023-01-10T12:16:05.63219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"view\"\nprint(\"CC counts:\", (df[\"view\"] == \"CC\").sum(), \"Percent:\", (df[\"view\"] == \"CC\").sum() / df.shape[0])\nprint(\"MLO counts:\", (df[\"view\"] == \"MLO\").sum(), \"Percent:\", (df[\"view\"] == \"MLO\").sum() / df.shape[0])\nprint(\"ML counts:\", (df[\"view\"] == \"ML\").sum(), \"Percent:\", (df[\"view\"] == \"ML\").sum() / df.shape[0])\nprint(\"LM counts:\", (df[\"view\"] == \"LM\").sum(), \"Percent:\", (df[\"view\"] == \"LM\").sum() / df.shape[0])\nprint(\"AT counts:\", (df[\"view\"] == \"AT\").sum(), \"Percent:\", (df[\"view\"] == \"AT\").sum() / df.shape[0])\nprint(\"LMO counts:\", (df[\"view\"] == \"LMO\").sum(), \"Percent:\", (df[\"view\"] == \"LMO\").sum() / df.shape[0])\ndf = df[df[\"view\"].isin([\"CC\", \"MLO\"])]","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:17:42.704359Z","iopub.execute_input":"2023-01-10T12:17:42.704803Z","iopub.status.idle":"2023-01-10T12:17:42.779243Z","shell.execute_reply.started":"2023-01-10T12:17:42.704765Z","shell.execute_reply":"2023-01-10T12:17:42.777953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"age\"\nprint(df[\"age\"].isna().value_counts())\ndf = df[~df[\"age\"].isna()]","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:17:52.929306Z","iopub.execute_input":"2023-01-10T12:17:52.930192Z","iopub.status.idle":"2023-01-10T12:17:52.946982Z","shell.execute_reply.started":"2023-01-10T12:17:52.930137Z","shell.execute_reply":"2023-01-10T12:17:52.945337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=df[\"machine_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:17:58.340923Z","iopub.execute_input":"2023-01-10T12:17:58.341368Z","iopub.status.idle":"2023-01-10T12:17:58.636526Z","shell.execute_reply.started":"2023-01-10T12:17:58.34133Z","shell.execute_reply":"2023-01-10T12:17:58.63521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x=df[\"site_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:13.319437Z","iopub.execute_input":"2023-01-10T12:18:13.319919Z","iopub.status.idle":"2023-01-10T12:18:13.536311Z","shell.execute_reply.started":"2023-01-10T12:18:13.319867Z","shell.execute_reply":"2023-01-10T12:18:13.534974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Count:\", df[df[\"site_id\"] == 1].shape[0])\ndf_select = df[df[\"site_id\"] == 1]\ntempnum = df_select[~df_select[\"machine_id\"].isin([49])].shape[0]\nprint(\"Count without 49:\", tempnum, \"Percent:\", tempnum / df.shape[0])\nsns.countplot(x=df[df[\"site_id\"] == 1][\"machine_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:16.935251Z","iopub.execute_input":"2023-01-10T12:18:16.936576Z","iopub.status.idle":"2023-01-10T12:18:17.216337Z","shell.execute_reply.started":"2023-01-10T12:18:16.936531Z","shell.execute_reply":"2023-01-10T12:18:17.214988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Count:\", df[df[\"site_id\"] == 2].shape[0])\nsns.countplot(x=df[df[\"site_id\"] == 2][\"machine_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:20.750869Z","iopub.execute_input":"2023-01-10T12:18:20.752246Z","iopub.status.idle":"2023-01-10T12:18:20.983961Z","shell.execute_reply.started":"2023-01-10T12:18:20.752187Z","shell.execute_reply":"2023-01-10T12:18:20.982676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df[df[\"cancer\"] == 1][\"machine_id\"].value_counts())\nsns.countplot(x=df[df[\"cancer\"] == 1][\"machine_id\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:27.022155Z","iopub.execute_input":"2023-01-10T12:18:27.022652Z","iopub.status.idle":"2023-01-10T12:18:27.305604Z","shell.execute_reply.started":"2023-01-10T12:18:27.022613Z","shell.execute_reply":"2023-01-10T12:18:27.304093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(15, 12)})\nsns.heatmap(df.corr(), cmap=\"Blues\", annot=True)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:34.081132Z","iopub.execute_input":"2023-01-10T12:18:34.081552Z","iopub.status.idle":"2023-01-10T12:18:35.204325Z","shell.execute_reply.started":"2023-01-10T12:18:34.081522Z","shell.execute_reply":"2023-01-10T12:18:35.202674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GetDataByPatient(10706)","metadata":{"execution":{"iopub.status.busy":"2023-01-11T13:05:18.515852Z","iopub.execute_input":"2023-01-11T13:05:18.516301Z","iopub.status.idle":"2023-01-11T13:05:23.85181Z","shell.execute_reply.started":"2023-01-11T13:05:18.516265Z","shell.execute_reply":"2023-01-11T13:05:23.850584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nmachine_arr = [21, 29, 48, 49, 93, 170, 190, 197, 210, 216]\nfor machine_id in machine_arr:\n    rand_sample = df[df[\"machine_id\"] == machine_id].sample(n=1, axis=\"rows\")\n    patient_id_sample = rand_sample[\"patient_id\"].values[0]\n    print(\"Machine id:\", machine_id, \"Patient id:\", patient_id_sample)\n    GetDataByPatient(patient_id_sample)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:18:50.393531Z","iopub.execute_input":"2023-01-10T12:18:50.394042Z","iopub.status.idle":"2023-01-10T12:20:25.354455Z","shell.execute_reply.started":"2023-01-10T12:18:50.393993Z","shell.execute_reply":"2023-01-10T12:20:25.353025Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### DEVICES AND DATA LOADING RESEARCH","metadata":{}},{"cell_type":"code","source":"# 29(INV) 93(OK) 190(OK) 197(OK) 210(INV) 216(OK)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.665713Z","iopub.status.idle":"2023-01-10T12:16:05.66634Z","shell.execute_reply.started":"2023-01-10T12:16:05.666015Z","shell.execute_reply":"2023-01-10T12:16:05.666042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"machine_id = 216\nn_samples = 3\nto_negative = False\nfor i in range(n_samples):\n    rand_sample = df[df[\"machine_id\"] == machine_id].sample(n=1, axis=\"rows\")\n    patient_id_sample = rand_sample[\"patient_id\"].values[0]\n    print(\"Machine id:\", machine_id, \"Patient id:\", patient_id_sample)\n    GetDataByPatientInvert(patient_id_sample, to_negative=to_negative)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.667625Z","iopub.status.idle":"2023-01-10T12:16:05.668214Z","shell.execute_reply.started":"2023-01-10T12:16:05.66789Z","shell.execute_reply":"2023-01-10T12:16:05.667934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### HIST OF SHAPES","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\nmachine_arr = [21, 29, 48, 49, 93, 170, 190, 197, 210, 216]\nmachine_arr_result_shapes = {}\nfor machine_id in machine_arr:\n    machine_arr_result_shapes[machine_id] = []\n    for i in tqdm(range(df[df[\"machine_id\"] == machine_id].shape[0]), leave=False, position=0):\n        rand_sample = df[df[\"machine_id\"] == machine_id].iloc[i]\n        patient_id_sample = rand_sample[\"patient_id\"]\n        #print(\"Machine id:\", machine_id, \"Patient id:\", patient_id_sample)\n        result_dict = GetParamsByPatient(patient_id_sample, [\"L_CC\", \"L_MLO\", \"R_CC\", \"R_MLO\"])\n        for i in result_dict.keys():\n            h = result_dict[i][0].Rows\n            w = result_dict[i][0].Columns\n            if (h, w) not in machine_arr_result_shapes[machine_id]:\n                machine_arr_result_shapes[machine_id].append((h, w))","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.670009Z","iopub.status.idle":"2023-01-10T12:16:05.670575Z","shell.execute_reply.started":"2023-01-10T12:16:05.670288Z","shell.execute_reply":"2023-01-10T12:16:05.670316Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"{21: [(2776, 2082)],\n 29: [(5355, 4915)],\n 48: [(4096, 3328)],\n 49: [(3328, 2560), (4096, 3328)],\n 93: [(3062, 2394), (2294, 1914)],\n 170: [(3328, 2560), (4096, 3328)],\n 190: [(2294, 1914), (3062, 2394)],\n 197: [(2850, 2394), (2294, 1914)],\n 210: [(5928, 4728), (4740, 3540)],\n 216: [(2294, 1914)]}","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.671828Z","iopub.status.idle":"2023-01-10T12:16:05.672427Z","shell.execute_reply.started":"2023-01-10T12:16:05.672133Z","shell.execute_reply":"2023-01-10T12:16:05.672161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Positive Data","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\nmachine_arr = [21, 29, 48, 49, 93, 170, 190, 197, 210, 216]\nfor machine_id in machine_arr:\n    rand_sample = df[(df[\"machine_id\"] == machine_id) & (df[\"cancer\"] == 1)]\n    plotted_patient = []\n    for row in rand_sample.values:\n        if row[1] not in plotted_patient and len(plotted_patient) < 3:\n            #print(row)\n            print(\"Machine_id:\", row[12], \"Patient_id:\", row[1])\n            GetDataByPatient(row[1])\n            plotted_patient.append(row[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:20:25.356857Z","iopub.execute_input":"2023-01-10T12:20:25.357607Z","iopub.status.idle":"2023-01-10T12:23:16.534311Z","shell.execute_reply.started":"2023-01-10T12:20:25.357564Z","shell.execute_reply":"2023-01-10T12:23:16.532938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### CREATE DF WITH METADATA (TODO)","metadata":{}},{"cell_type":"code","source":"df.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.675657Z","iopub.status.idle":"2023-01-10T12:16:05.676241Z","shell.execute_reply.started":"2023-01-10T12:16:05.675937Z","shell.execute_reply":"2023-01-10T12:16:05.675965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full = df.copy()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.677535Z","iopub.status.idle":"2023-01-10T12:16:05.678117Z","shell.execute_reply.started":"2023-01-10T12:16:05.677794Z","shell.execute_reply":"2023-01-10T12:16:05.67782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newcols = ['SOP Instance UID', 'Content Date', 'Content Time', 'Patient ID', 'Study Instance UID', 'Series Instance UID', 'Instance Number', 'Image Laterality', 'Samples per Pixel', 'Photometric Interpretation', 'Rows', 'Columns', 'Bits Allocated', 'Bits Stored', 'High Bit', 'Pixel Representation', 'Pixel Padding Value', 'Pixel Intensity Relationship', 'Pixel Intensity Relationship Sign', 'Window Center', 'Window Width', 'Rescale Intercept', 'Rescale Slope', 'Rescale Type', 'VOI LUT Function', 'Partial View', 'Lossy Image Compression']\nfor col in newcols:\n    df_full[col] = \"\"","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.679773Z","iopub.status.idle":"2023-01-10T12:16:05.680363Z","shell.execute_reply.started":"2023-01-10T12:16:05.680061Z","shell.execute_reply":"2023-01-10T12:16:05.680098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_full","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.68165Z","iopub.status.idle":"2023-01-10T12:16:05.682235Z","shell.execute_reply.started":"2023-01-10T12:16:05.681935Z","shell.execute_reply":"2023-01-10T12:16:05.681963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, row in tqdm(enumerate(df_full.values), leave=False, position=0):\n    img_path = \"../input/rsna-breast-cancer-detection/train_images/\" + str(row[1]) + \"/\" + str(row[2]) + \".dcm\"\n    ds = pydicom.dcmread(img_path)\n    for item in ds:\n        if item.name == \"Pixel Data\":\n            continue\n        df_full.loc[i, item.name] = str(item.value)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.683887Z","iopub.status.idle":"2023-01-10T12:16:05.684462Z","shell.execute_reply.started":"2023-01-10T12:16:05.68418Z","shell.execute_reply":"2023-01-10T12:16:05.684206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"convarr[0].keys()","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.685766Z","iopub.status.idle":"2023-01-10T12:16:05.686334Z","shell.execute_reply.started":"2023-01-10T12:16:05.686038Z","shell.execute_reply":"2023-01-10T12:16:05.686064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_dict = GetParamsByPatient(27847, [\"L_CC\", \"L_MLO\", \"R_CC\", \"R_MLO\"])\nfor i in result_dict.keys():\n    h = result_dict[i][0].Rows\n    w = result_dict[i][0].Columns\n    print(\"ID:\", i, (h, w))","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.691299Z","iopub.status.idle":"2023-01-10T12:16:05.691661Z","shell.execute_reply.started":"2023-01-10T12:16:05.691473Z","shell.execute_reply":"2023-01-10T12:16:05.691489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### F1 SCORE","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torchmetrics import Accuracy","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:26:08.26161Z","iopub.execute_input":"2023-01-10T12:26:08.262124Z","iopub.status.idle":"2023-01-10T12:26:13.02148Z","shell.execute_reply.started":"2023-01-10T12:26:08.262087Z","shell.execute_reply":"2023-01-10T12:26:13.019965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pfbeta(labels, predictions, beta):\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / y_true_count\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.696492Z","iopub.status.idle":"2023-01-10T12:16:05.696953Z","shell.execute_reply.started":"2023-01-10T12:16:05.696703Z","shell.execute_reply":"2023-01-10T12:16:05.696723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = torch.tensor([1, 0, 0.5, 0.3, 0.2, 1, 0.5, 0, 0, 1])\npreds = torch.tensor([1, 0, 0.5, 0.3, 0.2, 1, 0, 0, 0, 1])\npfbeta(target, preds, 1)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:16:05.698483Z","iopub.status.idle":"2023-01-10T12:16:05.698882Z","shell.execute_reply.started":"2023-01-10T12:16:05.698688Z","shell.execute_reply":"2023-01-10T12:16:05.698708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = torch.tensor([1.0, 0.0, 1.0, 0.0, 0.5, 0.5, 0.5])\npreds  = torch.tensor([1.0, 1.0, 0.0, 0.0, 0.0, 1.0, 0.5])\n\nbeta = 1\ny_true_count = 0\nctp = 0\ncfp = 0\n\nfor idx in range(len(target)):\n    prediction = min(max(preds[idx], 0), 1)\n    #print(target[idx])\n    if (target[idx]):\n        y_true_count += 1\n        ctp += prediction\n    else:\n        cfp += prediction\n\nprint(\"ctp:\", ctp)\nprint(\"cfp:\", cfp)\nprint(\"y_true_count:\", y_true_count)\n        \nbeta_squared = beta * beta\nc_precision = ctp / (ctp + cfp)\nc_recall = ctp / y_true_count\nprint(\"c_precision:\", c_precision)\nprint(\"c_recall:\", c_recall)\nif (c_precision > 0 and c_recall > 0):\n    result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n    print(result)\nelse:\n    print(0)","metadata":{"execution":{"iopub.status.busy":"2023-01-10T12:26:51.343071Z","iopub.execute_input":"2023-01-10T12:26:51.343475Z","iopub.status.idle":"2023-01-10T12:26:51.357669Z","shell.execute_reply.started":"2023-01-10T12:26:51.343442Z","shell.execute_reply":"2023-01-10T12:26:51.356265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}