{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"⚠️ **I use this notebook to investigate MRI scans and fit a simple model. <br/>Here is a notebook that was submitted to the competition: [Breast Cancer Detection: tf, CNN (test)](https://www.kaggle.com/code/maryiaznak/breast-cancer-detection-tf-cnn-test)**","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-26T21:43:29.900679Z","iopub.execute_input":"2023-02-26T21:43:29.901503Z","iopub.status.idle":"2023-02-26T21:43:29.912195Z","shell.execute_reply.started":"2023-02-26T21:43:29.901403Z","shell.execute_reply":"2023-02-26T21:43:29.910996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 1. Download data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:43:43.469027Z","iopub.execute_input":"2023-02-26T21:43:43.469751Z","iopub.status.idle":"2023-02-26T21:43:43.558177Z","shell.execute_reply.started":"2023-02-26T21:43:43.469714Z","shell.execute_reply":"2023-02-26T21:43:43.556988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:48:55.203634Z","iopub.execute_input":"2023-02-26T22:48:55.20399Z","iopub.status.idle":"2023-02-26T22:48:55.221339Z","shell.execute_reply.started":"2023-02-26T22:48:55.203958Z","shell.execute_reply":"2023-02-26T22:48:55.220412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Data investigation","metadata":{}},{"cell_type":"code","source":"print(f'Length of train dataframe: {len(train_df)}\\n')\nprint(f'Number of NaN values:\\n{train_df.isna().sum()}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:43:43.633755Z","iopub.execute_input":"2023-02-26T21:43:43.634124Z","iopub.status.idle":"2023-02-26T21:43:43.653758Z","shell.execute_reply.started":"2023-02-26T21:43:43.634092Z","shell.execute_reply":"2023-02-26T21:43:43.652435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['site_id', 'patient_id', 'laterality', 'view']\n\nfor column in cols:\n    print(f'Unique values of \"{column}\": {len(train_df[column].unique())}')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:43:43.715538Z","iopub.execute_input":"2023-02-26T21:43:43.715917Z","iopub.status.idle":"2023-02-26T21:43:43.735913Z","shell.execute_reply.started":"2023-02-26T21:43:43.715885Z","shell.execute_reply":"2023-02-26T21:43:43.734844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:43:43.794989Z","iopub.execute_input":"2023-02-26T21:43:43.795753Z","iopub.status.idle":"2023-02-26T21:44:46.898702Z","shell.execute_reply.started":"2023-02-26T21:43:43.795715Z","shell.execute_reply":"2023-02-26T21:44:46.897028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For simple model fitting and future predictions I'm gonna use only patient scans.","metadata":{}},{"cell_type":"code","source":"#let's consider one particular patient\npatient_id = train_df[train_df.cancer == 1].iloc[0].patient_id\n\none_patient_df = train_df[train_df.patient_id == patient_id]\none_patient_df","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:44:46.901139Z","iopub.execute_input":"2023-02-26T21:44:46.901545Z","iopub.status.idle":"2023-02-26T21:44:46.928402Z","shell.execute_reply.started":"2023-02-26T21:44:46.901513Z","shell.execute_reply":"2023-02-26T21:44:46.927351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir = '/kaggle/input/rsna-breast-cancer-detection/{}_images/{}/{}.dcm'\ntrain = 'train'\ntest = 'test'","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:44:46.930156Z","iopub.execute_input":"2023-02-26T21:44:46.930932Z","iopub.status.idle":"2023-02-26T21:44:46.936269Z","shell.execute_reply.started":"2023-02-26T21:44:46.930886Z","shell.execute_reply":"2023-02-26T21:44:46.935036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport dicomsdl\n\nn_rows = len(one_patient_df)\n\nplt.figure(figsize=(5 * n_rows, 5))\nfor i in range(n_rows):\n    row = one_patient_df.iloc[i]\n    \n    plt.subplot(1, n_rows, i + 1)\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, row['view'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})\n    plt.text(200, 700, row['cancer'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:44:46.939312Z","iopub.execute_input":"2023-02-26T21:44:46.939993Z","iopub.status.idle":"2023-02-26T21:45:02.671482Z","shell.execute_reply.started":"2023-02-26T21:44:46.93995Z","shell.execute_reply":"2023-02-26T21:45:02.670517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize=(5, 8))\nsns.countplot(data = train_df, x=\"laterality\", hue=\"cancer\", dodge = False)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:02.672577Z","iopub.execute_input":"2023-02-26T21:45:02.673432Z","iopub.status.idle":"2023-02-26T21:45:03.553108Z","shell.execute_reply.started":"2023-02-26T21:45:02.673393Z","shell.execute_reply":"2023-02-26T21:45:03.552067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nsns.countplot(data = train_df, x=\"view\", hue=\"cancer\", dodge = False)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:03.555012Z","iopub.execute_input":"2023-02-26T21:45:03.556238Z","iopub.status.idle":"2023-02-26T21:45:03.878178Z","shell.execute_reply.started":"2023-02-26T21:45:03.556189Z","shell.execute_reply":"2023-02-26T21:45:03.877087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:03.879678Z","iopub.execute_input":"2023-02-26T21:45:03.880751Z","iopub.status.idle":"2023-02-26T21:45:03.893321Z","shell.execute_reply.started":"2023-02-26T21:45:03.88071Z","shell.execute_reply":"2023-02-26T21:45:03.892195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here I'm plotting one scan sample of each view (without standardizing)\n\nLIST_OF_VIEWS = sorted(list(train_df.view.unique()))\n\nplt.figure(figsize=(5 * len(LIST_OF_VIEWS), 5))\n\nfor i, v in enumerate(LIST_OF_VIEWS, start = 1):\n    plt.subplot(1, len(LIST_OF_VIEWS), i)\n    row = train_df.loc[train_df.view == v].iloc[0]\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, v, fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:03.894925Z","iopub.execute_input":"2023-02-26T21:45:03.895774Z","iopub.status.idle":"2023-02-26T21:45:19.828955Z","shell.execute_reply.started":"2023-02-26T21:45:03.895735Z","shell.execute_reply":"2023-02-26T21:45:19.827725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split dataset in two groups depending on view (to investigate them a bit more):\n# 1. MLO and CC - major group\n# 2. all other views - minor group\n\nmajor_group = ['CC', 'MLO']\n\ntrain_df_major_group = train_df[train_df.view.isin(major_group)]\ntrain_df_minor_group = train_df[~train_df.view.isin(major_group)]","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:19.830767Z","iopub.execute_input":"2023-02-26T21:45:19.831758Z","iopub.status.idle":"2023-02-26T21:45:19.851Z","shell.execute_reply.started":"2023-02-26T21:45:19.831719Z","shell.execute_reply":"2023-02-26T21:45:19.849753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_minor_group.cancer.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:19.855481Z","iopub.execute_input":"2023-02-26T21:45:19.855805Z","iopub.status.idle":"2023-02-26T21:45:19.865611Z","shell.execute_reply.started":"2023-02-26T21:45:19.855776Z","shell.execute_reply":"2023-02-26T21:45:19.864312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# there are not so many examples of minor view group\n# let's display all of them\n\nplt.figure(figsize=(4, 4 * len(train_df_minor_group)))\n\nfor i, row in enumerate(train_df_minor_group.itertuples(index = False), start = 1):\n    plt.subplot(len(train_df_minor_group),1,i)\n    \n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n    \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id} {row.image_id}', fontsize = 13, bbox={'facecolor': 'white'})\n    \n# a lot of these scans look bad!!!","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:45:19.868341Z","iopub.execute_input":"2023-02-26T21:45:19.868821Z","iopub.status.idle":"2023-02-26T21:46:09.468439Z","shell.execute_reply.started":"2023-02-26T21:45:19.868779Z","shell.execute_reply":"2023-02-26T21:46:09.466467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i want to investigate patients which scans look weird\n\ndef display_scans_by_patient_id(ids):\n    plt.figure(figsize=(10 * 5, len(ids) * 7)) # let's consider max 10 pic per patient\n    for i, p_id in enumerate(ids):\n        df = train_df[train_df.patient_id == p_id].iloc[:10]\n\n        for j, row in enumerate(df.itertuples(index = False)):\n            plt.subplot(len(ids), 10, i * 10 + j + 1)\n\n            img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n            img_arr = img.pixelData()\n\n            # standardize all scans\n            img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            if img.PhotometricInterpretation == \"MONOCHROME1\":\n                img_arr = 1 - img_arr\n\n            plt.imshow(img_arr, cmap = plt.cm.bone)\n            plt.text(200, 300, f'{row.patient_id} {row.image_id}', fontsize = 20, bbox={'facecolor': 'white'})\n            plt.text(200, 800, f'{row.cancer}', fontsize = 20, bbox={'facecolor': 'white'})\n            \n\npatient_ids = [1511, 25323, 26530, 38739, 40317, 40832, 43377, 50454]\ndisplay_scans_by_patient_id(patient_ids)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:46:09.470488Z","iopub.execute_input":"2023-02-26T21:46:09.470919Z","iopub.status.idle":"2023-02-26T21:47:56.774842Z","shell.execute_reply.started":"2023-02-26T21:46:09.47088Z","shell.execute_reply":"2023-02-26T21:47:56.772251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bad_ids = patient_ids.copy()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:47:56.776784Z","iopub.execute_input":"2023-02-26T21:47:56.777322Z","iopub.status.idle":"2023-02-26T21:47:56.783577Z","shell.execute_reply.started":"2023-02-26T21:47:56.777267Z","shell.execute_reply":"2023-02-26T21:47:56.782196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ok, what's the goal:\n# I'm going to exclude patiens which have bad scans and don't have cancer (because we have a lot non-cancer scans)\n\n# all above patients have no cancer except 25323 patient\n# for this specific one I'm going to remove only 'bad' scans\nspecific_patient = 25_323\nbad_ids.remove(specific_patient)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:47:56.78576Z","iopub.execute_input":"2023-02-26T21:47:56.786318Z","iopub.status.idle":"2023-02-26T21:47:56.800801Z","shell.execute_reply.started":"2023-02-26T21:47:56.786282Z","shell.execute_reply":"2023-02-26T21:47:56.799518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# thank you\n# https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/373208\nbad_ids.append(27_770) ","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:47:56.802627Z","iopub.execute_input":"2023-02-26T21:47:56.802997Z","iopub.status.idle":"2023-02-26T21:47:56.817199Z","shell.execute_reply.started":"2023-02-26T21:47:56.802965Z","shell.execute_reply":"2023-02-26T21:47:56.816143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's check a few scans (30) which was marked as 'having cancer' from major view group\n# maybe we'll find something weird\n\ndf = train_df_major_group[~(train_df_major_group.patient_id.isin(patient_ids)) & (train_df_major_group.cancer == 1)]\ndf = df.groupby(['patient_id'])['image_id'].apply(lambda x : x.iloc[0]).to_frame().reset_index().sample(n = 30, random_state = 42)\n\nplt.figure(figsize=(6 * 5, 6 * 6)) # 6x5\n\nfor i, row in enumerate(df.itertuples(index = False), start = 1):\n    plt.subplot(6, 5, i)\n\n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n        \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id}', fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:47:56.819078Z","iopub.execute_input":"2023-02-26T21:47:56.819499Z","iopub.status.idle":"2023-02-26T21:49:12.88824Z","shell.execute_reply.started":"2023-02-26T21:47:56.819465Z","shell.execute_reply":"2023-02-26T21:49:12.887323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# and without cancer\n\ndf = train_df_major_group[~(train_df_major_group.patient_id.isin(patient_ids)) & (train_df_major_group.cancer == 0)]\ndf = df.groupby(['patient_id'])['image_id'].apply(lambda x : x.iloc[0]).to_frame().reset_index().sample(n = 30, random_state = 42)\n\nplt.figure(figsize=(6 * 5, 6 * 6)) # 6x5\n\nfor i, row in enumerate(df.itertuples(), start = 1):\n    plt.subplot(6, 5, i)\n\n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n        \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id}', fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:49:12.889598Z","iopub.execute_input":"2023-02-26T21:49:12.890295Z","iopub.status.idle":"2023-02-26T21:50:20.383514Z","shell.execute_reply.started":"2023-02-26T21:49:12.890239Z","shell.execute_reply":"2023-02-26T21:50:20.382255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# again investigate patients having 'weird' scans\n\npatient_ids = [33588, 12943]\ndisplay_scans_by_patient_id(patient_ids)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:20.385535Z","iopub.execute_input":"2023-02-26T21:50:20.385964Z","iopub.status.idle":"2023-02-26T21:50:40.297776Z","shell.execute_reply.started":"2023-02-26T21:50:20.385927Z","shell.execute_reply":"2023-02-26T21:50:40.296883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Preprocess the data","metadata":{}},{"cell_type":"code","source":"cat_cols = ['site_id', 'laterality', 'view']\nnum_cols = ['age']\n\nall_required_cols = cat_cols + num_cols + ['image_id', 'patient_id', 'cancer']","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.299194Z","iopub.execute_input":"2023-02-26T21:50:40.299792Z","iopub.status.idle":"2023-02-26T21:50:40.305502Z","shell.execute_reply.started":"2023-02-26T21:50:40.299737Z","shell.execute_reply":"2023-02-26T21:50:40.304404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(columns = train_df.columns[~train_df.columns.isin(all_required_cols)], inplace = True)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.307106Z","iopub.execute_input":"2023-02-26T21:50:40.307576Z","iopub.status.idle":"2023-02-26T21:50:40.336059Z","shell.execute_reply.started":"2023-02-26T21:50:40.307538Z","shell.execute_reply":"2023-02-26T21:50:40.334122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NaN values of categorical columns\ntrain_df[cat_cols] = train_df[cat_cols].fillna('NA')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.339067Z","iopub.execute_input":"2023-02-26T21:50:40.339984Z","iopub.status.idle":"2023-02-26T21:50:40.356572Z","shell.execute_reply.started":"2023-02-26T21:50:40.33995Z","shell.execute_reply":"2023-02-26T21:50:40.355591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# age column\nfrom sklearn.preprocessing import StandardScaler\n\ntrain_df['na_age'] = train_df.age.isna().astype('int')\n\nscaler = StandardScaler()\ntrain_df['age'] = scaler.fit_transform(train_df['age'].values.reshape(-1, 1))\n\ntrain_df['age'].fillna(0, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.359004Z","iopub.execute_input":"2023-02-26T21:50:40.359978Z","iopub.status.idle":"2023-02-26T21:50:40.406241Z","shell.execute_reply.started":"2023-02-26T21:50:40.359935Z","shell.execute_reply":"2023-02-26T21:50:40.405171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# OHE of categorical values\nohe = OneHotEncoder(drop = 'first', sparse = False) # ver 1.0.2\ntrain_ohe_df = pd.DataFrame(ohe.fit_transform(train_df[cat_cols]))\ntrain_ohe_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.407719Z","iopub.execute_input":"2023-02-26T21:50:40.408787Z","iopub.status.idle":"2023-02-26T21:50:40.478074Z","shell.execute_reply.started":"2023-02-26T21:50:40.408747Z","shell.execute_reply":"2023-02-26T21:50:40.476759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prefix = 'OHE_'\ntrain_ohe_df.columns = [prefix + str(col) for col in train_ohe_df.columns]\ntrain_ohe_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.479983Z","iopub.execute_input":"2023-02-26T21:50:40.48072Z","iopub.status.idle":"2023-02-26T21:50:40.499668Z","shell.execute_reply.started":"2023-02-26T21:50:40.480677Z","shell.execute_reply":"2023-02-26T21:50:40.49843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns = cat_cols)\ntrain_df = pd.concat([train_df, pd.DataFrame(train_ohe_df)], axis = 1)\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.50175Z","iopub.execute_input":"2023-02-26T21:50:40.502603Z","iopub.status.idle":"2023-02-26T21:50:40.539731Z","shell.execute_reply.started":"2023-02-26T21:50:40.502557Z","shell.execute_reply":"2023-02-26T21:50:40.538636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Remove 'bad' scans","metadata":{}},{"cell_type":"code","source":"# patient_ids - list of patients having 'bad' scans\ntrain_df = train_df[~train_df.patient_id.isin(bad_ids)]\n\nindex_to_remove = train_df.loc[(train_df.patient_id == specific_patient) & ( train_df.image_id == 1743461841), :].index\ntrain_df = train_df.drop(index = index_to_remove)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.543189Z","iopub.execute_input":"2023-02-26T21:50:40.543518Z","iopub.status.idle":"2023-02-26T21:50:40.563307Z","shell.execute_reply.started":"2023-02-26T21:50:40.543488Z","shell.execute_reply":"2023-02-26T21:50:40.562198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_count = train_df.cancer.value_counts()\ncancer_count","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.564872Z","iopub.execute_input":"2023-02-26T21:50:40.565859Z","iopub.status.idle":"2023-02-26T21:50:40.575391Z","shell.execute_reply.started":"2023-02-26T21:50:40.565826Z","shell.execute_reply":"2023-02-26T21:50:40.574288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_cancer_count = cancer_count.min()\n\ntrain_df = pd.concat([train_df[train_df.cancer == i].sample(n = min_cancer_count) \n                                  for i, _ in cancer_count.iteritems()])\n\nlen(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.582064Z","iopub.execute_input":"2023-02-26T21:50:40.583346Z","iopub.status.idle":"2023-02-26T21:50:40.601971Z","shell.execute_reply.started":"2023-02-26T21:50:40.583304Z","shell.execute_reply":"2023-02-26T21:50:40.60095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 5. Split the data into train and val sets","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val = train_test_split(train_df, test_size = 0.1, random_state = 42)\nlen(X_train), len(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.603689Z","iopub.execute_input":"2023-02-26T21:50:40.604347Z","iopub.status.idle":"2023-02-26T21:50:40.668584Z","shell.execute_reply.started":"2023-02-26T21:50:40.604306Z","shell.execute_reply":"2023-02-26T21:50:40.667426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. DataGenerator","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:40.67054Z","iopub.execute_input":"2023-02-26T21:50:40.67095Z","iopub.status.idle":"2023-02-26T21:50:45.679737Z","shell.execute_reply.started":"2023-02-26T21:50:40.670912Z","shell.execute_reply":"2023-02-26T21:50:45.678616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from enum import Enum, auto\n\nclass Mode(Enum):\n    TRAIN = auto()\n    TEST = auto()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:45.683109Z","iopub.execute_input":"2023-02-26T21:50:45.684479Z","iopub.status.idle":"2023-02-26T21:50:45.690452Z","shell.execute_reply.started":"2023-02-26T21:50:45.684445Z","shell.execute_reply":"2023-02-26T21:50:45.689217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height = 200\nimg_width = 150\nimg_shape = (img_height, img_width, 1)\n\nnot_required_cols = ['image_id', 'patient_id', 'cancer']\nrequired_cols = train_df.columns[~train_df.columns.isin(not_required_cols)]\n\nother_features_shape = (len(required_cols), )","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:50:45.69198Z","iopub.execute_input":"2023-02-26T21:50:45.692973Z","iopub.status.idle":"2023-02-26T21:50:45.70682Z","shell.execute_reply.started":"2023-02-26T21:50:45.69293Z","shell.execute_reply":"2023-02-26T21:50:45.705574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size,\n                 mode = Mode.TRAIN):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.mode = mode\n        self.mode_str = train if mode == Mode.TRAIN else test\n        \n        self.len = len(df)\n        \n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        X = [np.zeros((self.batch_size, ) + img_shape), np.zeros((self.batch_size, ) + other_features_shape)]\n        y = np.zeros((self.batch_size, 1))\n        \n        for i , pos in enumerate(range(start, end)):\n            if pos >= self.len: break\n                     \n            row = self.df.iloc[pos]\n            patient_id = np.int64(row.patient_id)\n            img_id = np.int64(row.image_id)\n            \n            file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            img_arr = self.__get_img(file_name)\n                 \n            X[0][i,...] = img_arr\n            X[1][i,...] = row.loc[required_cols].values\n            \n            if self.mode == Mode.TRAIN:\n                y[i] = np.int8(row.cancer)\n                \n        return (X, y) if self.mode == Mode.TRAIN else X\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size + bool(self.len % self.batch_size)\n    \n    \n    def __get_img(self, file_name):\n        \n        img = dicomsdl.open(file_name)\n        img_arr = img.pixelData()\n            \n        # standartize all scans\n        img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n        if img.PhotometricInterpretation == \"MONOCHROME1\":\n            img_arr = 1 - img_arr\n            \n        #crop image\n        img_arr = img_arr[:, ~np.all(img_arr == 0, axis = 0)]\n        img_arr = img_arr[~np.all(img_arr == 0, axis = 1), :]\n        \n        # resize image\n        img_arr = np.expand_dims(img_arr, axis = -1)\n        img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n        \n        return img_arr","metadata":{"execution":{"iopub.status.busy":"2023-02-26T21:58:27.411197Z","iopub.execute_input":"2023-02-26T21:58:27.41212Z","iopub.status.idle":"2023-02-26T21:58:27.427402Z","shell.execute_reply.started":"2023-02-26T21:58:27.412084Z","shell.execute_reply":"2023-02-26T21:58:27.426395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = ImageDataGen(X_train, 64)\nval_gen = ImageDataGen(X_val, 64)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:00:07.375176Z","iopub.execute_input":"2023-02-26T22:00:07.37589Z","iopub.status.idle":"2023-02-26T22:00:07.380966Z","shell.execute_reply.started":"2023-02-26T22:00:07.37585Z","shell.execute_reply":"2023-02-26T22:00:07.379949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Model","metadata":{}},{"cell_type":"code","source":"from keras.models import Model, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import (Conv2D, \n                                     MaxPooling2D,\n                                     Dense, \n                                     Dropout,\n                                     GlobalMaxPooling2D,\n                                     Flatten,\n                                     Concatenate)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:00:11.14817Z","iopub.execute_input":"2023-02-26T22:00:11.14853Z","iopub.status.idle":"2023-02-26T22:00:11.156411Z","shell.execute_reply.started":"2023-02-26T22:00:11.148499Z","shell.execute_reply":"2023-02-26T22:00:11.155219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a simple model\n\ninput_1 = Input(shape = img_shape)\n\nx = Conv2D(32, 5, activation = \"relu\", input_shape = img_shape)(input_1)\nx = Conv2D(64, 5, activation = \"relu\")(x)\nx = Conv2D(64, 5, activation = \"relu\")(x)\nx = MaxPooling2D()(x)\nx = Dropout(0.3)(x)\n\nx = Conv2D(64, 5, activation = \"relu\")(x)\nx = Conv2D(128, 5, activation = \"relu\")(x)\nx = Conv2D(128, 5, activation = \"relu\")(x)\nx = MaxPooling2D()(x)\nx = Dropout(0.3)(x)\n\nx = Conv2D(128, 5, activation = \"relu\")(x)\nx = Conv2D(256, 5, activation = \"relu\")(x)\nx = Conv2D(256, 5, activation = \"relu\")(x)\nx = Flatten()(x)\nx = Dropout(0.3)(x)\n\nx = Dense(1024, activation = 'relu')(x)\nx = Dropout(0.3)(x)\nx = Dense(256, activation = 'relu')(x)\nx = Dropout(0.3)(x)\nx = Dense(1, activation = 'sigmoid')(x)\n\ninput_2 = Input(shape = other_features_shape)\nconcat = Concatenate()([x, input_2])\ndense = Dense(16, activation = 'relu')(concat)\nout = Dense(1, activation = 'sigmoid')(dense)\n\nmodel = Model(inputs = [input_1, input_2], outputs = out)\n\nrecall_thresholds = [0.4, 0.5, 0.6, 0.8]\nmodel.compile(optimizer = Adam(learning_rate = 1e-5), \n              loss = 'binary_crossentropy', \n              metrics = [tf.keras.metrics.BinaryAccuracy(threshold = 0.5), \n                         tf.keras.metrics.Recall(thresholds = recall_thresholds),\n                         tf.keras.metrics.Precision(thresholds = recall_thresholds)])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:00:33.279509Z","iopub.execute_input":"2023-02-26T22:00:33.28004Z","iopub.status.idle":"2023-02-26T22:00:33.647421Z","shell.execute_reply.started":"2023-02-26T22:00:33.279993Z","shell.execute_reply":"2023-02-26T22:00:33.64622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# callbacks\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(patience = 5, restore_best_weights = True, verbose = 1) # val_loss\nreduce_lr = ReduceLROnPlateau(factor = 0.1, patience = 2, mode = 'min', verbose = 1) # val_loss ","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:00:42.952503Z","iopub.execute_input":"2023-02-26T22:00:42.952891Z","iopub.status.idle":"2023-02-26T22:00:42.960541Z","shell.execute_reply.started":"2023-02-26T22:00:42.952858Z","shell.execute_reply":"2023-02-26T22:00:42.958316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_gen,\n                    validation_data = val_gen,\n                    epochs = 18,\n                    verbose = 1, \n                    callbacks = [reduce_lr, early_stop])","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:00:47.005218Z","iopub.execute_input":"2023-02-26T22:00:47.005903Z","iopub.status.idle":"2023-02-26T22:32:36.430663Z","shell.execute_reply.started":"2023-02-26T22:00:47.005863Z","shell.execute_reply":"2023-02-26T22:32:36.429618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:46:16.425792Z","iopub.execute_input":"2023-02-26T22:46:16.426184Z","iopub.status.idle":"2023-02-26T22:46:21.225065Z","shell.execute_reply.started":"2023-02-26T22:46:16.426123Z","shell.execute_reply":"2023-02-26T22:46:21.22351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\n\nacc = history.history['binary_accuracy']\nval_acc = history.history['val_binary_accuracy']\n\n\nepochs = range(1, len(loss) + 1)\n\nplt.figure(figsize=(16, 5))\n#accuracy\nplt.subplot(1,2,1)\nplt.plot(epochs, acc, 'bo', label = 'Training accuracy')\nplt.plot(epochs, val_acc, 'r', label = 'Validation accuracy')\nplt.legend()\n\n#loss\nplt.subplot(1,2,2)\nplt.plot(epochs, loss, 'bo', label = 'Trainig loss')\nplt.plot(epochs, val_loss, 'r', label = 'Validation loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recall = history.history['recall']\nval_recall = history.history['val_recall']\n\nrecall_len = len(recall_thresholds)\nplt.figure(figsize=(10, 5 * recall_len))\n\nfor i, (r, vr) in enumerate(zip(zip(*recall), zip(*val_recall)), start = 1):\n    plt.subplot(recall_len, 1, i)\n    plt.plot(epochs, r, 'bo', label = 'Training recall')\n    plt.plot(epochs, vr, 'r', label = 'Validation recall')\n    plt.title(f'Recall threshold = {recall_thresholds[i - 1]}')\n    plt.legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"precision = history.history['precision']\nval_precision = history.history['val_precision']\n\nrecall_len = len(recall_thresholds)\nplt.figure(figsize=(10, 5 * recall_len))\n\nfor i, (p, vp) in enumerate(zip(zip(*precision), zip(*val_precision)), start = 1):\n    plt.subplot(recall_len, 1, i)\n    plt.plot(epochs, p, 'bo', label = 'Training precision')\n    plt.plot(epochs, vp, 'r', label = 'Validation precision')\n    plt.title(f'Precision threshold = {recall_thresholds[i - 1]}')\n    plt.legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Test dataset","metadata":{}},{"cell_type":"code","source":"cat_cols = ['site_id', 'laterality', 'view']\nnum_cols = ['age']\n\nall_required_cols = cat_cols + num_cols + ['image_id', 'patient_id', 'prediction_id']","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.709196Z","iopub.execute_input":"2023-02-26T22:49:08.709555Z","iopub.status.idle":"2023-02-26T22:49:08.715683Z","shell.execute_reply.started":"2023-02-26T22:49:08.709521Z","shell.execute_reply":"2023-02-26T22:49:08.714681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.drop(columns = test_df.columns[~test_df.columns.isin(all_required_cols)], inplace = True)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.737433Z","iopub.execute_input":"2023-02-26T22:49:08.738071Z","iopub.status.idle":"2023-02-26T22:49:08.753066Z","shell.execute_reply.started":"2023-02-26T22:49:08.738035Z","shell.execute_reply":"2023-02-26T22:49:08.752122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# NaN values of categorical columns\ntest_df[cat_cols] = test_df[cat_cols].fillna('NA')","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.755181Z","iopub.execute_input":"2023-02-26T22:49:08.755522Z","iopub.status.idle":"2023-02-26T22:49:08.764357Z","shell.execute_reply.started":"2023-02-26T22:49:08.755485Z","shell.execute_reply":"2023-02-26T22:49:08.763472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# age column\nfrom sklearn.preprocessing import StandardScaler\n\ntest_df['na_age'] = test_df.age.isna().astype('int')\n\ntest_df['age'] = scaler.transform(test_df['age'].values.reshape(-1, 1))\ntest_df['age'].fillna(0, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.766247Z","iopub.execute_input":"2023-02-26T22:49:08.766586Z","iopub.status.idle":"2023-02-26T22:49:08.778746Z","shell.execute_reply.started":"2023-02-26T22:49:08.766552Z","shell.execute_reply":"2023-02-26T22:49:08.777762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# OHE of categorical values\ntest_ohe_df = pd.DataFrame(ohe.transform(test_df[cat_cols]))\ntest_ohe_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.787356Z","iopub.execute_input":"2023-02-26T22:49:08.788119Z","iopub.status.idle":"2023-02-26T22:49:08.809154Z","shell.execute_reply.started":"2023-02-26T22:49:08.788083Z","shell.execute_reply":"2023-02-26T22:49:08.808005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ohe_df.columns = [prefix + str(col) for col in test_ohe_df.columns]\ntest_ohe_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.815624Z","iopub.execute_input":"2023-02-26T22:49:08.815891Z","iopub.status.idle":"2023-02-26T22:49:08.833177Z","shell.execute_reply.started":"2023-02-26T22:49:08.815867Z","shell.execute_reply":"2023-02-26T22:49:08.832391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_df.drop(columns = cat_cols)\ntest_df = pd.concat([test_df, pd.DataFrame(test_ohe_df)], axis = 1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.840352Z","iopub.execute_input":"2023-02-26T22:49:08.842282Z","iopub.status.idle":"2023-02-26T22:49:08.859366Z","shell.execute_reply.started":"2023-02-26T22:49:08.842256Z","shell.execute_reply":"2023-02-26T22:49:08.858218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGen(test_df, 16, mode = Mode.TEST)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:08.901281Z","iopub.execute_input":"2023-02-26T22:49:08.90153Z","iopub.status.idle":"2023-02-26T22:49:08.907742Z","shell.execute_reply.started":"2023-02-26T22:49:08.901507Z","shell.execute_reply":"2023-02-26T22:49:08.906804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_gen)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:09.01844Z","iopub.execute_input":"2023-02-26T22:49:09.018732Z","iopub.status.idle":"2023-02-26T22:49:13.613989Z","shell.execute_reply.started":"2023-02-26T22:49:09.018705Z","shell.execute_reply":"2023-02-26T22:49:13.612978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = pred[:len(test_df)]\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:13.615852Z","iopub.execute_input":"2023-02-26T22:49:13.616237Z","iopub.status.idle":"2023-02-26T22:49:13.63806Z","shell.execute_reply.started":"2023-02-26T22:49:13.616197Z","shell.execute_reply":"2023-02-26T22:49:13.63675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].max().to_frame().reset_index()\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:28.63805Z","iopub.execute_input":"2023-02-26T22:49:28.638756Z","iopub.status.idle":"2023-02-26T22:49:28.650618Z","shell.execute_reply.started":"2023-02-26T22:49:28.638712Z","shell.execute_reply":"2023-02-26T22:49:28.649514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-02-26T22:49:31.518757Z","iopub.execute_input":"2023-02-26T22:49:31.519119Z","iopub.status.idle":"2023-02-26T22:49:31.530071Z","shell.execute_reply.started":"2023-02-26T22:49:31.519086Z","shell.execute_reply":"2023-02-26T22:49:31.529007Z"},"trusted":true},"execution_count":null,"outputs":[]}]}