{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-18T02:38:10.903669Z","iopub.execute_input":"2023-02-18T02:38:10.904101Z","iopub.status.idle":"2023-02-18T02:38:10.908936Z","shell.execute_reply.started":"2023-02-18T02:38:10.904067Z","shell.execute_reply":"2023-02-18T02:38:10.908009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's download the data ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:10.910927Z","iopub.execute_input":"2023-02-18T02:38:10.911586Z","iopub.status.idle":"2023-02-18T02:38:11.0174Z","shell.execute_reply.started":"2023-02-18T02:38:10.911535Z","shell.execute_reply":"2023-02-18T02:38:11.01621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploring the data","metadata":{}},{"cell_type":"code","source":"print(f'Length of train dataframe: {len(train_df)}\\n')\nprint(f'Number of NaN values:\\n{train_df.isna().sum()}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:11.020187Z","iopub.execute_input":"2023-02-18T02:38:11.020739Z","iopub.status.idle":"2023-02-18T02:38:11.038342Z","shell.execute_reply.started":"2023-02-18T02:38:11.020692Z","shell.execute_reply":"2023-02-18T02:38:11.037202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> 37 missing value on Age, and more than  25000 value on BIRAFS and density ","metadata":{}},{"cell_type":"code","source":"#->For simple model fitting and future predictions I'm gonna use only patient scans.\n#let's consider one particular patient\npatient_id = train_df[train_df.cancer == 1].iloc[0].patient_id\n\none_patient_df = train_df[train_df.patient_id == patient_id]\none_patient_df","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:11.039918Z","iopub.execute_input":"2023-02-18T02:38:11.040269Z","iopub.status.idle":"2023-02-18T02:38:11.069055Z","shell.execute_reply.started":"2023-02-18T02:38:11.040239Z","shell.execute_reply":"2023-02-18T02:38:11.067719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install dicomsdl\n!pip install /kaggle/input/dicomsdl/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:11.072004Z","iopub.execute_input":"2023-02-18T02:38:11.072481Z","iopub.status.idle":"2023-02-18T02:38:34.178791Z","shell.execute_reply.started":"2023-02-18T02:38:11.072436Z","shell.execute_reply":"2023-02-18T02:38:34.177458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir = '/kaggle/input/rsna-breast-cancer-detection/{}_images/{}/{}.dcm'\ntrain = 'train'\ntest = 'test'\n\ntrain_df[\"image_path\"] =  '/kaggle/input/rsna-breast-cancer-detection/train_images/'+train_df[\"patient_id\"].astype(str) +\"/\"+ train_df [\"image_id\"].astype(str)+\".dcm\"","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:34.18102Z","iopub.execute_input":"2023-02-18T02:38:34.181538Z","iopub.status.idle":"2023-02-18T02:38:34.292166Z","shell.execute_reply.started":"2023-02-18T02:38:34.181472Z","shell.execute_reply":"2023-02-18T02:38:34.291145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport dicomsdl\n\nn_rows = len(one_patient_df)\n\nplt.figure(figsize=(5 * n_rows, 5))\nfor i in range(n_rows):\n    row = one_patient_df.iloc[i]\n    \n    plt.subplot(1, n_rows, i + 1)\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, row['view'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})\n    plt.text(200, 700, row['cancer'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:34.294032Z","iopub.execute_input":"2023-02-18T02:38:34.294828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\n\nplt.figure(figsize=(5, 8))\nsns.countplot(data = train_df, x=\"laterality\", hue=\"cancer\", dodge = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> number of left and right breast pictures is nearly the same","metadata":{}},{"cell_type":"code","source":"sns.displot(data=train_df, x='age', kde=True)\nplt.title(\"Distribution of Age\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution and count of the age column\nm = {\n    '20-30': 0,\n    '30-40': 0,\n    '40-50': 0,\n    '50-60': 0,\n    '60-70': 0,\n    '70-80': 0,\n    '80-90': 0\n}\nfor index, row in train_df.iterrows():\n    if row['age']>=20 and row['age']<30: m['20-30'] += 1 \n    elif row['age']>=30 and row['age']<40: m['30-40'] += 1\n    elif row['age']>=40 and row['age']<50: m['40-50'] += 1\n    elif row['age']>=50 and row['age']<60: m['50-60'] += 1\n    elif row['age']>=60 and row['age']<70: m['60-70'] += 1\n    elif row['age']>=70 and row['age']<80: m['70-80'] += 1\n    else: m['80-90'] += 1\nfig, ax = plt.subplots(figsize=(12,10))\nax = sns.barplot(x=list(m.keys()), y=list(m.values()))\nax.bar_label(ax.containers[0])\nplt.title(\"Patient Count of Different Age Groups\")\nplt.show()","metadata":{"execution":{"iopub.status.idle":"2023-02-18T02:38:54.412071Z","shell.execute_reply.started":"2023-02-18T02:38:49.38514Z","shell.execute_reply":"2023-02-18T02:38:54.411097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> From the above chart and plot it can be noticed that people from the age group of 50 to 70 have more chances of occurence of breast canncer. The distribution looks like a bell curve. Distribution of age almost like a normal distribution.","metadata":{}},{"cell_type":"markdown","source":"# Frequency of cancer and normal patients","metadata":{}},{"cell_type":"code","source":"# Let's visualize how many patient in the dataset have cancer and how many don't\ntemp = train_df.groupby('patient_id')['cancer'].max().to_frame()\ntemp.reset_index(inplace=True)\ntemp = temp['cancer'].value_counts().to_frame()\ntemp.reset_index(inplace=True)\ntemp.columns = ['cancer', 'patient_count']\nfig,ax = plt.subplots(figsize=(12, 8))\nax = sns.barplot(data=temp, x=temp.columns[0], y=temp.columns[1])\nax.bar_label(ax.containers[0])\nplt.title('Frequency of normal patients and breast cancer patients')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:54.413198Z","iopub.execute_input":"2023-02-18T02:38:54.41353Z","iopub.status.idle":"2023-02-18T02:38:54.658531Z","shell.execute_reply.started":"2023-02-18T02:38:54.413485Z","shell.execute_reply":"2023-02-18T02:38:54.657492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Among all the patients 486 patients were affected by cancer.","metadata":{}},{"cell_type":"markdown","source":"**Realtionship between age and cancer**","metadata":{}},{"cell_type":"code","source":"# Now let's identified cancer patients and relationship with age\ntemp = train_df.groupby('patient_id')[['age', 'cancer']].max()\nm = {\n    \"30-40\": [0, 0],\n    \"40-50\": [0, 0],\n    \"50-60\": [0, 0],\n    \"60-70\": [0, 0],\n    \"70-80\": [0, 0],\n    \"80-90\": [0, 0]\n}\nfor index, row in temp.iterrows():\n    age = row['age']\n    if age>=30 and age<40: \n        m['30-40'][0] += 1\n        if row['cancer']==1: m['30-40'][1] += 1\n    elif age>=40 and age<50: \n        m['40-50'][0] += 1\n        if row['cancer']==1: m['40-50'][1] += 1\n    elif age>=50 and age<60: \n        m['50-60'][0] += 1\n        if row['cancer']==1: m['50-60'][1] += 1\n    elif age>=60 and age<70: \n        m['60-70'][0] += 1\n        if row['cancer']==1: m['60-70'][1] += 1\n    elif age>=70 and age<80: \n        m['70-80'][0] += 1\n        if row['cancer']==1: m['70-80'][1] += 1\n    else: \n        m['80-90'][0] += 1\n        if row['cancer']==1: m['80-90'][1] += 1\ntemp = {}\nfor key, val in m.items(): temp[key] = val[1] / val[0]\nfig, ax = plt.subplots(figsize=(12,10))\nax = sns.barplot(x=list(temp.keys()), y=list(temp.values()))\nplt.title(\"Percentage of cancer patient of different age groups\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:54.659856Z","iopub.execute_input":"2023-02-18T02:38:54.660803Z","iopub.status.idle":"2023-02-18T02:38:55.585411Z","shell.execute_reply.started":"2023-02-18T02:38:54.660769Z","shell.execute_reply":"2023-02-18T02:38:55.584382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It is clear that, age is an important factor which influence that a patient can have cancer or not. With increasing age chances of breast cancer increases in woman.","metadata":{}},{"cell_type":"markdown","source":"# Biopsy\n","metadata":{}},{"cell_type":"markdown","source":"**Images labelled with biopsy**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(12,8))\nax = sns.countplot(x=train_df['biopsy'])\nax.bar_label(ax.containers[0])\nplt.title(\"Images with biopsy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:55.586874Z","iopub.execute_input":"2023-02-18T02:38:55.588038Z","iopub.status.idle":"2023-02-18T02:38:55.824817Z","shell.execute_reply.started":"2023-02-18T02:38:55.587986Z","shell.execute_reply":"2023-02-18T02:38:55.823604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Patients who treated with biopsy\ntemp = train_df[['patient_id', 'biopsy']]\ntemp = temp.groupby('patient_id')['biopsy'].max().to_frame()\nfig, ax = plt.subplots(figsize=(12,8))\nax = sns.countplot(x=temp['biopsy'])\nax.bar_label(ax.containers[0])\nplt.title(\"Images with biopsy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:55.826394Z","iopub.execute_input":"2023-02-18T02:38:55.826784Z","iopub.status.idle":"2023-02-18T02:38:56.069289Z","shell.execute_reply.started":"2023-02-18T02:38:55.826753Z","shell.execute_reply":"2023-02-18T02:38:56.068186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Relationship between cancer and biopsy**","metadata":{}},{"cell_type":"code","source":"temp = train_df[['cancer', 'biopsy']]\nfig, ax = plt.subplots(figsize=(12,6))\nsns.countplot(data=temp, x = 'biopsy', hue = 'cancer', ax=ax)\nax.bar_label(ax.containers[0])\nplt.title(\"Relation between cancer and biopsy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.07085Z","iopub.execute_input":"2023-02-18T02:38:56.071177Z","iopub.status.idle":"2023-02-18T02:38:56.334717Z","shell.execute_reply.started":"2023-02-18T02:38:56.071148Z","shell.execute_reply":"2023-02-18T02:38:56.333663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Biopsy is never performed on some patient who don't have cancer and what is the reason for this? A biopsy is a procedure in which a small sample of tissue is removed from the body and examined under a microscope to determine if cancer cells are present. Biopsies are typically performed on patients who have abnormal findings on imaging tests, such as mammography or ultrasound, or who have symptoms that may be indicative of cancer, such as a lump or mass in the breast. The decision to perform a biopsy is made by a doctor based on the individual patient's medical history, symptoms, and imaging test results. Biopsies are an important tool in the diagnosis of cancer, and can help doctors determine the type and extent of the disease, as well as the best course of treatment.","metadata":{}},{"cell_type":"markdown","source":"#  Reduce memory usage \n","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.339457Z","iopub.execute_input":"2023-02-18T02:38:56.339832Z","iopub.status.idle":"2023-02-18T02:38:56.354198Z","shell.execute_reply.started":"2023-02-18T02:38:56.3398Z","shell.execute_reply":"2023-02-18T02:38:56.353124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_mem_usage(train_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.35597Z","iopub.execute_input":"2023-02-18T02:38:56.356659Z","iopub.status.idle":"2023-02-18T02:38:56.390279Z","shell.execute_reply.started":"2023-02-18T02:38:56.356625Z","shell.execute_reply":"2023-02-18T02:38:56.388974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df = reduce_mem_usage(test_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.39166Z","iopub.execute_input":"2023-02-18T02:38:56.391996Z","iopub.status.idle":"2023-02-18T02:38:56.410826Z","shell.execute_reply.started":"2023-02-18T02:38:56.391966Z","shell.execute_reply":"2023-02-18T02:38:56.409956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# pre-processing the dataset","metadata":{}},{"cell_type":"markdown","source":"- Dealing with NaN values**\n***Age value relatively low therefore we fill it with median, we do imputation aswell for the others that have a lot of NaN***\n***For this , we will use the fancyimpute library to use KNN in filling the data and sklearn.impute for age***","metadata":{}},{"cell_type":"code","source":"# !pip install fancyimpute\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.412022Z","iopub.execute_input":"2023-02-18T02:38:56.412938Z","iopub.status.idle":"2023-02-18T02:38:56.416818Z","shell.execute_reply.started":"2023-02-18T02:38:56.412903Z","shell.execute_reply":"2023-02-18T02:38:56.415929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.impute import SimpleImputer\n# from fancyimpute import KNN\n\n# mean_imputer = SimpleImputer(strategy='mean')\n# train_df['age'] = mean_imputer.fit_transform(train_df[['age']])\n\n# train_df['BIRADS'] = train_df['BIRADS'].map({'0': 0, '1': 1, '2': 2})\n# train_df['density'] = train_df['density'].map({'A': 0, 'B': 1, 'C': 2, 'D': 3})\n\n# # Impute missing values for BIRADS and density using KNN imputation\n# imputed_data = KNN(k=5).fit_transform(train_df[['BIRADS', 'density']])\n# train_df[['BIRADS', 'density']] = imputed_data\n\n# # Convert imputed numerical columns back to categorical representation\n# train_df['BIRADS'] = train_df['BIRADS'].map({0: '0', 1: '1', 2: '2'})\n# train_df['density'] = train_df['density'].map({0: 'A', 1: 'B', 2: 'C', 3: 'D'})\n\n# train_df\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.418262Z","iopub.execute_input":"2023-02-18T02:38:56.41936Z","iopub.status.idle":"2023-02-18T02:38:56.430188Z","shell.execute_reply.started":"2023-02-18T02:38:56.419319Z","shell.execute_reply":"2023-02-18T02:38:56.428823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split the data into train and test**\n","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val = train_test_split(train_df, test_size = 0.4, random_state = 42)\nlen(X_train), len(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.431346Z","iopub.execute_input":"2023-02-18T02:38:56.431726Z","iopub.status.idle":"2023-02-18T02:38:56.753611Z","shell.execute_reply.started":"2023-02-18T02:38:56.431692Z","shell.execute_reply":"2023-02-18T02:38:56.752418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.754938Z","iopub.execute_input":"2023-02-18T02:38:56.755274Z","iopub.status.idle":"2023-02-18T02:38:56.786371Z","shell.execute_reply.started":"2023-02-18T02:38:56.755237Z","shell.execute_reply":"2023-02-18T02:38:56.785334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Generation**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nfrom enum import Enum, auto\n\nclass Mode(Enum):\n    TRAIN = auto()\n    TEST = auto()\n    \nimg_height = 300\nimg_width = 250\nimg_shape = (img_height, img_width, 1)\n\nclass ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size,\n                 mode = Mode.TRAIN):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.mode = mode\n        self.mode_str = train if mode == Mode.TRAIN else test\n        \n        self.len = len(df)\n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        X = np.zeros((self.batch_size, ) + img_shape)\n        y = np.zeros((self.batch_size, 1))\n        \n        for i , pos in enumerate(range(start, end)):\n            if pos >= self.len: break\n                     \n            row = self.df.iloc[pos]\n            patient_id = row.patient_id\n            img_id = row.image_id\n            \n            file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            \n            img = dicomsdl.open(file_name)\n            img_arr = img.pixelData()\n            \n            # standartize all scans ( Hyperparameter tuning )\n            img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            \n            if img.PhotometricInterpretation == \"MONOCHROME1\":\n                img_arr = 1 - img_arr\n\n            img_arr = np.expand_dims(img_arr, axis = -1)\n            img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n                 \n            X[i,...] = img_arr\n                \n            \n            if self.mode == Mode.TRAIN:\n                y[i] = row.cancer\n                \n        return (X, y) if self.mode == Mode.TRAIN else X\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size + bool(self.len % self.batch_size)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:38:56.787913Z","iopub.execute_input":"2023-02-18T02:38:56.788447Z","iopub.status.idle":"2023-02-18T02:39:05.215401Z","shell.execute_reply.started":"2023-02-18T02:38:56.788414Z","shell.execute_reply":"2023-02-18T02:39:05.214351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_shape[:-1]","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:05.216965Z","iopub.execute_input":"2023-02-18T02:39:05.217675Z","iopub.status.idle":"2023-02-18T02:39:05.225121Z","shell.execute_reply.started":"2023-02-18T02:39:05.21764Z","shell.execute_reply":"2023-02-18T02:39:05.224026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_gen = ImageDataGen(X_train, 50)\nval_gen = ImageDataGen(X_val, 50)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:05.226737Z","iopub.execute_input":"2023-02-18T02:39:05.227378Z","iopub.status.idle":"2023-02-18T02:39:05.235678Z","shell.execute_reply.started":"2023-02-18T02:39:05.227344Z","shell.execute_reply":"2023-02-18T02:39:05.234793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(val_gen)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:05.236779Z","iopub.execute_input":"2023-02-18T02:39:05.237496Z","iopub.status.idle":"2023-02-18T02:39:05.256897Z","shell.execute_reply.started":"2023-02-18T02:39:05.237465Z","shell.execute_reply":"2023-02-18T02:39:05.256043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"import tensorflow.keras as K\n\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import (Conv2D, \n                                     MaxPooling2D, \n                                     BatchNormalization, \n                                     Dense, \n                                     Dropout,\n                                     GlobalMaxPooling2D)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:05.25807Z","iopub.execute_input":"2023-02-18T02:39:05.259784Z","iopub.status.idle":"2023-02-18T02:39:05.270415Z","shell.execute_reply.started":"2023-02-18T02:39:05.259749Z","shell.execute_reply":"2023-02-18T02:39:05.269311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom keras import layers, optimizers\nfrom keras.layers import Input, Flatten\nfrom keras.models import Sequential, Model\nmodel = Sequential([\n    layers.Conv2D(filters=96, kernel_size=(11,11), strides=(4,4), activation='relu', input_shape=img_shape),\n    layers.BatchNormalization(),\n    layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    layers.Conv2D(filters=256, kernel_size=(5,5), strides=(1,1), activation='relu', padding=\"same\"),\n    layers.BatchNormalization(),\n    layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    layers.BatchNormalization(),\n    layers.Conv2D(filters=384, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    layers.BatchNormalization(),\n    layers.Conv2D(filters=256, kernel_size=(3,3), strides=(1,1), activation='relu', padding=\"same\"),\n    layers.BatchNormalization(),\n    layers.MaxPool2D(pool_size=(3,3), strides=(2,2)),\n    layers.Flatten(),\n    layers.Dense(4096, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(4096, activation='relu'),\n    layers.Dropout(0.5),\n    layers.Dense(2, activation='softmax')\n])\n    \nrecall_thresholds = [0.4, 0.5, 0.6, 0.8]\nmodel.compile(optimizer = Adam(learning_rate = 5e-5), \n              loss = 'binary_crossentropy', \n              metrics = [tf.keras.metrics.BinaryAccuracy(threshold = 0.5), \n                         tf.keras.metrics.Recall(thresholds = recall_thresholds),\n                         tf.keras.metrics.Precision(thresholds = recall_thresholds)])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:05.272062Z","iopub.execute_input":"2023-02-18T02:39:05.273069Z","iopub.status.idle":"2023-02-18T02:39:06.239661Z","shell.execute_reply.started":"2023-02-18T02:39:05.273034Z","shell.execute_reply":"2023-02-18T02:39:06.238451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# callbacks\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(patience = 5, restore_best_weights = True, verbose = 1) # val_loss\nreduce_lr = ReduceLROnPlateau(factor = 0.1, patience = 2, mode = 'min', verbose = 1) # val_loss ","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.241254Z","iopub.execute_input":"2023-02-18T02:39:06.241898Z","iopub.status.idle":"2023-02-18T02:39:06.248794Z","shell.execute_reply.started":"2023-02-18T02:39:06.241863Z","shell.execute_reply":"2023-02-18T02:39:06.247587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''history = model.fit(train_gen,\n                    validation_data = val_gen,\n                    epochs = 3,\n                    verbose = 1,\n                    workers = 8,\n                    callbacks = [reduce_lr])'''","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.250747Z","iopub.execute_input":"2023-02-18T02:39:06.251213Z","iopub.status.idle":"2023-02-18T02:39:06.262036Z","shell.execute_reply.started":"2023-02-18T02:39:06.251179Z","shell.execute_reply":"2023-02-18T02:39:06.260724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Resnet ","metadata":{}},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport matplotlib.pyplot as plt\nimport os\nimport glob\nimport gc\nimport time \nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom kaggle_datasets import KaggleDatasets\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.python.client import device_lib\nfrom tensorflow.keras import Model, layers, Input, Sequential\nfrom sklearn.metrics import log_loss\n\nfrom joblib import Parallel, delayed\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.264068Z","iopub.execute_input":"2023-02-18T02:39:06.264623Z","iopub.status.idle":"2023-02-18T02:39:06.283064Z","shell.execute_reply.started":"2023-02-18T02:39:06.264588Z","shell.execute_reply":"2023-02-18T02:39:06.281876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf\n\n# from enum import Enum, auto\n\n# class Mode(Enum):\n#     TRAIN = auto()\n#     TEST = auto()\n    \n# img_height = 300\n# img_width = 250\n# img_shape = (img_height, img_width, 3)\n\n# class ImageDataGen(tf.keras.utils.Sequence):\n    \n#     def __init__(self,\n#                  df,\n#                  batch_size,\n#                  mode = Mode.TRAIN):\n\n#         self.df = df\n#         self.batch_size = batch_size\n#         self.mode = mode\n#         self.mode_str = train if mode == Mode.TRAIN else test\n        \n#         self.len = len(df)\n        \n#     def __getitem__(self, index):\n        \n#         start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n#         X = np.zeros((self.batch_size, ) + img_shape)\n#         y = np.zeros((self.batch_size, 1))\n        \n#         for i , pos in enumerate(range(start, end)):\n#             if pos >= self.len: break\n                     \n#             row = self.df.iloc[pos]\n#             patient_id = row.patient_id\n#             img_id = row.image_id\n            \n#             file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            \n#             img = dicomsdl.open(file_name)\n#             img_arr = img.pixelData()\n            \n#             # standartize all scans ( Hyperparameter tuning )\n#             img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            \n#             if img.PhotometricInterpretation == \"RGB\":\n#                 img_arr = 1 - img_arr\n\n#             img_arr = np.expand_dims(img_arr, axis = -1)\n#             img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n                 \n#             X[i,...] = img_arr\n                \n            \n#             if self.mode == Mode.TRAIN:\n#                 y[i] = row.cancer\n                \n#         return (X, y) if self.mode == Mode.TRAIN else X\n                \n    \n#     def __len__(self):\n#         return self.len // self.batch_size + bool(self.len % self.batch_size)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.285004Z","iopub.execute_input":"2023-02-18T02:39:06.285399Z","iopub.status.idle":"2023-02-18T02:39:06.298577Z","shell.execute_reply.started":"2023-02-18T02:39:06.285365Z","shell.execute_reply":"2023-02-18T02:39:06.297288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df= train_df.sample(1000)\nX_train, X_val = train_test_split(train_df, test_size = 0.4, random_state = 42)\nlen(X_train), len(X_val)\ntrain_gen = ImageDataGen(X_train, 50)\nval_gen = ImageDataGen(X_val, 50)","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.299919Z","iopub.execute_input":"2023-02-18T02:39:06.300269Z","iopub.status.idle":"2023-02-18T02:39:06.323859Z","shell.execute_reply.started":"2023-02-18T02:39:06.300237Z","shell.execute_reply":"2023-02-18T02:39:06.322878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    Rescaling = layers.Rescaling(1./255,name = \"Rescaling0_1\")\n    RandomFlip = layers.RandomFlip('horizontal')\n    RandomRotation = layers.RandomRotation(0.2) ","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.325197Z","iopub.execute_input":"2023-02-18T02:39:06.325554Z","iopub.status.idle":"2023-02-18T02:39:06.364393Z","shell.execute_reply.started":"2023-02-18T02:39:06.325494Z","shell.execute_reply":"2023-02-18T02:39:06.363405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation  = Sequential([\n    #layers.InputLayer(input_shape=(RESIZE,RESIZE,3)), \n    RandomFlip,\n    RandomRotation])","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.365598Z","iopub.execute_input":"2023-02-18T02:39:06.365965Z","iopub.status.idle":"2023-02-18T02:39:06.375286Z","shell.execute_reply.started":"2023-02-18T02:39:06.365922Z","shell.execute_reply":"2023-02-18T02:39:06.374123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ResNet = ResNet50(include_top=False,weights=None,input_shape=img_shape,pooling = 'avg')\nResNet.trainable = False\n    \nmodel2 = Sequential(\n        [layers.InputLayer(input_shape=img_shape),         \n         layers.experimental.preprocessing.Rescaling(1./255,name = \"Rescaling0_1\"),\n         ResNet,\n         layers.Flatten(),\n         layers.Dense(32, activation = \"relu\"),\n            layers.Dense(1,activation =\"sigmoid\",dtype='float32')\n        ]\n    )\n    \nmodel2.compile(optimizer=\"adam\",loss= tf.keras.losses.BinaryCrossentropy())","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:06.376634Z","iopub.execute_input":"2023-02-18T02:39:06.377059Z","iopub.status.idle":"2023-02-18T02:39:08.390173Z","shell.execute_reply.started":"2023-02-18T02:39:06.37699Z","shell.execute_reply":"2023-02-18T02:39:08.388949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history =  model2.fit(train_gen,\n                    validation_data = val_gen,\n                    epochs = 2,\n                    verbose = 1,\n                    workers = 8,\n                    callbacks = [reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2023-02-18T02:39:08.391553Z","iopub.execute_input":"2023-02-18T02:39:08.392197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-18T03:55:29.821313Z","iopub.execute_input":"2023-02-18T03:55:29.822036Z","iopub.status.idle":"2023-02-18T03:55:48.981996Z","shell.execute_reply.started":"2023-02-18T03:55:29.821973Z","shell.execute_reply":"2023-02-18T03:55:48.70945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGen(test_df, 16, mode= Mode.TEST)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T03:55:48.993183Z","iopub.execute_input":"2023-02-18T03:55:48.9937Z","iopub.status.idle":"2023-02-18T03:55:49.08067Z","shell.execute_reply.started":"2023-02-18T03:55:48.993654Z","shell.execute_reply":"2023-02-18T03:55:49.078776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model2.predict(test_gen)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-18T03:55:51.18944Z","iopub.execute_input":"2023-02-18T03:55:51.190332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = pred[:len(test_df)]\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].max().to_frame().reset_index()\nsub.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Evaluation of the model**","metadata":{}}]}