{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Installing and Importing Necessary Libraries","metadata":{}},{"cell_type":"code","source":"! pip install /kaggle/input/gdcm-external-library/python_gdcm-3.0.21-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:04.735659Z","iopub.execute_input":"2023-02-24T09:00:04.736148Z","iopub.status.idle":"2023-02-24T09:00:36.316915Z","shell.execute_reply.started":"2023-02-24T09:00:04.73605Z","shell.execute_reply":"2023-02-24T09:00:36.315672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('darkgrid')\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\n\nimport cv2\nfrom PIL import Image\nfrom tqdm import tqdm\nimport pydicom\nimport gdcm\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-24T09:00:36.319493Z","iopub.execute_input":"2023-02-24T09:00:36.319903Z","iopub.status.idle":"2023-02-24T09:00:37.671984Z","shell.execute_reply.started":"2023-02-24T09:00:36.319858Z","shell.execute_reply":"2023-02-24T09:00:37.670642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Data and Pre-processing","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\nbase_img_dir = \"/kaggle/input/rsna-breast-cancer-detection/train_images\" ","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.673524Z","iopub.execute_input":"2023-02-24T09:00:37.674682Z","iopub.status.idle":"2023-02-24T09:00:37.763319Z","shell.execute_reply.started":"2023-02-24T09:00:37.674639Z","shell.execute_reply":"2023-02-24T09:00:37.762403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.765699Z","iopub.execute_input":"2023-02-24T09:00:37.766065Z","iopub.status.idle":"2023-02-24T09:00:37.788667Z","shell.execute_reply.started":"2023-02-24T09:00:37.766032Z","shell.execute_reply":"2023-02-24T09:00:37.787224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.791331Z","iopub.execute_input":"2023-02-24T09:00:37.791684Z","iopub.status.idle":"2023-02-24T09:00:37.80905Z","shell.execute_reply.started":"2023-02-24T09:00:37.791654Z","shell.execute_reply":"2023-02-24T09:00:37.807704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.811349Z","iopub.execute_input":"2023-02-24T09:00:37.81249Z","iopub.status.idle":"2023-02-24T09:00:37.838863Z","shell.execute_reply.started":"2023-02-24T09:00:37.812443Z","shell.execute_reply":"2023-02-24T09:00:37.837522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.840258Z","iopub.execute_input":"2023-02-24T09:00:37.840697Z","iopub.status.idle":"2023-02-24T09:00:37.903307Z","shell.execute_reply.started":"2023-02-24T09:00:37.840654Z","shell.execute_reply":"2023-02-24T09:00:37.901759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of values in data: {train.count().sum()}')\nprint(f'Number missing values in data: {sum(train.isna().sum())}')","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.905091Z","iopub.execute_input":"2023-02-24T09:00:37.905784Z","iopub.status.idle":"2023-02-24T09:00:37.936259Z","shell.execute_reply.started":"2023-02-24T09:00:37.905724Z","shell.execute_reply":"2023-02-24T09:00:37.935022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_missing_data(data):\n    total = data.isnull().sum().sort_values(ascending = False)\n    percentage = (data.isnull().sum()/data.isnull().count()).sort_values(ascending = False)\n    return pd.concat([total,percentage] , axis = 1 , keys = ['Total' , 'Percent'])\nfind_missing_data(train)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.937628Z","iopub.execute_input":"2023-02-24T09:00:37.937983Z","iopub.status.idle":"2023-02-24T09:00:37.985582Z","shell.execute_reply.started":"2023-02-24T09:00:37.937951Z","shell.execute_reply":"2023-02-24T09:00:37.984361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Total No. of Images: {len(train)}\")\nnum_patients = len(train[\"patient_id\"].unique())\nprint(f\"Total No. of Patients: {num_patients}.\")","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.987306Z","iopub.execute_input":"2023-02-24T09:00:37.987655Z","iopub.status.idle":"2023-02-24T09:00:37.994184Z","shell.execute_reply.started":"2023-02-24T09:00:37.987623Z","shell.execute_reply":"2023-02-24T09:00:37.993325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"def has_cancer(c):\n    if c > 0:\n        return True\n    else:\n        return False\n\ncancer_per_patient = train.groupby(\"patient_id\").cancer.sum().apply(lambda c: has_cancer(c)).values\n\nplt.figure(figsize=(8, 8))\nax = sns.countplot(data = cancer_per_patient, color='#3642E1')\nax.bar_label(ax.containers[0])\nplt.xlabel(\"Cancer\")\nplt.title(\"Results of cancer assessments\", fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:37.995248Z","iopub.execute_input":"2023-02-24T09:00:37.995548Z","iopub.status.idle":"2023-02-24T09:00:38.367924Z","shell.execute_reply.started":"2023-02-24T09:00:37.995521Z","shell.execute_reply":"2023-02-24T09:00:38.366605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* Of these 11913 patients, roughly 96% had no cancer whereas 4% (486 patients) have cancer.","metadata":{}},{"cell_type":"code","source":"ages = train[train.age.isnull() == False].groupby('patient_id').age.apply(lambda l: np.unique(l)[0])\n\nfig, ax = plt.subplots(1,2 ,figsize=(20,8))\n\n\nsns.histplot(ages, color='#3642E1', bins=60, ax=ax[0])\nsns.boxplot(x = train.cancer, y = train.age, palette='seismic',ax=ax[1])\nplt.suptitle('Age distribution of patients', fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:38.36929Z","iopub.execute_input":"2023-02-24T09:00:38.369663Z","iopub.status.idle":"2023-02-24T09:00:39.378072Z","shell.execute_reply.started":"2023-02-24T09:00:38.369627Z","shell.execute_reply":"2023-02-24T09:00:39.377197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* Patients with cancer are more likely to be older and between the age group of 60 to 70.\n* We appear to have two peaks around the ages of 50 and 70.\n* Beyond the age of 70, the number of patients decreases.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2 ,figsize=(20,8))\n\nsns.countplot(x = train[train.cancer==1].invasive, palette='seismic', ax=ax[0]).set_title(\"Number of patients with Invasive Cancer\", fontsize=15)\n\nsns.countplot(x = train[train.biopsy==1].cancer, palette='seismic', ax=ax[1]).set_title(\"Number of patients with a follow up Biopsy\", fontsize=15)\nfor i in range(0,2):\n    for container in ax[i].containers:\n        ax[i].bar_label(container) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:39.379147Z","iopub.execute_input":"2023-02-24T09:00:39.380095Z","iopub.status.idle":"2023-02-24T09:00:39.801281Z","shell.execute_reply.started":"2023-02-24T09:00:39.380059Z","shell.execute_reply":"2023-02-24T09:00:39.799974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* Around 70% of all patients with cancer reveal invasive cancer.\n* A follow-up biopsy was conducted on a few non-cancer patients, and every cancer patient had a biopsy.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2 ,figsize=(20,8))\n\nsns.countplot(data=train, x=\"implant\", hue=\"cancer\", palette='seismic', ax=ax[0]).set_title(\"Number of patients with Implants\", fontsize=15)\n\nsns.countplot(data=train, x=\"laterality\", hue=\"cancer\", palette='seismic', ax=ax[1]).set_title(\"Laterality Distribution\", fontsize=15)\nfor i in range(0,2):\n    for container in ax[i].containers:\n        ax[i].bar_label(container)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:39.806087Z","iopub.execute_input":"2023-02-24T09:00:39.80652Z","iopub.status.idle":"2023-02-24T09:00:40.318754Z","shell.execute_reply.started":"2023-02-24T09:00:39.806482Z","shell.execute_reply":"2023-02-24T09:00:40.317637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* Cancer was found in very few patients with implants, although quite a few patients without implants had developed the cancer.\n* Due to a lack of data, we are unable to determine whether implants are related to cancer in any way.\n* Laterality seems to have no connection to cancer.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2 ,figsize=(20,8))\n\nsns.countplot(data=train, x=\"density\", hue=\"cancer\", palette='seismic', ax=ax[0]).set_title(\"Density of breast tissue\", fontsize=15)\n\nsns.countplot(data=train, x=\"view\", palette='seismic', ax=ax[1]).set_title(\"Orientation of the Images\", fontsize=15)\nfor i in range(0,2):\n    for container in ax[i].containers:\n        ax[i].bar_label(container)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:40.320437Z","iopub.execute_input":"2023-02-24T09:00:40.3214Z","iopub.status.idle":"2023-02-24T09:00:40.915999Z","shell.execute_reply.started":"2023-02-24T09:00:40.321358Z","shell.execute_reply":"2023-02-24T09:00:40.914853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* The majority of the images feature category B and C densities. Yet, there are other instances that are either extremely dense D or not dense at all A.\n* The most common views are CC and MLO.","metadata":{}},{"cell_type":"markdown","source":"# BI-RADS Categories\n#### To communicate the findings and outcomes of mammograms, doctors use a prescribed format. The results are categorised into groups with numbers 0 through 6 by this system, also known as the Breast Imaging Reporting and Data System (BI-RADS). Because there are just three categories, we shall stick with them.\n   * 0 - Incomplete - Additional imaging evaluation and/or comparison to prior mammograms (or other imaging tests) is needed.\n   * 1 - Breast was rated as negative for cancer. In this case, negative means nothing new or abnormal was found.\n   * 2 - Breast was rated as normal. This is also a negative test result (there’s no sign of cancer), but the radiologist chooses to describe a finding that is not cancer, such as              benign calcifications, masses, or lymph nodes in the breast.\n","metadata":{}},{"cell_type":"code","source":"birads_C0 = [\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 0.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 1.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 0) & (train[\"BIRADS\"] == 2.0)][\"image_id\"].count(),\n]\n\nbirads_C1 = [\n    train[(train[\"cancer\"] == 1) & (train[\"BIRADS\"] == 0.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 1) & (train[\"BIRADS\"] == 1.0)][\"image_id\"].count(),\n    train[(train[\"cancer\"] == 1) & (train[\"BIRADS\"] == 2.0)][\"image_id\"].count(),\n]\nbirads_C0 = pd.DataFrame(birads_C0, columns =['values']) \nbirads_C1 = pd.DataFrame(birads_C1, columns =['values']) ","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:40.917619Z","iopub.execute_input":"2023-02-24T09:00:40.917973Z","iopub.status.idle":"2023-02-24T09:00:40.941924Z","shell.execute_reply.started":"2023-02-24T09:00:40.917941Z","shell.execute_reply":"2023-02-24T09:00:40.940209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = make_subplots(\n    rows=1, cols=2, specs=[[{\"type\": \"pie\"}, {\"type\": \"pie\"}]],\n    subplot_titles=(\"<b>Patients were Cancer was not detected<b>\",\"<b>Patients were Cancer was detected<b>\")\n)\n\nfig.add_trace(go.Pie(values = birads_C0['values'],labels=[\"BIRADS 0\", \"BIRADS 1\", \"BIRADS 2\"], textinfo='label+percent'),\n              row=1, col=1)\n\nfig.add_trace(go.Pie(values = birads_C1['values'],labels=[\"BIRADS 0\"],textinfo='label+value'),\n              row=1, col=2)\n\nfig.update_layout(\n                  title={\n                    'text': \"BIRADS score of Patients\",\n                    'y':0.95,\n                    'x':0.5,\n                    'xanchor': 'center',\n                    'yanchor': 'top'},\n                   font=dict(size=15),\n                  height=600,\n                  showlegend=True\n                  )\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:40.944023Z","iopub.execute_input":"2023-02-24T09:00:40.944527Z","iopub.status.idle":"2023-02-24T09:00:41.050154Z","shell.execute_reply.started":"2023-02-24T09:00:40.944478Z","shell.execute_reply":"2023-02-24T09:00:41.048997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n\n- Almost 8% of the patients whose cancer was undetected were evaluated as normal (1), while a majority of patients were rated as negative (2). \n\n- Nonetheless, roughly 30% of the images still indicate a follow up.\n\n- As anticipated, every images showing cancer required a follow up.\n","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2 ,figsize=(20,8))\n\nsns.countplot(data=train, x=\"implant\", hue=\"difficult_negative_case\", palette='seismic', ax=ax[0]).set_title(\"Breast Impants / Case was unusually difficult\", fontsize=15)\n\nsns.countplot(data=train, x=\"density\", hue=\"difficult_negative_case\", palette='seismic', ax=ax[1]).set_title(\"Density of breast tissue / Case was unusually difficult\", fontsize=15)\nfor i in range(0,2):\n    for container in ax[i].containers:\n        ax[i].bar_label(container) \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:41.051791Z","iopub.execute_input":"2023-02-24T09:00:41.052309Z","iopub.status.idle":"2023-02-24T09:00:41.611977Z","shell.execute_reply.started":"2023-02-24T09:00:41.052254Z","shell.execute_reply":"2023-02-24T09:00:41.610782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Observations:\n* We cannot determine if breast implants made the process exceptionally difficult due to a lack of data.                         \n\n\n`density` \n>A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. Only provided for train.\n\n\n* According to our prior observations, breast density levels of `B` and `C` were associated with the highest number of cancer incidences.\n* It is stated that `Extremely dense tissue can make diagnosis more difficult.` it appears that instances with breast densities of `B` and `C` made diagnosis challenging.\n* Nonetheless, there were relatively few occurrences where the breast density was `D` as well as lower than `A`.\n","metadata":{}},{"cell_type":"markdown","source":"# Exploring Image Data","metadata":{}},{"cell_type":"code","source":"def get_image_path(image_df, n):\n    ids = image_df[[\"patient_id\", \"image_id\"]].sample(n, random_state=77)\n    paths = []\n    for i in range(len(ids)):\n        path = os.path.join(base_img_dir, str(ids[\"patient_id\"].values[i]), str(ids[\"image_id\"].values[i])+'.dcm')\n        paths.append(path)\n    return paths\n\ndef image_plot(data_path, title):\n    plt.figure(figsize=(15, 8))\n    for j in range(n):\n        plt.subplot(1, n, j + 1)\n        im = pydicom.dcmread(data_path[j])\n        plt.imshow(im.pixel_array, cmap='bone')\n        plt.grid(False)\n        plt.title(title)\n        plt.axis(\"off\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:41.613014Z","iopub.execute_input":"2023-02-24T09:00:41.613381Z","iopub.status.idle":"2023-02-24T09:00:41.62312Z","shell.execute_reply.started":"2023-02-24T09:00:41.61335Z","shell.execute_reply":"2023-02-24T09:00:41.622264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_no_cancer = train[train[\"cancer\"]==0]\nimage_cancer = train[train[\"cancer\"]==1]\nimage_birads0 = train[train[\"BIRADS\"] == 0.0]\nimage_birads1 = train[train[\"BIRADS\"] == 1.0]\nimage_birads2 = train[train[\"BIRADS\"] == 2.0]\n\nn = 5\nno_cancer_paths = get_image_path(image_no_cancer, n)\ncancer_paths = get_image_path(image_cancer, n)\nbirads0_paths = get_image_path(image_birads0, n)\nbirads1_paths = get_image_path(image_birads1, n)\nbirads2_paths = get_image_path(image_birads2, n)\n\nimage_plot(no_cancer_paths, \"No Cancer Detected\")\nimage_plot(cancer_paths, \"Cancer Detected\")\nimage_plot(birads0_paths, \"BI-RADS 0\")\nimage_plot(birads1_paths, \"BI-RADS 1\")\nimage_plot(birads2_paths, \"BI-RADS 2\")","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:00:41.624696Z","iopub.execute_input":"2023-02-24T09:00:41.625392Z","iopub.status.idle":"2023-02-24T09:01:16.004045Z","shell.execute_reply.started":"2023-02-24T09:00:41.625355Z","shell.execute_reply":"2023-02-24T09:01:16.00282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 8))\nsns.heatmap(train.drop(['site_id', 'patient_id','image_id'], axis=1).corr(),annot=True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.00554Z","iopub.execute_input":"2023-02-24T09:01:16.006592Z","iopub.status.idle":"2023-02-24T09:01:16.706516Z","shell.execute_reply.started":"2023-02-24T09:01:16.00654Z","shell.execute_reply":"2023-02-24T09:01:16.705171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Building and Data Preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.708047Z","iopub.execute_input":"2023-02-24T09:01:16.708429Z","iopub.status.idle":"2023-02-24T09:01:16.713735Z","shell.execute_reply.started":"2023-02-24T09:01:16.708387Z","shell.execute_reply":"2023-02-24T09:01:16.712339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train.copy()\ny = train_df['cancer']\n\ntrain_df.drop(['patient_id','image_id','BIRADS','site_id','machine_id','density','difficult_negative_case','cancer','biopsy','invasive'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.715108Z","iopub.execute_input":"2023-02-24T09:01:16.715472Z","iopub.status.idle":"2023-02-24T09:01:16.734682Z","shell.execute_reply.started":"2023-02-24T09:01:16.71544Z","shell.execute_reply":"2023-02-24T09:01:16.733518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le=LabelEncoder()\ncol_list=['laterality','view']\nfor col in col_list:\n    train_df[col]=le.fit_transform(train_df[col])","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.736Z","iopub.execute_input":"2023-02-24T09:01:16.737096Z","iopub.status.idle":"2023-02-24T09:01:16.772702Z","shell.execute_reply.started":"2023-02-24T09:01:16.737054Z","shell.execute_reply":"2023-02-24T09:01:16.771485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.774145Z","iopub.execute_input":"2023-02-24T09:01:16.774488Z","iopub.status.idle":"2023-02-24T09:01:16.786796Z","shell.execute_reply.started":"2023-02-24T09:01:16.774459Z","shell.execute_reply":"2023-02-24T09:01:16.785675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"find_missing_data(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.788338Z","iopub.execute_input":"2023-02-24T09:01:16.788673Z","iopub.status.idle":"2023-02-24T09:01:16.812014Z","shell.execute_reply.started":"2023-02-24T09:01:16.788642Z","shell.execute_reply":"2023-02-24T09:01:16.810629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling the null values with mean value.\ntrain_df['age']=train_df['age'].fillna(train_df['age'].mean())","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.813415Z","iopub.execute_input":"2023-02-24T09:01:16.813918Z","iopub.status.idle":"2023-02-24T09:01:16.81951Z","shell.execute_reply.started":"2023-02-24T09:01:16.813886Z","shell.execute_reply":"2023-02-24T09:01:16.818638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 12))\nsns.heatmap(train_df.corr(),annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:16.820681Z","iopub.execute_input":"2023-02-24T09:01:16.821002Z","iopub.status.idle":"2023-02-24T09:01:17.23131Z","shell.execute_reply.started":"2023-02-24T09:01:16.820973Z","shell.execute_reply":"2023-02-24T09:01:17.230016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nx_train,x_test,y_train,y_test = train_test_split(train_df,y,random_state=777,test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:17.233128Z","iopub.execute_input":"2023-02-24T09:01:17.23363Z","iopub.status.idle":"2023-02-24T09:01:17.247366Z","shell.execute_reply.started":"2023-02-24T09:01:17.233582Z","shell.execute_reply":"2023-02-24T09:01:17.246103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import confusion_matrix,classification_report\nmodel = LogisticRegression()\n\nparameters = {\n    'penalty' : ['l1','l2'], \n    'C'       : np.logspace(-3,3,7),\n    'solver'  : ['liblinear']\n}\n\ngscv = GridSearchCV(model, param_grid = parameters, scoring='accuracy', cv=10) \ngscv.fit(x_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:17.248983Z","iopub.execute_input":"2023-02-24T09:01:17.249399Z","iopub.status.idle":"2023-02-24T09:01:49.605679Z","shell.execute_reply.started":"2023-02-24T09:01:17.249356Z","shell.execute_reply":"2023-02-24T09:01:49.604043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Tuned Hyperparameters :\", gscv.best_params_)\nprint(\"Accuracy :\",gscv.best_score_)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:49.613178Z","iopub.execute_input":"2023-02-24T09:01:49.61459Z","iopub.status.idle":"2023-02-24T09:01:49.621404Z","shell.execute_reply.started":"2023-02-24T09:01:49.614537Z","shell.execute_reply":"2023-02-24T09:01:49.620226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions","metadata":{}},{"cell_type":"code","source":"model = LogisticRegression(C = 0.001, penalty = \"l1\", solver = 'liblinear' )\nmodel.fit(x_train,y_train)\npredict = model.predict(x_test)\ncf_matrix = confusion_matrix(y_test,predict)\nprint(\"Accuracy:\",model.score(x_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:49.623339Z","iopub.execute_input":"2023-02-24T09:01:49.624076Z","iopub.status.idle":"2023-02-24T09:01:49.684785Z","shell.execute_reply.started":"2023-02-24T09:01:49.62403Z","shell.execute_reply":"2023-02-24T09:01:49.683559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.heatmap(cf_matrix/np.sum(cf_matrix), annot=True, \n            fmt='.2%', cmap='Blues')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:49.686516Z","iopub.execute_input":"2023-02-24T09:01:49.68724Z","iopub.status.idle":"2023-02-24T09:01:49.951528Z","shell.execute_reply.started":"2023-02-24T09:01:49.687191Z","shell.execute_reply":"2023-02-24T09:01:49.95036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test,predict))","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:49.953286Z","iopub.execute_input":"2023-02-24T09:01:49.953999Z","iopub.status.idle":"2023-02-24T09:01:49.980626Z","shell.execute_reply.started":"2023-02-24T09:01:49.953953Z","shell.execute_reply":"2023-02-24T09:01:49.979327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test.copy()\ntest_df.drop([\"site_id\",\"patient_id\",\"image_id\",\"machine_id\",\"prediction_id\"], axis=1, inplace=True)\n\ncol_list=['laterality','view']\nfor col in col_list:\n    test_df[col]=le.fit_transform(test_df[col])\ntest_df\n","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:49.981941Z","iopub.execute_input":"2023-02-24T09:01:49.982301Z","iopub.status.idle":"2023-02-24T09:01:50.001481Z","shell.execute_reply.started":"2023-02-24T09:01:49.982268Z","shell.execute_reply":"2023-02-24T09:01:49.999036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict_test = model.predict(test_df)\npredict_test","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:50.003635Z","iopub.execute_input":"2023-02-24T09:01:50.004121Z","iopub.status.idle":"2023-02-24T09:01:50.01532Z","shell.execute_reply.started":"2023-02-24T09:01:50.004081Z","shell.execute_reply":"2023-02-24T09:01:50.014181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:50.017465Z","iopub.execute_input":"2023-02-24T09:01:50.018094Z","iopub.status.idle":"2023-02-24T09:01:50.036039Z","shell.execute_reply.started":"2023-02-24T09:01:50.01806Z","shell.execute_reply":"2023-02-24T09:01:50.034614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"temp_df = test.copy()\ntemp_df['cancer'] = predict_test\nprediction_df = temp_df[['prediction_id','cancer']].groupby(['prediction_id']).mean()\nprediction_df","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:50.03721Z","iopub.execute_input":"2023-02-24T09:01:50.037727Z","iopub.status.idle":"2023-02-24T09:01:50.051825Z","shell.execute_reply.started":"2023-02-24T09:01:50.037695Z","shell.execute_reply":"2023-02-24T09:01:50.050711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df.to_csv('submission.csv',index=True)","metadata":{"execution":{"iopub.status.busy":"2023-02-24T09:01:50.053547Z","iopub.execute_input":"2023-02-24T09:01:50.054274Z","iopub.status.idle":"2023-02-24T09:01:50.061694Z","shell.execute_reply.started":"2023-02-24T09:01:50.054229Z","shell.execute_reply":"2023-02-24T09:01:50.060492Z"},"trusted":true},"execution_count":null,"outputs":[]}]}