{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T21:06:40.695135Z","iopub.execute_input":"2024-01-04T21:06:40.695562Z","iopub.status.idle":"2024-01-04T21:06:40.821106Z","shell.execute_reply.started":"2024-01-04T21:06:40.695527Z","shell.execute_reply":"2024-01-04T21:06:40.81955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the no. of rows and columns in our dataset\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.039898Z","iopub.execute_input":"2024-01-04T20:15:49.040684Z","iopub.status.idle":"2024-01-04T20:15:49.048622Z","shell.execute_reply.started":"2024-01-04T20:15:49.040637Z","shell.execute_reply":"2024-01-04T20:15:49.04717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the no. of unique values in each dataset\ndf.nunique()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.05034Z","iopub.execute_input":"2024-01-04T20:15:49.051389Z","iopub.status.idle":"2024-01-04T20:15:49.105857Z","shell.execute_reply.started":"2024-01-04T20:15:49.051333Z","shell.execute_reply":"2024-01-04T20:15:49.104624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Null Values in Dataset\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.109743Z","iopub.execute_input":"2024-01-04T20:15:49.110241Z","iopub.status.idle":"2024-01-04T20:15:49.13662Z","shell.execute_reply.started":"2024-01-04T20:15:49.110196Z","shell.execute_reply":"2024-01-04T20:15:49.135437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the datatype of each column\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.138369Z","iopub.execute_input":"2024-01-04T20:15:49.138763Z","iopub.status.idle":"2024-01-04T20:15:49.182934Z","shell.execute_reply.started":"2024-01-04T20:15:49.138729Z","shell.execute_reply":"2024-01-04T20:15:49.182021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting age column\ndf.iloc[:,5:6]","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.184399Z","iopub.execute_input":"2024-01-04T20:15:49.184754Z","iopub.status.idle":"2024-01-04T20:15:49.202183Z","shell.execute_reply.started":"2024-01-04T20:15:49.184723Z","shell.execute_reply":"2024-01-04T20:15:49.200621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.impute import SimpleImputer\n# Handling Null Values using SimpleImputer\nimputer = SimpleImputer(missing_values=np.nan,strategy=\"mean\")\nimputer.fit(df.iloc[:,5:6].values)\ndf.iloc[:,5:6] = imputer.transform(df.iloc[:,5:6].values)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:49.207942Z","iopub.execute_input":"2024-01-04T20:15:49.208406Z","iopub.status.idle":"2024-01-04T20:15:50.224692Z","shell.execute_reply.started":"2024-01-04T20:15:49.208366Z","shell.execute_reply":"2024-01-04T20:15:50.222836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for any duplicates in the dataset\ndf.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.226588Z","iopub.execute_input":"2024-01-04T20:15:50.228497Z","iopub.status.idle":"2024-01-04T20:15:50.295163Z","shell.execute_reply.started":"2024-01-04T20:15:50.228445Z","shell.execute_reply":"2024-01-04T20:15:50.293609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Changing the datatype\n# Convert image_id and patient_id to string\ndf[\"image_id\"] = df[\"image_id\"].astype(\"str\")\ndf[\"patient_id\"] = df[\"patient_id\"].astype(\"str\")","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.296863Z","iopub.execute_input":"2024-01-04T20:15:50.298373Z","iopub.status.idle":"2024-01-04T20:15:50.416137Z","shell.execute_reply.started":"2024-01-04T20:15:50.298271Z","shell.execute_reply":"2024-01-04T20:15:50.414917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping null values from dataset\ndf1 = df.dropna()\n# Finding correlation between 'BIRADS' and 'cancer' column \ncorrelation = df1['BIRADS'].corr(df1['cancer'])\nprint(correlation)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.424215Z","iopub.execute_input":"2024-01-04T20:15:50.424662Z","iopub.status.idle":"2024-01-04T20:15:50.476208Z","shell.execute_reply.started":"2024-01-04T20:15:50.424625Z","shell.execute_reply":"2024-01-04T20:15:50.474698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*** -0.2594312653716567** suggests a moderate negative correlation \n* Considering the fact that :\n    1. BIRADS has huge number of null values.\n    2. Its not greatly impacting the cancer prediction\n- We are dropping the column ","metadata":{"execution":{"iopub.status.busy":"2024-01-04T14:31:05.717199Z","iopub.execute_input":"2024-01-04T14:31:05.717637Z","iopub.status.idle":"2024-01-04T14:31:05.72598Z","shell.execute_reply.started":"2024-01-04T14:31:05.717603Z","shell.execute_reply":"2024-01-04T14:31:05.724644Z"}}},{"cell_type":"code","source":"df.drop(\"BIRADS\",axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.478088Z","iopub.execute_input":"2024-01-04T20:15:50.478905Z","iopub.status.idle":"2024-01-04T20:15:50.498995Z","shell.execute_reply.started":"2024-01-04T20:15:50.478835Z","shell.execute_reply":"2024-01-04T20:15:50.497967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.500752Z","iopub.execute_input":"2024-01-04T20:15:50.502323Z","iopub.status.idle":"2024-01-04T20:15:50.544777Z","shell.execute_reply.started":"2024-01-04T20:15:50.502268Z","shell.execute_reply":"2024-01-04T20:15:50.543601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Cramer's V**\n* Cramér's V is particularly useful when dealing with categorical variables and assessing the strength of association between them.\n* We use the following technique to find the correlation between 'density' and 'cancer' column in our dataset after dropping the null values.","metadata":{}},{"cell_type":"code","source":"from scipy.stats import chi2_contingency\n# Creating a contingency table\ncontingency_table = pd.crosstab(df1['cancer'],df1['density'])\n# Calculate Cramer's V\nchi2, _, _, _ = chi2_contingency(contingency_table)\nn = contingency_table.sum().sum()\nmin_dim = min(contingency_table.shape) - 1\ncramer_v = np.sqrt(chi2 / (n * min_dim))\nprint(cramer_v)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.546406Z","iopub.execute_input":"2024-01-04T20:15:50.547674Z","iopub.status.idle":"2024-01-04T20:15:50.583306Z","shell.execute_reply.started":"2024-01-04T20:15:50.547627Z","shell.execute_reply":"2024-01-04T20:15:50.581973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **0.021332267953364684** suggest a weak correlation \n* Considering the fact that :\n    1. BIRADS has huge number of null values.\n    2. Its not greatly impacting the cancer prediction\n- We are dropping the column ","metadata":{}},{"cell_type":"code","source":"df.drop(\"density\",axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.585938Z","iopub.execute_input":"2024-01-04T20:15:50.587252Z","iopub.status.idle":"2024-01-04T20:15:50.603149Z","shell.execute_reply.started":"2024-01-04T20:15:50.587201Z","shell.execute_reply":"2024-01-04T20:15:50.601876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping unrelated column\ndf.drop(\"site_id\",axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.604677Z","iopub.execute_input":"2024-01-04T20:15:50.605994Z","iopub.status.idle":"2024-01-04T20:15:50.619227Z","shell.execute_reply.started":"2024-01-04T20:15:50.605945Z","shell.execute_reply":"2024-01-04T20:15:50.617956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.621896Z","iopub.execute_input":"2024-01-04T20:15:50.622661Z","iopub.status.idle":"2024-01-04T20:15:50.66172Z","shell.execute_reply.started":"2024-01-04T20:15:50.622615Z","shell.execute_reply":"2024-01-04T20:15:50.659994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation between various columns in a dataset\nnumeric_df = df.select_dtypes(include=['number'])\nnumeric_df.corr()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.663655Z","iopub.execute_input":"2024-01-04T20:15:50.664468Z","iopub.status.idle":"2024-01-04T20:15:50.694268Z","shell.execute_reply.started":"2024-01-04T20:15:50.664417Z","shell.execute_reply":"2024-01-04T20:15:50.692861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nsns.heatmap(numeric_df.corr())","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:50.696111Z","iopub.execute_input":"2024-01-04T20:15:50.696877Z","iopub.status.idle":"2024-01-04T20:15:51.378757Z","shell.execute_reply.started":"2024-01-04T20:15:50.69682Z","shell.execute_reply":"2024-01-04T20:15:51.377434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.subplot(1,2,1)\nsns.distplot(df[\"age\"]) \n\nplt.subplot(1,2,2)\nsns.boxplot(df[\"machine_id\"])  ","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:51.380472Z","iopub.execute_input":"2024-01-04T20:15:51.381416Z","iopub.status.idle":"2024-01-04T20:15:52.285147Z","shell.execute_reply.started":"2024-01-04T20:15:51.381378Z","shell.execute_reply":"2024-01-04T20:15:52.283675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking for outlier in the dataset\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:52.286715Z","iopub.execute_input":"2024-01-04T20:15:52.287374Z","iopub.status.idle":"2024-01-04T20:15:52.326687Z","shell.execute_reply.started":"2024-01-04T20:15:52.287325Z","shell.execute_reply":"2024-01-04T20:15:52.32557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['age'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:52.328082Z","iopub.execute_input":"2024-01-04T20:15:52.328422Z","iopub.status.idle":"2024-01-04T20:15:52.339873Z","shell.execute_reply.started":"2024-01-04T20:15:52.328393Z","shell.execute_reply":"2024-01-04T20:15:52.339008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(df['age'])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:52.341407Z","iopub.execute_input":"2024-01-04T20:15:52.341842Z","iopub.status.idle":"2024-01-04T20:15:53.072069Z","shell.execute_reply.started":"2024-01-04T20:15:52.341809Z","shell.execute_reply":"2024-01-04T20:15:53.070787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Detecting and Handling Outlier","metadata":{}},{"cell_type":"code","source":"sns.boxplot(df['age'])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.073829Z","iopub.execute_input":"2024-01-04T20:15:53.074207Z","iopub.status.idle":"2024-01-04T20:15:53.323608Z","shell.execute_reply.started":"2024-01-04T20:15:53.074176Z","shell.execute_reply":"2024-01-04T20:15:53.322427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Q1 = df['age'].quantile(0.25)\nQ3 = df['age'].quantile(0.75)\nIQR = Q3 - Q1\nmin_value = Q1-1.5*IQR\nmax_value = Q3+1.5*IQR\nprint(\"min_value\",min_value,\" \",\"max_value=\",max_value)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.325044Z","iopub.execute_input":"2024-01-04T20:15:53.325387Z","iopub.status.idle":"2024-01-04T20:15:53.338711Z","shell.execute_reply.started":"2024-01-04T20:15:53.325359Z","shell.execute_reply":"2024-01-04T20:15:53.337133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[(df['age']<min_value) | (df['age']>max_value)]","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.341644Z","iopub.execute_input":"2024-01-04T20:15:53.342054Z","iopub.status.idle":"2024-01-04T20:15:53.369261Z","shell.execute_reply.started":"2024-01-04T20:15:53.34202Z","shell.execute_reply":"2024-01-04T20:15:53.36779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['age']==89]","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.371083Z","iopub.execute_input":"2024-01-04T20:15:53.371457Z","iopub.status.idle":"2024-01-04T20:15:53.394047Z","shell.execute_reply.started":"2024-01-04T20:15:53.371423Z","shell.execute_reply":"2024-01-04T20:15:53.392782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[(df['age']<29)].shape","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.395572Z","iopub.execute_input":"2024-01-04T20:15:53.396041Z","iopub.status.idle":"2024-01-04T20:15:53.4062Z","shell.execute_reply.started":"2024-01-04T20:15:53.395998Z","shell.execute_reply":"2024-01-04T20:15:53.404858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can observe that there are 3 outliers of age **26,28 and 89** also people of age 26 & 28 are not likely to get affected by cancer","metadata":{}},{"cell_type":"code","source":"df.groupby(\"age\")['cancer'].sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.414638Z","iopub.execute_input":"2024-01-04T20:15:53.415084Z","iopub.status.idle":"2024-01-04T20:15:53.427598Z","shell.execute_reply.started":"2024-01-04T20:15:53.415048Z","shell.execute_reply":"2024-01-04T20:15:53.426467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We define bins to create groups for people of each age group","metadata":{}},{"cell_type":"code","source":"bins=[20,30,40,50,60,70,80,90]\ndf['age-group'] =  pd.cut(df['age'], bins=bins)\nsns.countplot(x='age-group', data=df)\nplt.xlabel('Age Group')\nplt.ylabel('Count')\nplt.title('Distribution of Age Groups')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.428896Z","iopub.execute_input":"2024-01-04T20:15:53.4293Z","iopub.status.idle":"2024-01-04T20:15:53.77208Z","shell.execute_reply.started":"2024-01-04T20:15:53.429268Z","shell.execute_reply":"2024-01-04T20:15:53.770906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### The age group of people is mostly **'40-70'**","metadata":{}},{"cell_type":"code","source":"df.groupby('age')['cancer'].sum()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.773522Z","iopub.execute_input":"2024-01-04T20:15:53.774423Z","iopub.status.idle":"2024-01-04T20:15:53.786467Z","shell.execute_reply.started":"2024-01-04T20:15:53.774384Z","shell.execute_reply":"2024-01-04T20:15:53.785348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing people of age group less than 29 from our dataset\ndf = df[(df[\"age\"]>min_value)]","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.787838Z","iopub.execute_input":"2024-01-04T20:15:53.788374Z","iopub.status.idle":"2024-01-04T20:15:53.802019Z","shell.execute_reply.started":"2024-01-04T20:15:53.788343Z","shell.execute_reply":"2024-01-04T20:15:53.80069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_df = df.groupby('age-group')['cancer'].count().reset_index()\ngrouped_df.plot(x='age-group',y='cancer',kind='bar')","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:53.804012Z","iopub.execute_input":"2024-01-04T20:15:53.804928Z","iopub.status.idle":"2024-01-04T20:15:54.312158Z","shell.execute_reply.started":"2024-01-04T20:15:53.804758Z","shell.execute_reply":"2024-01-04T20:15:54.31099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(\"age-group\")[\"cancer\"].count().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.313715Z","iopub.execute_input":"2024-01-04T20:15:54.314233Z","iopub.status.idle":"2024-01-04T20:15:54.329685Z","shell.execute_reply.started":"2024-01-04T20:15:54.314197Z","shell.execute_reply":"2024-01-04T20:15:54.328246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We conclude that people of age group **40-70** are more likely to get affected by cancer","metadata":{}},{"cell_type":"code","source":"df.groupby(\"implant\")[\"cancer\"].count().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.331675Z","iopub.execute_input":"2024-01-04T20:15:54.33208Z","iopub.status.idle":"2024-01-04T20:15:54.344366Z","shell.execute_reply.started":"2024-01-04T20:15:54.332046Z","shell.execute_reply":"2024-01-04T20:15:54.34304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('implant')['cancer'].count().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.345709Z","iopub.execute_input":"2024-01-04T20:15:54.346432Z","iopub.status.idle":"2024-01-04T20:15:54.599256Z","shell.execute_reply.started":"2024-01-04T20:15:54.346386Z","shell.execute_reply":"2024-01-04T20:15:54.598079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No of patients not having implant but cancer positive\nlen(df[(df['implant']==0) & (df['cancer']==1)])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.601734Z","iopub.execute_input":"2024-01-04T20:15:54.602144Z","iopub.status.idle":"2024-01-04T20:15:54.614475Z","shell.execute_reply.started":"2024-01-04T20:15:54.60211Z","shell.execute_reply":"2024-01-04T20:15:54.613033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No of patients having implant and cancer positive\nlen(df[(df['implant']==1) & (df['cancer']==1)])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.616346Z","iopub.execute_input":"2024-01-04T20:15:54.617011Z","iopub.status.idle":"2024-01-04T20:15:54.628306Z","shell.execute_reply.started":"2024-01-04T20:15:54.616975Z","shell.execute_reply":"2024-01-04T20:15:54.627032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We notice that in relation to breast **implant** our data is highly imbalanced because **53200** patients in our training dataset didnot had an implant while only **1477** patients had implant\n* **13 out of 1477** patients who had implant is positive for cancer\n* **1145 out of 53200** patients who did not had implant is negative for cancer","metadata":{}},{"cell_type":"code","source":"df.groupby(\"biopsy\")[\"cancer\"].count().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.629915Z","iopub.execute_input":"2024-01-04T20:15:54.63091Z","iopub.status.idle":"2024-01-04T20:15:54.642733Z","shell.execute_reply.started":"2024-01-04T20:15:54.630847Z","shell.execute_reply":"2024-01-04T20:15:54.641278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# No of patients who had biopsy and also tested positive for cancer\nlen(df[(df['cancer']==1) & (df['biopsy']==1)])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.64527Z","iopub.execute_input":"2024-01-04T20:15:54.64576Z","iopub.status.idle":"2024-01-04T20:15:54.658609Z","shell.execute_reply.started":"2024-01-04T20:15:54.645715Z","shell.execute_reply":"2024-01-04T20:15:54.657363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='biopsy', hue='cancer', data=df)\nplt.xlabel('Biopsy')\nplt.ylabel('Count')\nplt.title('Biopsy vs. Cancer')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.660039Z","iopub.execute_input":"2024-01-04T20:15:54.660388Z","iopub.status.idle":"2024-01-04T20:15:54.961038Z","shell.execute_reply.started":"2024-01-04T20:15:54.660358Z","shell.execute_reply":"2024-01-04T20:15:54.959698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We conclude that all patients tested positive for cancer had biopsy","metadata":{}},{"cell_type":"code","source":"df.groupby(\"invasive\")[\"cancer\"].count().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.962816Z","iopub.execute_input":"2024-01-04T20:15:54.963201Z","iopub.status.idle":"2024-01-04T20:15:54.973364Z","shell.execute_reply.started":"2024-01-04T20:15:54.96317Z","shell.execute_reply":"2024-01-04T20:15:54.972078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df[(df['invasive']==1) & (df['cancer']==1)])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.974463Z","iopub.execute_input":"2024-01-04T20:15:54.975455Z","iopub.status.idle":"2024-01-04T20:15:54.987184Z","shell.execute_reply.started":"2024-01-04T20:15:54.975407Z","shell.execute_reply":"2024-01-04T20:15:54.98546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df[(df['invasive']==0) & (df['cancer']==1)])","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:54.989138Z","iopub.execute_input":"2024-01-04T20:15:54.98993Z","iopub.status.idle":"2024-01-04T20:15:55.001081Z","shell.execute_reply.started":"2024-01-04T20:15:54.989867Z","shell.execute_reply":"2024-01-04T20:15:54.999489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='invasive', hue='cancer', data=df)","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:15:55.003034Z","iopub.execute_input":"2024-01-04T20:15:55.003789Z","iopub.status.idle":"2024-01-04T20:15:55.295363Z","shell.execute_reply.started":"2024-01-04T20:15:55.003743Z","shell.execute_reply":"2024-01-04T20:15:55.293958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We conclude:\n* **340 out of 53859** patients having cancer pos has invasive=False\n* **818 out of 818** patients having cancer pos has invasive=True\ni.e in all the patients having cancer,invasive= True","metadata":{}},{"cell_type":"code","source":"# There are 11907 unique patients in the dataset\nlen(df['patient_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:16:34.633625Z","iopub.execute_input":"2024-01-04T20:16:34.634083Z","iopub.status.idle":"2024-01-04T20:16:34.647907Z","shell.execute_reply.started":"2024-01-04T20:16:34.63405Z","shell.execute_reply":"2024-01-04T20:16:34.646381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:19:40.077759Z","iopub.execute_input":"2024-01-04T20:19:40.078175Z","iopub.status.idle":"2024-01-04T20:19:40.085088Z","shell.execute_reply.started":"2024-01-04T20:19:40.078144Z","shell.execute_reply":"2024-01-04T20:19:40.084092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group columns having same patient_id\npatient_summary = df.groupby('patient_id')['image_id'].count().sort_values(ascending=False)\npatient_summary","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:34:28.851556Z","iopub.execute_input":"2024-01-04T20:34:28.851984Z","iopub.status.idle":"2024-01-04T20:34:28.889258Z","shell.execute_reply.started":"2024-01-04T20:34:28.851951Z","shell.execute_reply":"2024-01-04T20:34:28.887979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are no duplicate images in our dataset\nlen(df['image_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:35:34.781613Z","iopub.execute_input":"2024-01-04T20:35:34.782094Z","iopub.status.idle":"2024-01-04T20:35:34.79898Z","shell.execute_reply.started":"2024-01-04T20:35:34.782058Z","shell.execute_reply":"2024-01-04T20:35:34.797756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_per_patient = df.groupby(\"patient_id\")['cancer'].max().values\nnegative_count = (cancer_per_patient==0).sum()\npositive_count = (cancer_per_patient==1).sum()\nprint(f'There are {negative_count} no.of unique patients negative with cancer.')\nprint(f'There are {positive_count} no.of unique patients positive with cancer.')","metadata":{"execution":{"iopub.status.busy":"2024-01-04T20:51:06.082542Z","iopub.execute_input":"2024-01-04T20:51:06.082969Z","iopub.status.idle":"2024-01-04T20:51:06.10982Z","shell.execute_reply.started":"2024-01-04T20:51:06.082935Z","shell.execute_reply":"2024-01-04T20:51:06.107814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**cancer_per_patient**, contains the maximum 'cancer' value for each patient. If a patient has at least one record with 'cancer' equal to 1, the maximum value for that patient will be 1. If all entries for a patient are 0, the maximum value will be 0.","metadata":{}},{"cell_type":"code","source":"# Train-Test Split\ndf_train = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\nX = df_train.drop('cancer', axis=1)  \ny = df_train['cancer']  ","metadata":{"execution":{"iopub.status.busy":"2024-01-04T21:07:55.205304Z","iopub.execute_input":"2024-01-04T21:07:55.20577Z","iopub.status.idle":"2024-01-04T21:07:55.317946Z","shell.execute_reply.started":"2024-01-04T21:07:55.205734Z","shell.execute_reply":"2024-01-04T21:07:55.316701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test","metadata":{"execution":{"iopub.status.busy":"2024-01-04T21:09:06.286097Z","iopub.execute_input":"2024-01-04T21:09:06.286555Z","iopub.status.idle":"2024-01-04T21:09:06.301637Z","shell.execute_reply.started":"2024-01-04T21:09:06.286519Z","shell.execute_reply":"2024-01-04T21:09:06.300742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}