{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pydicom\n\nimport os\n\nfrom sklearn.utils import class_weight\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-30T14:32:43.332203Z","iopub.execute_input":"2022-11-30T14:32:43.332701Z","iopub.status.idle":"2022-11-30T14:32:43.491739Z","shell.execute_reply.started":"2022-11-30T14:32:43.332608Z","shell.execute_reply":"2022-11-30T14:32:43.490628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dataframe","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"../input/rsna-breast-cancer-detection/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:44.894616Z","iopub.execute_input":"2022-11-30T14:32:44.895029Z","iopub.status.idle":"2022-11-30T14:32:45.035457Z","shell.execute_reply.started":"2022-11-30T14:32:44.894998Z","shell.execute_reply":"2022-11-30T14:32:45.034396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# total number of rows is equal to number of unique image ids\nlen(train_df),train_df[\"image_id\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:45.909951Z","iopub.execute_input":"2022-11-30T14:32:45.911549Z","iopub.status.idle":"2022-11-30T14:32:45.933931Z","shell.execute_reply.started":"2022-11-30T14:32:45.911498Z","shell.execute_reply":"2022-11-30T14:32:45.933113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# total number of patients in train df\nlen(train_df.groupby(\"patient_id\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:46.54785Z","iopub.execute_input":"2022-11-30T14:32:46.548434Z","iopub.status.idle":"2022-11-30T14:32:46.803461Z","shell.execute_reply.started":"2022-11-30T14:32:46.548398Z","shell.execute_reply":"2022-11-30T14:32:46.802382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns, test_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:46.80577Z","iopub.execute_input":"2022-11-30T14:32:46.80624Z","iopub.status.idle":"2022-11-30T14:32:46.813467Z","shell.execute_reply.started":"2022-11-30T14:32:46.806177Z","shell.execute_reply":"2022-11-30T14:32:46.812358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the columns that are not present in test df\nset(train_df.columns) - set(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:46.996407Z","iopub.execute_input":"2022-11-30T14:32:46.997463Z","iopub.status.idle":"2022-11-30T14:32:47.007545Z","shell.execute_reply.started":"2022-11-30T14:32:46.997412Z","shell.execute_reply":"2022-11-30T14:32:47.005842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our model must not rely on above features because they wont be present during inference.","metadata":{}},{"cell_type":"code","source":"# lets check if every patient has same values for cancer property\ntrain_df.groupby(\"patient_id\")[\"cancer\"].nunique().value_counts()\n# 1 means all the patients that has same value for all images\n# 2 means that those patients has non cancer and cancer images at the same time","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:49.843813Z","iopub.execute_input":"2022-11-30T14:32:49.8448Z","iopub.status.idle":"2022-11-30T14:32:49.861188Z","shell.execute_reply.started":"2022-11-30T14:32:49.844759Z","shell.execute_reply":"2022-11-30T14:32:49.860389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see couple of patients with cancer and non cancer images\ngrouped = train_df.groupby(\"patient_id\")\n# filter patients that has both cancer or both non cancer images\n\ncount = 0\nfor patient_id, group in grouped:\n    if group[\"cancer\"].nunique() == 1:\n        continue\n\n    display(group)\n        \n    count += 1 # stop after 10 examples\n    if count == 10:\n        break\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:51.623203Z","iopub.execute_input":"2022-11-30T14:32:51.623648Z","iopub.status.idle":"2022-11-30T14:32:51.797924Z","shell.execute_reply.started":"2022-11-30T14:32:51.623612Z","shell.execute_reply":"2022-11-30T14:32:51.796926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As you can see from these examples we are getting same label for same breast.\n\nBut we should also verify it:","metadata":{}},{"cell_type":"code","source":"# make a unique id for breast of each patient\ntrain_df[\"breast_id\"] = train_df[\"patient_id\"].astype(str) + \"_\" + train_df[\"laterality\"]\n\n# lets check if every breast has same values for cancer property\ntrain_df.groupby(\"breast_id\")[\"cancer\"].nunique().value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:55.624993Z","iopub.execute_input":"2022-11-30T14:32:55.625817Z","iopub.status.idle":"2022-11-30T14:32:55.706517Z","shell.execute_reply.started":"2022-11-30T14:32:55.625778Z","shell.execute_reply":"2022-11-30T14:32:55.705258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now lets see the distribution of the views\n# a normal case both CC and MLO views\ntrain_df.groupby(\"breast_id\")[\"view\"].nunique().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:57.922702Z","iopub.execute_input":"2022-11-30T14:32:57.923671Z","iopub.status.idle":"2022-11-30T14:32:57.971584Z","shell.execute_reply.started":"2022-11-30T14:32:57.923632Z","shell.execute_reply":"2022-11-30T14:32:57.970032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now for the important part, lets plot the distribution of the cancer among the breasts\ntrain_df.groupby(\"breast_id\")[\"cancer\"].max().value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:32:59.435965Z","iopub.execute_input":"2022-11-30T14:32:59.436399Z","iopub.status.idle":"2022-11-30T14:32:59.67755Z","shell.execute_reply.started":"2022-11-30T14:32:59.436361Z","shell.execute_reply":"2022-11-30T14:32:59.676292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Wow, we are looking at a severely imbalanced dataset.\n","metadata":{}},{"cell_type":"code","source":"# lets look at the stats\nprint (\"Total Number of breasts:\", len(train_df.groupby(\"breast_id\")))\n\ndistribution = train_df.groupby(\"breast_id\")[\"cancer\"].max().value_counts()\nprint(\"no cancer: \", distribution[0])\nprint(\"cancer   : \", distribution[1])\n\nzeros = np.zeros(distribution[0])\nones = np.ones(distribution[1])\ny = np.concatenate([zeros, ones])\n\nclass_weights = class_weight.compute_class_weight('balanced', classes=np.unique(y), y=y)\nprint(\"class weights: \", class_weights)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T14:53:53.972724Z","iopub.execute_input":"2022-11-30T14:53:53.973128Z","iopub.status.idle":"2022-11-30T14:53:54.301254Z","shell.execute_reply.started":"2022-11-30T14:53:53.973096Z","shell.execute_reply":"2022-11-30T14:53:54.299975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# as you can see some cases has 3 different views\n# lets see the cases with 3 views\ngrouped = train_df.groupby(\"breast_id\")\n\nfor breast_id, group in grouped:\n    if group[\"view\"].nunique() == 3:\n        display(group)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T12:56:16.243288Z","iopub.execute_input":"2022-11-30T12:56:16.243702Z","iopub.status.idle":"2022-11-30T12:56:21.381006Z","shell.execute_reply.started":"2022-11-30T12:56:16.243667Z","shell.execute_reply":"2022-11-30T12:56:21.379732Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Even in above anomalies, there is always a MLO and a CC view.\nSo i suppose one can discard the rest of views.","metadata":{}},{"cell_type":"code","source":"# we found out that a breast can have MLO, ML, CC, AT, LM views.\n# We know that every breast has at least two views.\n# But we didnt check if they are spesifically CC and MLO views.\n# lets check that\n\nfor breast_id, group in train_df.groupby(\"breast_id\"):\n    if not (\"CC\" in group[\"view\"].values and \"MLO\" in group[\"view\"].values):\n        display(group)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T12:56:21.38299Z","iopub.execute_input":"2022-11-30T12:56:21.383898Z","iopub.status.idle":"2022-11-30T12:56:24.071531Z","shell.execute_reply.started":"2022-11-30T12:56:21.383845Z","shell.execute_reply":"2022-11-30T12:56:24.070223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There was no output. That means that every breast has at least one CC view and one MLO view.","metadata":{}},{"cell_type":"code","source":"# now for the hard part (inspecting image dimensions)\n# lets add the dicom image dimensions to the dataframe \n# WARNING: this cell takes about 50 minute to execute\n\ndef get_image_dimensions(image_path) -> tuple:\n    image_path = os.path.join(\"../input/rsna-breast-cancer-detection/train_images\", image_path)\n    image = pydicom.dcmread(image_path)\n    return image.Rows, image.Columns\n\ntrain_df[\"image_path\"] = train_df[\"patient_id\"].astype(str) + \"/\" + train_df[\"image_id\"].astype(str) + \".dcm\"\ntrain_df[\"image_dimension_x\"], train_df[\"image_dimension_y\"] = zip(*train_df[\"image_path\"].apply(get_image_dimensions))\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T12:56:24.073261Z","iopub.execute_input":"2022-11-30T12:56:24.073793Z","iopub.status.idle":"2022-11-30T13:09:52.391111Z","shell.execute_reply.started":"2022-11-30T12:56:24.073738Z","shell.execute_reply":"2022-11-30T13:09:52.388371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Well, this takes way too much time. I dont even know if above cell is working.\n\n300 gigabytes is way too much data to read, let alone preprocess and train, and make inference. I suppose one can upload their preprocessed training dataset to kaggle and use it for training.","metadata":{}},{"cell_type":"code","source":"# lets see random 20 rows\ntrain_df.sample(20)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T13:09:52.393537Z","iopub.status.idle":"2022-11-30T13:09:52.394108Z","shell.execute_reply.started":"2022-11-30T13:09:52.393873Z","shell.execute_reply":"2022-11-30T13:09:52.393899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mean image dimensions\ntrain_df[\"image_dimension_x\"].mean(), train_df[\"image_dimension_y\"].mean()","metadata":{"execution":{"iopub.status.busy":"2022-11-30T13:09:52.395528Z","iopub.status.idle":"2022-11-30T13:09:52.396023Z","shell.execute_reply.started":"2022-11-30T13:09:52.395791Z","shell.execute_reply":"2022-11-30T13:09:52.395829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# variance of image dimensions\ntrain_df[\"image_dimension_x\"].var(), train_df[\"image_dimension_y\"].var()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets draw normal distribution of image dimensions using mean and variance\nimport scipy.stats as stats\n\nx = np.linspace(0, 5000, 5000)\nplt.plot(x, stats.norm.pdf(x, train_df[\"image_dimension_x\"].mean(), train_df[\"image_dimension_x\"].var()))\nplt.show()\n\nx = np.linspace(0, 5000, 5000)\nplt.plot(x, stats.norm.pdf(x, train_df[\"image_dimension_y\"].mean(), train_df[\"image_dimension_y\"].var()))\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# min and max image dimensions\nprint(f\"x min: \\t {train_df['image_dimension_x'].min()} \\t x max: \\t {train_df['image_dimension_x'].max()}\")\nprint(f\"y min: \\t {train_df['image_dimension_y'].min()} \\t y max: \\t {train_df['image_dimension_y'].max()}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets plot the distribution of the image dimensions\ntrain_df[\"image_dimension_x\"].hist(bins=50)\ntrain_df[\"image_dimension_y\"].hist(bins=50)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-30T13:09:52.398135Z","iopub.status.idle":"2022-11-30T13:09:52.398903Z","shell.execute_reply.started":"2022-11-30T13:09:52.398677Z","shell.execute_reply":"2022-11-30T13:09:52.3987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(\"train_df_processed.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-30T13:09:52.40053Z","iopub.status.idle":"2022-11-30T13:09:52.401144Z","shell.execute_reply.started":"2022-11-30T13:09:52.40085Z","shell.execute_reply":"2022-11-30T13:09:52.400878Z"},"trusted":true},"execution_count":null,"outputs":[]}]}