{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n!pip install -qU python-gdcm pylibjpeg\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom datetime import datetime\nfrom tqdm.notebook import tqdm\nfrom concurrent.futures import ProcessPoolExecutor\n\nimport pydicom\nfrom PIL import Image\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-15T16:41:42.244116Z","iopub.execute_input":"2023-01-15T16:41:42.244755Z","iopub.status.idle":"2023-01-15T16:41:56.587438Z","shell.execute_reply.started":"2023-01-15T16:41:42.24462Z","shell.execute_reply":"2023-01-15T16:41:56.586078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Collection**","metadata":{}},{"cell_type":"code","source":"#Read in the test and train files, print first few rows of train file.\ntrain = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:42:55.888105Z","iopub.execute_input":"2023-01-15T16:42:55.888819Z","iopub.status.idle":"2023-01-15T16:42:56.087552Z","shell.execute_reply.started":"2023-01-15T16:42:55.888746Z","shell.execute_reply":"2023-01-15T16:42:56.086469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data definition**","metadata":{}},{"cell_type":"code","source":"#Summary of the train dataset:\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:43:02.402386Z","iopub.execute_input":"2023-01-15T16:43:02.403753Z","iopub.status.idle":"2023-01-15T16:43:02.448376Z","shell.execute_reply.started":"2023-01-15T16:43:02.403653Z","shell.execute_reply":"2023-01-15T16:43:02.446701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:43:06.42218Z","iopub.execute_input":"2023-01-15T16:43:06.422789Z","iopub.status.idle":"2023-01-15T16:43:06.432255Z","shell.execute_reply.started":"2023-01-15T16:43:06.422736Z","shell.execute_reply":"2023-01-15T16:43:06.430931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize the first 10 mammograms in the train dataset:\nfig, axs = plt.subplots(2,5)\n\nfor i in range(10):\n    patient_id = train['patient_id'][i]\n    image_id = train['image_id'][i]\n    path_img = '/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(patient_id) + '/' + str(image_id) + '.dcm'\n    im = pydicom.dcmread(path_img)\n    arr = im.pixel_array\n    axs[i//5][(i%5)].imshow(np.array(arr), interpolation='none', cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:43:10.54085Z","iopub.execute_input":"2023-01-15T16:43:10.541406Z","iopub.status.idle":"2023-01-15T16:43:25.776604Z","shell.execute_reply.started":"2023-01-15T16:43:10.541363Z","shell.execute_reply":"2023-01-15T16:43:25.77524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We see that there are 2 kinds of images: the ones with a white background, and\n#the ones with a black background. The two kinds of images correspond to two different \n#photometric interpretations (MONOCHROME1 or MONOCHROME2):\nfor i in range(10):\n    patient_id = train['patient_id'][i]\n    image_id = train['image_id'][i]\n    path_img = '/kaggle/input/rsna-breast-cancer-detection/train_images/' + str(patient_id) + '/' + str(image_id) + '.dcm'\n    im = pydicom.dcmread(path_img)\n    print(im.PhotometricInterpretation, im.pixel_array.shape)\n    \n#We also see that the two types of images have different number of pixels. See the website \n#https://dicom.nema.org/medical/dicom/current/output/chtml/part03/sect_C.7.6.3.html#sect_C.7.6.3.1\n#for more details about photometric interpretations.","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:43:49.038916Z","iopub.execute_input":"2023-01-15T16:43:49.039445Z","iopub.status.idle":"2023-01-15T16:43:58.938009Z","shell.execute_reply.started":"2023-01-15T16:43:49.039406Z","shell.execute_reply":"2023-01-15T16:43:58.93668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting the data type of a few columns from numerical to string or categorical.\ntrain['site_id'] = train.site_id.astype('str')\ntrain['patient_id'] = train.patient_id.astype('str')\ntrain['image_id'] = train.image_id.astype('str')\ntrain['cancer'] = train.cancer.astype('category')\ntrain['biopsy'] = train.biopsy.astype('category')\ntrain['invasive'] = train.invasive.astype('category')\ntrain['BIRADS'] = train.BIRADS.astype('category')\ntrain['implant'] = train.implant.astype('category')\ntrain['machine_id'] = train.machine_id.astype('str')\ntrain.info()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:14.623603Z","iopub.execute_input":"2023-01-15T16:44:14.624065Z","iopub.status.idle":"2023-01-15T16:44:14.820931Z","shell.execute_reply.started":"2023-01-15T16:44:14.624031Z","shell.execute_reply":"2023-01-15T16:44:14.819499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Summary statistics for all the columns except 'age'.\ntrain[['site_id','patient_id','image_id','laterality','view','cancer',\n      'biopsy','invasive','BIRADS','implant','density','machine_id',\n      'difficult_negative_case']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:18.934731Z","iopub.execute_input":"2023-01-15T16:44:18.935181Z","iopub.status.idle":"2023-01-15T16:44:19.072607Z","shell.execute_reply.started":"2023-01-15T16:44:18.935145Z","shell.execute_reply":"2023-01-15T16:44:19.07129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Summary statistics for 'age'.\ntrain['age'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:20.726612Z","iopub.execute_input":"2023-01-15T16:44:20.72709Z","iopub.status.idle":"2023-01-15T16:44:20.745126Z","shell.execute_reply.started":"2023-01-15T16:44:20.727051Z","shell.execute_reply":"2023-01-15T16:44:20.743503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#According to the Internet,the 'view' column should only have 2 values: 'CC'\n#for the cranio-caudal view, and 'MLO' for the mediolateral oblique view.\n#But it has 6 unique values, due to mispelling and mistakes.\n\nprint(train['view'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:24.519619Z","iopub.execute_input":"2023-01-15T16:44:24.521162Z","iopub.status.idle":"2023-01-15T16:44:24.534069Z","shell.execute_reply.started":"2023-01-15T16:44:24.521098Z","shell.execute_reply":"2023-01-15T16:44:24.532002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We'll fix the mispellings/mistakes above later.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data cleaning and organization**","metadata":{}},{"cell_type":"code","source":"#Count the number of missing values in each column and sort the columns by missing percentage.\nmissing = pd.concat([train.isnull().sum(), 100 * train.isnull().mean()], axis=1)\nmissing.columns=['count', '%']\nmissing.sort_values(by='%', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:30.943016Z","iopub.execute_input":"2023-01-15T16:44:30.94355Z","iopub.status.idle":"2023-01-15T16:44:31.003161Z","shell.execute_reply.started":"2023-01-15T16:44:30.943498Z","shell.execute_reply":"2023-01-15T16:44:31.001873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We see that half the rows have missing BIRADS or density data, and a small number of rows have missing age data.\n#Because we have so much data available, we can simply drop the rows with missing data.\ntrain_cleaned = train.copy().dropna().reset_index(drop=True)\nprint(train_cleaned.shape)\nprint(train_cleaned.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:35.080102Z","iopub.execute_input":"2023-01-15T16:44:35.080899Z","iopub.status.idle":"2023-01-15T16:44:35.171776Z","shell.execute_reply.started":"2023-01-15T16:44:35.080815Z","shell.execute_reply":"2023-01-15T16:44:35.169674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Drop rows where 'view' column has value 'AT'.\n#Also, where 'view' takes value 'LM' or 'ML' or 'LMO', reassign those values to 'MLO'.\ntrain_cleaned = train_cleaned[train_cleaned['view'] != 'AT']\ntrain_cleaned.loc[train_cleaned['view'].isin(['LM','ML','LMO']),'view'] = 'MLO'\ntrain_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:38.704465Z","iopub.execute_input":"2023-01-15T16:44:38.70694Z","iopub.status.idle":"2023-01-15T16:44:38.742775Z","shell.execute_reply.started":"2023-01-15T16:44:38.706874Z","shell.execute_reply":"2023-01-15T16:44:38.740993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Summary statistics for the cleaned train set, \n#for all columns except 'age'\ntrain_cleaned[['site_id','patient_id','image_id','laterality','view','cancer',\n      'biopsy','invasive','BIRADS','implant','density','machine_id',\n      'difficult_negative_case']].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:42.623259Z","iopub.execute_input":"2023-01-15T16:44:42.625892Z","iopub.status.idle":"2023-01-15T16:44:42.776241Z","shell.execute_reply.started":"2023-01-15T16:44:42.625766Z","shell.execute_reply":"2023-01-15T16:44:42.773333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Summary statistics for the 'age' column of the cleaned train set.\n#Summary statistics for 'age'.\ntrain_cleaned['age'].describe()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:46.260411Z","iopub.execute_input":"2023-01-15T16:44:46.262307Z","iopub.status.idle":"2023-01-15T16:44:46.281292Z","shell.execute_reply.started":"2023-01-15T16:44:46.262231Z","shell.execute_reply":"2023-01-15T16:44:46.279333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Turn the categorical columns into numerical data by using get_dummies.\ntrain_cleaned_final = pd.get_dummies(train_cleaned, columns=['laterality', 'view',\n                                      'cancer','biopsy', 'invasive', 'BIRADS',\n                                      'implant','density','difficult_negative_case']).reset_index(drop=True)\nprint(train_cleaned_final.shape)\ntrain_cleaned_final.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:51.419065Z","iopub.execute_input":"2023-01-15T16:44:51.419689Z","iopub.status.idle":"2023-01-15T16:44:51.498543Z","shell.execute_reply.started":"2023-01-15T16:44:51.41962Z","shell.execute_reply":"2023-01-15T16:44:51.497457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Clean up the images, and making a list of them. The code in this cell is\n#adapted from this Kaggle notebook:\n#https://www.kaggle.com/code/rimzakhama/\n#rsna-dcm-images-to-pngs-same-format-as-input\nimg_dict = {}\n\n#for i in tqdm(range(train_cleaned_final.shape[0])):\ndef clean_image(i):\n\n    patient_id = train_cleaned_final['patient_id'][i]\n    image_id = train_cleaned_final['image_id'][i]\n    path_img = '/kaggle/input/rsna-breast-cancer-detection/train_images/' + patient_id + '/' + image_id + '.dcm'\n    im = pydicom.dcmread(path_img)\n    arr = im.pixel_array\n    \n    #Rescaling pixel values to 255.\n    scaled_arr = (arr / arr.max()) * 255.0\n    \n    #Converting all images to white background.\n    if im.PhotometricInterpretation == \"MONOCHROME2\":\n        scaled_arr = 255.0 - scaled_arr\n        \n    #Casting the type of the pixel array to be uint8, and converting to image\n    new_im = Image.fromarray(scaled_arr.astype(np.uint8))\n     \n    #Resizing the image object to height=512 and width=512\n    new_im = new_im.resize((512,512))\n    \n    #Print progress report\n    if i % 1000==0:\n        print('time:', datetime.now(), 'i=', i)\n    \n    return (patient_id, image_id, new_im)\n    #pixel_list = [new_im.getpixel((i,j)) for i in range(512) for j in range(512)]\n    #pixel_dict = {i: pixel_list for i in range(512**2)}\n    #img_list.append(pixel_dict)\n\nwith ProcessPoolExecutor(max_workers=4) as executor:\n    for (patient_id, image_id, new_im) in executor.map(clean_image, [i for i in range(train_cleaned_final.shape[0])]):\n        img_dict[patient_id + '/' + image_id] = new_im","metadata":{"execution":{"iopub.status.busy":"2023-01-15T16:44:58.765489Z","iopub.execute_input":"2023-01-15T16:44:58.766038Z","iopub.status.idle":"2023-01-15T18:25:59.036884Z","shell.execute_reply.started":"2023-01-15T16:44:58.765998Z","shell.execute_reply":"2023-01-15T18:25:59.031168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize a few cleaned images, and compare them with the image before cleaning.\npatient_id_example_1 = '10289'\nimage_id_example_1 = '1616844775'\nim1 = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/' \n                      + patient_id_example_1 + '/' + image_id_example_1 + '.dcm')\nplt.imshow(im1.pixel_array, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:26:27.950044Z","iopub.execute_input":"2023-01-15T18:26:27.95076Z","iopub.status.idle":"2023-01-15T18:26:31.426271Z","shell.execute_reply.started":"2023-01-15T18:26:27.950697Z","shell.execute_reply":"2023-01-15T18:26:31.424842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_dict[patient_id_example_1 + '/' + image_id_example_1]","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:26:35.111311Z","iopub.execute_input":"2023-01-15T18:26:35.113868Z","iopub.status.idle":"2023-01-15T18:26:35.183445Z","shell.execute_reply.started":"2023-01-15T18:26:35.113673Z","shell.execute_reply":"2023-01-15T18:26:35.177198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id_example_2 = '22528'\nimage_id_example_2 = '1265735355'\nim1 = pydicom.dcmread('/kaggle/input/rsna-breast-cancer-detection/train_images/' \n                      + patient_id_example_2 + '/' + image_id_example_2 + '.dcm')\nplt.imshow(im1.pixel_array, cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:26:39.4356Z","iopub.execute_input":"2023-01-15T18:26:39.436807Z","iopub.status.idle":"2023-01-15T18:26:42.254683Z","shell.execute_reply.started":"2023-01-15T18:26:39.43675Z","shell.execute_reply":"2023-01-15T18:26:42.252451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_dict[patient_id_example_2 + '/' + image_id_example_2]","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:26:45.914861Z","iopub.execute_input":"2023-01-15T18:26:45.915346Z","iopub.status.idle":"2023-01-15T18:26:45.944228Z","shell.execute_reply.started":"2023-01-15T18:26:45.915304Z","shell.execute_reply":"2023-01-15T18:26:45.94279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save train_cleaned_final as a csv file.\ntrain_cleaned_final.to_csv('cleaned_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:28:39.171566Z","iopub.execute_input":"2023-01-15T18:28:39.173216Z","iopub.status.idle":"2023-01-15T18:28:39.334369Z","shell.execute_reply.started":"2023-01-15T18:28:39.173127Z","shell.execute_reply":"2023-01-15T18:28:39.332558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the dictionary of cleaned images (img_dict) as a pickle file.\nimport pickle\n\nfileObj = open('cleaned_images.pickle', 'wb')\npickle.dump(img_dict,fileObj)\nfileObj.close()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T18:29:43.795779Z","iopub.execute_input":"2023-01-15T18:29:43.796232Z","iopub.status.idle":"2023-01-15T18:30:15.773611Z","shell.execute_reply.started":"2023-01-15T18:29:43.796198Z","shell.execute_reply":"2023-01-15T18:30:15.771844Z"},"trusted":true},"execution_count":null,"outputs":[]}]}