{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom matplotlib import image as mpimg\nimport asyncio\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        os.path.join(dirname, filename)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n!pip install tensorflow #used for image recognition\n\nimport tensorflow as tf \nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport tensorflow_io as tfio\n\n#for decoding .dcm files\n!pip install pydicom\n!pip install pillow\n\nimport pydicom\nimport PIL","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-13T02:01:22.896501Z","iopub.execute_input":"2023-01-13T02:01:22.896935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Start Training","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find Basic Statistics/First Look","metadata":{}},{"cell_type":"markdown","source":"### Create New Data base joined on patient ID","metadata":{}},{"cell_type":"code","source":"patients = train_df.drop(['site_id', 'image_id', 'BIRADS', 'machine_id', 'difficult_negative_case', 'laterality', 'view', 'density'], axis=1)\npatients = patients.groupby(by=[\"patient_id\"]).max()\npatients","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"positive = patients[patients['cancer'] == 1]\nnegative = patients[patients['cancer'] == 0]\n\n\n#basic information\ntotal_positive = positive.age.count()\ntotal = patients.age.count()\nper_positive = total_positive/total\nprint(str(per_positive) + \"\\tpercent that tested Postive\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Control %: I will use this to determine if any factors can lead to an increased risk for breast cancer.**","metadata":{}},{"cell_type":"markdown","source":"### Implants","metadata":{}},{"cell_type":"code","source":"implant_positive = positive[positive['implant'] == 1].age.count()\nimplant_negative = negative[negative['implant'] == 1].age.count()\nprint(str(implant_positive/(implant_positive + implant_negative)) + \"\\tpercent w/ Implants that tested Postive\")\nprint(str(per_positive) + \"\\tpercent that tested Postive\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Since the % of patients with implants with cancer is significantly less than those in the control % we can conclude that there is a negative corrilation between implants and brest cancer.**","metadata":{}},{"cell_type":"markdown","source":"### Age","metadata":{}},{"cell_type":"code","source":"#groups by age\nage_grouped = patients.groupby(by=[\"age\"]).sum()\ntotal_age_grouped = train_df.groupby(by=[\"age\"]).patient_id.count()\n\n#creates new data frame using as idex\nage_grouped = pd.concat([age_grouped, total_age_grouped], axis=1)\nage_grouped = age_grouped.rename(columns={\"patient_id\": \"patients\", \"biopsy\": \"biopsy(s)\", \"implant\":\"implant(s)\"})\nage_grouped[\"per_cancer\"] = age_grouped[\"cancer\"]/age_grouped[\"patients\"]\nage_grouped = age_grouped[age_grouped[\"patients\"] != 0]\n\nage_grouped","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Find line of best fit","metadata":{}},{"cell_type":"code","source":"#line of best fit\na, b = np.polyfit(age_grouped[age_grouped['per_cancer'] != 0].index, age_grouped[age_grouped['per_cancer'] != 0].per_cancer, 1)\n#note: Does not take into account  ages that had zero positive tests, this elimated outlinears\n\n#create new dataset with estimations\nage_grouped_est = age_grouped.drop(['cancer', 'biopsy(s)', 'invasive', 'implant(s)', 'patients'], axis=1)\n\n\n#create new column based on line of best fit\nage_grouped_est['per_cancer'] = a*age_grouped_est.index+b\nage_grouped_est.loc[age_grouped_est['per_cancer'] < 0] = 0 #fixes below 0 errors\n\n#plot\nplt.plot(age_grouped_est.index, age_grouped_est['per_cancer'])\nplt.plot(age_grouped.index, age_grouped['per_cancer'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Standard Deviation and 95% estimation","metadata":{}},{"cell_type":"code","source":"#get deviation\nage_grouped_est['deviation'] = age_grouped_est['per_cancer'] - age_grouped['per_cancer']\n\n\n#find standard deviation\nstd = age_grouped_est[age_grouped_est.index >= 40].deviation.sum() / age_grouped_est[age_grouped_est.index >= 40].deviation.count()\nprint(std)\n\n#get upper and lower bounds with the 2std or 95% rule\nage_grouped_est['lower_bound_95'] = age_grouped_est['per_cancer'] - (2*std)\nage_grouped_est.loc[age_grouped_est['lower_bound_95'] < 0] = 0 #fixes below 0 errors\n\nage_grouped_est['upper_bound_95'] = age_grouped_est['per_cancer'] + (2*std)\n\nage_grouped_est","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot\nplt.plot(age_grouped.index, age_grouped['per_cancer'])\nplt.plot(age_grouped_est.index, age_grouped_est['per_cancer'])\nplt.plot(age_grouped_est.index, age_grouped_est['lower_bound_95'])\nplt.plot(age_grouped_est.index, age_grouped_est['upper_bound_95'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filter For Outliers","metadata":{}},{"cell_type":"code","source":"age_grouped_no_outliners = age_grouped[\n    (age_grouped['per_cancer'] < age_grouped_est['per_cancer'] + (4.5*std)) &\n    (age_grouped['per_cancer'] > age_grouped_est['per_cancer'] - (4.5*std))\n]\nage_grouped_no_outliners.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot\nplt.plot(age_grouped_no_outliners.index, age_grouped_no_outliners['per_cancer'])\n#plt.plot(age_grouped_est.index, age_grouped_est['per_cancer'])\nplt.plot(age_grouped_est.index, age_grouped_est['lower_bound_95'])\nplt.plot(age_grouped_est.index, age_grouped_est['upper_bound_95'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Density","metadata":{}},{"cell_type":"code","source":"#create dataframe\ndensity_df = train_df[train_df['density'].notna()].groupby(by='density').sum()\ndensity_df = density_df.drop(['site_id', 'patient_id', 'patient_id', 'machine_id', 'BIRADS', 'image_id'], axis=1)\n\n#add columns\ndensity_df['patients'] = train_df[train_df['density'].notna()].groupby(by='density').count().age\ndensity_df['age'] = density_df['age'] / density_df['patients']\ndensity_df['per_cancer'] = density_df['cancer']/density_df['patients']\n\n\n#rename columns\ndensity_df = density_df.rename(columns={\"age\": \"avg_age\", \"biopsy\": \"biopsy(s)\", \"implant\":\"implant(s)\"})\n#show dataframe\ndensity_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.bar(density_df.index, density_df['per_cancer'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There does not seem to be a corrilation between breast density and cancer. I will not be pursuing this dataframe further","metadata":{}},{"cell_type":"markdown","source":"# Start Training Image Scanning\nhttps://stackabuse.com/image-recognition-in-python-with-tensorflow-and-keras/","metadata":{}},{"cell_type":"markdown","source":"### Develop Dataset","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Functions:","metadata":{}},{"cell_type":"code","source":"# Loading in the data\n\n#allows us to decode all images. Not nessary just for reading purposes\ndef decode_img(img):\n    return tfio.image.decode_dicom_image(tf.io.read_file(img), dtype=tf.uint16)\n\ndef decode_img_los(img):\n    return tfio.image.decode_dicom_image(tf.io.read_file(img), scale='auto', on_error='lossy', dtype=tf.uint8)\n\n#creates an image column, this is also not needed its just to clean up the code\ndef img_col(pat_id, img_id):\n    series = []\n    for x,y in zip(pat_id, img_id):\n        #series.append(decode_img(x, y))\n        series.append(f'/kaggle/input/rsna-breast-cancer-detection/train_images/{x}/{y}.dcm')\n    return series\n\ndef decode_img_col(imgs):\n    series = []\n    for x in zip(imgs):\n        series.append(decode_img_los(x[0]))\n    return series\n    \n\n#similar to dump function but instead saves to directory instead\ndef decode_img_col_dump(imgs, extra_pathing):\n    for x in zip(imgs):\n        path = f'/kaggle/working/{extra_pathing}/{x.index}'\n        convert_dcm_jpg(x[0]).save(path)\n        \n        \ndef convert_dcm_jpg(path):\n    \n    im = pydicom.dcmread(path)\n\n    im = im.pixel_array.astype(float)\n\n    rescaled_image = (np.maximum(im,0)/im.max())*255 # float pixels\n    final_image = np.uint8(rescaled_image) # integers pixels\n\n    final_image = Image.fromarray(final_image)\n\n    return final_image\n        \n        \n\n#if given 2 images it will output them to show the user. This is not needed its just easier\ndef show_images(path1, path2, title1 = \"image\", title2 = \"image2\"):\n    image1 = decode_img(path1)\n    image2 = decode_img(path2)\n    \n    fig, axes = plt.subplots(1,2, figsize=(10,10))\n    axes[0].imshow(np.squeeze(image1.numpy()), cmap='gray')\n    axes[0].set_title(title1)\n    axes[1].imshow(np.squeeze(image2.numpy()), cmap='gray')\n    axes[1].set_title(title2);","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Start Actual Development of Bot and Data","metadata":{}},{"cell_type":"code","source":"train_df['image'] = img_col(train_df['patient_id'], train_df['image_id'])\n\n#test show image funciton\nshow_images(train_df['image'][91], train_df['image'][87], 'cancer negative', 'cancer positive')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split dataframe to prepare for tensorflow dataframe creation\ncancer_pos_df = train_df[train_df['cancer'] == 1].sort_index()\ncancer_neg_df = train_df[train_df['cancer'] == 0].sort_index()\n\n#gets rid of columsn that are not need. We only need the file path to the image for training the image bot\ncancer_pos_df = cancer_pos_df.drop(['site_id', 'image_id', 'BIRADS', 'machine_id', 'difficult_negative_case', 'laterality', 'view', 'density', 'age', 'cancer', 'patient_id', 'biopsy', 'invasive', 'implant'], axis=1)\ncancer_neg_df = cancer_neg_df.drop(['site_id', 'image_id', 'BIRADS', 'machine_id', 'difficult_negative_case', 'laterality', 'view', 'density', 'age', 'cancer', 'patient_id', 'biopsy', 'invasive', 'implant'], axis=1)\n\n#decode all images\ncancer_pos_df['decoded_img'] = decode_img_col_dump(cancer_pos_df['image'], 'pos')\ncancer_neg_df['decoded_img'] = decode_img_col_dump(cancer_neg_df['image'], 'neg')\n\n#show_images(cancer_pos_df['decoded_img'][88], cancer_pos_df['decoded_img'][89])\n\n\nimg_height = decode_img(cancer_neg_df['image'][0]).shape[1]\nimg_width = decode_img(cancer_neg_df['image'][0]).shape[2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}