{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom glob import glob\nimport matplotlib.pyplot as plt\nimport pydicom as dicom","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf_test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf_sample = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob as gb  \n# Initialize a variable  \n#glob(pathname, *, recursive = True)  \ngenVar = gb.glob('/kaggle/input/rsna-breast-cancer-detection/train_images/10**/*.dcm', recursive = True) # Set Pattern in glob() function  \n# Printing list of names of all files that matched the pattern  \nprint(\"List of the all the files in the directory having extension .dcm: \")  \nfor files in genVar:   \n    print(files)  ","metadata":{"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\n#ds = pydicom.dcmread(files)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pydicom import dcmread\n\nds = dcmread(files)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Representation of soem details\npat_name = ds\nprint(f\"Patient ID.......: {ds.PatientID}\")\nprint(f\"Image size.......: {ds.Rows} x {ds.Columns}\")\nprint(f\"Instance No ....: {ds.InstanceNumber }\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\nfig, axes = plt.subplots(6,6, figsize=(16,16)) \nfor i, ax in enumerate(axes.reshape(-1)):\n    img_path = genVar [i]\n    img = dicom.dcmread(img_path) \n    #ds.PhotometricInterpretation = 'YBR_FULL'\n#arr = ds.pixel_array\n    ax.imshow(ds.pixel_array)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n# distribution plot for FVC\nsns.distplot(df_train.cancer, hist = False,color = \"darkred\")\nplt.title(\"Cancer Distribution\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check the no of 0/1 label distribution in the dataset\nsizes =  [len(df_train[df_train.cancer == 0]),len(df_train[df_train.cancer == 1])]\nexplode = (0.1,0)  # explode 1st slice\ncolors = ['red',\"yellow\"]\nplt.pie(sizes, explode=explode, labels=df_train.cancer.unique(), colors=colors,autopct='%1.1f%%', shadow=True, startangle=140)\nplt.axis('equal')\nplt.title(\"Pie Chart for Cancer Distribution\")\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.age.min(),df_train.age.max()\n\n\nf, (ax1, ax2) = plt.subplots(1, 2, figsize = (16, 6))\n\n# Patient age group\nageGroupLabel = 'Below 60', '60-70', '70-80', 'Above 80'\n\nbelow60 = len(df_train[df_train.age<60])\nsixty_to_seventy = len(df_train[(df_train['age']>=60) & (df_train['age']<= 70)])\nseventy_to_eighty = len(df_train[(df_train['age']>70) & (df_train['age']<= 80)])\nabove80 = len(df_train[df_train.age>80])\n\n# Number of Guests expected in age group\npatientNumbers     = [below60, sixty_to_seventy, seventy_to_eighty,above80] \n\nexplode = (0, 0, 0, 0.1)\ncolors  = (\"green\",\"indigo\",\"blue\", \"red\")\n\n# Draw the pie chart\nax1.pie(patientNumbers,explode = explode,colors = colors,labels = ageGroupLabel,autopct = '%1.2f',startangle = 90)\n\n# Aspect ratio\nax1.axis('equal')\n\n\n# distribution plot for Age\nsns.distplot(df_train.age, hist = False, color = \"indigo\")\nplt.suptitle(\"Age Distribution\")\n\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets group Male & Female data\ngrp = df_train.groupby(\"cancer\")\n\n# draw a plot to display mean of FVC for Males and Females\nsplot = sns.barplot(x=df_train.cancer.unique(),y= grp['age'].mean())\n\nfor p in splot.patches:\n    splot.annotate(format(p.get_height(), '.2f'), (p.get_x() + p.get_width() / 2., p.get_height()), ha = 'center', va = 'center', xytext = (0, 10), textcoords = 'offset points')\n\nplt.xlabel(\"Cancer\",fontsize = 30)\nplt.ylabel(\"Mean Age\",fontsize = 30)\nplt.title (\"Age Mean for Female\",fontsize = 30) \nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display the image read above\nimg = ds.pixel_array\nimg[img == -2000] = 0\n\nplt.axis('off')\nplt.imshow(img)\nplt.show()\n\nplt.axis('off')\nplt.imshow(-img) # Invert colors with -\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# helper function\ndef dicom_to_image(filename):\n    img = dicom.dcmread(filename) \n    img = ds.pixel_array\n    img[img == -2000] = 0\n    return img","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n# lets display some 20 images at random\ntrain_dir ='/kaggle/input/rsna-breast-cancer-detection/train_images'\nfiles = glob.glob(train_dir + \"/*/*.dcm\")\n\nf, plots = plt.subplots(4, 5, sharex='col', sharey='row', figsize=(10, 8))\nfor i in range(20):\n    plots[i // 5, i % 5].axis('off')\n    plots[i // 5, i % 5].imshow(dicom_to_image(np.random.choice(files)), cmap=plt.cm.bone)","metadata":{},"execution_count":null,"outputs":[]}]}