{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619402,"sourceType":"datasetVersion","datasetId":2688773},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Table of contents\n\n* [Loading packages](#packages)\n* [Exploring the meta data](#eda)\n    * [Missing values](#missing_vals)\n    * [Understanding patients](#patients)\n    * [Image features](#image_features)\n    * [Inspecting target features](#target_features)\n    * [Machine Ids](#machine_ids)\n* [Exploring the images](#images)\n    * [Inspecting dicom files](#dicom)\n    * [Pixelarray distributions](#raw_values)\n    * [Background values and image sizes](#mid_dependence)\n    * [Exploring dicom images of machine 49](#mid_49)\n    * [Exploring images showing cancer](#cancer)\n* [Conclusion](#conclusion)","metadata":{}},{"cell_type":"code","source":"!pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-01-02T17:50:56.037401Z","iopub.execute_input":"2025-01-02T17:50:56.037803Z","iopub.status.idle":"2025-01-02T17:51:08.130176Z","shell.execute_reply.started":"2025-01-02T17:50:56.037697Z","shell.execute_reply":"2025-01-02T17:51:08.129109Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading packages <a class=\"anchor\" id=\"packages\"></a>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport pydicom\nfrom os import listdir\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('darkgrid')\nsns.set_color_codes('bright')\nimport plotly.express as px\n\nfrom PIL import Image\n\nfrom scipy.stats import mode, skew\n\nfrom sklearn.mixture import GaussianMixture\nfrom sklearn.preprocessing import StandardScaler\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:11.12753Z","iopub.execute_input":"2025-01-09T16:09:11.127822Z","iopub.status.idle":"2025-01-09T16:09:13.861107Z","shell.execute_reply.started":"2025-01-09T16:09:11.127756Z","shell.execute_reply":"2025-01-09T16:09:13.860379Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploring at the meta data <a class=\"anchor\" id=\"eda\"></a>","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain.head()\n","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:13.863613Z","iopub.execute_input":"2025-01-09T16:09:13.863963Z","iopub.status.idle":"2025-01-09T16:09:13.976502Z","shell.execute_reply.started":"2025-01-09T16:09:13.863929Z","shell.execute_reply":"2025-01-09T16:09:13.975428Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:13.977443Z","iopub.execute_input":"2025-01-09T16:09:13.977708Z","iopub.status.idle":"2025-01-09T16:09:13.983343Z","shell.execute_reply.started":"2025-01-09T16:09:13.977685Z","shell.execute_reply":"2025-01-09T16:09:13.982487Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ok, 54706 rows and 14 columns. Do we have missing values?","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:13.984449Z","iopub.execute_input":"2025-01-09T16:09:13.984736Z","iopub.status.idle":"2025-01-09T16:09:14.007237Z","shell.execute_reply.started":"2025-01-09T16:09:13.984713Z","shell.execute_reply":"2025-01-09T16:09:14.006361Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.shape","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:14.008161Z","iopub.execute_input":"2025-01-09T16:09:14.00862Z","iopub.status.idle":"2025-01-09T16:09:14.013945Z","shell.execute_reply.started":"2025-01-09T16:09:14.008596Z","shell.execute_reply":"2025-01-09T16:09:14.013064Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Aha! There are a lot of columns in train that are not present in test. Some of them are related to the target feature **cancer** and it's obvious that they are not given (biopsy, invasive, BIRADS and difficult_negative_case). But what makes me wonder is that the density feature is also not given in test... ","metadata":{}},{"cell_type":"markdown","source":"## Missing values <a class=\"anchor\" id=\"missing_vals\"></a>","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:14.014982Z","iopub.execute_input":"2025-01-09T16:09:14.015312Z","iopub.status.idle":"2025-01-09T16:09:14.044631Z","shell.execute_reply.started":"2025-01-09T16:09:14.015281Z","shell.execute_reply":"2025-01-09T16:09:14.043781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ok, we do have missing values for the age, BIRADS and density. ","metadata":{}},{"cell_type":"markdown","source":"## Understanding patients <a class=\"anchor\" id=\"patients\"></a>","metadata":{}},{"cell_type":"markdown","source":"Kaç hastamız olduğunu gösteriyor.","metadata":{}},{"cell_type":"code","source":"train.patient_id.nunique","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:14.048168Z","iopub.execute_input":"2025-01-09T16:09:14.04846Z","iopub.status.idle":"2025-01-09T16:09:14.055284Z","shell.execute_reply.started":"2025-01-09T16:09:14.048435Z","shell.execute_reply":"2025-01-09T16:09:14.054249Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"How old are the patients?","metadata":{}},{"cell_type":"code","source":"train.groupby('patient_id').age.nunique().unique()\n","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:14.056313Z","iopub.execute_input":"2025-01-09T16:09:14.056549Z","iopub.status.idle":"2025-01-09T16:09:14.076892Z","shell.execute_reply.started":"2025-01-09T16:09:14.056529Z","shell.execute_reply":"2025-01-09T16:09:14.076116Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Ok, that's good! We only have one age value per patient or none at all. ","metadata":{}},{"cell_type":"code","source":"ages = train[train.age.isnull() == False].groupby(\n    'patient_id').age.apply(lambda l: np.unique(l)[0])\n\nfig, ax = plt.subplots(1,2,figsize=(20,5))\n\nsns.histplot(ages, color='orange', bins=60, ax=ax[0])\nax[0].set_title('Age distribution of patients');\nsns.violinplot(train.cancer, train.age, ax=ax[1], palette='Set2');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:14.077748Z","iopub.execute_input":"2025-01-09T16:09:14.077999Z","iopub.status.idle":"2025-01-09T16:09:15.023364Z","shell.execute_reply.started":"2025-01-09T16:09:14.077976Z","shell.execute_reply":"2025-01-09T16:09:15.022508Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* Most of the patients are older than 40 years. \n* It seems that we have two peaks around the age of 50 and close to 70. \n* There is a drop of patients counts after the age of 70. \n* For patients with cancer it's more likely that they are older and above the age of 50. ","metadata":{}},{"cell_type":"markdown","source":"How many patients do have cancer and how many images were taken per patient?","metadata":{}},{"cell_type":"code","source":"def has_cancer(l):\n    if len(l) == 1:\n        if l[0] == 0:\n            return False\n        elif l[0] == 1:\n            return True\n        else:\n            raise Exception\n    elif len(l) == 2:\n        return True\n    else:\n        raise Exception\n\npatient_cancer_map = train.groupby('patient_id').cancer.unique().apply(lambda l: has_cancer(l))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:15.024659Z","iopub.execute_input":"2025-01-09T16:09:15.025321Z","iopub.status.idle":"2025-01-09T16:09:15.519748Z","shell.execute_reply.started":"2025-01-09T16:09:15.025282Z","shell.execute_reply":"2025-01-09T16:09:15.518961Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\n\nsns.countplot(patient_cancer_map, palette='Paired', ax=ax[0]);\nax[0].set_title('Number of patients with cancer');\n\nsns.countplot(train.groupby('patient_id').size(), ax=ax[1])\nax[1].set_title('Number of images per patient')\nax[1].set_xlabel('Number of images')\nax[1].set_ylabel('Counts of patients');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:15.520723Z","iopub.execute_input":"2025-01-09T16:09:15.520948Z","iopub.status.idle":"2025-01-09T16:09:15.907095Z","shell.execute_reply.started":"2025-01-09T16:09:15.520927Z","shell.execute_reply":"2025-01-09T16:09:15.906272Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* It's an imbalanced classification problem! \n* In most cases we have given 4 images per patient but there are a few cases with more than 10 images as well! Why?","metadata":{}},{"cell_type":"markdown","source":"## Image features <a class=\"anchor\" id=\"image_features\"></a>","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(20,10))\nsns.countplot(train.laterality, ax=ax[0,0], palette='Greens_r')\nsns.countplot(train.view, ax=ax[0,1], palette='Reds_r')\nsns.countplot(train.implant, ax=ax[1,0], palette='Blues_r')\nsns.countplot(train.density, ax=ax[1,1], palette='Purples_r');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:15.908541Z","iopub.execute_input":"2025-01-09T16:09:15.908937Z","iopub.status.idle":"2025-01-09T16:09:16.514325Z","shell.execute_reply.started":"2025-01-09T16:09:15.908897Z","shell.execute_reply":"2025-01-09T16:09:16.513436Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* The laterality is quite balanced. \n* In the data description we can find that there are usually two views per breast. The most common views are CC and MLO and given these 2 views for the right and left breast we end up with 4 images that most of the patients show.  \n* Only a very few images show implants. \n* Most of the images show medium dense images of category B and C. Nonetheless there are also cases that are very dense (D) or less dense (A). Given the information that it could be more difficult to identify cancer in dense tissues this could be an interesting feature when thinking about validation strategies. ","metadata":{}},{"cell_type":"markdown","source":"## Inspecting target features <a class=\"anchor\" id=\"target_features\"></a>\n\nWe have already seen that the number of patients with cancer is quiet low compared to the patients without. Let's see how it looks like on the image-level.","metadata":{}},{"cell_type":"code","source":"biopsy_counts = train.groupby('cancer').biopsy.value_counts().unstack().fillna(0) \nbiopsy_perc = biopsy_counts.transpose() / biopsy_counts.sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:16.516041Z","iopub.execute_input":"2025-01-09T16:09:16.516385Z","iopub.status.idle":"2025-01-09T16:09:16.527334Z","shell.execute_reply.started":"2025-01-09T16:09:16.516358Z","shell.execute_reply":"2025-01-09T16:09:16.526437Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\nsns.countplot(train.cancer, palette='Reds', ax=ax[0])\nax[0].set_title('Number of images displaying cancer');\nsns.heatmap(biopsy_perc.transpose(), ax=ax[1], annot=True, cmap='Oranges');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:16.52848Z","iopub.execute_input":"2025-01-09T16:09:16.528744Z","iopub.status.idle":"2025-01-09T16:09:16.888985Z","shell.execute_reply.started":"2025-01-09T16:09:16.528708Z","shell.execute_reply":"2025-01-09T16:09:16.887957Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* The number of images displaying cancer is very low. It should be even of lower percentage than the number of patients with cancer. \n* Looking at the biopsy feature we can say that all patients with cancer had a biopsy. But only around 3 % of images without cancer had resulted in a follow-up biopsy. Maybe we should better have a look at this feature on the patient-level. ","metadata":{}},{"cell_type":"code","source":"train.cancer.value_counts()/train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:16.889937Z","iopub.execute_input":"2025-01-09T16:09:16.890169Z","iopub.status.idle":"2025-01-09T16:09:16.898316Z","shell.execute_reply.started":"2025-01-09T16:09:16.890148Z","shell.execute_reply":"2025-01-09T16:09:16.897271Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patient_cancer_map.value_counts() / patient_cancer_map.shape[0]","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:16.899702Z","iopub.execute_input":"2025-01-09T16:09:16.900015Z","iopub.status.idle":"2025-01-09T16:09:16.919036Z","shell.execute_reply.started":"2025-01-09T16:09:16.89999Z","shell.execute_reply":"2025-01-09T16:09:16.917692Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As expected, the percentage of patients with cancer is higher than the percentage of images displaying cancer. How many images show invasive cancer that has spread beyond the layer of tissue in which it developed?","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(20,10))\nsns.countplot(train[train.cancer==1].invasive, ax=ax[0,0], palette='Reds')\nsns.countplot(train.BIRADS, order=[0., 1., 2.], ax=ax[0,1], palette='Blues')\nax[0,0].set_title('Images showing invasive cancer');\n\nsns.countplot(train[train.cancer==0].BIRADS, order=[0., 1., 2.], ax=ax[1,0], palette='Purples')\nsns.countplot(train[train.cancer==1].BIRADS, order=[0., 1., 2.], ax=ax[1,1], palette='Purples')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:16.92024Z","iopub.execute_input":"2025-01-09T16:09:16.921151Z","iopub.status.idle":"2025-01-09T16:09:17.406571Z","shell.execute_reply.started":"2025-01-09T16:09:16.921116Z","shell.execute_reply":"2025-01-09T16:09:17.405628Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[train.cancer==1].invasive.value_counts() / train[train.cancer==1].shape[0]","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.407766Z","iopub.execute_input":"2025-01-09T16:09:17.408098Z","iopub.status.idle":"2025-01-09T16:09:17.418167Z","shell.execute_reply.started":"2025-01-09T16:09:17.408063Z","shell.execute_reply":"2025-01-09T16:09:17.417284Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* Roughly 70 % of all images displaying cancer are showing invasive cancer thus the one that spreads into other tissues as well. :(\n* For BIRADS we need to remember that its value is\n    * 0 if the breast required follow-up\n    * 1 if the breast was rated as negative for cancer \n    * 2 if the breast was rated as normal\n* Given that information, we can see that most images were rated as negative which suites to the fact that most images are not positive for cancer. But nonetheless there is still a high number of images that lead into a follow-up. I'm a bit confused... what's the difference between 1 and 2? From the ordering I would suspect that 2 is better than 1. \n* Going one level deeper and showing the BIRADS counts for 'no cancer'-images and 'cancer'-images, we can at least say that all images showing cancer needed a follow up (what we expected! ;))","metadata":{}},{"cell_type":"markdown","source":"## Machine Ids <a class=\"anchor\" id=\"machine_ids\"></a>","metadata":{}},{"cell_type":"markdown","source":"How many different imaging devices were used?","metadata":{}},{"cell_type":"code","source":"train.machine_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.419409Z","iopub.execute_input":"2025-01-09T16:09:17.419668Z","iopub.status.idle":"2025-01-09T16:09:17.42917Z","shell.execute_reply.started":"2025-01-09T16:09:17.419645Z","shell.execute_reply":"2025-01-09T16:09:17.42826Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1,figsize=(20,5))\nsns.countplot(train.machine_id, palette='tab10', ax=ax);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:17.430064Z","iopub.execute_input":"2025-01-09T16:09:17.430332Z","iopub.status.idle":"2025-01-09T16:09:17.617124Z","shell.execute_reply.started":"2025-01-09T16:09:17.430303Z","shell.execute_reply":"2025-01-09T16:09:17.616287Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* 10 different machines but most of the images were take with device of id 49, 21, 29 and 48. \n* Personally I'm not sure whether this feature has some importance for us. To be honest, I think that the patient_id, the age and the density will be much more important than the machine. ","metadata":{}},{"cell_type":"markdown","source":"# Exploring the images <a class=\"anchor\" id=\"images\"></a>\n\nBrowsing through the train and test folder of the images, we can see that the data is given as dicom files. Long time ago I wrote a tutorial notebook on dicom files while I was learning more about it myself. Please take a look at it if dicom is unknown to you. I hope that it will help to get started with the data structure. ;) \n\nhttps://www.kaggle.com/code/allunia/pulmonary-dicom-preprocessing\n","metadata":{}},{"cell_type":"code","source":"train_path = '/kaggle/input/rsna-breast-cancer-detection/train_images/'\ntest_path = '/kaggle/input/rsna-breast-cancer-detection/test_images/'","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.618346Z","iopub.execute_input":"2025-01-09T16:09:17.618944Z","iopub.status.idle":"2025-01-09T16:09:17.624685Z","shell.execute_reply.started":"2025-01-09T16:09:17.618893Z","shell.execute_reply":"2025-01-09T16:09:17.623467Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"First of all, we need a method to get all scans of one patient:","metadata":{}},{"cell_type":"code","source":"def load_scans(path, patient_id):\n    dcm_path = path + str(patient_id)\n    slices = [pydicom.dcmread(dcm_path + '/' + file, force=True) for file in listdir(dcm_path)]\n    return slices","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.632571Z","iopub.execute_input":"2025-01-09T16:09:17.63316Z","iopub.status.idle":"2025-01-09T16:09:17.638549Z","shell.execute_reply.started":"2025-01-09T16:09:17.633131Z","shell.execute_reply":"2025-01-09T16:09:17.637416Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Pixelarray distributions <a class=\"anchor\" id=\"raw values\"></a>","metadata":{}},{"cell_type":"markdown","source":"* Usually we need to transform the raw pixelarray distributions to Hounsfield Units. The raw values depend on the measurement settings like acquisition parameters and tube voltage of the scanner. \n* By normalizing to values of water and air (water has HU 0 and air -1000) the images of different measurements are becoming comparable. \n* We also need to understand how background values were treated! In previous competitions it was often the case that background values were set to values smaller than -1000, but we need to check whether this was done in this competition as well!","metadata":{}},{"cell_type":"markdown","source":"Let's load an example:","metadata":{}},{"cell_type":"code","source":"patient_ids = train.patient_id.unique()","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.639802Z","iopub.execute_input":"2025-01-09T16:09:17.640147Z","iopub.status.idle":"2025-01-09T16:09:17.647906Z","shell.execute_reply.started":"2025-01-09T16:09:17.640111Z","shell.execute_reply":"2025-01-09T16:09:17.64725Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scans = load_scans(train_path, patient_ids[0])","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.648873Z","iopub.execute_input":"2025-01-09T16:09:17.649473Z","iopub.status.idle":"2025-01-09T16:09:17.972076Z","shell.execute_reply.started":"2025-01-09T16:09:17.649449Z","shell.execute_reply":"2025-01-09T16:09:17.971258Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And let's plot the raw pixelarray distribution for this example and display the related image:","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2,figsize=(20,5))\nsns.histplot(scans[0].pixel_array.flatten(), ax=ax[0])\nax[1].imshow(scans[0].pixel_array)","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:17.973143Z","iopub.execute_input":"2025-01-09T16:09:17.973417Z","iopub.status.idle":"2025-01-09T16:09:35.824408Z","shell.execute_reply.started":"2025-01-09T16:09:17.973392Z","shell.execute_reply":"2025-01-09T16:09:35.823477Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Insights\n\n* Uhh! That's interesting! The background seems to be above 3000 this time! So it's competely different to previous competitions.\n* As the raw values depend on the machine, it could be worth it to explore background values dependent on the machine id. ","metadata":{}},{"cell_type":"markdown","source":"Let's get an impression by extracting the mode per scan for a few patients (let's say 50) for each machine id.","metadata":{}},{"cell_type":"code","source":"machine_ids = train.machine_id.unique()\nmachine_ids","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:35.825724Z","iopub.execute_input":"2025-01-09T16:09:35.826001Z","iopub.status.idle":"2025-01-09T16:09:35.832665Z","shell.execute_reply.started":"2025-01-09T16:09:35.825977Z","shell.execute_reply":"2025-01-09T16:09:35.831773Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scans[0].Columns","metadata":{"execution":{"iopub.status.busy":"2025-01-09T16:09:35.833686Z","iopub.execute_input":"2025-01-09T16:09:35.833927Z","iopub.status.idle":"2025-01-09T16:09:35.850535Z","shell.execute_reply.started":"2025-01-09T16:09:35.833905Z","shell.execute_reply":"2025-01-09T16:09:35.849513Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SINIFLANDIRMA ","metadata":{}},{"cell_type":"markdown","source":"## Kütüphanelerin import edilmesi","metadata":{}},{"cell_type":"code","source":"! pip install -U pylibjpeg pylibjpeg-openjpeg pylibjpeg-libjpeg pydicom python-gdcm\n! pip install --upgrade pydicom","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:09:35.851691Z","iopub.execute_input":"2025-01-09T16:09:35.852014Z","iopub.status.idle":"2025-01-09T16:10:02.41374Z","shell.execute_reply.started":"2025-01-09T16:09:35.851982Z","shell.execute_reply":"2025-01-09T16:10:02.412541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# dosya yollarını veriyoruz\ncsv_path = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ntrain_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\ntest_path = \"/kaggle/input/rsna-breast-cancer-detection/test_images\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:46.60354Z","iopub.status.idle":"2025-01-09T16:10:46.603844Z","shell.execute_reply.started":"2025-01-09T16:10:46.603699Z","shell.execute_reply":"2025-01-09T16:10:46.603713Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nimport pandas as pd\nfrom tqdm import tqdm\nimport cv2\nfrom concurrent.futures import ThreadPoolExecutor\n\n# Mamografi görüntüsünü kırpan fonksiyon\ndef crop_mammogram(image_path):\n    img = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n    if img is None:\n        return None, 0, 0  # Görüntü okunamadıysa None döndür\n    _, thresh = cv2.threshold(img, 1, 255, cv2.THRESH_BINARY)\n    contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    max_contour = max(contours, key=cv2.contourArea)\n    x, y, w, h = cv2.boundingRect(max_contour)\n    cropped_img = img[y:y+h, x:x+w]\n    return cropped_img, h, w\n\ndef closest_value(x, *args):\n    closest = min(args, key=lambda num: abs(x - num))\n    return closest\n\ndef resize_image(cropped_img, target_height, target_width):\n    return cv2.resize(cropped_img, (target_width, target_height))\n\n# CLAHE uygulayan fonksiyon\ndef apply_clahe(image):\n    if len(image.shape) == 3:\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n    clahe = cv2.createCLAHE(clipLimit=5.0, tileGridSize=(8, 8))\n    return clahe.apply(image)\n\n# CSV dosyasını oku\ncsv_file = \"/kaggle/input/rsna-breast-cancer-detection/train.csv\"\ndata = pd.read_csv(csv_file)\n\n# NaN BI-RADS değerlerini atla\ndata = data[data[\"BIRADS\"].notna()]\n\n# BI-RADS değerini tam sayıya çevir\ndata[\"BIRADS\"] = data[\"BIRADS\"].astype(int)\n\n# Görsellerin bulunduğu klasör ve hedef ana klasör\nimage_folder = \"/kaggle/input/rsna-breast-cancer-512-pngs\"\noutput_folder = \"/kaggle/working/split_data\"\n\n# Eğitim, doğrulama ve test oranları\ntrain_ratio = 0.7\nval_ratio = 0.2\ntest_ratio = 0.1\n\n# Klasör yapısını oluştur\nfor split in [\"train\", \"val\", \"test\"]:\n    for birads in [0, 1, 2]:\n        for laterality in ['L', 'R']:  # 'L' ve 'R' laterality değerleri\n            os.makedirs(os.path.join(output_folder, split, f\"birads_{birads}\", f\"laterality_{laterality}\"), exist_ok=True)\n\n# Kanser durumu 1 olanları seç\ncancer_1_data = data[data['cancer'] == 1]\n\n# Kanser durumu 0 olanları rastgele seç\ncancer_0_data = data[data['cancer'] == 0]\nsampled_cancer_0_data = cancer_0_data.groupby(\"BIRADS\").apply(lambda x: x.sample(n=min(len(x), 1158), random_state=42)).reset_index(drop=True)\n\n# Birleştirilmiş veri seti\ncombined_data = pd.concat([cancer_1_data, sampled_cancer_0_data])\n\n# Görselleri sınıflara göre gruplandır\nclass_groups = combined_data.groupby([\"BIRADS\", \"laterality\"])\n\n# Tüm işlemler için genel bir ilerleme çubuğu\ntotal_files = len(combined_data)\nprogress_bar = tqdm(total=total_files, desc=\"Kırpma işlemi\", unit=\"file\", ncols=100, dynamic_ncols=True)\n\n# Her sınıf için verileri rastgele dağıt\nall_heights = []\nall_widths = []\n\n# ThreadPoolExecutor ile paralel işlem yap\nwith ThreadPoolExecutor() as executor:\n    futures = []\n    \n    for (birads, laterality), group in class_groups:\n        file_names = [f\"{row['patient_id']}_{row['image_id']}.png\" for _, row in group.iterrows()]\n        random.shuffle(file_names)\n\n        # Bölme sınırlarını belirle\n        train_end = int(len(file_names) * train_ratio)\n        val_end = train_end + int(len(file_names) * val_ratio)\n\n        # Görselleri böl\n        splits = {\n            \"train\": file_names[:train_end],\n            \"val\": file_names[train_end:val_end],\n            \"test\": file_names[val_end:]\n        }\n\n        # Görselleri ilgili klasörlere taşı\n        for split, split_files in splits.items():\n            for file_name in split_files:\n                source_path = os.path.join(image_folder, file_name)\n                target_dir = os.path.join(output_folder, split, f\"birads_{birads}\", f\"laterality_{laterality}\")\n                target_path = os.path.join(target_dir, file_name)\n\n                # Hedef dizinin varlığını kontrol et, yoksa oluştur\n                os.makedirs(target_dir, exist_ok=True)\n\n                if os.path.exists(source_path):\n                    # Görseli kırp\n                    future = executor.submit(crop_mammogram, source_path)\n                    futures.append((future, target_path))\n\n    # Sonuçları işle ve ilerleme çubuğunu güncelle\n    for future, target_path in futures:\n        cropped_image, h, w = future.result()\n\n        if cropped_image is None:\n            continue\n\n        # İlerleme çubuğunu güncelle\n        progress_bar.update(1)\n\nprogress_bar.close()\n\n# Yeniden boyutlandırma işlemi için ilerleme çubuğu ekleyelim\nresize_progress_bar = tqdm(total=len(futures), desc=\"Yeniden Boyutlandırma\", unit=\"file\", ncols=100, dynamic_ncols=True)\n\n# Görselleri yeniden boyutlandır\nwith ThreadPoolExecutor() as executor:\n    for future, target_path in futures:\n        cropped_image, h, w = future.result()\n\n        clahe_image = apply_clahe(cropped_image)\n        \n        # Görüntüyü yeniden boyutlandır\n        resized_image = resize_image(cropped_image, 512, 512)\n        clahe_resized_image = resize_image(clahe_image, 512, 512)\n\n        # Yeni boyutlardaki görseli kaydet\n        if resized_image is not None:\n            cv2.imwrite(target_path, resized_image, [cv2.IMWRITE_PNG_COMPRESSION, 0])\n            cv2.imwrite(target_path.replace(\".png\", \"_clahe.png\"), clahe_resized_image, [cv2.IMWRITE_PNG_COMPRESSION, 0])\n        \n        # İlerleme çubuğunu güncelle\n        resize_progress_bar.update(1)\n\nresize_progress_bar.close()\nprint(\"Kırpma, işleme ve yeniden boyutlandırma işlemi tamamlandı.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:02.416276Z","iopub.execute_input":"2025-01-09T16:10:02.416939Z","iopub.status.idle":"2025-01-09T16:10:46.392306Z","shell.execute_reply.started":"2025-01-09T16:10:02.416883Z","shell.execute_reply":"2025-01-09T16:10:46.391354Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bu split işlemini sürekli olarak tekrarlamamak için output'umuzu bir \"Kaggle Dataset'ine\" dönüştüreceğiz ve input olarak ekleyeceğiz.","metadata":{}},{"cell_type":"code","source":"shutil.make_archive(\"rsna_birads_512_train\", 'zip','/kaggle/working/birads_train')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:46.393763Z","iopub.execute_input":"2025-01-09T16:10:46.394597Z","iopub.status.idle":"2025-01-09T16:10:46.561525Z","shell.execute_reply.started":"2025-01-09T16:10:46.39456Z","shell.execute_reply":"2025-01-09T16:10:46.560188Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import FileLink\nFileLink(r'rsna_birads_512_train.zip')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:46.562254Z","iopub.status.idle":"2025-01-09T16:10:46.562587Z","shell.execute_reply.started":"2025-01-09T16:10:46.562428Z","shell.execute_reply":"2025-01-09T16:10:46.562444Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Eğitim veri setini oluşturduk, şimdi sınıflandırma için hangi modelin kullanılacağına karar verelim.","metadata":{}},{"cell_type":"markdown","source":"### Nasıl Karar Verilir\nKullanılacak olan veriler 512x512 boyutlarında olduğu için model seçiminde bu boyutlara yakın girdi kabul eden mimarileri kullanmamız daha efektif sonuçlar almamızı sağlayacaktır.\n\nMedikal alanda en yaygın kullanılan ve en yüksek doğruluk oranlarına ulaşan mimariler incelendiğinde EfficientNet ailesinin öne çıktığı görülmüştür.\n\nBu nedenle çalışmamızda EfficientNetB5 mimarisinin kullanılmasına karar verilmiştir.","metadata":{}},{"cell_type":"code","source":"import numpy as np  # lineer cebir için gerekli olan kütüphane\nimport pandas as pd #  verileri tablolama ve .csv formatlı dosyaları kullanılabilir hale getirmek için kullanacağımız kütüphane\nimport matplotlib.pyplot as plt  #  veri görselleştirme, grafik ve tablo oluşturmak için kullanacağımız kütüphane\nimport plotly\nimport tensorflow   \n\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator  #  ImageDataGenerator, Data Augmentation yaparak eğitim verilerini çeşitli işlemler uygulayarak\n                                                                     #  tespiti istenen nesnenin her durumunu eğitim verilerinin içine dahil eder\nfrom tensorflow.keras.applications import EfficientNetB5   #  kullanacağımız yapay sinir ağı katmanı modelidir\nfrom tensorflow.keras.layers import Flatten, Dense, Dropout   #   yapay sinir ağında kullanılması gereken katmanları ve içerdikleri fonksiyonlar\nfrom tensorflow.keras.models import Sequential   #  oluşturacağımız modele katmanlar eklememizi sağlayan bir keras fonksiyonudur, model = Sequential() şeklinde çağırılır\nfrom tensorflow.keras.optimizers import RMSprop, SGD, Adam, Nadam  #   bunlar optimizerlarımız\nfrom tensorflow.keras.regularizers import l1, l2, L1L2   # bunlar \nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau  # EarlyStopping, model aynı doğruluk oranını ardarda almaya başlayınca eğitimi sonlandırır\nimport scipy\n\ntensorflow.random.set_seed(0) # tensorflow kodu tekrar kullanılabilir olsun diye rastgele tohumlar atar\nnp.random.seed(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:51.263848Z","iopub.execute_input":"2025-01-09T16:10:51.264471Z","iopub.status.idle":"2025-01-09T16:10:57.786966Z","shell.execute_reply.started":"2025-01-09T16:10:51.264439Z","shell.execute_reply":"2025-01-09T16:10:57.786147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path = \"/kaggle/working/split_data\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:57.788833Z","iopub.execute_input":"2025-01-09T16:10:57.789707Z","iopub.status.idle":"2025-01-09T16:10:57.797027Z","shell.execute_reply.started":"2025-01-09T16:10:57.789672Z","shell.execute_reply":"2025-01-09T16:10:57.795265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale = 1/255,\n    rotation_range = 20,\n    width_shift_range = 0.2,\n    height_shift_range = 0.2,\n    horizontal_flip = True,\n    vertical_flip = True,\n    fill_mode = 'nearest')\n\nvalidation_datagen = ImageDataGenerator(rescale = 1./255)\ntest_datagen = ImageDataGenerator(rescale = 1./255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:10:57.799488Z","iopub.execute_input":"2025-01-09T16:10:57.800652Z","iopub.status.idle":"2025-01-09T16:10:57.815061Z","shell.execute_reply.started":"2025-01-09T16:10:57.800608Z","shell.execute_reply":"2025-01-09T16:10:57.814127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_shape = (512, 512, 3) # VGG16 Modelinin istediği shape değerleri\n\ntrain_batch_size = 16 # train için verileri 64'lı gruplar olarak alıyoruz\nvalidation_batch_size = 8 # doğrulama için verileri 16'lı gruplar olarak veriyoruz\n\ntrain_generator = train_datagen.flow_from_directory(\n    path + '/train',\n    target_size = (img_shape[0],img_shape[1]),\n    batch_size = train_batch_size,\n    class_mode = 'categorical')\n\nvalidation_generator = validation_datagen.flow_from_directory(\n    path+'/val',\n    target_size = (img_shape[0],img_shape[1]),\n    batch_size = validation_batch_size,\n    class_mode = 'categorical',\n    shuffle = False)\n\ntest_generator = test_datagen.flow_from_directory(\n    path+'/test',\n    target_size = (img_shape[0],img_shape[1]),\n    batch_size = validation_batch_size,\n    class_mode = 'categorical',\n    shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:05.834943Z","iopub.execute_input":"2025-01-09T16:29:05.835688Z","iopub.status.idle":"2025-01-09T16:29:06.263928Z","shell.execute_reply.started":"2025-01-09T16:29:05.835655Z","shell.execute_reply":"2025-01-09T16:29:06.262792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"efb5 = EfficientNetB5(weights = 'imagenet',\n            include_top = False,\n           input_shape = img_shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:06.864666Z","iopub.execute_input":"2025-01-09T16:29:06.865645Z","iopub.status.idle":"2025-01-09T16:29:11.496929Z","shell.execute_reply.started":"2025-01-09T16:29:06.86561Z","shell.execute_reply":"2025-01-09T16:29:11.496109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in efb5.layers[:-3]:   #  vgg16 modeline ait olan son 3 nörondan öncesini öğrenemez\n    layer.trainable = False     #  hale getiriyoruz ve eğitime dahil etmiyoruz","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:11.49847Z","iopub.execute_input":"2025-01-09T16:29:11.498746Z","iopub.status.idle":"2025-01-09T16:29:11.515708Z","shell.execute_reply.started":"2025-01-09T16:29:11.498721Z","shell.execute_reply":"2025-01-09T16:29:11.514857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential() #  modelimizi çağırdık\n\nmodel.add(efb5) # modele VGG16'nın son halini çağırdık\n               # VGG16'nın üzerine kendi katmanlarımızı ekliyoruz\nmodel.add(Flatten()) # \nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(3,activation='softmax'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:11.516743Z","iopub.execute_input":"2025-01-09T16:29:11.517083Z","iopub.status.idle":"2025-01-09T16:29:12.836453Z","shell.execute_reply.started":"2025-01-09T16:29:11.517057Z","shell.execute_reply":"2025-01-09T16:29:12.835437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()  # modele ait olan katmanları ve bu katmanlar hakkındaki bilgileri veriyor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:12.838809Z","iopub.execute_input":"2025-01-09T16:29:12.839645Z","iopub.status.idle":"2025-01-09T16:29:12.862231Z","shell.execute_reply.started":"2025-01-09T16:29:12.839605Z","shell.execute_reply":"2025-01-09T16:29:12.861236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(loss='categorical_crossentropy',\n             optimizer=Adam(learning_rate=1e-4),\n             metrics=['acc'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:12.863405Z","iopub.execute_input":"2025-01-09T16:29:12.863755Z","iopub.status.idle":"2025-01-09T16:29:12.880923Z","shell.execute_reply.started":"2025-01-09T16:29:12.863722Z","shell.execute_reply":"2025-01-09T16:29:12.880174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\ntf.config.optimizer.set_jit(True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T16:29:12.881787Z","iopub.execute_input":"2025-01-09T16:29:12.882004Z","iopub.status.idle":"2025-01-09T16:29:12.888378Z","shell.execute_reply.started":"2025-01-09T16:29:12.881975Z","shell.execute_reply":"2025-01-09T16:29:12.88763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"es = EarlyStopping(monitor='val_acc',mode='min', verbose=1, patience=10)\nmc = ModelCheckpoint('/kaggle/working/modelcheckpoint.keras', monitor='val_acc',mode='max',verbose=1, save_best_only=True)\nrlo = ReduceLROnPlateau(monitor='val_loss', factor=0.2,\n                            patience=5, min_lr=0.001)\n\nhistory = model.fit(\n    train_generator,\n    steps_per_epoch = train_generator.samples//train_generator.batch_size,\n    epochs=100,\n    validation_data = validation_generator,\n    validation_steps = validation_generator.samples//validation_generator.batch_size,\n    verbose=1,\n    callbacks = [es,mc,rlo])\n\nmodel.save('/kaggle/working/effb5.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T18:54:09.146174Z","iopub.execute_input":"2025-01-09T18:54:09.14663Z","iopub.status.idle":"2025-01-09T19:57:57.195724Z","shell.execute_reply.started":"2025-01-09T18:54:09.146601Z","shell.execute_reply":"2025-01-09T19:57:57.194066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('/kaggle/working/effb5.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T19:57:57.199088Z","iopub.execute_input":"2025-01-09T19:57:57.199926Z","iopub.status.idle":"2025-01-09T19:58:03.136671Z","shell.execute_reply.started":"2025-01-09T19:57:57.199877Z","shell.execute_reply":"2025-01-09T19:58:03.13562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.evaluate(validation_generator, batch_size=32)\nscore\nprint('Score Accuracy : {:.2f}%'.format(score[1]*100))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T19:58:03.138837Z","iopub.execute_input":"2025-01-09T19:58:03.139313Z","iopub.status.idle":"2025-01-09T19:58:39.301263Z","shell.execute_reply.started":"2025-01-09T19:58:03.139266Z","shell.execute_reply":"2025-01-09T19:58:39.300441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_acc = history.history['acc']\nval_acc = history.history['val_acc']\ntrain_loss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1,len(train_acc)+1)\n\nplt.plot(epochs, train_acc, 'b*-', label='Training Accuracy')\nplt.plot(epochs, val_acc, 'r', label='Validation Accuracy')\nplt.title('Training and Validation Accuracy')\nplt.legend()\n         \nplt.figure()\n\nplt.plot(epochs, train_loss, 'b*-', label = 'Training Loss') # b*- mavi renkle ilgili bişey\nplt.plot(epochs, val_loss, 'r', label = 'Validation Loss')\nplt.title('Training and Validation Loss')\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T19:58:39.302919Z","iopub.execute_input":"2025-01-09T19:58:39.303198Z","iopub.status.idle":"2025-01-09T19:58:39.749504Z","shell.execute_reply.started":"2025-01-09T19:58:39.303172Z","shell.execute_reply":"2025-01-09T19:58:39.748652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_Y = model.predict(validation_generator, batch_size=32, verbose=True) \n# verbose bize işlemin ilerleyişinin gösterilip gösterilmeyeceğini belirletir\n# verbose=0 ilerleme gösterilmez sadece eğitim bitsin diye beklersin\n# verbose=1 ilerleme 1/10 epoch ============== %58 şeklinde gösterilir\n\npred_Y_cat = np.argmax(pred_Y, -1) #  burada bize pred_Y np.array'indeki elemanları tersten almamızı ve öyle argmax uygulamamızı söylüyor\n                                   #  np.argmax(np.array, axis=0/1)\n                                   #  eğer dizi çok boyutluysa\n                                   #  np.argmax() fonksiyonu diziye Flatten işlemi uygular ve sonra \n                                   #  aralarından en büyüğünü seçer ancak\n                                   #  np.argmax()'ın aldığı ikinci argüman yani axis argümanı diziyi\n                                   #  +x ve -y eksenlerine dizer, x=0 -y ekseni, x=1 +x eksenine denk gelir\n                                   #  ve argmax bize sonucu indis olarak verir. Yani en büyük değer 156 ise\n                                   #  bize 156 yerine 156nın denk geldiği indisi verir\nprint(pred_Y_cat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T19:59:09.124982Z","iopub.execute_input":"2025-01-09T19:59:09.125293Z","iopub.status.idle":"2025-01-09T19:59:32.734389Z","shell.execute_reply.started":"2025-01-09T19:59:09.125264Z","shell.execute_reply":"2025-01-09T19:59:32.733472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import roc_curve, auc, confusion_matrix\nfrom sklearn.preprocessing import label_binarize\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\nnum_classes = 3\n\n# Test verisi ile tahmin yapma\ny_pred_prob = model.predict(test_generator)\ny_pred = np.argmax(y_pred_prob, axis=1)  # Sınıf etiketlerini al\n\n# Gerçek etiketleri almak için test_generator'dan veri çekin\ny_true = test_generator.classes\ny_true_bin = label_binarize(y_true, classes=[0, 1, 2])\n\n# AUC-ROC hesaplama\nfpr = {}\ntpr = {}\nroc_auc = {}\n\nfor i in range(num_classes):\n    fpr[i], tpr[i], _ = roc_curve(y_true_bin[:, i], y_pred_prob[:, i])\n    roc_auc[i] = auc(fpr[i], tpr[i])\n\n# AUC-ROC eğrilerini çizme\nplt.figure()\nfor i in range(num_classes):\n    plt.plot(fpr[i], tpr[i], label='Sınıf {} (AUC = {:.2f})'.format(i, roc_auc[i]))\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('Yanlış Pozitif Oranı')\nplt.ylabel('Doğru Pozitif Oranı')\nplt.title('ROC Eğrisi')\nplt.legend(loc='lower right')\nplt.show()\n\n# Karışıklık matrisini hesapla\ncm = confusion_matrix(y_true, y_pred)\n\n# Karışıklık matrisini görselleştir\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', cbar=False,\n            xticklabels=['Sınıf 0', 'Sınıf 1', 'Sınıf 2'],\n            yticklabels=['Sınıf 0', 'Sınıf 1', 'Sınıf 2'])\nplt.xlabel('Tahmin Edilen Sınıf')\nplt.ylabel('Gerçek Sınıf')\nplt.title('Karışıklık Matrisi')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-09T19:59:32.735634Z","iopub.execute_input":"2025-01-09T19:59:32.736005Z","iopub.status.idle":"2025-01-09T19:59:56.100449Z","shell.execute_reply.started":"2025-01-09T19:59:32.73597Z","shell.execute_reply":"2025-01-09T19:59:56.09929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}