{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-01T19:57:38.18224Z","iopub.execute_input":"2023-12-01T19:57:38.182798Z","iopub.status.idle":"2023-12-01T19:57:50.948714Z","shell.execute_reply.started":"2023-12-01T19:57:38.182756Z","shell.execute_reply":"2023-12-01T19:57:50.947461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Spécification du chemin vers le dossier contenant les fichiers DICOM\ndata_path = '/kaggle/input/rsna-breast-cancer-detection'\n\n# Chargement des  métadonnées\nmetadata = pd.read_csv(os.path.join(data_path, '/kaggle/input/rsna-breast-cancer-detection/train.csv'))\n\n# Affichage  des premières lignes du DataFrame\nmetadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:59:04.938179Z","iopub.execute_input":"2023-12-01T19:59:04.939795Z","iopub.status.idle":"2023-12-01T19:59:05.076113Z","shell.execute_reply.started":"2023-12-01T19:59:04.939733Z","shell.execute_reply":"2023-12-01T19:59:05.074221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_statistics = metadata.groupby('cancer').describe()\nprint(class_statistics)","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:01:12.108265Z","iopub.execute_input":"2023-12-01T19:01:12.108665Z","iopub.status.idle":"2023-12-01T19:01:12.239689Z","shell.execute_reply.started":"2023-12-01T19:01:12.108626Z","shell.execute_reply":"2023-12-01T19:01:12.238097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Distribution de l'âge en fonction de la classe\nsns.boxplot(x='cancer', y='age', data=metadata)\nplt.title('Distribution de l\\'âge en fonction de la classe')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:01:12.242838Z","iopub.execute_input":"2023-12-01T19:01:12.243251Z","iopub.status.idle":"2023-12-01T19:01:13.515957Z","shell.execute_reply.started":"2023-12-01T19:01:12.243214Z","shell.execute_reply":"2023-12-01T19:01:13.514615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calcul des statistiques descriptives pour l'âge en fonction de la classe\nage_statistics = metadata.groupby('cancer')['age'].describe()\nprint(age_statistics)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:01:13.517371Z","iopub.execute_input":"2023-12-01T19:01:13.517763Z","iopub.status.idle":"2023-12-01T19:01:13.545763Z","shell.execute_reply.started":"2023-12-01T19:01:13.517729Z","shell.execute_reply":"2023-12-01T19:01:13.544162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Identification des valeurs manquantes dans le DataFrame\nmissing_values = metadata.isnull().sum()\n\n# Affichage des colonnes avec des valeurs manquantes et le nombre de valeurs manquantes\nprint(\"Valeurs manquantes par colonne :\")\nprint(missing_values[missing_values > 0])\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:01:13.547227Z","iopub.execute_input":"2023-12-01T19:01:13.547572Z","iopub.status.idle":"2023-12-01T19:01:13.574958Z","shell.execute_reply.started":"2023-12-01T19:01:13.54754Z","shell.execute_reply":"2023-12-01T19:01:13.574118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> **normalisation des valeurs des pixels entre 0 et 1, et affichage des images originales et normalisées côte à côte.**","metadata":{}},{"cell_type":"code","source":"!pip install pydicom[gdcm] pylibjpeg\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:04:05.458168Z","iopub.execute_input":"2023-12-01T19:04:05.459752Z","iopub.status.idle":"2023-12-01T19:04:21.368315Z","shell.execute_reply.started":"2023-12-01T19:04:05.459655Z","shell.execute_reply":"2023-12-01T19:04:21.366838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pylibjpeg pylibjpeg-libjpeg","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:05:24.985547Z","iopub.execute_input":"2023-12-01T19:05:24.985984Z","iopub.status.idle":"2023-12-01T19:05:39.970008Z","shell.execute_reply.started":"2023-12-01T19:05:24.985946Z","shell.execute_reply":"2023-12-01T19:05:39.965601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pydicom\nimport os\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef normalize_images_for_patient(patient_id, directory_path):\n    patient_dir = os.path.join(directory_path, str(patient_id))\n\n    for filename in os.listdir(patient_dir):\n        file_path = os.path.join(patient_dir, filename)\n        dicom_data = pydicom.dcmread(file_path)\n        pixel_array = dicom_data.pixel_array\n\n        # Normaliser les valeurs des pixels entre 0 et 1\n        pixel_min = pixel_array.min()\n        pixel_max = pixel_array.max()\n        pixel_normalized = (pixel_array - pixel_min) / (pixel_max - pixel_min)\n\n        # Afficher l'image originale et normalisée pour illustration\n        plt.figure(figsize=(8, 4))\n        plt.subplot(1, 2, 1)\n        plt.imshow(pixel_array, cmap='gray')\n        plt.title('Image Originale')\n\n        plt.subplot(1, 2, 2)\n        plt.imshow(pixel_normalized, cmap='gray')\n        plt.title('Image Normalisée')\n\n        plt.show()\n\n# Spécifiez le chemin vers le répertoire contenant les fichiers DICOM (train_images)\ndirectory_path = '/kaggle/input/rsna-breast-cancer-detection/train_images'\n\n# Liste des patients\npatients = ['10006', '10011', '10025', '10038', '10042', '10048', '10049', '10050', '10051', '10086',\n            '10095', '10097', '10102', '10106', '10116', '10119', '10122', '10124', '10126', '10130',\n            '10132', '10136', '1014', '10144', '1015', '10151', '10152', '10153', '10175', '10179']\n\n# Normaliser les images pour chaque patient\nfor patient_id in patients:\n    normalize_images_for_patient(patient_id, directory_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T19:11:10.298287Z","iopub.execute_input":"2023-12-01T19:11:10.299235Z","iopub.status.idle":"2023-12-01T19:21:01.896697Z","shell.execute_reply.started":"2023-12-01T19:11:10.299165Z","shell.execute_reply":"2023-12-01T19:21:01.894736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Afficher les premières lignes du DataFrame après l'encodage one-hot\nprint(metadata_encoded.head())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T20:07:01.056604Z","iopub.execute_input":"2023-12-01T20:07:01.057167Z","iopub.status.idle":"2023-12-01T20:07:01.075522Z","shell.execute_reply.started":"2023-12-01T20:07:01.057125Z","shell.execute_reply":"2023-12-01T20:07:01.073727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Afficher les valeurs uniques dans les colonnes 'laterality' et 'view'\nprint(\"Unique values in 'laterality':\", metadata['laterality'].unique())\nprint(\"Unique values in 'view':\", metadata['view'].unique())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-01T20:04:56.056656Z","iopub.execute_input":"2023-12-01T20:04:56.057157Z","iopub.status.idle":"2023-12-01T20:04:56.074549Z","shell.execute_reply.started":"2023-12-01T20:04:56.057122Z","shell.execute_reply":"2023-12-01T20:04:56.073341Z"},"trusted":true},"execution_count":null,"outputs":[]}]}