{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"},{"sourceId":4619805,"sourceType":"datasetVersion","datasetId":2688675}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#importing the libraries\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pandas as pd\nimport seaborn as sn","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:35.945798Z","iopub.execute_input":"2023-12-14T22:25:35.946714Z","iopub.status.idle":"2023-12-14T22:25:36.77613Z","shell.execute_reply.started":"2023-12-14T22:25:35.946679Z","shell.execute_reply":"2023-12-14T22:25:36.775167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing our cancer dataset\ndf = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ndf.columns\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:36.778008Z","iopub.execute_input":"2023-12-14T22:25:36.778871Z","iopub.status.idle":"2023-12-14T22:25:36.919492Z","shell.execute_reply.started":"2023-12-14T22:25:36.778833Z","shell.execute_reply":"2023-12-14T22:25:36.918751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing our cancer dataset\ndf2 = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ndf2.columns","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:36.920466Z","iopub.execute_input":"2023-12-14T22:25:36.920711Z","iopub.status.idle":"2023-12-14T22:25:36.931216Z","shell.execute_reply.started":"2023-12-14T22:25:36.920688Z","shell.execute_reply":"2023-12-14T22:25:36.930329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Informations from the Dataset Description:**\n- site_id - ID code for the source hospital.\n- patient_id - ID code for the patient.\n- image_id - ID code for the image.\n- laterality - Whether the image is of the left or right breast.\n- view - The orientation of the image. The default for a screening exam is to capture two views per breast.\n- age - The patient's age in years.\n- implant - Whether or not the patient had breast implants. Site 1 only provides breast implant information at the patient level, not at the breast level.\n- density - A rating for how dense the breast tissue is, with A being the least dense and D being the most dense. Extremely dense tissue can make diagnosis more difficult. (Only provided for train).\n- machine_id - An ID code for the imaging device.\n- cancer - Whether or not the breast was positive for malignant cancer. The target value. (Only provided for train).\n- biopsy - Whether or not a follow-up biopsy was performed on the breast. (Only provided for train).\n- invasive - If the breast is positive for cancer, whether or not the cancer proved to be invasive. (Only provided for train).\n- BIRADS - 0 if the breast required follow-up, 1 if the breast was rated as negative for cancer, and 2 if the breast was rated as normal. Only provided for train.\n- prediction_id - The ID for the matching submission row. Multiple images will share the same prediction ID. (Test only).\n- difficult_negative_case - True if the case was unusually difficult. (Only provided for train)","metadata":{}},{"cell_type":"code","source":"#find the dimensions of the data set using the panda dataset ‘shape’ attribute.\nprint(\"Cancer data set dimensions : {}\".format(df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:36.933209Z","iopub.execute_input":"2023-12-14T22:25:36.933785Z","iopub.status.idle":"2023-12-14T22:25:36.938344Z","shell.execute_reply.started":"2023-12-14T22:25:36.93375Z","shell.execute_reply":"2023-12-14T22:25:36.937563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:37.604603Z","iopub.execute_input":"2023-12-14T22:25:37.60507Z","iopub.status.idle":"2023-12-14T22:25:37.634976Z","shell.execute_reply.started":"2023-12-14T22:25:37.605037Z","shell.execute_reply":"2023-12-14T22:25:37.634118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sample images:**","metadata":{}},{"cell_type":"code","source":"import os\nimport pydicom\n\ndef load_dicom_images(directory):\n    dicom_images = []\n    for root, dirs, files in os.walk(directory):\n        for file in files:\n            dicom_images.append(pydicom.dcmread(os.path.join(root, file)))\n    return dicom_images\n\n# Step 2: Visualize DICOM Images in Black and White\ndef visualize_dicom_images(images, num_samples):\n    sample_images = images[:num_samples]\n    for i, image in enumerate(sample_images):\n        plt.subplot(1, num_samples, i + 1)\n        plt.imshow(image.pixel_array, cmap='gray')  # Utilisez 'gray' pour afficher en noir et blanc\n        plt.title(f\"Image {i + 1}\")\n        plt.axis('off')\n    plt.show()\n\n# Changez le chemin du dossier en fonction de votre situation\ntrain_dcm_folder = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10006\"\ndcm_img = load_dicom_images(train_dcm_folder)\nvisualize_dicom_images(dcm_img, 4)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:38.254575Z","iopub.execute_input":"2023-12-14T22:25:38.254925Z","iopub.status.idle":"2023-12-14T22:25:50.012087Z","shell.execute_reply.started":"2023-12-14T22:25:38.254895Z","shell.execute_reply":"2023-12-14T22:25:50.011056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = df[['site_id', 'patient_id', 'image_id', 'laterality', 'view', 'age',\n       'cancer', 'biopsy', 'invasive', 'BIRADS', 'implant', 'density',\n       'machine_id', 'difficult_negative_case']]\ndf1[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.013958Z","iopub.execute_input":"2023-12-14T22:25:50.014329Z","iopub.status.idle":"2023-12-14T22:25:50.039795Z","shell.execute_reply.started":"2023-12-14T22:25:50.014278Z","shell.execute_reply":"2023-12-14T22:25:50.038854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note :The value NaN (Not a Number) is not equivalent to the null value (0). NaN is used to represent the absence of data or an undefined value in a numeric array","metadata":{}},{"cell_type":"code","source":"#Missing or Null Data points\ndf1.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.041136Z","iopub.execute_input":"2023-12-14T22:25:50.041483Z","iopub.status.idle":"2023-12-14T22:25:50.062987Z","shell.execute_reply.started":"2023-12-14T22:25:50.041458Z","shell.execute_reply":"2023-12-14T22:25:50.061985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation Matrix:**\n\nCorrelation is a statistical measure of the relationship between two variables. The correlation coefficient, ranging from -1 to 1, indicates the strength and direction of this connection.","metadata":{}},{"cell_type":"code","source":"# Select only numeric columns from the DataFrame:\nnumeric_columns = df1.select_dtypes(include=['int64', 'float64'])\n\n# Create the correlation matrix\ncorrelation_matrix = numeric_columns.corr()\n\n# Plot the correlation matrix using seaborn\nplt.figure(figsize=(20, 15))\nsn.heatmap(correlation_matrix, annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.066021Z","iopub.execute_input":"2023-12-14T22:25:50.066387Z","iopub.status.idle":"2023-12-14T22:25:50.75969Z","shell.execute_reply.started":"2023-12-14T22:25:50.066354Z","shell.execute_reply":"2023-12-14T22:25:50.758776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The correlation of cancer with another variable:**","metadata":{}},{"cell_type":"code","source":"# Select only numeric columns from the DataFrame\nnumeric_columns = df1.select_dtypes(include=['int64', 'float64'])\n\n# Create the correlation matrix\ncorr_matrix = numeric_columns.corr()\n\n# Sort the correlations between the \"cancer\" column and other numeric columns\ncancer_correlations = corr_matrix['cancer'].sort_values(ascending=False)\n\nprint(cancer_correlations)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.760879Z","iopub.execute_input":"2023-12-14T22:25:50.761199Z","iopub.status.idle":"2023-12-14T22:25:50.785758Z","shell.execute_reply.started":"2023-12-14T22:25:50.761169Z","shell.execute_reply":"2023-12-14T22:25:50.784877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : The matrix demonstrates a positive relationship between cancerand : invasive,biopsy and age","metadata":{}},{"cell_type":"markdown","source":"**Number of patients / the youngest and the oldest patient / negative and positive cases :**","metadata":{}},{"cell_type":"code","source":"num_patients = df1['patient_id'].nunique()\nmin_patient_age = df1['age'].min()\nmax_patient_age = df1['age'].max()\n\ngrouped = df1.groupby('patient_id')['cancer'].max()\nn_negative = grouped.value_counts().get(0, 0)\nn_positive = grouped.value_counts().get(1, 0)\n\nprint(f\"There are {num_patients} different patients in the train set.\\n\")\nprint(f\"The youngest patient is {int(min_patient_age)} years old.\")\nprint(f\"The oldest patient is {int(max_patient_age)} years old.\\n\")\nprint(f\"{n_negative} patients ({n_negative / num_patients:.1%}) are negative to breast cancer.\")\nprint(f\"{n_positive} patients ({n_positive / num_patients:.1%}) are positive to breast cancer.\")","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.786899Z","iopub.execute_input":"2023-12-14T22:25:50.78717Z","iopub.status.idle":"2023-12-14T22:25:50.812644Z","shell.execute_reply.started":"2023-12-14T22:25:50.787146Z","shell.execute_reply":"2023-12-14T22:25:50.811782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Figure represents the number of patients with/without cancer:**","metadata":{}},{"cell_type":"code","source":"cancer_per_patient = df1.groupby(\"patient_id\")[\"cancer\"].max().values\nn_negative = (cancer_per_patient == 0).sum()\nn_positive = (cancer_per_patient == 1).sum()\n\nfig, ax = plt.subplots()\nbars = ax.bar([\"No cancer\", \"Cancer\"], [n_negative, n_positive], color='mediumvioletred')\nax.set(xlabel=\"\", ylabel=\"Count\", title=\"Number of patients with/without cancer\")\n\n# Adding labels on top of the bars\nfor bar, count in zip(bars, [n_negative, n_positive]):\n    height = bar.get_height()\n    ax.text(bar.get_x() + bar.get_width() / 2, height, count, ha=\"center\", va=\"bottom\")\n\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:50.813786Z","iopub.execute_input":"2023-12-14T22:25:50.814387Z","iopub.status.idle":"2023-12-14T22:25:51.036809Z","shell.execute_reply.started":"2023-12-14T22:25:50.814352Z","shell.execute_reply":"2023-12-14T22:25:51.036004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Age distribution:**\n","metadata":{}},{"cell_type":"code","source":"ages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\ncancer_ages = df1[df1['cancer'] == 1].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\nno_cancer_ages = df1[df1['cancer'] == 0].groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nplt.figure(figsize=(14, 7))\n\nplt.subplot(1, 2, 2)\nsn.histplot(cancer_ages, bins=51, color='mediumvioletred', kde=True)  \nsn.histplot(no_cancer_ages, bins=63, color='indigo', kde=True)  \nplt.title(\"Patients with/without cancer\", fontsize=14)\nplt.xlabel(\"Age\", fontsize=12)\nplt.ylabel(\"Count\", fontsize=12)\nplt.xticks(fontsize=10)\nplt.yticks(fontsize=10)\nplt.xlim(33, 89)\nplt.legend([\"Cancer\", \"No cancer\"], fontsize=10)\n\nplt.suptitle(\"Age distribution of the patients\", fontsize=16)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:51.03813Z","iopub.execute_input":"2023-12-14T22:25:51.038415Z","iopub.status.idle":"2023-12-14T22:25:53.261049Z","shell.execute_reply.started":"2023-12-14T22:25:51.03839Z","shell.execute_reply":"2023-12-14T22:25:53.260318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The average age:**","metadata":{}},{"cell_type":"code","source":"# Statistics\nimport pandas as pd\n\nages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nstats = pd.DataFrame({\n    #moyenne\n    \"Mean\": [ages.mean()],\n    #écart type\n    \"Std\": [ages.std()],\n   \n})\n\nformatted_stats = stats.style.format(\"{:.2f}\")\nformatted_stats.set_caption(\"Statistics for Age\")\nformatted_stats.set_properties(**{'text-align': 'center'})\n\ndisplay(formatted_stats)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:53.262543Z","iopub.execute_input":"2023-12-14T22:25:53.26291Z","iopub.status.idle":"2023-12-14T22:25:54.168142Z","shell.execute_reply.started":"2023-12-14T22:25:53.262876Z","shell.execute_reply":"2023-12-14T22:25:54.167204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**For cancer age :**","metadata":{}},{"cell_type":"code","source":"# Statistics for Cancer Age\nimport pandas as pd\n\nages = df1.groupby('patient_id')['age'].apply(lambda x: x.unique()[0])\n\nstats = pd.DataFrame({\n    \"Mean\": [cancer_ages.mean()],\n    \"Std\": [cancer_ages.std()],\n    \"Minimum Cancer Age\": [cancer_ages.min()]\n})\n\nformatted_stats = stats.style.format(\"{:.2f}\")\nformatted_stats.set_caption(\"Statistics for Cancer Age\")\nformatted_stats.set_properties(**{'text-align': 'center'})\n\ndisplay(formatted_stats)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:54.172086Z","iopub.execute_input":"2023-12-14T22:25:54.172461Z","iopub.status.idle":"2023-12-14T22:25:55.006704Z","shell.execute_reply.started":"2023-12-14T22:25:54.172434Z","shell.execute_reply":"2023-12-14T22:25:55.005817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Cancer Positive Age Distribution:**","metadata":{}},{"cell_type":"code","source":"import seaborn as sn\nmy_colors = [\"#8a4dbf\", \"#9f76e6\", \"#9569e6\", \"#aa92e6\", \"#bba1e6\", \"#a991e6\"]\n# Create a figure with two subplots arranged vertically\nf, (a0, a1) = plt.subplots(2, 1, gridspec_kw={'height_ratios': [3, 1]}, figsize=(14, 10))\n# Plot a histogram with a kernel density estimate (KDE) for the 'cancer_ages' data, using the color from 'my_colors'\nsn.histplot(data=cancer_ages, kde=True, color=my_colors[5], ax=a0)\n\n# Add vertical lines and text annotations for mean, min, and max values on the histogram\na0.axvline(x=cancer_ages.mean(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.mean()+1, y=0.018, s=f\"Mean: {cancer_ages.mean():.2f}\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=cancer_ages.min(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.min()+1, y=0.008, s=f\"Min: {cancer_ages.min()}\", size=17, color=\"black\", weight=\"bold\")\na0.axvline(x=cancer_ages.max(), ls=\":\", lw=2, color=\"black\")\na0.text(x=cancer_ages.max()-7, y=0.037, s=f\"Max: {cancer_ages.max()}\", size=17, color=\"black\", weight=\"bold\")\n\n# Plot a boxen plot for the 'cancer_ages' data on the second subplot\nsn.boxenplot(x=cancer_ages, ax=a1, color=my_colors[2])\n# Set labels for the x-axis and y-axis on the second subplot\na1.set(xlabel=\"Age\", ylabel=\"\")\n# Set tick label font size on the second subplot\na1.tick_params(labelsize=12)\n\n# Set the overall title for the entire figure\nplt.suptitle(\"Cancer Positive Age Distribution\", weight=\"bold\", size=20)\n# Set labels for the x-axis and y-axis on the first subplot\na0.set(xlabel=\"Age\", ylabel=\"Density\")\n# Set tick label font size on the first subplot\na0.tick_params(labelsize=12)\n# Configure spines (axes borders) for the first subplot\na0.spines[\"top\"].set_visible(False)\na0.spines[\"right\"].set_visible(False)\na0.spines[\"left\"].set_linewidth(2)\na0.spines[\"bottom\"].set_linewidth(2)\n\n# Adjust layout to prevent clipping of titles and labels\nplt.tight_layout()\n# Display the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:55.007945Z","iopub.execute_input":"2023-12-14T22:25:55.008289Z","iopub.status.idle":"2023-12-14T22:25:55.467118Z","shell.execute_reply.started":"2023-12-14T22:25:55.008256Z","shell.execute_reply":"2023-12-14T22:25:55.466226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Number of Images Taken per Patient:**","metadata":{}},{"cell_type":"code","source":"num_images_per_patient = df1['patient_id'].value_counts()\n\nplt.figure(figsize=(14, 8))\nsn.countplot(x=num_images_per_patient, palette=['mediumvioletred'], saturation=0.7)\n\nplt.title(\"Number of Images Taken per Patient\", fontsize=20, fontweight=\"bold\")\nplt.xlabel('Number of Images Taken', fontsize=14, fontweight=\"bold\")\nplt.ylabel('Count of Patients', fontsize=14, fontweight=\"bold\")\n\nplt.xticks(rotation=90)\nplt.tick_params(labelsize=12)\nplt.tight_layout()\nsn.despine(trim=True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:55.468344Z","iopub.execute_input":"2023-12-14T22:25:55.468641Z","iopub.status.idle":"2023-12-14T22:25:55.785323Z","shell.execute_reply.started":"2023-12-14T22:25:55.468615Z","shell.execute_reply":"2023-12-14T22:25:55.784428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2, 3, figsize=(18, 12))\nsn.countplot(x=df1['laterality'], palette='Blues_r', ax=ax[0, 0])\nsn.countplot(x=df1['implant'], palette='Greens_r', ax=ax[0, 1])\nsn.countplot(x=df1['difficult_negative_case'], palette='Reds_r', ax=ax[0, 2])\nsn.countplot(x=df1['view'], palette='Oranges_r', ax=ax[1, 0])\nsn.countplot(x=df1['density'], palette='Purples_r', order=['A', 'B', 'C', 'D'], ax=ax[1, 1])\nsn.countplot(x=df1['site_id'], palette='Greys_r', ax=ax[1, 2])\n\nax[0, 0].set_title('Laterality', fontsize=18, fontweight='bold')\nax[0, 1].set_title('Implant', fontsize=18, fontweight='bold')\nax[0, 2].set_title('Difficult Negative Case', fontsize=18, fontweight='bold')\nax[1, 0].set_title('View', fontsize=18, fontweight='bold')\nax[1, 1].set_title('Density', fontsize=18, fontweight='bold')\nax[1, 2].set_title('Site ID', fontsize=18, fontweight='bold')\n\nfor i in range(2):\n    for j in range(3):\n        ax[i, j].set_xlabel('')\n        ax[i, j].set_ylabel('')\n        ax[i, j].tick_params(axis='both', which='major', labelsize=14)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:55.786565Z","iopub.execute_input":"2023-12-14T22:25:55.786846Z","iopub.status.idle":"2023-12-14T22:25:57.024422Z","shell.execute_reply.started":"2023-12-14T22:25:55.78682Z","shell.execute_reply":"2023-12-14T22:25:57.023495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**--->Observations:**\n1. In terms of laterality, the photos are balanced, suggesting that the same amount of images were captured for both the left and right breast.\n2. Implants are uncommon, with just a few photos indicating their presence.\n3. Several photos were difficult to diagnose, which may indicate the existence of breast tissue that is more complicated or unusual.\n4. The bulk of photos consist of two different types of views: CC (craniocaudal) and MLO (mediolateral oblique).\n5. The density of breast tissue in the photos is often in the medium range (B and C), with a minority of photographs showing breast tissue density that is either extremely dense (D) or extremely low (A).\n6. The photos were obtained in a balanced manner at two distinct locations, which may indicate that the data came from two distinct medical institutions or imaging centers.","metadata":{}},{"cell_type":"code","source":"## unstack(): function reshapes the result into a pivot table with 'cancer' as the index and 'biopsy' as columns.\nbiopsy_counts = df1.groupby(['cancer', 'biopsy']).size().unstack(fill_value=0)\nbiopsy_perc = biopsy_counts.apply(lambda x: x / x.sum(), axis=1)\n\n# create fig\nfig, ax = plt.subplots(1, 2, figsize=(12, 6))\n\n# fig1 (countplot)\nsn.countplot(x='biopsy', hue='cancer', data=df1, palette=['mediumvioletred', 'green'], ax=ax[0])\nax[0].bar_label(ax[0].containers[0], label_type='edge')\nax[0].bar_label(ax[0].containers[1], label_type='edge')\n\n# fig2 (heatmap) \nsn.heatmap(biopsy_perc, square=True, annot=True, fmt='.1%', cmap='Purples', ax=ax[1], cbar=False)\nax[1].set_yticklabels(ax[1].get_yticklabels(), rotation=0)\n\nplt.subplots_adjust(wspace=0.3)\nax[0].set_title(\"Number of images resulting in a biopsy\")\nax[1].set_title(\"Percentage of images resulting in a biopsy\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:57.025705Z","iopub.execute_input":"2023-12-14T22:25:57.026017Z","iopub.status.idle":"2023-12-14T22:25:57.364806Z","shell.execute_reply.started":"2023-12-14T22:25:57.025989Z","shell.execute_reply":"2023-12-14T22:25:57.363888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note: A biopsy is a medical procedure in which a sample of tissue or cells is taken from a person's body for further analysis","metadata":{}},{"cell_type":"markdown","source":"**Image Analysis:**","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 3, figsize=(16, 4))\n\nsn.countplot(x='invasive', data=df1[df1['cancer'] == True], ax=ax[0], color='mediumvioletred')\nsn.countplot(x='BIRADS', data=df1[df1['cancer'] == False], order=[0, 1, 2], ax=ax[1], color='green')\nsn.countplot(x='BIRADS', data=df1[df1['cancer'] == True], order=[0, 1, 2], ax=ax[2], color='purple')\n\nax[0].set_title(\"Count of Invasive Cancer Images\")\nax[0].set_xlabel(\"Invasive\")\nax[0].set_ylabel(\"Count\")\n\nax[1].set_title(\"BIRADS for Healthy Images\")\nax[1].set_xlabel(\"BIRADS\")\nax[1].set_ylabel(\"Count\")\n\nax[2].set_title(\"BIRADS for Cancer Images\")\nax[2].set_xlabel(\"BIRADS\")\nax[2].set_ylabel(\"Count\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:57.36626Z","iopub.execute_input":"2023-12-14T22:25:57.366649Z","iopub.status.idle":"2023-12-14T22:25:57.896712Z","shell.execute_reply.started":"2023-12-14T22:25:57.366612Z","shell.execute_reply":"2023-12-14T22:25:57.895842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : BIRADS value signification:\n\n0: required follow-up\n\n1: rated as negative for cancer\n\n2: rated as normal","metadata":{}},{"cell_type":"markdown","source":"Note2 : \"invasive\" means something that aggressively spreads .","metadata":{}},{"cell_type":"markdown","source":"**Machine id:**\n","metadata":{}},{"cell_type":"code","source":"count_machine=df1.groupby(by=\"machine_id\").count()[\"patient_id\"]\ncount_machine.reset_index().head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:57.897961Z","iopub.execute_input":"2023-12-14T22:25:57.89832Z","iopub.status.idle":"2023-12-14T22:25:57.926386Z","shell.execute_reply.started":"2023-12-14T22:25:57.898269Z","shell.execute_reply":"2023-12-14T22:25:57.925615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nplt.title(\"Number of images per machine_id\")\nsn.barplot(data=count_machine.reset_index(), x=\"machine_id\", y=\"patient_id\", color='mediumvioletred')\nplt.ylabel(\"Number of images\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:57.927645Z","iopub.execute_input":"2023-12-14T22:25:57.927924Z","iopub.status.idle":"2023-12-14T22:25:58.267018Z","shell.execute_reply.started":"2023-12-14T22:25:57.9279Z","shell.execute_reply":"2023-12-14T22:25:58.266097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : \"not-malignant cancer\" cases should be selected from \"patients with biopsy but without malignant cancer.\"\n\nWe took this choice because it seems more logical, but another choice might be correct.","metadata":{}},{"cell_type":"code","source":"# The not-malignant cancer cases were limited into biopsy cases.\nDF_train = df[df['biopsy'] == 1].reset_index(drop = True)\nDF_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:58.268338Z","iopub.execute_input":"2023-12-14T22:25:58.268706Z","iopub.status.idle":"2023-12-14T22:25:58.289351Z","shell.execute_reply.started":"2023-12-14T22:25:58.268669Z","shell.execute_reply":"2023-12-14T22:25:58.28837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Creating a balanced dataset :**","metadata":{}},{"cell_type":"markdown","source":"**Training data:**","metadata":{}},{"cell_type":"markdown","source":"**Test:**","metadata":{}},{"cell_type":"markdown","source":"#model","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:58.290699Z","iopub.execute_input":"2023-12-14T22:25:58.290982Z","iopub.status.idle":"2023-12-14T22:25:58.296426Z","shell.execute_reply.started":"2023-12-14T22:25:58.290957Z","shell.execute_reply":"2023-12-14T22:25:58.29544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"download data : ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:58.297675Z","iopub.execute_input":"2023-12-14T22:25:58.29795Z","iopub.status.idle":"2023-12-14T22:25:58.392432Z","shell.execute_reply.started":"2023-12-14T22:25:58.297926Z","shell.execute_reply":"2023-12-14T22:25:58.391512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"investigation des données \n","metadata":{}},{"cell_type":"code","source":"print(f'Length of train dataframe: {len(train_df)}\\n')\nprint(f'Number of NaN values:\\n{train_df.isna().sum()}\\n')","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:58.393761Z","iopub.execute_input":"2023-12-14T22:25:58.394132Z","iopub.status.idle":"2023-12-14T22:25:58.414557Z","shell.execute_reply.started":"2023-12-14T22:25:58.394095Z","shell.execute_reply":"2023-12-14T22:25:58.413612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\n\n!pip install /kaggle/input/rsnamodules/dicomsdl-0.109.1-cp37-cp37m-manylinux_2_12_x86_64.manylinux2010_x86_64.whl \n\ntry:\n    import pylibjpeg\nexcept:\n    !pip install /kaggle/input/rsna-2022-whl/{pylibjpeg-1.4.0-py3-none-any.whl,python_gdcm-3.0.15-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl}","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:25:58.419655Z","iopub.execute_input":"2023-12-14T22:25:58.420075Z","iopub.status.idle":"2023-12-14T22:26:02.509844Z","shell.execute_reply.started":"2023-12-14T22:25:58.420048Z","shell.execute_reply":"2023-12-14T22:26:02.508528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's consider one particular patient\npatient_id = train_df[train_df.cancer == 1].iloc[0].patient_id\n\none_patient_df = train_df[train_df.patient_id == patient_id]\none_patient_df","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:02.51165Z","iopub.execute_input":"2023-12-14T22:26:02.512566Z","iopub.status.idle":"2023-12-14T22:26:02.535383Z","shell.execute_reply.started":"2023-12-14T22:26:02.512525Z","shell.execute_reply":"2023-12-14T22:26:02.534528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_dir = '/kaggle/input/rsna-breast-cancer-detection/{}_images/{}/{}.dcm'\ntrain = 'train'\ntest = 'test'\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:02.536528Z","iopub.execute_input":"2023-12-14T22:26:02.536766Z","iopub.status.idle":"2023-12-14T22:26:02.54096Z","shell.execute_reply.started":"2023-12-14T22:26:02.536744Z","shell.execute_reply":"2023-12-14T22:26:02.539975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dicomsdl\n\nimport matplotlib.pyplot as plt\nimport dicomsdl\n\nn_rows = len(one_patient_df)\n\nplt.figure(figsize=(5 * n_rows, 5))\nfor i in range(n_rows):\n    row = one_patient_df.iloc[i]\n    \n    plt.subplot(1, n_rows, i + 1)\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, row['view'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})\n    plt.text(200, 700, row['cancer'], fontsize = 14, bbox={'facecolor': 'white', 'pad' : 5})","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:02.542411Z","iopub.execute_input":"2023-12-14T22:26:02.542774Z","iopub.status.idle":"2023-12-14T22:26:26.461113Z","shell.execute_reply.started":"2023-12-14T22:26:02.542741Z","shell.execute_reply":"2023-12-14T22:26:26.460123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\nplt.figure(figsize=(5, 8))\nsns.countplot(data = train_df, x=\"laterality\", hue=\"cancer\", dodge = False)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:26.463116Z","iopub.execute_input":"2023-12-14T22:26:26.463463Z","iopub.status.idle":"2023-12-14T22:26:26.704089Z","shell.execute_reply.started":"2023-12-14T22:26:26.463429Z","shell.execute_reply":"2023-12-14T22:26:26.70327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 10))\nsns.countplot(data = train_df, x=\"view\", hue=\"cancer\", dodge = False)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:26.705175Z","iopub.execute_input":"2023-12-14T22:26:26.705505Z","iopub.status.idle":"2023-12-14T22:26:27.073597Z","shell.execute_reply.started":"2023-12-14T22:26:26.705476Z","shell.execute_reply":"2023-12-14T22:26:27.072711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.view.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:27.074861Z","iopub.execute_input":"2023-12-14T22:26:27.075211Z","iopub.status.idle":"2023-12-14T22:26:27.088721Z","shell.execute_reply.started":"2023-12-14T22:26:27.075175Z","shell.execute_reply":"2023-12-14T22:26:27.087929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here I'm plotting one scan sample of each view (without standardizing)\n\nLIST_OF_VIEWS = sorted(list(train_df.view.unique()))\n\nplt.figure(figsize=(5 * len(LIST_OF_VIEWS), 5))\n\nfor i, v in enumerate(LIST_OF_VIEWS, start = 1):\n    plt.subplot(1, len(LIST_OF_VIEWS), i)\n    row = train_df.loc[train_df.view == v].iloc[0]\n    \n    img_arr = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id)).pixelData()\n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, v, fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:27.089689Z","iopub.execute_input":"2023-12-14T22:26:27.089919Z","iopub.status.idle":"2023-12-14T22:26:40.310459Z","shell.execute_reply.started":"2023-12-14T22:26:27.089898Z","shell.execute_reply":"2023-12-14T22:26:40.309495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split dataset in two groups depending on view (to investigate them a bit more):\n# 1. MLO and CC - major group\n# 2. all other views - minor group\n\nmajor_group = ['CC', 'MLO']\n\ntrain_df_major_group = train_df[train_df.view.isin(major_group)]\ntrain_df_minor_group = train_df[~train_df.view.isin(major_group)]","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:40.31209Z","iopub.execute_input":"2023-12-14T22:26:40.312821Z","iopub.status.idle":"2023-12-14T22:26:40.334155Z","shell.execute_reply.started":"2023-12-14T22:26:40.312785Z","shell.execute_reply":"2023-12-14T22:26:40.333285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_minor_group.cancer.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:40.335889Z","iopub.execute_input":"2023-12-14T22:26:40.336505Z","iopub.status.idle":"2023-12-14T22:26:40.344968Z","shell.execute_reply.started":"2023-12-14T22:26:40.336459Z","shell.execute_reply":"2023-12-14T22:26:40.34403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# there are not so many examples of minor view group\n# let's display all of them\n\nplt.figure(figsize=(4, 4 * len(train_df_minor_group)))\n\nfor i, row in enumerate(train_df_minor_group.itertuples(index = False), start = 1):\n    plt.subplot(len(train_df_minor_group),1,i)\n    \n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n    \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id} {row.image_id}', fontsize = 13, bbox={'facecolor': 'white'})\n    \n# a lot of these scans look bad!!!\n","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:26:40.351161Z","iopub.execute_input":"2023-12-14T22:26:40.351449Z","iopub.status.idle":"2023-12-14T22:27:28.549354Z","shell.execute_reply.started":"2023-12-14T22:26:40.351424Z","shell.execute_reply":"2023-12-14T22:27:28.548354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i want to investigate patients which scans look weird\n\ndef display_scans_by_patient_id(ids):\n    plt.figure(figsize=(10 * 5, len(ids) * 7)) # let's consider max 10 pic per patient\n    for i, p_id in enumerate(ids):\n        df = train_df[train_df.patient_id == p_id].iloc[:10]\n\n        for j, row in enumerate(df.itertuples(index = False)):\n            plt.subplot(len(ids), 10, i * 10 + j + 1)\n\n            img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n            img_arr = img.pixelData()\n\n            # standardize all scans\n            img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            if img.PhotometricInterpretation == \"MONOCHROME1\":\n                img_arr = 1 - img_arr\n\n            plt.imshow(img_arr, cmap = plt.cm.bone)\n            plt.text(200, 300, f'{row.patient_id} {row.image_id}', fontsize = 20, bbox={'facecolor': 'white'})\n            plt.text(200, 800, f'{row.cancer}', fontsize = 20, bbox={'facecolor': 'white'})\n            \n\npatient_ids = [1511, 25323, 26530, 38739, 40317, 40832, 43377, 50454]\ndisplay_scans_by_patient_id(patient_ids)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:27:28.550892Z","iopub.execute_input":"2023-12-14T22:27:28.551658Z","iopub.status.idle":"2023-12-14T22:28:44.101061Z","shell.execute_reply.started":"2023-12-14T22:27:28.551621Z","shell.execute_reply":"2023-12-14T22:28:44.100138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bad_ids = patient_ids.copy()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:28:44.102317Z","iopub.execute_input":"2023-12-14T22:28:44.102616Z","iopub.status.idle":"2023-12-14T22:28:44.106802Z","shell.execute_reply.started":"2023-12-14T22:28:44.102589Z","shell.execute_reply":"2023-12-14T22:28:44.105894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ok, what's the goal:\n# I'm going to exclude patiens which have bad scans and don't have cancer (because we have a lot non-cancer scans)\n\n# all above patients have no cancer except 25323 patient\n# for this specific one I'm going to remove only 'bad' scans\nspecific_patient = 25_323\nbad_ids.remove(specific_patient)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:28:44.10825Z","iopub.execute_input":"2023-12-14T22:28:44.109036Z","iopub.status.idle":"2023-12-14T22:28:44.120031Z","shell.execute_reply.started":"2023-12-14T22:28:44.108997Z","shell.execute_reply":"2023-12-14T22:28:44.119149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# thank you\n# https://www.kaggle.com/competitions/rsna-breast-cancer-detection/discussion/373208\nbad_ids.append(27_770) ","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:28:44.121267Z","iopub.execute_input":"2023-12-14T22:28:44.121609Z","iopub.status.idle":"2023-12-14T22:28:44.130036Z","shell.execute_reply.started":"2023-12-14T22:28:44.121583Z","shell.execute_reply":"2023-12-14T22:28:44.129263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# maybe we'll find something weird\n\ndf = train_df_major_group[~(train_df_major_group.patient_id.isin(patient_ids)) & (train_df_major_group.cancer == 1)]\ndf = df.groupby(['patient_id'])['image_id'].apply(lambda x : x.iloc[0]).to_frame().reset_index().sample(n = 30, random_state = 42)\n\nplt.figure(figsize=(6 * 5, 6 * 6)) # 6x5\n\nfor i, row in enumerate(df.itertuples(index = False), start = 1):\n    plt.subplot(6, 5, i)\n\n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n        \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id}', fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:28:44.131419Z","iopub.execute_input":"2023-12-14T22:28:44.131727Z","iopub.status.idle":"2023-12-14T22:29:47.068012Z","shell.execute_reply.started":"2023-12-14T22:28:44.131703Z","shell.execute_reply":"2023-12-14T22:29:47.066999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# and without cancer\n\ndf = train_df_major_group[~(train_df_major_group.patient_id.isin(patient_ids)) & (train_df_major_group.cancer == 0)]\ndf = df.groupby(['patient_id'])['image_id'].apply(lambda x : x.iloc[0]).to_frame().reset_index().sample(n = 30, random_state = 42)\n\nplt.figure(figsize=(6 * 5, 6 * 6)) # 6x5\n\nfor i, row in enumerate(df.itertuples(), start = 1):\n    plt.subplot(6, 5, i)\n\n    img = dicomsdl.open(images_dir.format(train, row.patient_id, row.image_id))\n    img_arr = img.pixelData()\n    \n    # standardize all scans\n    img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n    if img.PhotometricInterpretation == \"MONOCHROME1\":\n        img_arr = 1 - img_arr\n        \n    plt.imshow(img_arr, cmap = plt.cm.bone)\n    plt.text(200, 300, f'{row.patient_id}', fontsize = 13, bbox={'facecolor': 'white'})","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:29:47.069276Z","iopub.execute_input":"2023-12-14T22:29:47.069591Z","iopub.status.idle":"2023-12-14T22:30:38.783467Z","shell.execute_reply.started":"2023-12-14T22:29:47.069565Z","shell.execute_reply":"2023-12-14T22:30:38.782165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# again investigate patients having 'weird' scans\n\npatient_ids = [33588, 12943]\ndisplay_scans_by_patient_id(patient_ids)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:38.784722Z","iopub.execute_input":"2023-12-14T22:30:38.784991Z","iopub.status.idle":"2023-12-14T22:30:52.783417Z","shell.execute_reply.started":"2023-12-14T22:30:38.784967Z","shell.execute_reply":"2023-12-14T22:30:52.78234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"enlever les mauvais scans","metadata":{}},{"cell_type":"code","source":"# patient_ids - list of patients having 'bad' scans\ntrain_df = train_df[~train_df.patient_id.isin(bad_ids)]\n\nindex_to_remove = train_df.loc[(train_df.patient_id == specific_patient) & ( train_df.image_id == 1743461841), :].index\ntrain_df.drop(index = index_to_remove, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:52.784801Z","iopub.execute_input":"2023-12-14T22:30:52.78513Z","iopub.status.idle":"2023-12-14T22:30:52.803554Z","shell.execute_reply.started":"2023-12-14T22:30:52.785101Z","shell.execute_reply":"2023-12-14T22:30:52.802554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cancer_count = train_df.cancer.value_counts()\ncancer_count","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:52.804715Z","iopub.execute_input":"2023-12-14T22:30:52.804991Z","iopub.status.idle":"2023-12-14T22:30:52.813802Z","shell.execute_reply.started":"2023-12-14T22:30:52.804966Z","shell.execute_reply":"2023-12-14T22:30:52.812756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"min_cancer_count = cancer_count.min()\n\ntrain_df = pd.concat([train_df[train_df.cancer == i].sample(n=min_cancer_count) \n                     for i, _ in cancer_count.items()])\n\nlen(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:52.815071Z","iopub.execute_input":"2023-12-14T22:30:52.815407Z","iopub.status.idle":"2023-12-14T22:30:52.833352Z","shell.execute_reply.started":"2023-12-14T22:30:52.815375Z","shell.execute_reply":"2023-12-14T22:30:52.832599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"split data into train and val sets","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val = train_test_split(train_df, test_size = 0.2, random_state = 42)\nlen(X_train), len(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:52.834505Z","iopub.execute_input":"2023-12-14T22:30:52.835128Z","iopub.status.idle":"2023-12-14T22:30:53.030525Z","shell.execute_reply.started":"2023-12-14T22:30:52.83509Z","shell.execute_reply":"2023-12-14T22:30:53.02955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"generateur des données ","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:30:53.031767Z","iopub.execute_input":"2023-12-14T22:30:53.032073Z","iopub.status.idle":"2023-12-14T22:31:05.177357Z","shell.execute_reply.started":"2023-12-14T22:30:53.032046Z","shell.execute_reply":"2023-12-14T22:31:05.176469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from enum import Enum, auto\n\nclass Mode(Enum):\n    TRAIN = auto()\n    TEST = auto()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:31:05.178539Z","iopub.execute_input":"2023-12-14T22:31:05.179088Z","iopub.status.idle":"2023-12-14T22:31:05.183849Z","shell.execute_reply.started":"2023-12-14T22:31:05.17906Z","shell.execute_reply":"2023-12-14T22:31:05.182866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height = 300\nimg_width = 250\nimg_shape = (img_height, img_width, 1)\n\nclass ImageDataGen(tf.keras.utils.Sequence):\n    \n    def __init__(self,\n                 df,\n                 batch_size,\n                 mode = Mode.TRAIN):\n\n        self.df = df\n        self.batch_size = batch_size\n        self.mode = mode\n        self.mode_str = train if mode == Mode.TRAIN else test\n        \n        self.len = len(df)\n        \n    def __getitem__(self, index):\n        \n        start, end = index * self.batch_size, (index + 1) * self.batch_size\n        \n        X = np.zeros((self.batch_size, ) + img_shape)\n        y = np.zeros((self.batch_size, 1))\n        \n        for i , pos in enumerate(range(start, end)):\n            if pos >= self.len: break\n                     \n            row = self.df.iloc[pos]\n            patient_id = row.patient_id\n            img_id = row.image_id\n            \n            file_name = images_dir.format(self.mode_str, patient_id, img_id)\n            \n            img = dicomsdl.open(file_name)\n            img_arr = img.pixelData()\n            \n            # standartize all scans\n            img_arr = (img_arr - img_arr.min()) / (img_arr.max() - img_arr.min())\n            \n            if img.PhotometricInterpretation == \"MONOCHROME1\":\n                img_arr = 1 - img_arr\n\n            img_arr = np.expand_dims(img_arr, axis = -1)\n            img_arr = tf.image.resize(img_arr, img_shape[:-1], method = 'nearest').numpy()\n                 \n            X[i,...] = img_arr\n                \n            \n            if self.mode == Mode.TRAIN:\n                y[i] = row.cancer\n                \n        return (X, y) if self.mode == Mode.TRAIN else X\n                \n    \n    def __len__(self):\n        return self.len // self.batch_size + bool(self.len % self.batch_size)\ntrain_gen = ImageDataGen(X_train, 50)\nval_gen = ImageDataGen(X_val, 50)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:31:05.185154Z","iopub.execute_input":"2023-12-14T22:31:05.185684Z","iopub.status.idle":"2023-12-14T22:31:05.338631Z","shell.execute_reply.started":"2023-12-14T22:31:05.185649Z","shell.execute_reply":"2023-12-14T22:31:05.337731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model","metadata":{}},{"cell_type":"code","source":"import tensorflow.keras as K\n\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import (Conv2D, \n                                     MaxPooling2D, \n                                     BatchNormalization, \n                                     Dense, \n                                     Dropout,\n                                     GlobalMaxPooling2D)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:31:05.339671Z","iopub.execute_input":"2023-12-14T22:31:05.339947Z","iopub.status.idle":"2023-12-14T22:31:05.346536Z","shell.execute_reply.started":"2023-12-14T22:31:05.339922Z","shell.execute_reply":"2023-12-14T22:31:05.345579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# build a simple model\n\nmodel = Sequential()\n\nmodel.add(Conv2D(32, 5, activation = \"relu\", input_shape = img_shape))\nmodel.add(Conv2D(64, 5, activation = \"relu\"))\nmodel.add(Conv2D(64, 5, activation = \"relu\"))\nmodel.add(MaxPooling2D())\nmodel.add(Dropout(0.3))\n\nmodel.add(Conv2D(64, 5, activation = \"relu\"))\nmodel.add(Conv2D(128, 5, activation = \"relu\"))\nmodel.add(Conv2D(128, 5, activation = \"relu\"))\nmodel.add(MaxPooling2D())\nmodel.add(Dropout(0.3))\n\nmodel.add(Conv2D(128, 5, activation = \"relu\"))\nmodel.add(Conv2D(256, 5, activation = \"relu\"))\nmodel.add(Conv2D(256, 5, activation = \"relu\"))\nmodel.add(GlobalMaxPooling2D())\nmodel.add(Dropout(0.3))\n\nmodel.add(Dense(64, activation = 'relu'))\nmodel.add(Dropout(0.3))\nmodel.add(Dense(1, activation = 'sigmoid'))\n\nrecall_thresholds = [0.4, 0.5, 0.6, 0.8]\nmodel.compile(optimizer = Adam(learning_rate = 5e-5), \n              loss = 'binary_crossentropy', \n              metrics = [tf.keras.metrics.BinaryAccuracy(threshold = 0.5), \n                         tf.keras.metrics.Recall(thresholds = recall_thresholds),\n                         tf.keras.metrics.Precision(thresholds = recall_thresholds)])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:31:05.3477Z","iopub.execute_input":"2023-12-14T22:31:05.347941Z","iopub.status.idle":"2023-12-14T22:31:09.482979Z","shell.execute_reply.started":"2023-12-14T22:31:05.347917Z","shell.execute_reply":"2023-12-14T22:31:09.482059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# callbacks\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nearly_stop = EarlyStopping(patience = 5, restore_best_weights = True, verbose = 1) # val_loss\nreduce_lr = ReduceLROnPlateau(factor = 0.1, patience = 2, mode = 'min', verbose = 1) # val_loss ","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:31:09.484158Z","iopub.execute_input":"2023-12-14T22:31:09.484467Z","iopub.status.idle":"2023-12-14T22:31:09.490175Z","shell.execute_reply.started":"2023-12-14T22:31:09.484441Z","shell.execute_reply":"2023-12-14T22:31:09.489204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ntf.keras.backend.clear_session()\n\n\nhistory = model.fit(train_gen,\n                    validation_data = val_gen,\n                    epochs = 15,\n                    verbose = 1,\n                    workers = 8,\n                    callbacks = [reduce_lr])","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:41:11.449485Z","iopub.execute_input":"2023-12-14T22:41:11.449943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.808619Z","iopub.status.idle":"2023-12-14T22:39:32.808948Z","shell.execute_reply.started":"2023-12-14T22:39:32.808791Z","shell.execute_reply":"2023-12-14T22:39:32.808806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = history.history['loss']\nval_loss = history.history['val_loss']\n\nacc = history.history['binary_accuracy']\nval_acc = history.history['val_binary_accuracy']\n\n\nepochs = range(1, len(loss) + 1)\n\nplt.figure(figsize=(16, 5))\n#accuracy\nplt.subplot(1,2,1)\nplt.plot(epochs, acc, 'bo', label = 'Training accuracy')\nplt.plot(epochs, val_acc, 'r', label = 'Validation accuracy')\nplt.legend()\n\n#loss\nplt.subplot(1,2,2)\nplt.plot(epochs, loss, 'bo', label = 'Trainig loss')\nplt.plot(epochs, val_loss, 'r', label = 'Validation loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.810539Z","iopub.status.idle":"2023-12-14T22:39:32.81091Z","shell.execute_reply.started":"2023-12-14T22:39:32.810726Z","shell.execute_reply":"2023-12-14T22:39:32.810743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"recall = history.history['recall']\nval_recall = history.history['val_recall']\n\nrecall_len = len(recall_thresholds)\nplt.figure(figsize=(10, 5 * recall_len))\n\nfor i, (r, vr) in enumerate(zip(zip(*recall), zip(*val_recall)), start = 1):\n    plt.subplot(recall_len, 1, i)\n    plt.plot(epochs, r, 'bo', label = 'Training recall')\n    plt.plot(epochs, vr, 'r', label = 'Validation recall')\n    plt.title(f'Recall threshold = {recall_thresholds[i - 1]}')\n    plt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.812822Z","iopub.status.idle":"2023-12-14T22:39:32.813167Z","shell.execute_reply.started":"2023-12-14T22:39:32.813Z","shell.execute_reply":"2023-12-14T22:39:32.813017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"precision = history.history['precision']\nval_precision = history.history['val_precision']\n\nrecall_len = len(recall_thresholds)\nplt.figure(figsize=(10, 5 * recall_len))\n\nfor i, (p, vp) in enumerate(zip(zip(*precision), zip(*val_precision)), start = 1):\n    plt.subplot(recall_len, 1, i)\n    plt.plot(epochs, p, 'bo', label = 'Training precision')\n    plt.plot(epochs, vp, 'r', label = 'Validation precision')\n    plt.title(f'Precision threshold = {recall_thresholds[i - 1]}')\n    plt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.814222Z","iopub.status.idle":"2023-12-14T22:39:32.814579Z","shell.execute_reply.started":"2023-12-14T22:39:32.814414Z","shell.execute_reply":"2023-12-14T22:39:32.81443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"tester sur dataset","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.815932Z","iopub.status.idle":"2023-12-14T22:39:32.816268Z","shell.execute_reply.started":"2023-12-14T22:39:32.816104Z","shell.execute_reply":"2023-12-14T22:39:32.816119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGen(test_df, 16, mode = Mode.TEST)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.817976Z","iopub.status.idle":"2023-12-14T22:39:32.818291Z","shell.execute_reply.started":"2023-12-14T22:39:32.818136Z","shell.execute_reply":"2023-12-14T22:39:32.818151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_gen)","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.819361Z","iopub.status.idle":"2023-12-14T22:39:32.81969Z","shell.execute_reply.started":"2023-12-14T22:39:32.819526Z","shell.execute_reply":"2023-12-14T22:39:32.819541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['cancer'] = pred[:len(test_df)]\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.821156Z","iopub.status.idle":"2023-12-14T22:39:32.8215Z","shell.execute_reply.started":"2023-12-14T22:39:32.821331Z","shell.execute_reply":"2023-12-14T22:39:32.821352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = test_df.groupby('prediction_id')['cancer'].max().to_frame().reset_index()\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-12-14T22:39:32.822387Z","iopub.status.idle":"2023-12-14T22:39:32.822719Z","shell.execute_reply.started":"2023-12-14T22:39:32.822557Z","shell.execute_reply":"2023-12-14T22:39:32.822574Z"},"trusted":true},"execution_count":null,"outputs":[]}]}