{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":32048.637823,"end_time":"2024-12-18T18:38:46.240181","environment_variables":{},"exception":true,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-18T09:44:37.602358","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"a39948ba","cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2024-12-18T09:44:40.311383Z","iopub.status.busy":"2024-12-18T09:44:40.311059Z","iopub.status.idle":"2024-12-18T09:44:40.315587Z","shell.execute_reply":"2024-12-18T09:44:40.314793Z"},"papermill":{"duration":0.019132,"end_time":"2024-12-18T09:44:40.317279","exception":false,"start_time":"2024-12-18T09:44:40.298147","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8dfd7e2d","cell_type":"code","source":"!pip install -U pydicom dicom pylibjpeg pylibjpeg-libjpeg pylibjpeg-openjpeg --no-deps","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:40.337824Z","iopub.status.busy":"2024-12-18T09:44:40.337596Z","iopub.status.idle":"2024-12-18T09:44:44.050934Z","shell.execute_reply":"2024-12-18T09:44:44.050078Z"},"papermill":{"duration":3.725712,"end_time":"2024-12-18T09:44:44.052918","exception":false,"start_time":"2024-12-18T09:44:40.327206","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"fe1d188b","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.preprocessing import StandardScaler\nimport joblib\nimport pydicom\nfrom PIL import Image\nimport sys\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport io","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:44.07705Z","iopub.status.busy":"2024-12-18T09:44:44.076076Z","iopub.status.idle":"2024-12-18T09:44:59.028049Z","shell.execute_reply":"2024-12-18T09:44:59.027276Z"},"papermill":{"duration":14.966058,"end_time":"2024-12-18T09:44:59.030106","exception":false,"start_time":"2024-12-18T09:44:44.064048","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8087fa5c","cell_type":"code","source":"# Detect GPUs and create MirroredStrategy\ngpus = tf.config.list_physical_devices('GPU')\nif gpus:\n    try:\n        # Currently, memory growth needs to be the same across GPUs\n        for gpu in gpus:\n            tf.config.experimental.set_memory_growth(gpu, True)\n        strategy = tf.distribute.MirroredStrategy()\n        print('Running on multiple GPUs:', gpus)\n    except RuntimeError as e:\n        # Memory growth must be set before GPUs have been initialized\n        print(e)\nelse:\n    strategy = tf.distribute.get_strategy()  # Default strategy for CPU or single GPU\n    print('Running on CPU or single GPU')\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.052928Z","iopub.status.busy":"2024-12-18T09:44:59.052415Z","iopub.status.idle":"2024-12-18T09:44:59.64494Z","shell.execute_reply":"2024-12-18T09:44:59.643847Z"},"papermill":{"duration":0.60576,"end_time":"2024-12-18T09:44:59.646718","exception":false,"start_time":"2024-12-18T09:44:59.040958","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"0cf046f7","cell_type":"code","source":"tf.config.optimizer.set_experimental_options({'layout_optimizer': False})","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.669758Z","iopub.status.busy":"2024-12-18T09:44:59.669188Z","iopub.status.idle":"2024-12-18T09:44:59.673134Z","shell.execute_reply":"2024-12-18T09:44:59.672339Z"},"papermill":{"duration":0.017118,"end_time":"2024-12-18T09:44:59.67472","exception":false,"start_time":"2024-12-18T09:44:59.657602","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"54c29b98","cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.699329Z","iopub.status.busy":"2024-12-18T09:44:59.699059Z","iopub.status.idle":"2024-12-18T09:44:59.812063Z","shell.execute_reply":"2024-12-18T09:44:59.8113Z"},"papermill":{"duration":0.128723,"end_time":"2024-12-18T09:44:59.814158","exception":false,"start_time":"2024-12-18T09:44:59.685435","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"aaf76300","cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.837432Z","iopub.status.busy":"2024-12-18T09:44:59.836805Z","iopub.status.idle":"2024-12-18T09:44:59.858049Z","shell.execute_reply":"2024-12-18T09:44:59.857049Z"},"papermill":{"duration":0.03446,"end_time":"2024-12-18T09:44:59.859645","exception":false,"start_time":"2024-12-18T09:44:59.825185","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8cd5fec1","cell_type":"code","source":"desc = train_df.describe().T\ndesc = desc[['count', 'min', 'max']]\nprint(desc.to_markdown())","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.882344Z","iopub.status.busy":"2024-12-18T09:44:59.881686Z","iopub.status.idle":"2024-12-18T09:44:59.944732Z","shell.execute_reply":"2024-12-18T09:44:59.943474Z"},"papermill":{"duration":0.076064,"end_time":"2024-12-18T09:44:59.946452","exception":false,"start_time":"2024-12-18T09:44:59.870388","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e4abf718","cell_type":"code","source":"import os\n\n# Specify the path to the directory\npath = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n\n# Count the number of subdirectories\nsubdirectories = [d for d in os.listdir(path) if os.path.isdir(os.path.join(path, d))]\nnum_subdirectories = len(subdirectories)\n\nprint(f\"Number of subdirectories: {num_subdirectories}\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:44:59.969146Z","iopub.status.busy":"2024-12-18T09:44:59.968631Z","iopub.status.idle":"2024-12-18T09:45:48.178748Z","shell.execute_reply":"2024-12-18T09:45:48.17769Z"},"papermill":{"duration":48.234912,"end_time":"2024-12-18T09:45:48.19218","exception":false,"start_time":"2024-12-18T09:44:59.957268","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dab7b9af","cell_type":"code","source":"subdirectories[:10]","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:48.21653Z","iopub.status.busy":"2024-12-18T09:45:48.215886Z","iopub.status.idle":"2024-12-18T09:45:48.221765Z","shell.execute_reply":"2024-12-18T09:45:48.220818Z"},"papermill":{"duration":0.020391,"end_time":"2024-12-18T09:45:48.22349","exception":false,"start_time":"2024-12-18T09:45:48.203099","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8a6e2c46","cell_type":"code","source":"cols = [\"laterality\", \"view\", \"implant\"]\nfor col in cols:\n    print(train_df[col].value_counts())","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:48.246888Z","iopub.status.busy":"2024-12-18T09:45:48.246307Z","iopub.status.idle":"2024-12-18T09:45:48.265558Z","shell.execute_reply":"2024-12-18T09:45:48.264518Z"},"papermill":{"duration":0.032613,"end_time":"2024-12-18T09:45:48.267246","exception":false,"start_time":"2024-12-18T09:45:48.234633","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"961ec03d","cell_type":"code","source":"# Add image path to each column in csv\nbase_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n\ntrain_df[\"image_path\"] =  train_df.apply( lambda row : f\"{base_path}/{row['patient_id']}/{row['image_id']}.dcm\", axis = 1)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:48.291023Z","iopub.status.busy":"2024-12-18T09:45:48.290796Z","iopub.status.idle":"2024-12-18T09:45:48.727835Z","shell.execute_reply":"2024-12-18T09:45:48.727132Z"},"papermill":{"duration":0.450676,"end_time":"2024-12-18T09:45:48.729653","exception":false,"start_time":"2024-12-18T09:45:48.278977","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d994006a","cell_type":"code","source":"train_df.isna().sum()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:48.753512Z","iopub.status.busy":"2024-12-18T09:45:48.753217Z","iopub.status.idle":"2024-12-18T09:45:48.770049Z","shell.execute_reply":"2024-12-18T09:45:48.769267Z"},"papermill":{"duration":0.030656,"end_time":"2024-12-18T09:45:48.771706","exception":false,"start_time":"2024-12-18T09:45:48.74105","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b8a00ad7","cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Heatmap to visualize nulls\nsns.heatmap(train_df.isna(), cbar=False, cmap='viridis')\nplt.title('Heatmap of Null Values')\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:48.794412Z","iopub.status.busy":"2024-12-18T09:45:48.794168Z","iopub.status.idle":"2024-12-18T09:45:50.331031Z","shell.execute_reply":"2024-12-18T09:45:50.329952Z"},"papermill":{"duration":1.550609,"end_time":"2024-12-18T09:45:50.333191","exception":false,"start_time":"2024-12-18T09:45:48.782582","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"435c3819","cell_type":"code","source":"train_df['age'] = train_df['age'].fillna(train_df['age'].mean())","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.361695Z","iopub.status.busy":"2024-12-18T09:45:50.361046Z","iopub.status.idle":"2024-12-18T09:45:50.368144Z","shell.execute_reply":"2024-12-18T09:45:50.367321Z"},"papermill":{"duration":0.022661,"end_time":"2024-12-18T09:45:50.369837","exception":false,"start_time":"2024-12-18T09:45:50.347176","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"77a60bd7","cell_type":"code","source":"train_df.sample()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.398819Z","iopub.status.busy":"2024-12-18T09:45:50.398469Z","iopub.status.idle":"2024-12-18T09:45:50.417361Z","shell.execute_reply":"2024-12-18T09:45:50.416524Z"},"papermill":{"duration":0.035824,"end_time":"2024-12-18T09:45:50.419066","exception":false,"start_time":"2024-12-18T09:45:50.383242","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"8dc43240","cell_type":"code","source":"train_df.drop([\"site_id\", \"machine_id\", \"biopsy\", \"BIRADS\", \"density\"], axis = 1, inplace = True)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.445027Z","iopub.status.busy":"2024-12-18T09:45:50.444721Z","iopub.status.idle":"2024-12-18T09:45:50.452298Z","shell.execute_reply":"2024-12-18T09:45:50.451643Z"},"papermill":{"duration":0.022094,"end_time":"2024-12-18T09:45:50.454087","exception":false,"start_time":"2024-12-18T09:45:50.431993","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"374dfe99","cell_type":"code","source":"train_df['difficult_negative_case'] = train_df['difficult_negative_case'].map({False: 0, True: 1})","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.479384Z","iopub.status.busy":"2024-12-18T09:45:50.478866Z","iopub.status.idle":"2024-12-18T09:45:50.486426Z","shell.execute_reply":"2024-12-18T09:45:50.485447Z"},"papermill":{"duration":0.022337,"end_time":"2024-12-18T09:45:50.488348","exception":false,"start_time":"2024-12-18T09:45:50.466011","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"21eba642","cell_type":"code","source":"cols = [\"laterality\", \"view\", \"invasive\", \"cancer\", \"implant\", \"difficult_negative_case\"]\nfor col in cols:\n    print(train_df[col].value_counts())","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.514399Z","iopub.status.busy":"2024-12-18T09:45:50.514122Z","iopub.status.idle":"2024-12-18T09:45:50.529763Z","shell.execute_reply":"2024-12-18T09:45:50.528813Z"},"papermill":{"duration":0.030185,"end_time":"2024-12-18T09:45:50.531613","exception":false,"start_time":"2024-12-18T09:45:50.501428","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b1dc7dd6","cell_type":"code","source":"train_df = train_df[train_df[\"view\"].isin([\"MLO\", \"CC\"])]","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.556808Z","iopub.status.busy":"2024-12-18T09:45:50.556544Z","iopub.status.idle":"2024-12-18T09:45:50.567305Z","shell.execute_reply":"2024-12-18T09:45:50.566426Z"},"papermill":{"duration":0.025414,"end_time":"2024-12-18T09:45:50.569047","exception":false,"start_time":"2024-12-18T09:45:50.543633","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a4e53abb","cell_type":"code","source":"encoded_df = pd.get_dummies(train_df, columns=[\"view\", \"laterality\"], prefix=[\"view\", \"laterality\"])","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.593985Z","iopub.status.busy":"2024-12-18T09:45:50.593423Z","iopub.status.idle":"2024-12-18T09:45:50.614011Z","shell.execute_reply":"2024-12-18T09:45:50.612898Z"},"papermill":{"duration":0.035288,"end_time":"2024-12-18T09:45:50.615976","exception":false,"start_time":"2024-12-18T09:45:50.580688","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"aebfe3fd","cell_type":"code","source":"# Display the resulting dataframe\nencoded_df.sample(10)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.641598Z","iopub.status.busy":"2024-12-18T09:45:50.641251Z","iopub.status.idle":"2024-12-18T09:45:50.657467Z","shell.execute_reply":"2024-12-18T09:45:50.656683Z"},"papermill":{"duration":0.030772,"end_time":"2024-12-18T09:45:50.659176","exception":false,"start_time":"2024-12-18T09:45:50.628404","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3b420d7c","cell_type":"code","source":"encoded_df.shape","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.686637Z","iopub.status.busy":"2024-12-18T09:45:50.686299Z","iopub.status.idle":"2024-12-18T09:45:50.691338Z","shell.execute_reply":"2024-12-18T09:45:50.690552Z"},"papermill":{"duration":0.020678,"end_time":"2024-12-18T09:45:50.693132","exception":false,"start_time":"2024-12-18T09:45:50.672454","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d4757a28","cell_type":"code","source":"encoded_df.columns","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.719242Z","iopub.status.busy":"2024-12-18T09:45:50.718714Z","iopub.status.idle":"2024-12-18T09:45:50.724033Z","shell.execute_reply":"2024-12-18T09:45:50.723128Z"},"papermill":{"duration":0.020458,"end_time":"2024-12-18T09:45:50.726005","exception":false,"start_time":"2024-12-18T09:45:50.705547","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"852d301e","cell_type":"code","source":"cols = ['view_CC', 'view_MLO',\n       'laterality_L', 'laterality_R']\n\nfor col in cols:\n    encoded_df[col] = encoded_df[col].map({False: 0, True: 1})","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.751817Z","iopub.status.busy":"2024-12-18T09:45:50.751352Z","iopub.status.idle":"2024-12-18T09:45:50.759905Z","shell.execute_reply":"2024-12-18T09:45:50.759286Z"},"papermill":{"duration":0.022974,"end_time":"2024-12-18T09:45:50.761469","exception":false,"start_time":"2024-12-18T09:45:50.738495","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6fe5c5b3","cell_type":"code","source":"encoded_df.head(1)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.787457Z","iopub.status.busy":"2024-12-18T09:45:50.787227Z","iopub.status.idle":"2024-12-18T09:45:50.797738Z","shell.execute_reply":"2024-12-18T09:45:50.796896Z"},"papermill":{"duration":0.02555,"end_time":"2024-12-18T09:45:50.799423","exception":false,"start_time":"2024-12-18T09:45:50.773873","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e690d4d4","cell_type":"markdown","source":"# Checking Image Compatibility","metadata":{"papermill":{"duration":0.012124,"end_time":"2024-12-18T09:45:50.824884","exception":false,"start_time":"2024-12-18T09:45:50.81276","status":"completed"},"tags":[]}},{"id":"2af84b9c","cell_type":"code","source":"image_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10102/1245250349.dcm\"\ndicom = pydicom.dcmread(image_path)\nimage = dicom.pixel_array\n\n# image = image - np.min(image)  # Shift minimum to 0\n# image = (image / np.max(image) * 255).astype(np.uint8)  # Scale to 0-255 and convert to uint8\n\n# image = Image.fromarray(image)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:50.851579Z","iopub.status.busy":"2024-12-18T09:45:50.851083Z","iopub.status.idle":"2024-12-18T09:45:51.87089Z","shell.execute_reply":"2024-12-18T09:45:51.870145Z"},"papermill":{"duration":1.035289,"end_time":"2024-12-18T09:45:51.872906","exception":false,"start_time":"2024-12-18T09:45:50.837617","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"ef5f1986","cell_type":"code","source":"image","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:51.899673Z","iopub.status.busy":"2024-12-18T09:45:51.899329Z","iopub.status.idle":"2024-12-18T09:45:51.90522Z","shell.execute_reply":"2024-12-18T09:45:51.904279Z"},"papermill":{"duration":0.021071,"end_time":"2024-12-18T09:45:51.907115","exception":false,"start_time":"2024-12-18T09:45:51.886044","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4a35a9e7","cell_type":"code","source":"image = tf.expand_dims(image, axis=-1)\nimage = tf.image.resize(image, [512, 512])  # Resize for EfficientNetV2\nimage = tf.image.grayscale_to_rgb(image)  # Convert to RGB\nimage = image / 255.0","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:51.933814Z","iopub.status.busy":"2024-12-18T09:45:51.933583Z","iopub.status.idle":"2024-12-18T09:45:52.001538Z","shell.execute_reply":"2024-12-18T09:45:52.000814Z"},"papermill":{"duration":0.083378,"end_time":"2024-12-18T09:45:52.003369","exception":false,"start_time":"2024-12-18T09:45:51.919991","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c5315f90","cell_type":"code","source":"img = image.numpy()  # Convert TensorFlow tensor to NumPy array\nimg = img - np.min(img)  # Shift minimum to 0\nimg = (img / np.max(img) * 255).astype(np.uint8)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.030098Z","iopub.status.busy":"2024-12-18T09:45:52.029779Z","iopub.status.idle":"2024-12-18T09:45:52.036953Z","shell.execute_reply":"2024-12-18T09:45:52.036347Z"},"papermill":{"duration":0.022176,"end_time":"2024-12-18T09:45:52.038425","exception":false,"start_time":"2024-12-18T09:45:52.016249","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f9dd0e4c","cell_type":"code","source":"plt.imshow(img)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.065301Z","iopub.status.busy":"2024-12-18T09:45:52.06449Z","iopub.status.idle":"2024-12-18T09:45:52.373058Z","shell.execute_reply":"2024-12-18T09:45:52.372075Z"},"papermill":{"duration":0.32484,"end_time":"2024-12-18T09:45:52.375963","exception":false,"start_time":"2024-12-18T09:45:52.051123","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"cbc3c63e","cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.407898Z","iopub.status.busy":"2024-12-18T09:45:52.407633Z","iopub.status.idle":"2024-12-18T09:45:52.413002Z","shell.execute_reply":"2024-12-18T09:45:52.412133Z"},"papermill":{"duration":0.023232,"end_time":"2024-12-18T09:45:52.414707","exception":false,"start_time":"2024-12-18T09:45:52.391475","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"58fd8218","cell_type":"code","source":"encoded_df[['cancer',\n       'invasive', 'implant']].value_counts(normalize=True)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.444879Z","iopub.status.busy":"2024-12-18T09:45:52.444601Z","iopub.status.idle":"2024-12-18T09:45:52.459035Z","shell.execute_reply":"2024-12-18T09:45:52.45825Z"},"papermill":{"duration":0.031421,"end_time":"2024-12-18T09:45:52.460804","exception":false,"start_time":"2024-12-18T09:45:52.429383","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d748fee4","cell_type":"markdown","source":"# Reduce Dataset","metadata":{"papermill":{"duration":0.014964,"end_time":"2024-12-18T09:45:52.491798","exception":false,"start_time":"2024-12-18T09:45:52.476834","status":"completed"},"tags":[]}},{"id":"2c2f1637","cell_type":"code","source":"# Example DataFrame (use your full dataset instead)\ndf = encoded_df.copy()\n\n# Create a combined stratification label\ndf['stratify_label'] = df[['cancer', 'invasive', 'implant']].astype(str).agg('_'.join, axis=1)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.52205Z","iopub.status.busy":"2024-12-18T09:45:52.521746Z","iopub.status.idle":"2024-12-18T09:45:52.796583Z","shell.execute_reply":"2024-12-18T09:45:52.795852Z"},"papermill":{"duration":0.291881,"end_time":"2024-12-18T09:45:52.798535","exception":false,"start_time":"2024-12-18T09:45:52.506654","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a6e6f73f","cell_type":"code","source":"df['stratify_label'].value_counts()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.830924Z","iopub.status.busy":"2024-12-18T09:45:52.83032Z","iopub.status.idle":"2024-12-18T09:45:52.842999Z","shell.execute_reply":"2024-12-18T09:45:52.841953Z"},"papermill":{"duration":0.030358,"end_time":"2024-12-18T09:45:52.844761","exception":false,"start_time":"2024-12-18T09:45:52.814403","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4ca8fa7f","cell_type":"code","source":"# Filter rows where stratify_label == '0_0_0'\nrows_to_reduce = df[df['stratify_label'] == '0_0_0']\n\n# Randomly sample 40,000 rows from those rows\nrows_to_remove = rows_to_reduce.sample(n=50000, random_state=42)\n\n# Drop the sampled rows from the original DataFrame\ndf_reduced = df.drop(rows_to_remove.index)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.875771Z","iopub.status.busy":"2024-12-18T09:45:52.87545Z","iopub.status.idle":"2024-12-18T09:45:52.898926Z","shell.execute_reply":"2024-12-18T09:45:52.898256Z"},"papermill":{"duration":0.041618,"end_time":"2024-12-18T09:45:52.900884","exception":false,"start_time":"2024-12-18T09:45:52.859266","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c243a546","cell_type":"code","source":"# Check the new value counts\ndf_reduced['stratify_label'].value_counts(normalize=False)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.932856Z","iopub.status.busy":"2024-12-18T09:45:52.932567Z","iopub.status.idle":"2024-12-18T09:45:52.939722Z","shell.execute_reply":"2024-12-18T09:45:52.938879Z"},"papermill":{"duration":0.024702,"end_time":"2024-12-18T09:45:52.941566","exception":false,"start_time":"2024-12-18T09:45:52.916864","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3c5906d3","cell_type":"code","source":"df_reduced.shape","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:52.973184Z","iopub.status.busy":"2024-12-18T09:45:52.972632Z","iopub.status.idle":"2024-12-18T09:45:52.977612Z","shell.execute_reply":"2024-12-18T09:45:52.976793Z"},"papermill":{"duration":0.02278,"end_time":"2024-12-18T09:45:52.979395","exception":false,"start_time":"2024-12-18T09:45:52.956615","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"d8633fe1","cell_type":"code","source":"age_scaler = StandardScaler()\ndf_reduced['age'] = age_scaler.fit_transform(df_reduced[['age']])\n\n# Save Artifact\njoblib.dump(age_scaler, '/kaggle/working/age_scaler_artifact.pkl')\n\n# Load Artifact\nage_scaler = joblib.load('/kaggle/working/age_scaler_artifact.pkl')\nprint(\"Scaler loaded successfully!\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.010361Z","iopub.status.busy":"2024-12-18T09:45:53.009763Z","iopub.status.idle":"2024-12-18T09:45:53.020763Z","shell.execute_reply":"2024-12-18T09:45:53.019777Z"},"papermill":{"duration":0.028057,"end_time":"2024-12-18T09:45:53.022439","exception":false,"start_time":"2024-12-18T09:45:52.994382","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"21c7dece","cell_type":"code","source":"print(f\"mean: {age_scaler.mean_}\")\nprint(f\"std: {age_scaler.scale_}\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.05284Z","iopub.status.busy":"2024-12-18T09:45:53.052252Z","iopub.status.idle":"2024-12-18T09:45:53.05721Z","shell.execute_reply":"2024-12-18T09:45:53.056334Z"},"papermill":{"duration":0.021792,"end_time":"2024-12-18T09:45:53.058888","exception":false,"start_time":"2024-12-18T09:45:53.037096","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"dee56478","cell_type":"markdown","source":"# Train Test Val Split","metadata":{"papermill":{"duration":0.014404,"end_time":"2024-12-18T09:45:53.088357","exception":false,"start_time":"2024-12-18T09:45:53.073953","status":"completed"},"tags":[]}},{"id":"95718e63","cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# Splitting train+val and test\ntrain_val_df, test_df = train_test_split(\n    df_reduced,\n    test_size=0.15,  # 15% for test\n    stratify=df_reduced['stratify_label'],  # Stratify by combined label\n    random_state=42\n)\n\n# Splitting train and val\ntrain_df, val_df = train_test_split(\n    train_val_df,\n    test_size=0.1765,  # 15% of total = 17.65% of train+val\n    stratify=train_val_df['stratify_label'],\n    random_state=42\n)\n\n# Sanity check distributions\nprint(\"Train distribution:\")\nprint(train_df['stratify_label'].value_counts(normalize=False))\nprint(\"\\nValidation distribution:\")\nprint(val_df['stratify_label'].value_counts(normalize=False))\nprint(\"\\nTest distribution:\")\nprint(test_df['stratify_label'].value_counts(normalize=False))","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.11907Z","iopub.status.busy":"2024-12-18T09:45:53.118609Z","iopub.status.idle":"2024-12-18T09:45:53.233678Z","shell.execute_reply":"2024-12-18T09:45:53.232764Z"},"papermill":{"duration":0.133029,"end_time":"2024-12-18T09:45:53.235982","exception":false,"start_time":"2024-12-18T09:45:53.102953","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"126add8b","cell_type":"markdown","source":"# Visualize Mammogram Samples","metadata":{"papermill":{"duration":0.015676,"end_time":"2024-12-18T09:45:53.267993","exception":false,"start_time":"2024-12-18T09:45:53.252317","status":"completed"},"tags":[]}},{"id":"f6bae969","cell_type":"code","source":"x = \"/kaggle/input/rsna-breast-cancer-detection/train_images/10185\"\ndir_images = os.listdir(x)\ndir_images_path = [os.path.join(x,img) for img in dir_images]","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.300654Z","iopub.status.busy":"2024-12-18T09:45:53.300046Z","iopub.status.idle":"2024-12-18T09:45:53.311986Z","shell.execute_reply":"2024-12-18T09:45:53.311375Z"},"papermill":{"duration":0.029697,"end_time":"2024-12-18T09:45:53.313705","exception":false,"start_time":"2024-12-18T09:45:53.284008","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"57442b0c","cell_type":"code","source":"def process_image(image_path):\n    # Read the DICOM file\n    dicom = pydicom.dcmread(image_path)\n    image = dicom.pixel_array.astype(np.float32)  # Ensure the array is in float32 for normalization\n\n    # Normalize pixel values to range [0, 1] based on min and max pixel intensity\n    min_val = np.min(image)\n    max_val = np.max(image)\n    image = (image - min_val) / (max_val - min_val)\n\n    # Add a channel dimension if the image is grayscale\n    image = tf.expand_dims(image, axis=-1)\n\n    # Resize the image to 512x512 for EfficientNetV2\n    image = tf.image.resize(image, [512, 512])\n\n    # Convert to RGB (grayscale images need three channels for RGB models)\n    image = tf.image.grayscale_to_rgb(image)\n    \n    return image\ndef get_viewable_img(img):\n    img = img.numpy()  # Convert TensorFlow tensor to NumPy array\n    img = img - np.min(img)  # Shift minimum to 0\n    img = (img / np.max(img) * 255).astype(np.uint8)\n    img = Image.fromarray(img)\n    return img","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.345603Z","iopub.status.busy":"2024-12-18T09:45:53.345265Z","iopub.status.idle":"2024-12-18T09:45:53.351132Z","shell.execute_reply":"2024-12-18T09:45:53.350312Z"},"papermill":{"duration":0.023617,"end_time":"2024-12-18T09:45:53.352814","exception":false,"start_time":"2024-12-18T09:45:53.329197","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"e0a1ead9","cell_type":"code","source":"n = len(dir_images_path)\n# Determine grid size (adjust as needed for better layout)\ncols = 3\nrows = (n + cols - 1) // cols  # Round up rows to fit all images\n\n# Create subplots\nfig, axes = plt.subplots(rows, cols, figsize=(10, 5 * rows))  # Adjust figsize for better spacing\n\n# Flatten axes for easier iteration (works even if rows/cols > 1)\naxes = axes.flatten() if rows * cols > 1 else [axes]\n\n# Loop through images and display them\nfor i, img_path in enumerate(dir_images_path):\n    image = get_viewable_img(process_image(img_path))\n    axes[i].imshow(image)\n    axes[i].axis('off')  # Turn off axes\n    axes[i].set_title(f\"Image {i + 1}\")  # Optional title\n\n# Hide any unused subplots\nfor j in range(i + 1, len(axes)):\n    axes[j].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:53.384894Z","iopub.status.busy":"2024-12-18T09:45:53.384592Z","iopub.status.idle":"2024-12-18T09:45:57.231067Z","shell.execute_reply":"2024-12-18T09:45:57.230072Z"},"papermill":{"duration":3.868177,"end_time":"2024-12-18T09:45:57.236787","exception":false,"start_time":"2024-12-18T09:45:53.36861","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"a278b168","cell_type":"markdown","source":"# Preprocessing Functions","metadata":{"papermill":{"duration":0.02232,"end_time":"2024-12-18T09:45:57.2819","exception":false,"start_time":"2024-12-18T09:45:57.25958","status":"completed"},"tags":[]}},{"id":"1f1d18bf","cell_type":"code","source":"def preprocess_image(image_path_tensor):\n    def _read_and_preprocess_image(image_path):\n        # Read file using TensorFlow (handles tensors)\n        image_data = tf.io.read_file(image_path)\n        \n        # Convert to numpy array for pydicom (outside of tf.function)\n        image_data_np = image_data.numpy()\n\n        # Use BytesIO to simulate a file object\n        with io.BytesIO(image_data_np) as f:\n            dicom = pydicom.dcmread(f)\n            image = dicom.pixel_array.astype(np.float32)\n        \n        # Normalize pixel values\n        min_val = np.min(image)\n        max_val = np.max(image)\n        image = (image - min_val) / (max_val - min_val)\n\n        # Add channel dimension and resize\n        image = np.expand_dims(image, axis=-1)\n        image = tf.image.resize(image, [512, 512]).numpy()\n\n        # Convert to RGB\n        image = np.concatenate([image, image, image], axis=-1)\n        return image\n\n    # Use tf.py_function to call the Python function\n    image = tf.py_function(_read_and_preprocess_image, [image_path_tensor], tf.float32)\n\n    # Set the shape for the output tensor (if known)\n    image.set_shape([512, 512, 3])\n\n    return image","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.326864Z","iopub.status.busy":"2024-12-18T09:45:57.326039Z","iopub.status.idle":"2024-12-18T09:45:57.333263Z","shell.execute_reply":"2024-12-18T09:45:57.332287Z"},"papermill":{"duration":0.031332,"end_time":"2024-12-18T09:45:57.335041","exception":false,"start_time":"2024-12-18T09:45:57.303709","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"af5292c6","cell_type":"code","source":"def preprocess_tabular_data(row):\n    # Safely extract and convert 'age'\n    age = tf.convert_to_tensor(row.get('age', 0.0), dtype=tf.float32)\n    age = tf.expand_dims(age, axis=0)  # Ensure shape [1]\n\n    # Safely extract binary features and convert them to tensors\n    binary_features = tf.stack([\n        tf.convert_to_tensor(row.get('view_CC', 0), dtype=tf.float32),\n        tf.convert_to_tensor(row.get('view_MLO', 0), dtype=tf.float32),\n        tf.convert_to_tensor(row.get('laterality_L', 0), dtype=tf.float32),\n        tf.convert_to_tensor(row.get('laterality_R', 0), dtype=tf.float32),\n    ], axis=0)\n    \n    # # Debugging output\n    # tf.print(\"Age:\", age)\n    # tf.print(\"Binary Features:\", binary_features)\n\n    # Combine features\n    tabular_features = tf.concat([age, binary_features], axis=0)\n    return tabular_features\n\ndef preprocess_label(row):\n    return {\n        'cancer_output': tf.convert_to_tensor(float(row['cancer']), dtype=tf.float32),\n        'invasive_output': tf.convert_to_tensor(float(row['invasive']), dtype=tf.float32),\n        'difficult_negative_case_output': tf.convert_to_tensor(float(row['difficult_negative_case']), dtype=tf.float32)\n    }\ndef preprocess_row(row):\n    image = preprocess_image(row['image_path'])\n    tabular_data = preprocess_tabular_data(row)\n    labels = preprocess_label(row)\n    return {\n        'image_input': image,\n        'tabular_input': tabular_data\n    }, labels","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.378065Z","iopub.status.busy":"2024-12-18T09:45:57.377371Z","iopub.status.idle":"2024-12-18T09:45:57.385833Z","shell.execute_reply":"2024-12-18T09:45:57.384859Z"},"papermill":{"duration":0.032355,"end_time":"2024-12-18T09:45:57.387776","exception":false,"start_time":"2024-12-18T09:45:57.355421","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"f96a093b","cell_type":"code","source":"BATCH_SIZE_PER_REPLICA = 50  # Adjust as needed, considering GPU memory\nGLOBAL_BATCH_SIZE = BATCH_SIZE_PER_REPLICA * strategy.num_replicas_in_sync  # Should be 2 for T4 x2","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.430703Z","iopub.status.busy":"2024-12-18T09:45:57.430078Z","iopub.status.idle":"2024-12-18T09:45:57.434519Z","shell.execute_reply":"2024-12-18T09:45:57.433627Z"},"papermill":{"duration":0.027117,"end_time":"2024-12-18T09:45:57.43615","exception":false,"start_time":"2024-12-18T09:45:57.409033","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"c8295630","cell_type":"code","source":"# Dataset creation (modified to return a single dictionary)\ndef create_dataset(df):\n    dataset = tf.data.Dataset.from_tensor_slices(dict(df))\n    dataset = dataset.map(preprocess_row, num_parallel_calls=tf.data.AUTOTUNE)\n    dataset = dataset.shuffle(buffer_size=1000).batch(GLOBAL_BATCH_SIZE).prefetch(tf.data.AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.477851Z","iopub.status.busy":"2024-12-18T09:45:57.477546Z","iopub.status.idle":"2024-12-18T09:45:57.482499Z","shell.execute_reply":"2024-12-18T09:45:57.481685Z"},"papermill":{"duration":0.027767,"end_time":"2024-12-18T09:45:57.484326","exception":false,"start_time":"2024-12-18T09:45:57.456559","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"6463cf8c","cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.52575Z","iopub.status.busy":"2024-12-18T09:45:57.525222Z","iopub.status.idle":"2024-12-18T09:45:57.530664Z","shell.execute_reply":"2024-12-18T09:45:57.529798Z"},"papermill":{"duration":0.02787,"end_time":"2024-12-18T09:45:57.532296","exception":false,"start_time":"2024-12-18T09:45:57.504426","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"fcb2e8a1","cell_type":"code","source":"float_cols = ['age', 'cancer', 'invasive', 'implant',\n       'difficult_negative_case', 'view_CC', 'view_MLO',\n       'laterality_L', 'laterality_R']\ntrain_df[float_cols] = train_df[float_cols].astype(np.float32)\nval_df[float_cols] = val_df[float_cols].astype(np.float32) #make sure to apply to all datasets\ntest_df[float_cols] = test_df[float_cols].astype(np.float32) #make sure to apply to all datasets","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.572488Z","iopub.status.busy":"2024-12-18T09:45:57.572175Z","iopub.status.idle":"2024-12-18T09:45:57.583781Z","shell.execute_reply":"2024-12-18T09:45:57.582731Z"},"papermill":{"duration":0.033844,"end_time":"2024-12-18T09:45:57.585664","exception":false,"start_time":"2024-12-18T09:45:57.55182","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"83ca91ce","cell_type":"code","source":"with strategy.scope():\n    train_dataset = create_dataset(train_df)\n    val_dataset = create_dataset(val_df)\n    test_dataset = create_dataset(test_df)\n    # Distribute the datasets\n    train_dataset_distributed = strategy.experimental_distribute_dataset(train_dataset)\n    val_dataset_distributed = strategy.experimental_distribute_dataset(val_dataset)\n    test_dataset_distributed = strategy.experimental_distribute_dataset(test_dataset)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:57.626095Z","iopub.status.busy":"2024-12-18T09:45:57.625791Z","iopub.status.idle":"2024-12-18T09:45:58.059319Z","shell.execute_reply":"2024-12-18T09:45:58.058557Z"},"papermill":{"duration":0.456257,"end_time":"2024-12-18T09:45:58.061545","exception":false,"start_time":"2024-12-18T09:45:57.605288","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"3abbabb1","cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\n\n# Create the ModelCheckpoint callback\ncheckpoint_filepath = 'best_model_v2.keras'  # Filepath to save the best model\nmodel_checkpoint_callback = ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=False,  # Save the entire model (not just weights)\n    monitor='val_loss',  # Monitor validation loss\n    mode='min',  # The best model has the lowest validation loss\n    save_best_only=True,  # Only save the best model\n    verbose=1\n)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:58.103075Z","iopub.status.busy":"2024-12-18T09:45:58.102749Z","iopub.status.idle":"2024-12-18T09:45:58.36733Z","shell.execute_reply":"2024-12-18T09:45:58.366364Z"},"papermill":{"duration":0.287488,"end_time":"2024-12-18T09:45:58.369426","exception":false,"start_time":"2024-12-18T09:45:58.081938","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b87b76f4","cell_type":"code","source":"from tensorflow.keras.applications import EfficientNetV2B3\nfrom tensorflow.keras.layers import Input, Dense, Concatenate, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.metrics import AUC","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:58.413537Z","iopub.status.busy":"2024-12-18T09:45:58.412724Z","iopub.status.idle":"2024-12-18T09:45:58.422761Z","shell.execute_reply":"2024-12-18T09:45:58.421719Z"},"papermill":{"duration":0.033704,"end_time":"2024-12-18T09:45:58.424646","exception":false,"start_time":"2024-12-18T09:45:58.390942","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"cc67a894","cell_type":"code","source":"with strategy.scope():\n    model = EfficientNetV2B3(include_top=False, pooling='avg', weights='imagenet')","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:45:58.464924Z","iopub.status.busy":"2024-12-18T09:45:58.464639Z","iopub.status.idle":"2024-12-18T09:46:04.201988Z","shell.execute_reply":"2024-12-18T09:46:04.201047Z"},"papermill":{"duration":5.760151,"end_time":"2024-12-18T09:46:04.204358","exception":false,"start_time":"2024-12-18T09:45:58.444207","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"83d5867c","cell_type":"code","source":"# Unfreeze the layers from block7 onwards\nunfreeze = False\nfor layer in model.layers:\n    if 'block6' in layer.name:\n        unfreeze = True\n    if unfreeze:\n        layer.trainable = True\n    else:\n        layer.trainable = False","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:04.252195Z","iopub.status.busy":"2024-12-18T09:46:04.251777Z","iopub.status.idle":"2024-12-18T09:46:04.262907Z","shell.execute_reply":"2024-12-18T09:46:04.261998Z"},"papermill":{"duration":0.037695,"end_time":"2024-12-18T09:46:04.26527","exception":false,"start_time":"2024-12-18T09:46:04.227575","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"17cd6ce9","cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:04.324208Z","iopub.status.busy":"2024-12-18T09:46:04.323838Z","iopub.status.idle":"2024-12-18T09:46:04.72561Z","shell.execute_reply":"2024-12-18T09:46:04.72464Z"},"papermill":{"duration":0.450637,"end_time":"2024-12-18T09:46:04.747526","exception":false,"start_time":"2024-12-18T09:46:04.296889","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"31c6c2af","cell_type":"code","source":"# model.save(\"efficientnetv2b3.keras\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:04.807229Z","iopub.status.busy":"2024-12-18T09:46:04.806891Z","iopub.status.idle":"2024-12-18T09:46:04.810842Z","shell.execute_reply":"2024-12-18T09:46:04.809886Z"},"papermill":{"duration":0.035268,"end_time":"2024-12-18T09:46:04.81256","exception":false,"start_time":"2024-12-18T09:46:04.777292","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"033ed3a2","cell_type":"code","source":"with strategy.scope():\n    # Image input and feature extractor\n    image_input = Input(shape=(512, 512, 3), name='image_input')\n    base_model = model\n    image_features = base_model(image_input)\n\n    # Tabular input\n    tabular_input = Input(shape=(5,), name='tabular_input')\n    tabular_features = Dense(64, activation='relu')(tabular_input)  # Increased units\n    tabular_features = Dropout(0.3)(tabular_features)  # Added dropout\n    tabular_features = Dense(32, activation='relu')(tabular_features)\n    tabular_features = Dropout(0.2)(tabular_features)\n\n    # Concatenate image features and tabular features\n    combined_features = Concatenate()([image_features, tabular_features])\n    combined_features = Dense(512, activation='relu')(combined_features)\n    combined_features = Dropout(0.4)(combined_features)\n\n    # Separate FC layers for each output with more layers and dropout\n    # Cancer output branch\n    fc1_cancer = Dense(512, activation='relu')(combined_features)\n    fc1_cancer = Dropout(0.4)(fc1_cancer)\n    fc2_cancer = Dense(256, activation='relu')(fc1_cancer)\n    fc2_cancer = Dropout(0.3)(fc2_cancer)\n    fc3_cancer = Dense(128, activation='relu')(fc2_cancer)\n    fc3_cancer = Dropout(0.2)(fc3_cancer)\n    fc4_cancer = Dense(64, activation='relu')(fc3_cancer)\n    cancer_output = Dense(1, activation='sigmoid', name='cancer_output')(fc4_cancer)\n\n    # Invasive output branch\n    fc1_invasive = Dense(256, activation='relu')(combined_features)\n    fc1_invasive = Dropout(0.3)(fc1_invasive)\n    fc2_invasive = Dense(128, activation='relu')(fc1_invasive)\n    fc2_invasive = Dropout(0.2)(fc2_invasive)\n    fc3_invasive = Dense(64, activation='relu')(fc2_invasive)\n    invasive_output = Dense(1, activation='sigmoid', name='invasive_output')(fc3_invasive)\n\n    # Difficult negative case output branch\n    fc1_difficult = Dense(256, activation='relu')(combined_features)\n    fc1_difficult = Dropout(0.3)(fc1_difficult)\n    fc2_difficult = Dense(128, activation='relu')(fc1_difficult)\n    fc2_difficult = Dropout(0.2)(fc2_difficult)\n    fc3_difficult = Dense(64, activation='relu')(fc2_difficult)\n    difficult_negative_case_output = Dense(1, activation='sigmoid', name='difficult_negative_case_output')(fc3_difficult)\n\n    # Final model\n    model = Model(\n        inputs=[image_input, tabular_input],\n        outputs=[cancer_output, invasive_output, difficult_negative_case_output]\n    )\n\n    model.compile(\n        optimizer='adam',\n        loss={\n            'cancer_output': 'binary_crossentropy',\n            'invasive_output': 'binary_crossentropy',\n            'difficult_negative_case_output': 'binary_crossentropy'\n        },\n        loss_weights={\n            'cancer_output': 1.0,\n            'invasive_output': 1.0,\n            'difficult_negative_case_output': 0.5\n        },\n        metrics={\n            'cancer_output': \"accuracy\",\n            'invasive_output': \"accuracy\",\n            'difficult_negative_case_output': \"accuracy\",\n        })","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:04.909134Z","iopub.status.busy":"2024-12-18T09:46:04.908802Z","iopub.status.idle":"2024-12-18T09:46:05.262235Z","shell.execute_reply":"2024-12-18T09:46:05.261321Z"},"papermill":{"duration":0.42362,"end_time":"2024-12-18T09:46:05.264541","exception":false,"start_time":"2024-12-18T09:46:04.840921","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"be60981b","cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:05.326073Z","iopub.status.busy":"2024-12-18T09:46:05.325123Z","iopub.status.idle":"2024-12-18T09:46:05.379174Z","shell.execute_reply":"2024-12-18T09:46:05.378252Z"},"papermill":{"duration":0.08725,"end_time":"2024-12-18T09:46:05.381784","exception":false,"start_time":"2024-12-18T09:46:05.294534","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b4cef635","cell_type":"code","source":"# model.save(\"model_v2.keras\")","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:05.451312Z","iopub.status.busy":"2024-12-18T09:46:05.450999Z","iopub.status.idle":"2024-12-18T09:46:05.454951Z","shell.execute_reply":"2024-12-18T09:46:05.454074Z"},"papermill":{"duration":0.035311,"end_time":"2024-12-18T09:46:05.456528","exception":false,"start_time":"2024-12-18T09:46:05.421217","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"4bb4b0eb","cell_type":"code","source":"history = model.fit(\n    train_dataset_distributed,\n    validation_data=val_dataset_distributed,\n    epochs=10,\n    verbose=1,\n    callbacks=[model_checkpoint_callback]\n)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T09:46:05.516706Z","iopub.status.busy":"2024-12-18T09:46:05.516371Z","iopub.status.idle":"2024-12-18T18:38:22.257335Z","shell.execute_reply":"2024-12-18T18:38:22.253759Z"},"papermill":{"duration":31936.781074,"end_time":"2024-12-18T18:38:22.267243","exception":false,"start_time":"2024-12-18T09:46:05.486169","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"b62562e5","cell_type":"code","source":"# Load the best saved model\nbest_model = tf.keras.models.load_model(checkpoint_filepath)","metadata":{"execution":{"iopub.execute_input":"2024-12-18T18:38:22.36602Z","iopub.status.busy":"2024-12-18T18:38:22.364185Z","iopub.status.idle":"2024-12-18T18:38:40.347587Z","shell.execute_reply":"2024-12-18T18:38:40.346551Z"},"papermill":{"duration":18.034826,"end_time":"2024-12-18T18:38:40.349881","exception":false,"start_time":"2024-12-18T18:38:22.315055","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"id":"27f8a84a","cell_type":"code","source":"from sklearn.metrics import confusion_matrix, f1_score\n\n# Assuming you have a distributed test dataset: test_dataset_distributed\n# And the corresponding strategy: strategy\n\n@tf.function\ndef predict_step(inputs):\n    images, tabular = inputs\n    predictions = model([images, tabular], training=False)\n    return predictions\n\ndef get_distributed_predictions(dataset):\n    predictions_list = []\n    \n    for inputs, _ in dataset:\n        predictions = strategy.run(predict_step, args=(inputs,))\n        # Concatenate predictions from different replicas\n        predictions = {\n            name: strategy.gather(tensor, axis=0) for name, tensor in predictions.items()\n        }\n        predictions_list.append(predictions)\n    \n    # Concatenate predictions from all steps\n    final_predictions = {\n        name: np.concatenate([pred[name] for pred in predictions_list], axis=0)\n        for name in predictions_list[0].keys()\n    }\n    return final_predictions\n\n# Get predictions\ntest_predictions = get_distributed_predictions(test_dataset_distributed)\n\n# Get true labels (assuming your dataset provides labels)\ntest_labels = []\nfor _, labels in test_dataset_distributed:\n    test_labels.append(labels)\ntest_labels = np.concatenate(test_labels, axis=0)\n\n# Convert one-hot encoded labels to single integers if needed\nif len(test_labels.shape) > 1 and test_labels.shape[1] > 1:\n    test_labels = np.argmax(test_labels, axis=1)\n\n# Threshold predictions to get binary values (for confusion matrix and F1)\nthreshold = 0.5\ntest_predictions_binary = {\n    name: (preds > threshold).astype(int) for name, preds in test_predictions.items()\n}","metadata":{"papermill":{"duration":1.700904,"end_time":"2024-12-18T18:38:42.096609","exception":true,"start_time":"2024-12-18T18:38:40.395705","status":"failed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null}]}