{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39272,"databundleVersionId":4629629,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11550468,"sourceType":"datasetVersion","datasetId":7243360},{"sourceId":11559448,"sourceType":"datasetVersion","datasetId":7247889},{"sourceId":11559559,"sourceType":"datasetVersion","datasetId":7247972}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install --no-index /kaggle/input/utilss/pylibjpeg-2.0.1-py3-none-any.whl --no-deps\n!pip install --no-index /kaggle/input/utilss/pylibjpeg_libjpeg-2.3.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install --no-index /kaggle/input/utilss/pylibjpeg_openjpeg-2.4.0-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n\n!pip install --no-index /kaggle/input/utilss/pydicom-3.0.1-py3-none-any.whl --no-deps\n!pip install --no-index /kaggle/input/utils-numpy/numpy-1.26.4-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install --no-index /kaggle/input/utils-numpy/scikit_image-0.25.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:29.577068Z","iopub.execute_input":"2025-05-27T13:58:29.577406Z","iopub.status.idle":"2025-05-27T13:58:46.329098Z","shell.execute_reply.started":"2025-05-27T13:58:29.57738Z","shell.execute_reply":"2025-05-27T13:58:46.327708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain_df","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:46.330472Z","iopub.execute_input":"2025-05-27T13:58:46.330765Z","iopub.status.idle":"2025-05-27T13:58:46.905347Z","shell.execute_reply.started":"2025-05-27T13:58:46.330739Z","shell.execute_reply":"2025-05-27T13:58:46.904265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_df.info())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:46.90764Z","iopub.execute_input":"2025-05-27T13:58:46.907977Z","iopub.status.idle":"2025-05-27T13:58:46.949385Z","shell.execute_reply.started":"2025-05-27T13:58:46.90795Z","shell.execute_reply":"2025-05-27T13:58:46.948295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Class distribution\nprint(train_df['cancer'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:46.950463Z","iopub.execute_input":"2025-05-27T13:58:46.950757Z","iopub.status.idle":"2025-05-27T13:58:46.960635Z","shell.execute_reply.started":"2025-05-27T13:58:46.950731Z","shell.execute_reply":"2025-05-27T13:58:46.959376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nprint(train_df.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:46.961902Z","iopub.execute_input":"2025-05-27T13:58:46.962392Z","iopub.status.idle":"2025-05-27T13:58:46.998385Z","shell.execute_reply.started":"2025-05-27T13:58:46.962354Z","shell.execute_reply":"2025-05-27T13:58:46.997167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"age\"] = train_df[\"age\"].fillna(train_df[\"age\"].median())\n#train_df[\"BIRADS\"] = train_df[\"BIRADS\"].fillna(train_df[\"BIRADS\"].median())\n#train_df = train_df.drop(columns=[\"BIRADS\"])\n#train_df = train_df.drop(columns=[\"density\"])\ntrain_df = train_df.drop(columns=[\"machine_id\"])\ntrain_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:46.999651Z","iopub.execute_input":"2025-05-27T13:58:47.00001Z","iopub.status.idle":"2025-05-27T13:58:47.028574Z","shell.execute_reply.started":"2025-05-27T13:58:46.999967Z","shell.execute_reply":"2025-05-27T13:58:47.02729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Add image paths to the train dataframe","metadata":{}},{"cell_type":"code","source":"import os\n\n# Define the base directory where images are stored\nBASE_DIR = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n\n# Function to construct full image path\ndef get_image_path(row):\n    return os.path.join(BASE_DIR, str(row[\"patient_id\"]), f\"{row['image_id']}.dcm\")\n\n# Apply the function to each row\ntrain_df[\"image_path\"] = train_df.apply(get_image_path, axis=1)\n\n# Preview the updated DataFrame\nprint(train_df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:47.029642Z","iopub.execute_input":"2025-05-27T13:58:47.029896Z","iopub.status.idle":"2025-05-27T13:58:47.572062Z","shell.execute_reply.started":"2025-05-27T13:58:47.029876Z","shell.execute_reply":"2025-05-27T13:58:47.570857Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_base_path = \"/kaggle/input/rsna-breast-cancer-detection/train_images\"\n\ntrain_df['image_path'] = train_df.apply(\n    lambda row: f\"{image_base_path}/{row['patient_id']}/{row['image_id']}.dcm\", \n    axis=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:47.573359Z","iopub.execute_input":"2025-05-27T13:58:47.573745Z","iopub.status.idle":"2025-05-27T13:58:48.017561Z","shell.execute_reply.started":"2025-05-27T13:58:47.5737Z","shell.execute_reply":"2025-05-27T13:58:48.016355Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data engineering","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\n\ndf = pd.get_dummies(train_df, columns=['laterality', 'view'], drop_first=False)\n\n# Normalize numerical features\nnumerical_cols = ['age']  # add more if needed\nscaler = StandardScaler()\ndf[numerical_cols] = scaler.fit_transform(df[numerical_cols])\n\ntabular_data = df.drop(columns=['site_id','invasive', 'biopsy', 'image_id','difficult_negative_case', 'patient_id', 'image_path', 'cancer', 'BIRADS', 'density'])\ntabular_data = tabular_data.astype(np.float32).values  # 👈 this is key!\nlabels = df['cancer'].values\nlabels = labels.astype(np.float32)  # Binary classification","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:58:48.020265Z","iopub.execute_input":"2025-05-27T13:58:48.020616Z","iopub.status.idle":"2025-05-27T13:59:06.631966Z","shell.execute_reply.started":"2025-05-27T13:58:48.020592Z","shell.execute_reply":"2025-05-27T13:59:06.63087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\n\ndf = pd.get_dummies(train_df, columns=['laterality', 'view'], drop_first=False)\n\n# Normalize numerical features\nnumerical_cols = ['age']  # add more if needed\nscaler = StandardScaler()\ndf[numerical_cols] = scaler.fit_transform(df[numerical_cols])\n\n# Include all positive samples\npositive_df = df[df['cancer'] == 1]\n\n# Sample equal or smaller number of negative examples (or a 1:3 ratio)\nnegative_df = df[df['cancer'] == 0].sample(n=len(positive_df) * 3, random_state=42)\n\n# Combine and shuffle\ndf_balanced = pd.concat([positive_df, negative_df]).sample(frac=1, random_state=42).reset_index(drop=True)\n\ntabular_data = df_balanced.drop(columns=['site_id','invasive', 'biopsy', 'image_id','difficult_negative_case', 'patient_id', 'image_path', 'cancer', 'BIRADS', 'density'])\ntabular_data = tabular_data.astype(np.float32).values  # 👈 this is key!\nlabels = df_balanced['cancer'].astype(np.float32).values\nimage_paths = df_balanced['image_path'].values.astype(str)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:59:06.659598Z","iopub.execute_input":"2025-05-27T13:59:06.660224Z","iopub.status.idle":"2025-05-27T13:59:06.768501Z","shell.execute_reply.started":"2025-05-27T13:59:06.660189Z","shell.execute_reply":"2025-05-27T13:59:06.767155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(tabular_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:59:06.769762Z","iopub.execute_input":"2025-05-27T13:59:06.770123Z","iopub.status.idle":"2025-05-27T13:59:06.777125Z","shell.execute_reply.started":"2025-05-27T13:59:06.770099Z","shell.execute_reply":"2025-05-27T13:59:06.776113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# import pydicom\n# import tensorflow as tf\n# from tensorflow.keras import layers, models, applications, Input\n# from sklearn.preprocessing import StandardScaler\n# from tensorflow.keras.preprocessing import image\n# import numpy as np\n# import pandas as pd\n# from sklearn.model_selection import train_test_split\n# from skimage.transform import resize\n\n# # Function to build the multimodal model\n# def create_multimodal_model(image_shape=(224, 224, 1), tabular_input_dim=4):\n#     # IMAGE INPUT BRANCH\n#     image_input = layers.Input(shape=image_shape, name=\"image_input\")\n\n#     # Convert grayscale image to RGB by using a Conv2D layer with 3 filters\n#     x = layers.Conv2D(3, (1, 1), padding=\"same\")(image_input)  # Convert grayscale (1 channel) to 3 channels\n\n#     # Use EfficientNetB0 without pre-trained weights\n#     base_model = applications.EfficientNetB0(\n#         include_top=False,\n#         weights=None,  # No pre-trained weights\n#         input_tensor=x,  # Use the modified input tensor\n#         pooling=\"avg\"  # Global average pooling\n#     )\n\n#     # Freeze the pre-trained layers (although there are none since we're training from scratch)\n#     base_model.trainable = False\n\n#     # Add custom layers on top of EfficientNetB0\n#     x = base_model.output\n#     x = layers.Dense(256, activation=\"relu\")(x)\n#     x = layers.Dropout(0.5)(x)\n\n#     # TABULAR INPUT BRANCH\n#     tabular_input = layers.Input(shape=(tabular_input_dim,), name=\"tabular_input\")\n#     tabular_x = layers.Dense(64, activation=\"relu\")(tabular_input)\n#     tabular_x = layers.Dropout(0.5)(tabular_x)\n\n#     # Concatenate image and tabular branches\n#     combined = layers.concatenate([x, tabular_x])\n\n#     # Output layer\n#     output = layers.Dense(1, activation=\"sigmoid\")(combined)\n\n#     # Create the model\n#     model = models.Model(inputs=[image_input, tabular_input], outputs=output)\n\n#     return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:59:06.778489Z","iopub.execute_input":"2025-05-27T13:59:06.779347Z","iopub.status.idle":"2025-05-27T13:59:07.766417Z","shell.execute_reply.started":"2025-05-27T13:59:06.779313Z","shell.execute_reply":"2025-05-27T13:59:07.765226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Creating submission.csv file","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport pydicom\nfrom skimage.transform import resize\n\n# ---------- 1. LOAD TEST METADATA ----------\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\n\n# ---------- 2. CREATE IMAGE PATHS ----------\ntest_df[\"image_path\"] = test_df.apply(\n    lambda row: f\"/kaggle/input/rsna-breast-cancer-detection/test_images/{row['patient_id']}/{row['image_id']}.dcm\",\n    axis=1\n)\n\n# ---------- 3. ONE-HOT ENCODE CATEGORICALS ----------\ntest_df = pd.get_dummies(test_df, columns=[\"view\", \"laterality\"], drop_first=False)\n\n# ---------- 4. ENSURE EXPECTED FEATURES ----------\nexpected_features = [\n    \"age\", \"implant\",\n    \"view_CC\", \"view_MLO\", \"view_AT\", \"view_LM\", \"view_LMO\", \"view_ML\",\n    \"laterality_L\", \"laterality_R\"\n]\n\n# Add missing one-hot columns with 0\nfor col in expected_features:\n    if col not in test_df.columns:\n        test_df[col] = 0\n\n# ---------- 5. PREPARE INPUTS ----------\ntabular_features = expected_features  # Already complete\nX_tab_test = test_df[tabular_features].astype(np.float32).values\nX_img_test_paths = test_df[\"image_path\"].values\nprediction_ids = test_df[\"prediction_id\"].values\n\nNUM_TABULAR_FEATURES = X_tab_test.shape[1]\n\n# ---------- 6. LOAD DICOM ----------\ndef load_dicom_image(path):\n    if isinstance(path, tf.Tensor):\n        path = path.numpy().decode(\"utf-8\")\n    dicom = pydicom.dcmread(path)\n    img = dicom.pixel_array.astype(np.float32)\n    img = resize(img, (224, 224), mode='constant', preserve_range=True)\n    img = np.expand_dims(img, axis=-1)\n    img /= 255.0\n    return img\n\ndef preprocess_test(image_path, tabular_row):\n    image = tf.py_function(load_dicom_image, [image_path], tf.float32)\n    image.set_shape([224, 224, 1])\n    tabular_row.set_shape([NUM_TABULAR_FEATURES])\n    return (image, tabular_row)  # <-- This is good\n\n\ndef create_test_dataset(image_paths, tabular_data, batch_size=32):\n    image_paths = tf.constant(image_paths)\n    tabular_data = tf.convert_to_tensor(tabular_data, dtype=tf.float32)\n    ds = tf.data.Dataset.from_tensor_slices((image_paths, tabular_data))\n    ds = ds.map(preprocess_test, num_parallel_calls=tf.data.AUTOTUNE)\n    ds = ds.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n    return ds\n\n# ---------- 8. CREATE TEST DATASET ----------\ntest_ds = create_test_dataset(X_img_test_paths, X_tab_test)\n\n# Debug one batch\nfor images, tabular in test_ds.take(1):\n    print(\"Image batch shape:\", images.shape)   # (batch_size, 224, 224, 1)\n    print(\"Tabular batch shape:\", tabular.shape) # (batch_size, 13)\n\n# ---------- 9. LOAD TRAINED MODEL ----------\nmodel = tf.keras.models.load_model(\"/kaggle/input/best-model-1/best_model (5).keras\")\n\n# ---------- 10. PREDICT ----------\ntest_ds_for_pred = test_ds.map(lambda img, tab: ((img, tab),))  # comma ensures it's a tuple of inputs\npreds = model.predict(test_ds_for_pred, verbose=1).squeeze()\n\n\n# ---------- 11. AGGREGATE ----------\nsubmission_df = pd.DataFrame({\n    \"prediction_id\": prediction_ids,\n    \"cancer\": preds\n})\nsubmission_df = submission_df.groupby(\"prediction_id\", as_index=False).mean()\n\n# ---------- 12. SAVE ----------\nsubmission_df.to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv is saved.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T13:59:07.767535Z","iopub.execute_input":"2025-05-27T13:59:07.769262Z","iopub.status.idle":"2025-05-27T13:59:22.030767Z","shell.execute_reply.started":"2025-05-27T13:59:07.769219Z","shell.execute_reply":"2025-05-27T13:59:22.029788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}