{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":36363,"databundleVersionId":4050810,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# RSNA 2022 Cerical Spine Fracture Detection\n**CSCI217 Project**","metadata":{}},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nimport tensorflow.keras.layers as tfl\nfrom tensorflow.keras import backend as K\nfrom sklearn.model_selection import StratifiedKFold\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.applications import EfficientNetB0\n\nimport os\nimport cv2\nimport glob\nimport pydicom as dicom\nimport nibabel as nib\nimport sys","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-09T12:29:35.86691Z","iopub.execute_input":"2024-05-09T12:29:35.86735Z","iopub.status.idle":"2024-05-09T12:29:35.874442Z","shell.execute_reply.started":"2024-05-09T12:29:35.867317Z","shell.execute_reply":"2024-05-09T12:29:35.873231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"markdown","source":"#### Load dataframes","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train.csv\")\ndf_test = pd.DataFrame({\"row_id\": ['1.2.826.0.1.3680043.22327_C1', '1.2.826.0.1.3680043.25399_C1', '1.2.826.0.1.3680043.5876_C1'], \n                        \"StudyInstanceUID\": ['1.2.826.0.1.3680043.22327', '1.2.826.0.1.3680043.25399', '1.2.826.0.1.3680043.5876'], \n                        \"prediction_type\": [\"C1\", \"C1\", \"C1\"]})  \n\ntrain_images_dir = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'\ntest_images_dir = '/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test_images'\n\nnew_submission = []\nmeans = dict(zip(df_train.columns[1:], np.average(df_train.iloc[:,1:], axis=0, weights=df_train[\"patient_overall\"] + 1)))\nprediction_type = df_test['prediction_type'].tolist()\nsubmission = pd.read_csv('/kaggle/input/rsna-2022-cervical-spine-fracture-detection/sample_submission.csv')\nfor i in range(len(submission)):        \n    new_submission.append(means[prediction_type[i]])\nsubmission['fractured'] = new_submission\n\nprediction_type_mapping = df_test['prediction_type'].map({'C1': 0, 'C2': 1, 'C3': 2, 'C4': 3, 'C5': 4, 'C6': 5, 'C7': 6}).values\n\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:36.900982Z","iopub.execute_input":"2024-05-09T12:29:36.901738Z","iopub.status.idle":"2024-05-09T12:29:36.941153Z","shell.execute_reply.started":"2024-05-09T12:29:36.901704Z","shell.execute_reply":"2024-05-09T12:29:36.939806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Load Dicom Helper Function","metadata":{}},{"cell_type":"code","source":"def load_dicom(path, size = 64):\n    img=dicom.dcmread(path)\n    img.PhotometricInterpretation = 'YBR_FULL'\n    data=img.pixel_array\n    data=data-np.min(data)\n    if np.max(data) != 0:\n        data=data/np.max(data)\n    data=(data*255).astype(np.uint8)        \n    return cv2.cvtColor(data.reshape(512, 512), cv2.COLOR_GRAY2RGB)\n\n    \npatients = sorted(os.listdir(train_images_dir))","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:38.932145Z","iopub.execute_input":"2024-05-09T12:29:38.932538Z","iopub.status.idle":"2024-05-09T12:29:38.941896Z","shell.execute_reply.started":"2024-05-09T12:29:38.932508Z","shell.execute_reply":"2024-05-09T12:29:38.940707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### visualize Images","metadata":{}},{"cell_type":"code","source":"image_file = glob.glob(\"/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images/1.2.826.0.1.3680043.10001/*.dcm\")\nplt.figure(figsize=(20, 10))\n\nfor i in range(16):\n    ax = plt.subplot(4, 4, i + 1)\n    image_path = image_file[i]\n    image = load_dicom(image_path)\n    plt.axis('off')   \n    plt.imshow(image)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:40.966329Z","iopub.execute_input":"2024-05-09T12:29:40.966704Z","iopub.status.idle":"2024-05-09T12:29:42.776784Z","shell.execute_reply.started":"2024-05-09T12:29:40.966675Z","shell.execute_reply":"2024-05-09T12:29:42.775747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Create Data Generator\n(As data is so large that it can't fit in memory)","metadata":{}},{"cell_type":"code","source":"#resizing & normalization","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:42.779065Z","iopub.execute_input":"2024-05-09T12:29:42.779452Z","iopub.status.idle":"2024-05-09T12:29:42.784305Z","shell.execute_reply.started":"2024-05-09T12:29:42.779419Z","shell.execute_reply":"2024-05-09T12:29:42.783203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def RSNATrainGenerator(train_df, batch_size, infinite = True, base_path = train_images_dir):\n    while True:\n        trainset = []\n        trainidt = []\n        trainlabel = []\n        for i in (range(len(train_df))):\n            idt = train_df.loc[i, 'StudyInstanceUID']\n            path = os.path.join(base_path, idt)\n            for im in os.listdir(path):\n                dc = dicom.read_file(os.path.join(path,im))\n                if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n                    continue\n                img = load_dicom(os.path.join(path , im))\n                img = cv2.resize(img, (128 , 128))\n                image = img_to_array(img)\n                image = image / 255.0\n                trainset += [image]\n                cur_label = [train_df.loc[i,f'C{j}'] for j in range(1,8)]\n                trainlabel += [cur_label]\n                trainidt += [idt]\n                if len(trainidt) == batch_size:                    \n                    yield np.array(trainset), np.array(trainlabel)\n                    trainset, trainlabel, trainidt = [], [], []\n            i+=1","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:42.785667Z","iopub.execute_input":"2024-05-09T12:29:42.786056Z","iopub.status.idle":"2024-05-09T12:29:42.798224Z","shell.execute_reply.started":"2024-05-09T12:29:42.786008Z","shell.execute_reply":"2024-05-09T12:29:42.797007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport cv2\nfrom tensorflow.keras.preprocessing.image import img_to_array\nimport pydicom as dicom\n\ndef RSNATrainGenerator(train_df, batch_size, infinite=True, base_path='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/train_images'):\n    while True:\n        trainset = []\n        trainidt = []\n        trainlabel = []\n        for i in range(len(train_df)):\n            idt = train_df.loc[i, 'StudyInstanceUID']\n            path = os.path.join(base_path, idt)\n            for im in os.listdir(path):\n                dc = dicom.read_file(os.path.join(path,im))\n                if dc.file_meta.TransferSyntaxUID.name =='JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n                    continue\n                img = load_dicom(os.path.join(path, im))\n                img = cv2.resize(img, (128, 128))\n                image = img_to_array(img)\n                image = image / 255.0\n                trainset += [image]\n                cur_label = [train_df.loc[i,f'C{j}'] for j in range(1,8)]\n                trainlabel += [cur_label]\n                trainidt += [idt]\n                if len(trainset) == batch_size:                    \n                    yield np.array(trainset), np.array(trainlabel)\n                    trainset, trainlabel, trainidt = [], [], []\n            if not infinite:\n                break\n\n# Ensure this is defined in your environment and try to run your model setup again.\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T11:46:53.809613Z","iopub.execute_input":"2024-05-09T11:46:53.810001Z","iopub.status.idle":"2024-05-09T11:46:53.821614Z","shell.execute_reply.started":"2024-05-09T11:46:53.809974Z","shell.execute_reply":"2024-05-09T11:46:53.820662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def RSNATestGenerator(test_df, batch_size, infinite=True, base_path='/kaggle/input/rsna-2022-cervical-spine-fracture-detection/test_images'):\n    while True:\n        testset = []\n        testidt = []\n        for i in range(len(test_df)):\n            idt = test_df.iloc[i]['StudyInstanceUID'] if isinstance(test_df, pd.DataFrame) else test_df[i]\n            path = os.path.join(base_path, idt)\n            if os.path.exists(path):\n                for im in os.listdir(path):\n                    dc = dicom.read_file(os.path.join(path, im))\n                    if dc.file_meta.TransferSyntaxUID.name == 'JPEG Lossless, Non-Hierarchical, First-Order Prediction (Process 14 [Selection Value 1])':\n                        continue\n                    img = load_dicom(os.path.join(path, im))\n                    img = cv2.resize(img, (128, 128))\n                    image = img_to_array(img)\n                    image = image / 255.0\n                    testset.append(image)\n                    testidt.append(idt)\n                    if len(testset) == batch_size:\n                        yield np.array(testset)\n                        testset = []\n        if len(testset) > 0:  # Yield remaining images if there are any\n            yield np.array(testset)\n        if not infinite:\n            break\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:47.891034Z","iopub.execute_input":"2024-05-09T12:29:47.891438Z","iopub.status.idle":"2024-05-09T12:29:47.902718Z","shell.execute_reply.started":"2024-05-09T12:29:47.891406Z","shell.execute_reply":"2024-05-09T12:29:47.901551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = RSNATrainGenerator(df_train, 64)\nsample = next(train_data)\nprint(\"input_shape:\", sample[0].shape)\nprint(\"target_shape:\", sample[1].shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:49.701311Z","iopub.execute_input":"2024-05-09T12:29:49.701703Z","iopub.status.idle":"2024-05-09T12:29:50.77579Z","shell.execute_reply.started":"2024-05-09T12:29:49.701674Z","shell.execute_reply":"2024-05-09T12:29:50.774602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Model","metadata":{"execution":{"iopub.status.busy":"2022-12-24T08:45:28.342629Z","iopub.execute_input":"2022-12-24T08:45:28.342996Z","iopub.status.idle":"2022-12-24T08:45:28.347812Z","shell.execute_reply.started":"2022-12-24T08:45:28.342963Z","shell.execute_reply":"2022-12-24T08:45:28.346634Z"}}},{"cell_type":"code","source":"###########CNN################################################################################","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:52.62056Z","iopub.execute_input":"2024-05-09T12:29:52.621306Z","iopub.status.idle":"2024-05-09T12:29:52.625937Z","shell.execute_reply.started":"2024-05-09T12:29:52.621266Z","shell.execute_reply":"2024-05-09T12:29:52.62478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow.keras.layers as tfl\n\ndef get_basic_cnn():\n    '''Basic Convolutional Neural Network to train on CT scan images for spine fracture detection.'''\n    \n    inp = tfl.Input(shape=(128, 128, 3))\n    x = tfl.Conv2D(32, (3, 3), activation='relu')(inp)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Conv2D(64, (3, 3), activation='relu')(x)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Conv2D(128, (3, 3), activation='relu')(x)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Flatten()(x)\n    x = tfl.Dense(128, activation='relu')(x)\n    x = tfl.Dropout(0.5)(x)\n    out = tfl.Dense(7, activation='sigmoid')(x)\n    \n    model = tf.keras.models.Model(inputs=inp, outputs=out)\n    \n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n                  loss='binary_crossentropy',\n                  metrics=[tf.keras.metrics.BinaryAccuracy(), tf.keras.metrics.Recall(), tf.keras.metrics.Precision()])\n    \n    return model\n\n# Create the model\nmodel = get_basic_cnn()\n\n# Print the model summary to understand its architecture\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:53.893156Z","iopub.execute_input":"2024-05-09T12:29:53.893861Z","iopub.status.idle":"2024-05-09T12:29:53.99751Z","shell.execute_reply.started":"2024-05-09T12:29:53.893812Z","shell.execute_reply":"2024-05-09T12:29:53.996458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming 'model' is your trained model and 'test_data' is your preprocessed test dataset\npredictions = model.predict(test_data)\n\n# Define a threshold for deciding fractures\nthreshold = 0.5\n\n# Function to interpret and print results\ndef report_fractures(predictions, threshold):\n    segments = ['C1', 'C2', 'C3', 'C4', 'C5', 'C6', 'C7']\n    result = []\n    for i, pred in enumerate(predictions):\n        fractures = [segments[j] for j in range(len(pred)) if pred[j] > threshold]\n        if fractures:\n            result.append(f\"Image {i+1} shows potential fractures at: {', '.join(fractures)}\")\n        else:\n            result.append(f\"Image {i+1} shows no fractures.\")\n    return result\n\n# Generate and print the fracture report\nfracture_report = report_fractures(predictions, threshold)\nfor report in fracture_report:\n    print(report)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-09T12:29:55.484202Z","iopub.execute_input":"2024-05-09T12:29:55.484607Z","iopub.status.idle":"2024-05-09T12:29:55.53575Z","shell.execute_reply.started":"2024-05-09T12:29:55.484567Z","shell.execute_reply":"2024-05-09T12:29:55.53431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####Random forest#############################################################","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:14:21.008182Z","iopub.execute_input":"2024-05-07T16:14:21.008793Z","iopub.status.idle":"2024-05-07T16:14:21.01282Z","shell.execute_reply.started":"2024-05-07T16:14:21.008762Z","shell.execute_reply":"2024-05-07T16:14:21.011825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n# Load the data\ntrain_df = df_train.copy()  # Make a copy of the original dataframe\n\n# Extract features (images) using the existing function\n# Note: You may need to adjust the batch size based on your system's memory\ntrain_data = RSNATrainGenerator(train_df, batch_size=64)\nX_images, y_labels = [], []\n\n# Iterate through the data generator and collect features and labels\nfor _ in range(len(train_df) // 64):  # Adjust the range based on the number of batches\n    images, labels = next(train_data)\n    X_images.extend(images)\n    y_labels.extend(labels)\n\nX_images = np.array(X_images)\ny_labels = np.array(y_labels)\n\n# Flatten the image arrays\nX_images_flat = X_images.reshape(X_images.shape[0], -1)\n\n# Flatten the labels to represent fracture presence (binary classification)\ny_fracture = np.any(y_labels > 0, axis=1).astype(int)\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_images_flat, y_fracture, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:14:21.64227Z","iopub.execute_input":"2024-05-07T16:14:21.642665Z","iopub.status.idle":"2024-05-07T16:14:59.785945Z","shell.execute_reply.started":"2024-05-07T16:14:21.642636Z","shell.execute_reply":"2024-05-07T16:14:59.785104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize and train the Random Forest classifier\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=42)\nrf_classifier.fit(X_train, y_train)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:14:59.787488Z","iopub.execute_input":"2024-05-07T16:14:59.787764Z","iopub.status.idle":"2024-05-07T16:15:11.17629Z","shell.execute_reply.started":"2024-05-07T16:14:59.78774Z","shell.execute_reply":"2024-05-07T16:15:11.175339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict fracture presence on the test set\ny_pred = rf_classifier.predict(X_test)\n\n# Evaluate the classifier\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:15:13.523342Z","iopub.execute_input":"2024-05-07T16:15:13.52374Z","iopub.status.idle":"2024-05-07T16:15:13.560342Z","shell.execute_reply.started":"2024-05-07T16:15:13.523711Z","shell.execute_reply":"2024-05-07T16:15:13.5594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train a Random Forest classifier for each cervical spine level\nrf_classifiers = {}\nfor level in range(1, 8):  # Levels C1 to C7\n    y_labels_level = y_labels[:, level - 1]  # Labels for the specific level\n    \n    # Split the data into train and test sets\n    X_train_level, X_test_level, y_train_level, y_test_level = train_test_split(\n        X_images_flat, y_labels_level, test_size=0.2, random_state=42)\n    \n    # Initialize and train the Random Forest classifier for the level\n    rf_classifier_level = RandomForestClassifier(n_estimators=100, random_state=42)\n    rf_classifier_level.fit(X_train_level, y_train_level)\n    \n    # Store the trained classifier\n    rf_classifiers[f'C{level}'] = rf_classifier_level\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:15:15.582389Z","iopub.execute_input":"2024-05-07T16:15:15.583061Z","iopub.status.idle":"2024-05-07T16:15:39.00289Z","shell.execute_reply.started":"2024-05-07T16:15:15.583029Z","shell.execute_reply":"2024-05-07T16:15:39.002099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict fracture presence and location using the trained classifiers\nfracture_presence = rf_classifier.predict(X_images_flat)\nfracture_location = {}\n\nfor level, rf_classifier_level in rf_classifiers.items():\n    fracture_location[level] = rf_classifier_level.predict(X_images_flat)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:15:48.741337Z","iopub.execute_input":"2024-05-07T16:15:48.741942Z","iopub.status.idle":"2024-05-07T16:15:49.187266Z","shell.execute_reply.started":"2024-05-07T16:15:48.741911Z","shell.execute_reply":"2024-05-07T16:15:49.186492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\n\n# Load the data\ntrain_df = df_train.copy()  # Make a copy of the original dataframe\n\n# Extract features (images) using the existing function\n# Note: You may need to adjust the batch size based on your system's memory\ntrain_data = RSNATrainGenerator(train_df, batch_size=64)\nX_images, y_labels = [], []\n\n# Iterate through the data generator and collect features and labels\nfor _ in range(len(train_df) // 64):  # Adjust the range based on the number of batches\n    images, labels = next(train_data)\n    X_images.extend(images)\n    y_labels.extend(labels)\n\nX_images = np.array(X_images)\ny_labels = np.array(y_labels)\n\n# Flatten the image arrays\nX_images_flat = X_images.reshape(X_images.shape[0], -1)\n\n# Flatten the labels to represent fracture presence (binary classification)\ny_fracture = np.any(y_labels > 0, axis=1).astype(int)\n\n# Split the data into train and test sets\nX_train, X_test, y_train, y_test = train_test_split(X_images_flat, y_fracture, test_size=0.2, random_state=42)\n\n# Initialize and train the Random Forest classifier\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=42)\nrf_classifier.fit(X_train, y_train)\n\n# Predict fracture presence on the test set\ny_pred = rf_classifier.predict(X_test)\n\n# Evaluate the classifier\nprint(\"Classification Report for Fracture Presence:\")\nprint(classification_report(y_test, y_pred))\nprint(\"Confusion Matrix for Fracture Presence:\")\nprint(confusion_matrix(y_test, y_pred))\n\n# Train a Random Forest classifier for each cervical spine level\nrf_classifiers = {}\nfor level in range(1, 8):  # Levels C1 to C7\n    y_labels_level = y_labels[:, level - 1]  # Labels for the specific level\n    \n    # Split the data into train and test sets\n    X_train_level, X_test_level, y_train_level, y_test_level = train_test_split(\n        X_images_flat, y_labels_level, test_size=0.2, random_state=42)\n    \n    # Initialize and train the Random Forest classifier for the level\n    rf_classifier_level = RandomForestClassifier(n_estimators=100, random_state=42)\n    rf_classifier_level.fit(X_train_level, y_train_level)\n    \n    # Store the trained classifier\n    rf_classifiers[f'C{level}'] = rf_classifier_level\n\n# Predict fracture presence and location using the trained classifiers\nfracture_presence = rf_classifier.predict(X_images_flat)\nfracture_location = {}\n\nfor level, rf_classifier_level in rf_classifiers.items():\n    fracture_location[level] = rf_classifier_level.predict(X_images_flat)\n\n# Output predictions\nprint(\"\\nPredictions for Fracture Presence:\")\nprint(fracture_presence)\n\nprint(\"\\nPredictions for Fracture Location:\")\nfor level, location in fracture_location.items():\n    print(f\"{level}: {location}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:15:54.356335Z","iopub.execute_input":"2024-05-07T16:15:54.357248Z","iopub.status.idle":"2024-05-07T16:16:41.005073Z","shell.execute_reply.started":"2024-05-07T16:15:54.357215Z","shell.execute_reply":"2024-05-07T16:16:41.004151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############################################################","metadata":{"execution":{"iopub.status.busy":"2024-05-07T16:16:41.006825Z","iopub.execute_input":"2024-05-07T16:16:41.00753Z","iopub.status.idle":"2024-05-07T16:16:41.011411Z","shell.execute_reply.started":"2024-05-07T16:16:41.007494Z","shell.execute_reply":"2024-05-07T16:16:41.010469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"############Nadeeennn###################################################","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_A1():       \n    eff_model = tf.keras.applications.EfficientNetB0(\n                    include_top=False,\n                    weights=\"imagenet\",\n                    pooling=\"max\")\n    \n    for layer in eff_model.layers[:-10]:\n        layer.trainable = False\n        \n        \n    inp = tfl.Input((128, 128 ,3))\n    x = eff_model(inp)\n    x = tfl.Dense(128, 'relu')(x)\n    x = tfl.Dropout(0.5)(x)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    \n    model = tf.keras.models.Model(inp, out)\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 1e-5),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    return model\n\nget_model_A1()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:25.428161Z","iopub.execute_input":"2023-01-06T18:17:25.428863Z","iopub.status.idle":"2023-01-06T18:17:30.73234Z","shell.execute_reply.started":"2023-01-06T18:17:25.428826Z","shell.execute_reply":"2023-01-06T18:17:30.731165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_A2():\n    inp = tfl.Input((128, 128 ,3))\n    x = tfl.Conv2D(32, (3, 3), activation='relu')(inp)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Conv2D(64, (3, 3), activation='relu')(inp)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Conv2D(128, (3, 3), activation='relu')(inp)\n    x = tfl.MaxPooling2D((2, 2))(x)\n    x = tfl.Flatten()(x)\n    x = tfl.Dense(128, 'relu')(x)\n    x = tfl.Dropout(0.5)(x)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    \n    model = tf.keras.models.Model(inp, out)\n        \n    model.compile(loss=\"binary_crossentropy\",\n                  optimizer = tf.keras.optimizers.Adam(learning_rate = 1e-4),\n                  metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    \n    return model\n\nget_model_A2()","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:30.734774Z","iopub.execute_input":"2023-01-06T18:17:30.735703Z","iopub.status.idle":"2023-01-06T18:17:30.80336Z","shell.execute_reply.started":"2023-01-06T18:17:30.735674Z","shell.execute_reply":"2023-01-06T18:17:30.802425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_M1():       \n    MobileNet_model = tf.keras.applications.MobileNet(\n    include_top=False,\n    weights='imagenet',\n    pooling=\"max\",\n)\n    \n    for layer in MobileNet_model.layers[:-10]:\n        layer.trainable = False\n        \n        \n    inp = tfl.Input((128, 128 ,3))\n    x = MobileNet_model(inp)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    model = tf.keras.models.Model(inp, out)\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 0.0001),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:30.804895Z","iopub.execute_input":"2023-01-06T18:17:30.80525Z","iopub.status.idle":"2023-01-06T18:17:30.812325Z","shell.execute_reply.started":"2023-01-06T18:17:30.805216Z","shell.execute_reply":"2023-01-06T18:17:30.811117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_M2():\n    DenseNet_model = tf.keras.applications.DenseNet121(\n    include_top=False,\n    weights='imagenet',\n    pooling=\"max\",\n)\n    \n    for layer in DenseNet_model.layers[:-10]:\n        layer.trainable = False\n        \n        \n    inp = tfl.Input((128, 128 ,3))\n    x = DenseNet_model(inp)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    model = tf.keras.models.Model(inp, out)\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 0.0001),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:31.050355Z","iopub.execute_input":"2023-01-06T18:17:31.05098Z","iopub.status.idle":"2023-01-06T18:17:31.059024Z","shell.execute_reply.started":"2023-01-06T18:17:31.050936Z","shell.execute_reply":"2023-01-06T18:17:31.057967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_O1():       \n    eff_model = tf.keras.applications.Xception(\n                    include_top=False,\n                    weights=\"imagenet\",\n                    pooling=\"max\"\n                    )\n    \n    for layer in eff_model.layers[:-10]:\n        layer.trainable = False\n        \n        \n    inp = tfl.Input((128, 128 ,3))\n    x = eff_model(inp)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    model = tf.keras.models.Model(inp, out)\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 0.0001),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:31.656382Z","iopub.execute_input":"2023-01-06T18:17:31.65705Z","iopub.status.idle":"2023-01-06T18:17:31.66426Z","shell.execute_reply.started":"2023-01-06T18:17:31.657016Z","shell.execute_reply":"2023-01-06T18:17:31.663074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nfrom tensorflow.keras import layers\ndef get_model_O2():\n    model = keras.Sequential([\n        layers.Dense(128, activation='relu', input_shape=[128,128,3]),\n        layers.Dropout(0.4),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.6),\n        layers.Dense(64, activation='relu'),\n        layers.Flatten(),\n        layers.Dense(7, activation='sigmoid'),\n    ])\n    \n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 1e-4),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:17:32.192895Z","iopub.execute_input":"2023-01-06T18:17:32.193386Z","iopub.status.idle":"2023-01-06T18:17:32.20885Z","shell.execute_reply.started":"2023-01-06T18:17:32.193344Z","shell.execute_reply":"2023-01-06T18:17:32.207923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_K1():       \n    eff_model = tf.keras.applications.InceptionV3(\n                    include_top=False,\n                    weights=\"imagenet\",\n                    pooling=\"max\"\n                    )\n    \n    for layer in eff_model.layers[:-10]:\n        layer.trainable = False\n        \n        \n    inp = tfl.Input((128, 128 ,3))\n    x = eff_model(inp)\n    # x = tfl.Conv2D(3, 3, padding = 'SAME')(x)\n    out = tfl.Dense(7, 'sigmoid')(x)\n    model = tf.keras.models.Model(inp, out)\n    model.layers[2].trainable = False\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 0.0001),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T18:27:56.331723Z","iopub.execute_input":"2023-01-06T18:27:56.332096Z","iopub.status.idle":"2023-01-06T18:27:56.341708Z","shell.execute_reply.started":"2023-01-06T18:27:56.332057Z","shell.execute_reply":"2023-01-06T18:27:56.340743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_K2():\n    num_classes = 7\n    image_size = 128\n    model = keras.Sequential([\n                    layers.experimental.preprocessing.Rescaling(1./255, input_shape=(image_size, image_size, 3)),\n                    layers.Conv2D(16, 3, padding='same', activation='relu'),\n                    layers.MaxPooling2D(),\n                    layers.Conv2D(32, 3, padding='same', activation='relu'),\n                    layers.MaxPooling2D(),\n                    layers.Conv2D(64, 3, padding='same', activation='relu'),\n                    layers.MaxPooling2D(),\n                    layers.Flatten(),\n                    layers.Dense(128, activation='relu'),\n                    layers.Dense(64, activation='relu'),\n                    layers.Dense(num_classes)])\n    model.compile(loss=\"binary_crossentropy\", optimizer = tf.keras.optimizers.Adam(learning_rate = 1e-4),\n                 metrics=[tf.keras.metrics.BinaryAccuracy()])\n    model.summary()\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T20:44:00.616811Z","iopub.execute_input":"2023-01-06T20:44:00.617211Z","iopub.status.idle":"2023-01-06T20:44:00.625188Z","shell.execute_reply.started":"2023-01-06T20:44:00.61718Z","shell.execute_reply":"2023-01-06T20:44:00.624182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_N1():\n    base_model = keras.applications.ResNet50(\n    weights='imagenet',  # Load weights pre-trained on ImageNet.\n    input_shape=(128, 128, 3),\n    include_top=False)  # include the ImageNet classifier at the top.\n    base_model.trainable = False\n\n    model = tf.keras.models.Sequential()\n    model.add(base_model)\n    model.add(tf.keras.layers.Flatten())\n#     model.add(tf.keras.layers.Dropout(0.5))\n    model.add(tf.keras.layers.Dense(128, activation='relu'))\n    model.add(tf.keras.layers.Dense(64, activation='relu'))\n    model.add(tf.keras.layers.Dense(32, activation='relu'))\n    model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n\n#     model.layers[0].trainable = False\n    \n    model.compile(\n        loss='binary_crossentropy',\n        optimizer=tf.keras.optimizers.Adam(learning_rate = 1e-4),\n        metrics=[tf.keras.metrics.BinaryAccuracy()]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T21:26:42.163242Z","iopub.execute_input":"2023-01-06T21:26:42.164368Z","iopub.status.idle":"2023-01-06T21:26:42.174246Z","shell.execute_reply.started":"2023-01-06T21:26:42.16431Z","shell.execute_reply":"2023-01-06T21:26:42.172987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_N2():\n    base_model = keras.applications.VGG16(\n    weights='imagenet',  # Load weights pre-trained on ImageNet.\n    input_shape=(128, 128, 3),\n    include_top=False)  # Do not include the ImageNet classifier at the top.\n    base_model.trainable = False\n\n    model = tf.keras.models.Sequential()\n    model.add(base_model)\n    model.add(tf.keras.layers.Flatten())\n#     model.add(tf.keras.layers.Dropout(0.5))\n    model.add(tf.keras.layers.Dense(128, activation='relu'))\n    model.add(tf.keras.layers.Dense(64, activation='relu'))\n    model.add(tf.keras.layers.Dense(32, activation='relu'))\n    model.add(tf.keras.layers.Dense(1, activation='sigmoid'))\n\n#     model.layers[0].trainable = False\n    \n    model.compile(\n        loss='binary_crossentropy',\n        optimizer=tf.keras.optimizers.Adam(learning_rate = 1e-4),\n        metrics=[tf.keras.metrics.BinaryAccuracy()]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-01-06T21:26:42.501959Z","iopub.execute_input":"2023-01-06T21:26:42.502294Z","iopub.status.idle":"2023-01-06T21:26:42.510576Z","shell.execute_reply.started":"2023-01-06T21:26:42.502246Z","shell.execute_reply":"2023-01-06T21:26:42.509472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"functions = [('Abdallah', get_model_A1, get_model_A2), ('Mohammad', get_model_M1, get_model_M2), ('Omar Ahmed', get_model_O1, get_model_O2), ('Omar Khaled', get_model_K1, get_model_K2), ('Nadeen', get_model_N1, get_model_N2)]","metadata":{"execution":{"iopub.status.busy":"2023-01-06T21:27:09.831545Z","iopub.execute_input":"2023-01-06T21:27:09.831927Z","iopub.status.idle":"2023-01-06T21:27:09.838107Z","shell.execute_reply.started":"2023-01-06T21:27:09.831894Z","shell.execute_reply":"2023-01-06T21:27:09.836975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracies = []\nfor person in functions:\n    print(person[0])\n    for train_idx, val_idx in StratifiedKFold(5).split(df_train, df_train['patient_overall']):    \n        K.clear_session()\n        x_train = df_train.iloc[train_idx].reset_index()\n        x_val = df_train.iloc[val_idx].reset_index()\n\n        train_gen = RSNATrainGenerator(x_train, min(len(x_train), 64), infinite = False, base_path = train_images_dir)\n        val_gen = RSNATrainGenerator(x_val, min(len(x_val), 64), infinite = False, base_path = train_images_dir)\n\n        model1 = person[1]()\n        model2 = person[2]()\n\n\n        hist1 = model1.fit_generator(                            \n            train_gen,\n            epochs = 5,\n            callbacks = [tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', patience = 2, restore_best_weights = True)],\n            validation_steps = max((len(x_val) // 64), 1),\n            steps_per_epoch = max((len(x_train) // 64), 1),\n            validation_data = val_gen,\n          )\n\n        hist2 = model2.fit_generator(                            \n            train_gen,\n            epochs = 5,\n            callbacks = [tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', patience = 2, restore_best_weights = True)],\n            validation_steps = max((len(x_val) // 64), 1),\n            steps_per_epoch = max((len(x_train) // 64), 1),\n            validation_data = val_gen,\n          )\n        accuracies.append(model1.evaluate(val_gen, steps = max((len(x_val) // 64), 1))[1])\n        accuracies.append(model2.evaluate(val_gen, steps = max((len(x_val) // 64), 1))[1])\n        try: # the best we can do at the moment..\n            preds1 = model1.predict_generator(RSNATestGenerator(df_test, min(len(df_test), 64), infinite = False, base_path = test_images_dir), steps = max((len(df_test) // 64), 1))\n            preds2 = model2.predict_generator(RSNATestGenerator(df_test, min(len(df_test), 64), infinite = False, base_path = test_images_dir), steps = max((len(df_test) // 64), 1))\n\n            new_preds = []\n            for pred_idx in range(len(preds1)):\n                new_preds.append(preds1[pred_idx][prediction_type_mapping[pred_idx]])\n            submission['fractured'] += np.array(new_preds) / 10\n\n            new_preds = []\n            for pred_idx in range(len(preds2)):\n                new_preds.append(preds2[pred_idx][prediction_type_mapping[pred_idx]])\n            submission['fractured'] += np.array(new_preds) / 10\n\n        except: traceback.print_exc()    ","metadata":{"execution":{"iopub.status.busy":"2023-01-06T21:27:13.680226Z","iopub.execute_input":"2023-01-06T21:27:13.680699Z","iopub.status.idle":"2023-01-06T21:29:19.034192Z","shell.execute_reply.started":"2023-01-06T21:27:13.680665Z","shell.execute_reply":"2023-01-06T21:29:19.031597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(accuracies).mean()","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:20.638153Z","iopub.execute_input":"2024-05-04T23:09:20.638889Z","iopub.status.idle":"2024-05-04T23:09:20.675995Z","shell.execute_reply.started":"2024-05-04T23:09:20.638847Z","shell.execute_reply":"2024-05-04T23:09:20.674818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:24.566154Z","iopub.execute_input":"2024-05-04T23:09:24.566536Z","iopub.status.idle":"2024-05-04T23:09:24.576822Z","shell.execute_reply.started":"2024-05-04T23:09:24.566506Z","shell.execute_reply":"2024-05-04T23:09:24.575867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index = 0)","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:26.962519Z","iopub.execute_input":"2024-05-04T23:09:26.963428Z","iopub.status.idle":"2024-05-04T23:09:26.970504Z","shell.execute_reply.started":"2024-05-04T23:09:26.963392Z","shell.execute_reply":"2024-05-04T23:09:26.969617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_preds","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:28.96349Z","iopub.execute_input":"2024-05-04T23:09:28.963839Z","iopub.status.idle":"2024-05-04T23:09:28.999505Z","shell.execute_reply.started":"2024-05-04T23:09:28.963812Z","shell.execute_reply":"2024-05-04T23:09:28.998383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:30.698145Z","iopub.execute_input":"2024-05-04T23:09:30.69853Z","iopub.status.idle":"2024-05-04T23:09:30.70809Z","shell.execute_reply.started":"2024-05-04T23:09:30.698499Z","shell.execute_reply":"2024-05-04T23:09:30.707142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.array(new_preds) / 5","metadata":{"execution":{"iopub.status.busy":"2024-05-04T23:09:31.974763Z","iopub.execute_input":"2024-05-04T23:09:31.975497Z","iopub.status.idle":"2024-05-04T23:09:32.010624Z","shell.execute_reply.started":"2024-05-04T23:09:31.975462Z","shell.execute_reply":"2024-05-04T23:09:32.009456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}