{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd  \nimport seaborn as sns \nimport matplotlib.pyplot as plt  \n\nimport pandas.api.types\nimport sklearn.metrics  \n\nimport nibabel as nib\n\nimport os\nimport pydicom\nfrom glob import glob\nfrom tqdm import tqdm, trange","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:34.954695Z","iopub.execute_input":"2023-09-24T17:48:34.955023Z","iopub.status.idle":"2023-09-24T17:48:36.958899Z","shell.execute_reply.started":"2023-09-24T17:48:34.954995Z","shell.execute_reply":"2023-09-24T17:48:36.956923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:36.960754Z","iopub.execute_input":"2023-09-24T17:48:36.966728Z","iopub.status.idle":"2023-09-24T17:48:36.986274Z","shell.execute_reply.started":"2023-09-24T17:48:36.966683Z","shell.execute_reply":"2023-09-24T17:48:36.985303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nTarget_cols = [\"bowel_healthy\", \"bowel_injury\", \"extravasation_healthy\",\n                   \"extravasation_injury\", \"kidney_healthy\", \"kidney_low\",\n                   \"kidney_high\", \"liver_healthy\", \"liver_low\", \"liver_high\",\n                   \"spleen_healthy\", \"spleen_low\", \"spleen_high\",\"any_injury\"]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:36.991029Z","iopub.execute_input":"2023-09-24T17:48:36.99174Z","iopub.status.idle":"2023-09-24T17:48:37.001748Z","shell.execute_reply.started":"2023-09-24T17:48:36.991702Z","shell.execute_reply":"2023-09-24T17:48:37.000604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reading train.csv**","metadata":{}},{"cell_type":"code","source":"file_path = \"/kaggle/input/rsna-2023-abdominal-trauma-detection\"\ntrain_csv=f\"{file_path}/train.csv\" \ntrain=pd.read_csv(train_csv) \ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.006658Z","iopub.execute_input":"2023-09-24T17:48:37.006975Z","iopub.status.idle":"2023-09-24T17:48:37.051547Z","shell.execute_reply.started":"2023-09-24T17:48:37.006951Z","shell.execute_reply":"2023-09-24T17:48:37.050596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.isnull().sum())\n\nprint(train.shape)\n\nprint(train.info())","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.055894Z","iopub.execute_input":"2023-09-24T17:48:37.058666Z","iopub.status.idle":"2023-09-24T17:48:37.086765Z","shell.execute_reply.started":"2023-09-24T17:48:37.058621Z","shell.execute_reply":"2023-09-24T17:48:37.085596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"No missing values in data","metadata":{}},{"cell_type":"code","source":"# statstical analysis:\nf = train.describe()  \nf.iloc[1]","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.091129Z","iopub.execute_input":"2023-09-24T17:48:37.091605Z","iopub.status.idle":"2023-09-24T17:48:37.185791Z","shell.execute_reply.started":"2023-09-24T17:48:37.091569Z","shell.execute_reply":"2023-09-24T17:48:37.184589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.apply(np.max)","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.187134Z","iopub.execute_input":"2023-09-24T17:48:37.188187Z","iopub.status.idle":"2023-09-24T17:48:37.203596Z","shell.execute_reply.started":"2023-09-24T17:48:37.188149Z","shell.execute_reply":"2023-09-24T17:48:37.2021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"organs_healthy = [ \n          'bowel_healthy',\n          'extravasation_healthy',\n          'kidney_healthy',\n          'liver_healthy',\n          'spleen_healthy'\n] \n#calculate and plot \ncorr_matrix1=train[organs_healthy].corr()\nsns.heatmap(corr_matrix1,annot=True); \nplt.title('Correlation Heatmap for healthy organs')","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.208659Z","iopub.execute_input":"2023-09-24T17:48:37.210173Z","iopub.status.idle":"2023-09-24T17:48:37.84644Z","shell.execute_reply.started":"2023-09-24T17:48:37.210136Z","shell.execute_reply":"2023-09-24T17:48:37.845541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# similarlly calculate for low and high addominal trauma disease in healthy organs  \n\nlow_high = [\n            'bowel_injury',\n            'extravasation_injury',\n            'kidney_low',\n            'kidney_high',\n            'liver_high',\n            'liver_low' , \n            'spleen_low',\n            'spleen_high',\n            'any_injury'\n] \n#calculate and plot \ncorr_matrix2 = train[low_high].corr()\nsns.heatmap(corr_matrix2,annot=True , linewidths=1) \nplt.title('Correlation heatmap for injury organs')","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:37.848066Z","iopub.execute_input":"2023-09-24T17:48:37.849067Z","iopub.status.idle":"2023-09-24T17:48:38.742196Z","shell.execute_reply.started":"2023-09-24T17:48:37.84903Z","shell.execute_reply":"2023-09-24T17:48:38.741223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So We Can easily compare correaltion of any organ with any_injury as like:\n\n1.liver:\n\n liver_low with any_injury  :0.49 \n liver_high with any_injury:0.23\n \n2.kidney: \n \n kidney_low with any_injury : 0.32\n kidney_high with any_injury:0.24 \n \n3.spleen: \n  \n  spleen_low with any_injury: 0.43\n  spleen_high with any_injury:0.37 \n  \n4.bowel_injury and any_injury: 0.24\n\n5.extravasation_injury and any_injury:0.43\n  ","metadata":{}},{"cell_type":"markdown","source":"**Reading: train_series_meta.csv**","metadata":{}},{"cell_type":"code","source":"train_series_meta = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train_series_meta.csv')\ntrain_series_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.746913Z","iopub.execute_input":"2023-09-24T17:48:38.749254Z","iopub.status.idle":"2023-09-24T17:48:38.779167Z","shell.execute_reply.started":"2023-09-24T17:48:38.749216Z","shell.execute_reply":"2023-09-24T17:48:38.778172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_series_meta.info())\n\nprint(train_series_meta.isnull().sum())  \n\n# Statistical Analysis  \nf= train_series_meta.describe()\nf.iloc[1]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.780659Z","iopub.execute_input":"2023-09-24T17:48:38.781281Z","iopub.status.idle":"2023-09-24T17:48:38.814821Z","shell.execute_reply.started":"2023-09-24T17:48:38.781247Z","shell.execute_reply":"2023-09-24T17:48:38.813797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_series_meta=pd.read_csv(\"/kaggle/input/rsna-2023-abdominal-trauma-detection/test_series_meta.csv\") \ntest_series_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.816248Z","iopub.execute_input":"2023-09-24T17:48:38.817907Z","iopub.status.idle":"2023-09-24T17:48:38.841187Z","shell.execute_reply.started":"2023-09-24T17:48:38.817871Z","shell.execute_reply":"2023-09-24T17:48:38.840204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_series_meta.isnull().sum()) \n\nf=test_series_meta.describe()\nf.iloc[1]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.842655Z","iopub.execute_input":"2023-09-24T17:48:38.843322Z","iopub.status.idle":"2023-09-24T17:48:38.87387Z","shell.execute_reply.started":"2023-09-24T17:48:38.843287Z","shell.execute_reply":"2023-09-24T17:48:38.872707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Reading image_level_labels.csv**","metadata":{}},{"cell_type":"code","source":"image_labels = pd.read_csv( \"/kaggle/input/rsna-2023-abdominal-trauma-detection/image_level_labels.csv\")\nimage_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.875033Z","iopub.execute_input":"2023-09-24T17:48:38.875288Z","iopub.status.idle":"2023-09-24T17:48:38.913612Z","shell.execute_reply.started":"2023-09-24T17:48:38.875265Z","shell.execute_reply":"2023-09-24T17:48:38.911588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_labels.info()) \n\nprint(image_labels.isnull().sum())\n\n# Statistical Analysis\nf=image_labels.describe()\nf.iloc[1]","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.915083Z","iopub.execute_input":"2023-09-24T17:48:38.915835Z","iopub.status.idle":"2023-09-24T17:48:38.957257Z","shell.execute_reply.started":"2023-09-24T17:48:38.915801Z","shell.execute_reply":"2023-09-24T17:48:38.956365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Training Model****","metadata":{}},{"cell_type":"markdown","source":"**Weighted Mean Baseline**","metadata":{}},{"cell_type":"markdown","source":"**Goal**: Create a simple baseline using a constant prediction for all test sample. This will come from the mean value of each target variable scaled by proposed sample weights:\n\na. 1 for all healthy labels.\n\nb. 2 for low grade solid organ injuries (liver, spleen, kidney).\n\nc. 4 for high grade solid organ injuries.\n\nd. 2 for bowel injuries.\n\ne. 6 for extravasation.\n\nf. 6 for the auto-generated any_injury label.\n\nIn addition, we provide a method to evaluate the score for the training data and investigate alternative scale factors. This will be used to highlight the challenges of unbalanced data and weighted scoring metrics.","metadata":{}},{"cell_type":"code","source":"train[Target_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:38.958812Z","iopub.execute_input":"2023-09-24T17:48:38.95984Z","iopub.status.idle":"2023-09-24T17:48:39.036946Z","shell.execute_reply.started":"2023-09-24T17:48:38.959807Z","shell.execute_reply":"2023-09-24T17:48:39.0349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass ParticipantVisibleError(Exception):\n    pass\n\ndef normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    # Normalize the sum of each row's probabilities to 100%.\n    # 0.75, 0.75 => 0.5, 0.5\n    # 0.1, 0.1 => 0.5, 0.5\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    '''\n    Pseudocode:\n    1. For every label group (liver, bowel, etc):\n        - Normalize the sum of each row's probabilities to 100%.\n        - Calculate the sample weighted log loss.\n    2. Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    3. Calculate the sample weighted log loss for the new label group\n    4. Return the average of all of the label group log losses as the final score.\n    '''\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n    # Run basic QC checks on the inputs\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    # Calculate the label group log losses\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n            \n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n        \n    # Derive a new any_injury label by taking the max of 1 - p(healthy) for each label group\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses) \n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:39.041112Z","iopub.execute_input":"2023-09-24T17:48:39.04144Z","iopub.status.idle":"2023-09-24T17:48:39.064739Z","shell.execute_reply.started":"2023-09-24T17:48:39.041414Z","shell.execute_reply":"2023-09-24T17:48:39.063454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign the appropriate weights to each category\ndef create_training_solution(y_train):\n    sol_train = y_train.copy()\n    \n    # bowel healthy|injury sample weight = 1|2/1\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    \n    # extravasation healthy/injury sample weight = 1|6/1\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    \n    # kidney healthy|low|high sample weight = 1|2|4\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    \n    # liver healthy|low|high sample weight = 1|2|4\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    \n    # spleen healthy|low|high sample weight = 1|2|4\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    \n    # any healthy|injury sample weight = 1|6/1\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    return sol_train","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:39.066193Z","iopub.execute_input":"2023-09-24T17:48:39.066603Z","iopub.status.idle":"2023-09-24T17:48:39.083883Z","shell.execute_reply.started":"2023-09-24T17:48:39.066565Z","shell.execute_reply":"2023-09-24T17:48:39.082589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_train = create_training_solution(train)\n\n# predict a constant using the mean of the training data\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\nno_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score without scaling: {no_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:48:39.085745Z","iopub.execute_input":"2023-09-24T17:48:39.086842Z","iopub.status.idle":"2023-09-24T17:48:39.192848Z","shell.execute_reply.started":"2023-09-24T17:48:39.086808Z","shell.execute_reply":"2023-09-24T17:48:39.191692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by different sample weights\nscale_by_2 = ['kidney_low','liver_low','spleen_low','spleen_high']\nscale_by_4 = ['bowel_injury','kidney_high','liver_high']\nscale_by_6 = ['extravasation_injury','any_injury']\nscale_healthy = ['bowel_healthy', 'extravasation_healthy', 'kidney_healthy', 'liver_healthy', 'spleen_healthy']\n\nsf_2 = 2.8461531332\nsf_4 = 4.841531\nsf_6 = 20.81635153\nscale_h = 0.99519515313\n\n# The score function deletes the ID column so we remake it\nsolution_train = create_training_solution(train)\n\n# Reset the prediction\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each target \ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6\ny_pred[scale_healthy] *=scale_h\n\nweight_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with weight scaling: {weight_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:49:43.397844Z","iopub.execute_input":"2023-09-24T17:49:43.398213Z","iopub.status.idle":"2023-09-24T17:49:43.474646Z","shell.execute_reply.started":"2023-09-24T17:49:43.398184Z","shell.execute_reply":"2023-09-24T17:49:43.473562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update scale factors to improve score\n# sf_2 = 4\n# sf_4 = 6\n# sf_6 = 28\n\n# The score function deletes the ID column so we remake it\nsolution_train = create_training_solution(train)\n\n# Reset the prediction, again\ny_pred = train.copy()\ny_pred[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each target \ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6 \ny_pred[scale_healthy] *=scale_h\n\nimproved_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with better scaling: {improved_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:49:44.904355Z","iopub.execute_input":"2023-09-24T17:49:44.905459Z","iopub.status.idle":"2023-09-24T17:49:44.993482Z","shell.execute_reply.started":"2023-09-24T17:49:44.905417Z","shell.execute_reply":"2023-09-24T17:49:44.992394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')\n\n# Set output to mean of training data\nsubmission[Target_cols] = train[Target_cols].mean().tolist()\n\n# Scale each category by desired scale factor\nsubmission[scale_by_2] *=sf_2\nsubmission[scale_by_4] *=sf_4\nsubmission[scale_by_6] *=sf_6 \nsubmission[scale_healthy] *=scale_h\n\n# Save Submission!\nsubmission.to_csv('submission.csv', index=False) \n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T17:49:46.378133Z","iopub.execute_input":"2023-09-24T17:49:46.378685Z","iopub.status.idle":"2023-09-24T17:49:46.40137Z","shell.execute_reply.started":"2023-09-24T17:49:46.378643Z","shell.execute_reply":"2023-09-24T17:49:46.400437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}