{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pandas.api.types\nimport sklearn.metrics","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-05T12:13:34.674949Z","iopub.execute_input":"2023-09-05T12:13:34.676133Z","iopub.status.idle":"2023-09-05T12:13:34.68017Z","shell.execute_reply.started":"2023-09-05T12:13:34.676084Z","shell.execute_reply":"2023-09-05T12:13:34.679257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/train.csv')\ny_train.head()\nInjuries = ['bowel_healthy', 'bowel_injury', \n            'extravasation_healthy', 'extravasation_injury', \n            'kidney_healthy', 'kidney_low', 'kidney_high', \n            'liver_healthy', 'liver_low', 'liver_high', \n            'spleen_healthy', 'spleen_low', 'spleen_high', \n            'any_injury']","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.681841Z","iopub.execute_input":"2023-09-05T12:13:34.682311Z","iopub.status.idle":"2023-09-05T12:13:34.701387Z","shell.execute_reply.started":"2023-09-05T12:13:34.682285Z","shell.execute_reply":"2023-09-05T12:13:34.700252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalize_probabilities_to_one(df: pd.DataFrame, group_columns: list) -> pd.DataFrame:\n    row_totals = df[group_columns].sum(axis=1)\n    if row_totals.min() == 0:\n        raise ParticipantVisibleError('All rows must contain at least one non-zero prediction')\n    for col in group_columns:\n        df[col] /= row_totals\n    return df\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    del solution[row_id_column_name]\n    del submission[row_id_column_name]\n\n    if not pandas.api.types.is_numeric_dtype(submission.values):\n        raise ParticipantVisibleError('All submission values must be numeric')\n\n    if not np.isfinite(submission.values).all():\n        raise ParticipantVisibleError('All submission values must be finite')\n\n    if solution.min().min() < 0:\n        raise ParticipantVisibleError('All labels must be at least zero')\n    if submission.min().min() < 0:\n        raise ParticipantVisibleError('All predictions must be at least zero')\n\n    binary_targets = ['bowel', 'extravasation']\n    triple_level_targets = ['kidney', 'liver', 'spleen']\n    all_target_categories = binary_targets + triple_level_targets\n\n    label_group_losses = []\n    for category in all_target_categories:\n        if category in binary_targets:\n            col_group = [f'{category}_healthy', f'{category}_injury']\n        else:\n            col_group = [f'{category}_healthy', f'{category}_low', f'{category}_high']\n\n        solution = normalize_probabilities_to_one(solution, col_group)\n\n        for col in col_group:\n            if col not in submission.columns:\n                raise ParticipantVisibleError(f'Missing submission column {col}')\n        submission = normalize_probabilities_to_one(submission, col_group)\n        label_group_losses.append(\n            sklearn.metrics.log_loss(\n                y_true=solution[col_group].values,\n                y_pred=submission[col_group].values,\n                sample_weight=solution[f'{category}_weight'].values\n            )\n        )\n\n    healthy_cols = [x + '_healthy' for x in all_target_categories]\n    any_injury_labels = (1 - solution[healthy_cols]).max(axis=1)\n    any_injury_predictions = (1 - submission[healthy_cols]).max(axis=1)\n    any_injury_loss = sklearn.metrics.log_loss(\n        y_true=any_injury_labels.values,\n        y_pred=any_injury_predictions.values,\n        sample_weight=solution['any_injury_weight'].values\n    )\n\n    label_group_losses.append(any_injury_loss)\n    return np.mean(label_group_losses)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.751644Z","iopub.execute_input":"2023-09-05T12:13:34.751984Z","iopub.status.idle":"2023-09-05T12:13:34.764329Z","shell.execute_reply.started":"2023-09-05T12:13:34.751952Z","shell.execute_reply":"2023-09-05T12:13:34.763431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_training_solution(y_train):\n    sol_train = y_train.copy()\n    sol_train['bowel_weight'] = np.where(sol_train['bowel_injury'] == 1, 2, 1)\n    sol_train['extravasation_weight'] = np.where(sol_train['extravasation_injury'] == 1, 6, 1)\n    sol_train['kidney_weight'] = np.where(sol_train['kidney_low'] == 1, 2, np.where(sol_train['kidney_high'] == 1, 4, 1))\n    sol_train['liver_weight'] = np.where(sol_train['liver_low'] == 1, 2, np.where(sol_train['liver_high'] == 1, 4, 1))\n    sol_train['spleen_weight'] = np.where(sol_train['spleen_low'] == 1, 2, np.where(sol_train['spleen_high'] == 1, 4, 1))\n    sol_train['any_injury_weight'] = np.where(sol_train['any_injury'] == 1, 6, 1)\n    return sol_train","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.766423Z","iopub.execute_input":"2023-09-05T12:13:34.771072Z","iopub.status.idle":"2023-09-05T12:13:34.780826Z","shell.execute_reply.started":"2023-09-05T12:13:34.77104Z","shell.execute_reply":"2023-09-05T12:13:34.779519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_train = create_training_solution(y_train)\n\ny_pred = y_train.copy()\ny_pred[Injuries] = y_train[Injuries].mean().tolist()\n\nno_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score without scaling: {no_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.781853Z","iopub.execute_input":"2023-09-05T12:13:34.782072Z","iopub.status.idle":"2023-09-05T12:13:34.848895Z","shell.execute_reply.started":"2023-09-05T12:13:34.782053Z","shell.execute_reply":"2023-09-05T12:13:34.847702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scale_by_2 = ['kidney_low','liver_low','spleen_low','spleen_high']\nscale_by_4 = ['bowel_injury','kidney_high','liver_high']\nscale_by_6 = ['extravasation_injury','any_injury']\nscale_healthy = ['bowel_healthy', 'extravasation_healthy', 'kidney_healthy', 'liver_healthy', 'spleen_healthy']\nsf_2 = 2.8461531332\nsf_4 = 4.841531\nsf_6 = 20.81635153\nscale_h = 0.99519515313\n\nsolution_train = create_training_solution(y_train)\n\ny_pred = y_train.copy()\ny_pred[Injuries] = y_train[Injuries].mean().tolist()\n\ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6\ny_pred[scale_healthy] *=scale_h\n\nweight_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with weight scaling: {weight_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.85027Z","iopub.execute_input":"2023-09-05T12:13:34.850529Z","iopub.status.idle":"2023-09-05T12:13:34.919068Z","shell.execute_reply.started":"2023-09-05T12:13:34.850507Z","shell.execute_reply":"2023-09-05T12:13:34.917604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"solution_train = create_training_solution(y_train)\n\ny_pred = y_train.copy()\ny_pred[Injuries] = y_train[Injuries].mean().tolist()\n\n# Scale each target \ny_pred[scale_by_2] *=sf_2\ny_pred[scale_by_4] *=sf_4\ny_pred[scale_by_6] *=sf_6\ny_pred[scale_healthy] *=scale_h\n\nimproved_scale_score = score(solution_train,y_pred,'patient_id')\nprint(f'Training score with better scaling: {improved_scale_score}')","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.921637Z","iopub.execute_input":"2023-09-05T12:13:34.922342Z","iopub.status.idle":"2023-09-05T12:13:34.987677Z","shell.execute_reply.started":"2023-09-05T12:13:34.922307Z","shell.execute_reply":"2023-09-05T12:13:34.986797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/rsna-2023-abdominal-trauma-detection/sample_submission.csv')\n\nsubmission[Injuries] = y_train[Injuries].mean().tolist()\n\nsubmission[scale_by_2] *=sf_2\nsubmission[scale_by_4] *=sf_4\nsubmission[scale_by_6] *=sf_6\nsubmission[scale_healthy] *=scale_h\n\n# Save Submission!\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-05T12:13:34.98933Z","iopub.execute_input":"2023-09-05T12:13:34.989691Z","iopub.status.idle":"2023-09-05T12:13:35.008671Z","shell.execute_reply.started":"2023-09-05T12:13:34.989662Z","shell.execute_reply":"2023-09-05T12:13:35.007356Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}