{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\ntrain = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/train.csv')\ntest = pd.read_csv('/kaggle/input/rsna-breast-cancer-detection/test.csv')\n\nfrom sklearn.isotonic import IsotonicRegression\n\ndef create_iso_model(df):\n    df = df[~df['age'].isna()].reset_index(drop=True)\n    df_train = df.groupby(['patient_id', 'age']).cancer.max().reset_index()\n    isr = IsotonicRegression(out_of_bounds='clip')\n    isr.fit(df_train['age'], df_train['cancer'])\n    return isr    \n\nimplants = train[train['implant'] == 1].copy()\nimp_model = create_iso_model(implants)\n\nnon_implants = train[train['implant'] == 0].copy()\nnon_imp_model = create_iso_model(non_implants)\n\ndef predict(test):\n    test_imp = test[test['implant'] == 1].copy()\n    if test_imp.shape[0] > 0:\n        test_imp['age'] = test_imp['age'].fillna(test_imp['age'].mean())\n        test_imp['cancer'] = imp_model.predict(test_imp['age'])\n\n    test_non_imp = test[test['implant'] == 0].copy()\n    if test_non_imp.shape[0] > 0:\n        test_non_imp['age'] = test_non_imp['age'].fillna(test_non_imp['age'].mean())\n        test_non_imp['cancer'] = imp_model.predict(test_non_imp['age'])\n    return pd.concat([test_imp, test_non_imp]).reset_index(drop=True)\n\n# test_imp = test.copy()\n# test_imp['implant'] = 1\n# test_imp['patient_id'] = test_imp['patient_id'] + 10000\n\nsub_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")\n\npred_df = predict(test)\n\nprediction_id = pred_df['patient_id'].astype(str) + \"_\" + pred_df['laterality']\n\ndata = {\"prediction_id\": np.array(list((prediction_id))),\n        \"cancer\": pred_df['cancer'].values\n}\n\nsub_df = pd.DataFrame(data=data)\n\nsubb = sub_df.groupby('prediction_id')['cancer'].mean().to_frame().reset_index()\n\nsubb.to_csv('submission.csv', index=False)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-15T01:07:06.499132Z","iopub.execute_input":"2023-01-15T01:07:06.500052Z","iopub.status.idle":"2023-01-15T01:07:06.527307Z","shell.execute_reply.started":"2023-01-15T01:07:06.499909Z","shell.execute_reply":"2023-01-15T01:07:06.5258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}