{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/39272/logos/header.png?t=2022-11-28-17-29-35\">\n\n<h1>A Simple Age-Based Baseline Submission</h1>\n\n---\n\n<br>\n\nPlease note that a full EDA is coming... but I just wanted to throw this out there.","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n**IMPORTS**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nimport pandas as pd\nimport numpy as np\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-30T23:28:45.950566Z","iopub.execute_input":"2022-11-30T23:28:45.951058Z","iopub.status.idle":"2022-11-30T23:28:46.701346Z","shell.execute_reply.started":"2022-11-30T23:28:45.951019Z","shell.execute_reply":"2022-11-30T23:28:46.700422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n**LOAD DATAFRAMES**","metadata":{}},{"cell_type":"code","source":"# Load train data and add prediction ID (not used in train)\ntrain_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/train.csv\")\ntrain_df[\"prediction_id\"] = train_df[\"patient_id\"].astype(str)+\"_\"+train_df[\"laterality\"].astype(str)\n\n# Load test data and add prediction ID for mapping to submission file\ntest_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/test.csv\")\ntest_df[\"prediction_id\"] = test_df[\"patient_id\"].astype(str)+\"_\"+test_df[\"laterality\"].astype(str)\n\n# Load submission dataframe\nss_df = pd.read_csv(\"/kaggle/input/rsna-breast-cancer-detection/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-11-30T23:28:20.940344Z","iopub.execute_input":"2022-11-30T23:28:20.94085Z","iopub.status.idle":"2022-11-30T23:28:21.166946Z","shell.execute_reply.started":"2022-11-30T23:28:20.940811Z","shell.execute_reply":"2022-11-30T23:28:21.165667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n**FIT THE BASIC CLASSIFIER**","metadata":{}},{"cell_type":"code","source":"# Define train age data and weighting\n#   --> SCALE_FACTOR just improves raw correlation by adjusting distribution\nSCALE_FACTOR = 3.3333\ntrain_x = np.expand_dims(train_df[\"age\"].fillna(train_df[\"age\"].mean())**SCALE_FACTOR, axis=-1)\ntrain_y = train_df[\"cancer\"].to_numpy()\n\n# Set weight of cancer instances to about 250x larger than non-cancer cases\n#    - In the v2 version we set this to be closer to the real distribution (~50)\n#    - In the v3 version we set this to to be even MORE extreme 500 \n#    - In the v4 version we set this to to be EVEN even MORE extreme 1000x \n#    - In the v4 version we set this to to be even EVEN even MORE extreme 5000x \nAPPROX_WT = 5000\ntrain_wt = (train_df[\"cancer\"]+(1/APPROX_WT)).to_numpy()\n\n# Define test age data\ntest_x = np.expand_dims(test_df[\"age\"].fillna(train_df[\"age\"].mean())**SCALE_FACTOR, axis=-1)\n\n# Define regressor and fit\nclf = LogisticRegression().fit(train_x, train_y, sample_weight=train_wt)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n**PREDICT PROBABILITIES AND SUBMIT**","metadata":{}},{"cell_type":"code","source":"# get preds and take is_cancer side of probability\ntest_preds = clf.predict_proba(test_x)[:, 1]\n\n# Get mapping from prediction_id to cancer probability\ntest_df[\"cancer_prob\"] = test_preds\ntest_pred_cancer_map = test_df.groupby(\"prediction_id\")[\"cancer_prob\"].first().to_dict()\n\n# Update our submission dataframe and save\nss_df[\"cancer\"] = ss_df[\"prediction_id\"].map(test_pred_cancer_map)\nss_df.to_csv(\"submission.csv\", index=False)\n\ndisplay(ss_df)","metadata":{},"execution_count":null,"outputs":[]}]}