{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch as t\nimport torchtext as tt\nimport torch.nn as nn\nimport sklearn\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport collections, itertools\n\n%cd /kaggle/input/rsna-breast-cancer-detection","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:14:56.400532Z","iopub.execute_input":"2023-01-29T21:14:56.401868Z","iopub.status.idle":"2023-01-29T21:14:59.407389Z","shell.execute_reply.started":"2023-01-29T21:14:56.401731Z","shell.execute_reply":"2023-01-29T21:14:59.406588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_df(train = True):\n    if train: return pd.read_csv(\"train.csv\")\n    else: return pd.read_csv(\"test.csv\")\ndf = load_df()","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:14:59.408774Z","iopub.execute_input":"2023-01-29T21:14:59.410252Z","iopub.status.idle":"2023-01-29T21:14:59.525329Z","shell.execute_reply.started":"2023-01-29T21:14:59.410193Z","shell.execute_reply":"2023-01-29T21:14:59.523669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's examine the tabular features.","metadata":{}},{"cell_type":"code","source":"sns.heatmap(df[[\"age\",\"cancer\",\"biopsy\",\"invasive\",\"implant\"]].corr(method=\"pearson\"), annot = True)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:14:59.527288Z","iopub.execute_input":"2023-01-29T21:14:59.527989Z","iopub.status.idle":"2023-01-29T21:14:59.796144Z","shell.execute_reply.started":"2023-01-29T21:14:59.527953Z","shell.execute_reply":"2023-01-29T21:14:59.794772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cancer, biopsy status, and invasiveness all heavily predict each other, as expected (though they are useless as predictive features due to them being only present in the training set). Advanced age is linked with cancer, and the presence of implants have a negligible impact.","metadata":{}},{"cell_type":"code","source":"sns.catplot(data=df, x=\"laterality\", y=\"cancer\", kind=\"bar\",\n           palette=\"dark\", alpha=.6, height=5)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:16.032453Z","iopub.execute_input":"2023-01-29T21:16:16.032855Z","iopub.status.idle":"2023-01-29T21:16:16.956168Z","shell.execute_reply.started":"2023-01-29T21:16:16.032828Z","shell.execute_reply":"2023-01-29T21:16:16.955257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data = df, x=\"view\")","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:29.440364Z","iopub.execute_input":"2023-01-29T21:16:29.440679Z","iopub.status.idle":"2023-01-29T21:16:29.609789Z","shell.execute_reply.started":"2023-01-29T21:16:29.440655Z","shell.execute_reply":"2023-01-29T21:16:29.608598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"CC (Crainocaudal view) and MLO (Mediolateral Oblique view) are the most common. Together, they make up the vast majority (99.93%) of views.","metadata":{}},{"cell_type":"code","source":"(len(df[df[\"view\"] == \"CC\"]) + len(df[df[\"view\"] == \"MLO\"])) / len(df)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:32.659305Z","iopub.execute_input":"2023-01-29T21:16:32.659668Z","iopub.status.idle":"2023-01-29T21:16:32.68147Z","shell.execute_reply.started":"2023-01-29T21:16:32.659643Z","shell.execute_reply":"2023-01-29T21:16:32.679508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.catplot(data= df, kind=\"bar\",\n    x=\"view\", y=\"cancer\",\n    palette=\"dark\", alpha=.6, height=6\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:37.738597Z","iopub.execute_input":"2023-01-29T21:16:37.738941Z","iopub.status.idle":"2023-01-29T21:16:38.656622Z","shell.execute_reply.started":"2023-01-29T21:16:37.738896Z","shell.execute_reply":"2023-01-29T21:16:38.655967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Alarming if true. Could it be that AT scans are done as re-scans for particularly suspicious cases?","metadata":{}},{"cell_type":"code","source":"len(df[df[\"view\"] == \"AT\"])","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:41.374333Z","iopub.execute_input":"2023-01-29T21:16:41.375148Z","iopub.status.idle":"2023-01-29T21:16:41.387469Z","shell.execute_reply.started":"2023-01-29T21:16:41.375112Z","shell.execute_reply":"2023-01-29T21:16:41.386246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not really.","metadata":{}},{"cell_type":"markdown","source":"### Tabular features benchmark","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn import preprocessing\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import KFold\n\ndef preprocess(df, to_remove = [\"site_id\",\"patient_id\",\"image_id\",\"biopsy\",\"invasive\",\"BIRADS\",\"machine_id\",\"difficult_negative_case\",\"density\",\"view\"]):\n    to_remove = list(set(df.columns).intersection(set(to_remove)))\n    p_df = df.drop(to_remove, axis=1)\n    X = p_df.drop(\"cancer\", axis=1)\n    y = p_df[\"cancer\"]\n    return X, y\n\ndef get_cat_feats(df):\n    return [col for col in df if df[col].dtype == object]\n\ndef pfbeta(labels, predictions, beta=1, gamma = 1e-8):\n    y_true_count = 0\n    ctp = 0\n    cfp = 0\n\n    for idx in range(len(labels)):\n        prediction = min(max(predictions[idx], 0), 1)\n        if (labels[idx]):\n            y_true_count += 1\n            ctp += prediction\n        else:\n            cfp += prediction\n\n    beta_squared = beta * beta\n    c_precision = ctp / (ctp + cfp)\n    c_recall = ctp / (y_true_count + gamma)\n    if (c_precision > 0 and c_recall > 0):\n        result = (1 + beta_squared) * (c_precision * c_recall) / (beta_squared * c_precision + c_recall)\n        return result\n    else:\n        return 0\n    \nkf = KFold(n_splits=5, shuffle = True)\nkf.get_n_splits(df)\nfor i, (train_index, test_index) in enumerate(kf.split(df)):\n    print(f\"Fold {i}:\")\n    print(f\"  Training on = {len(train_index)}\")\n    print(f\"  Testing on = {len(test_index)}\")\n    \n    lbl = preprocessing.LabelEncoder()\n    train_df = df.iloc[train_index]; test_df = df.iloc[test_index]\n    train_X, train_y = preprocess(train_df)\n\n    cat_feats = get_cat_feats(train_X)\n    for cat in cat_feats:\n        train_X[cat] = lbl.fit_transform(train_X[cat].astype(str))\n\n    my_model = XGBClassifier()\n    # Add silent=True to avoid printing out updates with each cycle\n    my_model.fit(train_X, train_y, verbose=False)\n\n    test_X, test_y = preprocess(test_df)\n    cat_feats = get_cat_feats(test_X)\n    for cat in cat_feats:\n        test_X[cat] = lbl.transform(test_X[cat].astype(str))\n\n    predictions = my_model.predict(test_X)\n\n    print(\"Mean Absolute Error : \" + str(mean_absolute_error(predictions, test_y)))\n    print(\"Score:\", pfbeta(predictions, test_y.values))","metadata":{"execution":{"iopub.status.busy":"2023-01-29T21:16:43.394416Z","iopub.execute_input":"2023-01-29T21:16:43.394761Z","iopub.status.idle":"2023-01-29T21:16:50.350731Z","shell.execute_reply.started":"2023-01-29T21:16:43.394736Z","shell.execute_reply":"2023-01-29T21:16:50.349408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evidently it's extremely difficult to glean any information from the tabular data alone, so a pure imaging model is the way to go. \n\nI hope you found this EDA helpful. Happy Kaggling! 😊","metadata":{}}]}