{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n1. Distribution of predictions\n</div>","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\n\ndf = pd.read_csv('../input/rsna-validation/validation-fold0.csv')\n# consider CC and MLO views\ndf = df[(df['view'] == 'CC') | (df['view'] == 'MLO')]\nsns.histplot(data=df, x='preds', bins=20)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:47:48.821732Z","iopub.execute_input":"2023-02-19T07:47:48.822202Z","iopub.status.idle":"2023-02-19T07:47:50.413569Z","shell.execute_reply.started":"2023-02-19T07:47:48.82211Z","shell.execute_reply":"2023-02-19T07:47:50.412164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n2. Distribution of errors\n</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\ndf['err'] = df['cancer'].astype(int) - df['preds']\nsns.histplot(data=df, x='err', bins=20)\n\npreds = df['preds']\nthresh = np.percentile(preds, 98)\nprint(f'threshold={thresh:.2}')\nfp_df = df[df['err'] < -thresh]\nfn_df = df[df['err'] > (1 - thresh)]\n\nprint(f'{fp_df.shape[0]} false positives')\nprint(f'{fn_df.shape[0]} false negatives')","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:48:14.10315Z","iopub.execute_input":"2023-02-19T07:48:14.103763Z","iopub.status.idle":"2023-02-19T07:48:14.433482Z","shell.execute_reply.started":"2023-02-19T07:48:14.103714Z","shell.execute_reply":"2023-02-19T07:48:14.43261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n3. Confusion matrix (sanity check)\n</div>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\nlabels = df['cancer'].astype(int)\nint_labels = labels.astype(int)\nint_preds = (preds > thresh).astype(int)\ncm = confusion_matrix(int_labels, int_preds, labels=np.arange(2))\ntn, fp, fn, tp = cm.ravel()\nprint(cm)\nprint(f'tn {tn} fp {fp} fn {fn} tp {tp}')","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:48:20.524531Z","iopub.execute_input":"2023-02-19T07:48:20.524915Z","iopub.status.idle":"2023-02-19T07:48:20.647126Z","shell.execute_reply.started":"2023-02-19T07:48:20.524885Z","shell.execute_reply":"2023-02-19T07:48:20.645728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n4. Distribution of false positives\n</div>\n","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=fp_df, x='err', kde=True, bins=20)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:48:28.603402Z","iopub.execute_input":"2023-02-19T07:48:28.604766Z","iopub.status.idle":"2023-02-19T07:48:28.889544Z","shell.execute_reply.started":"2023-02-19T07:48:28.604717Z","shell.execute_reply":"2023-02-19T07:48:28.888347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n4. Distribution of false negatives\n</div>\n","metadata":{}},{"cell_type":"code","source":"sns.histplot(data=fn_df, x='err', kde=True, bins=20)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:48:35.924347Z","iopub.execute_input":"2023-02-19T07:48:35.924736Z","iopub.status.idle":"2023-02-19T07:48:36.164526Z","shell.execute_reply.started":"2023-02-19T07:48:35.924705Z","shell.execute_reply":"2023-02-19T07:48:36.163317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****The distribution of false negatives is strange. The predicted probability of cancer is very close to zero for many cancer cases****.","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block\" style=\"font-size:20px;background-color: #CCC8E3\">\n5. Examples with the biggest differences in predictions between CC and MLO views\n</div>","metadata":{}},{"cell_type":"code","source":"# select positive examples\ndf = df[df['cancer'] == 1]\nprint(f'analyzing {df.shape[0]} positive cases...')\nview_groups = df.groupby(['patient_id', 'laterality', 'view'])\nmean_preds = view_groups.agg(view_mean_pred=('preds', 'mean'))\n\npred_vals = mean_preds['view_mean_pred'].values\ncc_diff = pred_vals[::2] - pred_vals[1::2]\nmlo_diff = pred_vals[1::2] - pred_vals[::2]\nmean_preds['diff'] = 0\n\nmean_preds['diff'].iloc[::2] = cc_diff\nmean_preds['diff'].iloc[1::2] = mlo_diff\n\nview_dict = {}\nfor view in ['CC', 'MLO']:\n    group = df[df['view'] == view].groupby(['patient_id', 'laterality'])\n    view_dict[view] = group.agg(mean_pred=('preds', 'mean'))\n\npred_df = view_dict['CC']\npred_df['diff'] = (pred_df['mean_pred'] - view_dict['MLO']['mean_pred']).abs()\n\n# sort in descending order of deviation between CC and MLO predictions\npred_df.sort_values(by='diff', ascending=False, inplace=True)\nprint(pred_df.head(5))","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:48:57.096822Z","iopub.execute_input":"2023-02-19T07:48:57.097215Z","iopub.status.idle":"2023-02-19T07:48:57.145052Z","shell.execute_reply.started":"2023-02-19T07:48:57.097183Z","shell.execute_reply":"2023-02-19T07:48:57.14373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nfrom matplotlib import pyplot as plt\n\ninput_dir = '../input/rsna-breast-cancer-1024-pngs/output'\ncount = 5\npatients = pred_df.index.get_level_values(0)[:count]\nfilenames = []\nfor patient_id in patients:\n    patient_df = df[df['patient_id'] == patient_id]\n    print(patient_df)\n    fig = plt.figure(figsize=(20, 20))\n    nrows = patient_df.shape[0]\n    ax = fig.subplots(1, nrows)\n    for i in range(nrows):\n        row = patient_df.iloc[i]\n        image_id = row['image_id']\n        filename = f'{patient_id}_{image_id}.png'\n        filenames.append(filename)\n        im = cv2.imread(f'{input_dir}/{filename}')\n        ax[i].imshow(im)\n        ax[i].set_title(filename)\nplt.show()\nprint(filenames)","metadata":{"execution":{"iopub.status.busy":"2023-02-19T07:50:02.134399Z","iopub.execute_input":"2023-02-19T07:50:02.134815Z","iopub.status.idle":"2023-02-19T07:50:07.666068Z","shell.execute_reply.started":"2023-02-19T07:50:02.134775Z","shell.execute_reply":"2023-02-19T07:50:07.664868Z"},"trusted":true},"execution_count":null,"outputs":[]}]}