{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"test = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# IMPORT EXAMPLE SUBMISSION\n# sub = pd.read_csv('../input/peinferencelstm/submission_seresnext26_ver3.csv')\nsub = pd.read_csv('../input/lightgbm-on-meta-features/submission.csv')\nsub.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# CHECKING CONSISTENCY RULES"},{"metadata":{},"cell_type":"markdown","source":"this function is taken from [this notebook](https://www.kaggle.com/anthracene/host-confirmed-label-consistency-check)"},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_consistency(sub, test):\n    \n    '''\n    Checks label consistency and returns the errors\n    \n    Args:\n    sub   = submission dataframe (pandas)\n    test  = test.csv dataframe (pandas)\n    '''\n    \n    # EXAM LEVEL\n    df = None\n    for i in test['StudyInstanceUID'].unique():\n        df_tmp = sub.loc[sub.id.str.contains(i, regex = False)].reset_index(drop = True)\n        df_tmp['StudyInstanceUID'] = df_tmp['id'].str.split('_').str[0]\n        df_tmp['label_type']       = df_tmp['id'].str.split('_').str[1:].apply(lambda x: '_'.join(x))\n        del df_tmp['id']\n        \n        df = pd.concat([df, df_tmp], axis = 0)\n    \n    df_exam = df.pivot(index = 'StudyInstanceUID', columns = 'label_type', values = 'label')\n    \n    # IMAGE LEVEL\n    df_image = sub.loc[sub.id.isin(test.SOPInstanceUID)].reset_index(drop = True)\n    df_image = df_image.merge(test, how = 'left', left_on = 'id', right_on = 'SOPInstanceUID')\n    df_image.rename(columns = {\"label\": \"pe_present_on_image\"}, inplace = True)\n    del df_image['id']\n    \n    # MERGER\n    df = df_exam.merge(df_image, how = 'left', on = 'StudyInstanceUID')\n    ids    = ['StudyInstanceUID', 'SeriesInstanceUID', 'SOPInstanceUID']\n    labels = [c for c in df.columns if c not in ids]\n    df = df[ids + labels]\n    \n    # SPLIT NEGATIVE AND POSITIVE EXAMS\n    df['positive_images_in_exam'] = df['StudyInstanceUID'].map(df.groupby(['StudyInstanceUID']).pe_present_on_image.max())\n    df_pos = df.loc[df.positive_images_in_exam >  0.5]\n    df_neg = df.loc[df.positive_images_in_exam <= 0.5]\n    \n    # CHECKING CONSISTENCY OF POSITIVE EXAM LABELS\n    rule1a = df_pos.loc[((df_pos.rv_lv_ratio_lt_1  >  0.5)  & \n                         (df_pos.rv_lv_ratio_gte_1 >  0.5)) | \n                        ((df_pos.rv_lv_ratio_lt_1  <= 0.5)  & \n                         (df_pos.rv_lv_ratio_gte_1 <= 0.5))].reset_index(drop = True)\n    rule1a['broken_rule'] = '1a'\n    rule1b = df_pos.loc[(df_pos.central_pe    <= 0.5) & \n                        (df_pos.rightsided_pe <= 0.5) & \n                        (df_pos.leftsided_pe  <= 0.5)].reset_index(drop = True)\n    rule1b['broken_rule'] = '1b'\n    rule1c = df_pos.loc[(df_pos.acute_and_chronic_pe > 0.5) & \n                        (df_pos.chronic_pe           > 0.5)].reset_index(drop = True)\n    rule1c['broken_rule'] = '1c'\n    rule1d = df_pos.loc[(df_pos.indeterminate        > 0.5) | \n                        (df_pos.negative_exam_for_pe > 0.5)].reset_index(drop = True)\n    rule1d['broken_rule'] = '1d'\n\n    # CHECKING CONSISTENCY OF NEGATIVE EXAM LABELS\n    rule2a = df_neg.loc[((df_neg.indeterminate        >  0.5)  & \n                         (df_neg.negative_exam_for_pe >  0.5)) | \n                        ((df_neg.indeterminate        <= 0.5)  & \n                         (df_neg.negative_exam_for_pe <= 0.5))].reset_index(drop = True)\n    rule2a['broken_rule'] = '2a'\n    rule2b = df_neg.loc[(df_neg.rv_lv_ratio_lt_1     > 0.5) | \n                        (df_neg.rv_lv_ratio_gte_1    > 0.5) |\n                        (df_neg.central_pe           > 0.5) | \n                        (df_neg.rightsided_pe        > 0.5) | \n                        (df_neg.leftsided_pe         > 0.5) |\n                        (df_neg.acute_and_chronic_pe > 0.5) | \n                        (df_neg.chronic_pe           > 0.5)].reset_index(drop = True)\n    rule2b['broken_rule'] = '2b'\n    \n    # MERGING INCONSISTENT PREDICTIONS\n    errors = pd.concat([rule1a, rule1b, rule1c, rule1d, rule2a, rule2b], axis = 0)\n    \n    # OUTPUT\n    print('Found', len(errors), 'inconsistent predictions')\n    return errors","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# CHECK\nerrors = check_consistency(sub, test)\nerrors['broken_rule'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Post Processing"},{"metadata":{"trusted":true},"cell_type":"code","source":"test = pd.read_csv('../input/rsna-str-pulmonary-embolism-detection/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# IMPORT EXAMPLE SUBMISSION\nsub = pd.read_csv('../input/peinferencelstm/submission_ver12.csv')\nsub.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\n# EXAM LEVEL\ndf = None\nfor i in test['StudyInstanceUID'].unique():\n\n    df_tmp = sub.loc[sub.id.str.contains(i, regex = False)].reset_index(drop = True)\n    df_tmp['StudyInstanceUID'] = df_tmp['id'].str.split('_').str[0]\n    df_tmp['label_type']       = df_tmp['id'].str.split('_').str[1:].apply(lambda x: '_'.join(x))\n    del df_tmp['id']\n    \n    df = pd.concat([df, df_tmp], axis = 0)\n        \ndf_exam = df.pivot(index = 'StudyInstanceUID', columns = 'label_type', values = 'label')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# IMAGE LEVEL\ndf_image = sub.loc[sub.id.isin(test.SOPInstanceUID)].reset_index(drop = True)\ndf_image = df_image.merge(test, how = 'left', left_on = 'id', right_on = 'SOPInstanceUID')\ndf_image.rename(columns = {\"label\": \"pe_present_on_image\"}, inplace = True)\ndel df_image['id']\ndf_image.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\n# MERGER\ndf = df_exam.merge(df_image, how = 'left', on = 'StudyInstanceUID')\nids    = ['StudyInstanceUID', 'SeriesInstanceUID', 'SOPInstanceUID']\nlabels = [c for c in df.columns if c not in ids]\ndf = df[ids + labels]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# SPLIT NEGATIVE AND POSITIVE EXAMS\n\ndf['positive_images_in_exam'] = df['StudyInstanceUID'].map(df.groupby(['StudyInstanceUID']).pe_present_on_image.max())\n\npos_indices = df.positive_images_in_exam > 0.5\nneg_indices = df.positive_images_in_exam <= 0.5\ndf_pos = df.loc[pos_indices]\ndf_neg = df.loc[neg_indices]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"delta = 1e-4","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 1a ~ 1d rules"},{"metadata":{"trusted":true},"cell_type":"code","source":"# either rv_lv_ratio_lt_1 or rv_lv_ratio_gte_1 must have p > 0.5; both cannot have p > 0.5.\ndef rule1a(row):\n    rv_lv_ratio_lt_1 = row[\"rv_lv_ratio_lt_1\"]\n    rv_lv_ratio_gte_1 = row[\"rv_lv_ratio_gte_1\"]\n    if rv_lv_ratio_lt_1 > rv_lv_ratio_gte_1:\n        rv_lv_ratio_lt_1 = max(0.5 + delta, rv_lv_ratio_lt_1)\n        rv_lv_ratio_gte_1 = min(0.5 - delta, rv_lv_ratio_gte_1)\n    else:\n        rv_lv_ratio_lt_1 = min(0.5 - delta, rv_lv_ratio_lt_1)\n        rv_lv_ratio_gte_1 = max(0.5 + delta, rv_lv_ratio_gte_1)\n    return rv_lv_ratio_lt_1, rv_lv_ratio_gte_1\n\ndef postprocess_rule1a(df, df_pos, pos_indices):\n    indices = ((df_pos.rv_lv_ratio_lt_1  >  0.5)  & \n                 (df_pos.rv_lv_ratio_gte_1 >  0.5)) | \\\n                ((df_pos.rv_lv_ratio_lt_1  <= 0.5)  & \\\n                 (df_pos.rv_lv_ratio_gte_1 <= 0.5))\n\n    if np.any(indices):\n        columns = [\"rv_lv_ratio_lt_1\", \"rv_lv_ratio_gte_1\"]\n        tmp_val = df_pos.copy().loc[indices, columns].apply(lambda row: rule1a(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(pos_indices & indices), col] = val\n\npostprocess_rule1a(df, df_pos, pos_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# at least one of central_pe, rightsided_pe and leftsided_pe must have p > 0.5; multiple having p > 0.5 is allowed.\ndef rule1b(row):\n    central_pe = row[\"central_pe\"]\n    rightsided_pe = row[\"rightsided_pe\"]\n    leftsided_pe = row[\"leftsided_pe\"]\n    l = [central_pe, rightsided_pe, leftsided_pe]\n    max_idx = np.argmax(l)\n    l[max_idx] = 0.5 + delta\n    return l[0], l[1], l[2]\n\ndef postprocess_rule1b(df, df_pos, pos_indices):\n    indices = (df_pos.central_pe    <= 0.5) & \\\n              (df_pos.rightsided_pe <= 0.5) & \\\n              (df_pos.leftsided_pe  <= 0.5)\n    \n    if np.any(indices):\n        columns = [\"central_pe\", \"rightsided_pe\", \"leftsided_pe\"]\n        tmp_val = df_pos.copy().loc[indices, columns].apply(lambda row: rule1b(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(pos_indices & indices), col] = val\n\npostprocess_rule1b(df, df_pos, pos_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# acute_and_chronic_pe and chronic_pe: only one of them can have p > 0.5; neither having p > 0.5 is allowed.\ndef rule1c(row):\n    acute_and_chronic_pe = row[\"acute_and_chronic_pe\"]\n    chronic_pe = row[\"chronic_pe\"]\n    if acute_and_chronic_pe > chronic_pe:\n        chronic_pe = 0.5 - delta\n    else:\n        acute_and_chronic_pe = 0.5 - delta\n    return acute_and_chronic_pe, chronic_pe\n\ndef postprocess_rule1c(df, df_pos, pos_indices):\n    indices = (df_pos.acute_and_chronic_pe > 0.5) & \\\n              (df_pos.chronic_pe           > 0.5)\n\n    if np.any(indices):\n        columns = [\"acute_and_chronic_pe\", \"chronic_pe\"]\n        tmp_val = df_pos.copy().loc[indices, columns].apply(lambda row: rule1c(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(pos_indices & indices), col] = val\n\npostprocess_rule1c(df, df_pos, pos_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rule1d(row):\n    negative_exam_for_pe = row[\"negative_exam_for_pe\"]\n    indeterminate = row[\"indeterminate\"]\n    negative_exam_for_pe = min(0.5 - delta, negative_exam_for_pe)\n    indeterminate = min(0.5 - delta, indeterminate)\n    return negative_exam_for_pe, indeterminate\n\ndef postprocess_rule1d(df, df_pos, pos_indices):\n    indices = (df_pos.indeterminate        > 0.5) | \\\n              (df_pos.negative_exam_for_pe > 0.5)\n    \n    if np.any(indices):\n        columns = [\"negative_exam_for_pe\", \"indeterminate\"]\n        tmp_val = df_pos.copy().loc[indices, columns].apply(lambda row: rule1d(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(pos_indices & indices), col] = val\n\npostprocess_rule1d(df, df_pos, pos_indices)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2a ~ 2b rules"},{"metadata":{"trusted":true},"cell_type":"code","source":"# either indeterminate or negative_exam_for_pe must have p > 0.5; both cannot have p > 0.5.\ndef rule2a(row):\n    negative_exam_for_pe = row[\"negative_exam_for_pe\"]\n    indeterminate = row[\"indeterminate\"]\n    if negative_exam_for_pe > 0.5:\n        if negative_exam_for_pe > indeterminate:\n            indeterminate = 0.5 - delta\n        else:\n            negative_exam_for_pe = 0.5 - delta\n    else: \n        if negative_exam_for_pe > indeterminate:\n            negative_exam_for_pe = 0.5 + delta\n        else:\n            indeterminate = 0.5 + delta\n    return negative_exam_for_pe, indeterminate\n        \n\ndef postprocess_rule2a(df, df_neg, neg_indices):\n    indices = ((df_neg.indeterminate        >  0.5)  & \n               (df_neg.negative_exam_for_pe >  0.5)) | \\\n              ((df_neg.indeterminate        <= 0.5)  & \n               (df_neg.negative_exam_for_pe <= 0.5))\n    \n    if np.any(indices):\n        columns = [\"negative_exam_for_pe\", \"indeterminate\"]\n        tmp_val = df_neg.copy().loc[indices, columns].apply(lambda row: rule2a(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(neg_indices & indices), col] = val\n\npostprocess_rule2a(df, df_neg, neg_indices)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# all positive-related labels: rv_lv_ratio_lt_1, rv_lv_ratio_gte_1, central_pe, rightsided_pe, leftsided_pe, acute_and_chronic_pe and chronic_pe must have p < 0.5.\ndef rule2b(row):\n    rv_lv_ratio_lt_1 = row[\"rv_lv_ratio_lt_1\"]\n    rv_lv_ratio_gte_1 = row[\"rv_lv_ratio_gte_1\"]\n    central_pe = row[\"central_pe\"]\n    rightsided_pe = row[\"rightsided_pe\"]\n    leftsided_pe = row[\"leftsided_pe\"]\n    acute_and_chronic_pe = row[\"acute_and_chronic_pe\"]\n    chronic_pe = row[\"chronic_pe\"]\n    \n    rv_lv_ratio_lt_1 = min(0.5 - delta, rv_lv_ratio_lt_1)\n    rv_lv_ratio_gte_1 = min(0.5 - delta, rv_lv_ratio_gte_1)\n    central_pe = min(0.5 - delta, central_pe)\n    rightsided_pe = min(0.5 - delta, rightsided_pe)\n    leftsided_pe = min(0.5 - delta, leftsided_pe)\n    acute_and_chronic_pe = min(0.5 - delta, acute_and_chronic_pe)\n    chronic_pe = min(0.5 - delta, chronic_pe)\n    return rv_lv_ratio_lt_1, rv_lv_ratio_gte_1, \\\n           central_pe, rightsided_pe, leftsided_pe, \\\n           acute_and_chronic_pe, chronic_pe \n\ndef postprocess_rule2b(df, df_neg, neg_indices):\n    indices = (df_neg.rv_lv_ratio_lt_1     > 0.5) | \\\n              (df_neg.rv_lv_ratio_gte_1    > 0.5) | \\\n              (df_neg.central_pe           > 0.5) | \\\n              (df_neg.rightsided_pe        > 0.5) | \\\n              (df_neg.leftsided_pe         > 0.5) | \\\n              (df_neg.acute_and_chronic_pe > 0.5) | \\\n              (df_neg.chronic_pe           > 0.5)\n    \n    if np.any(indices):\n        columns = [\n            \"rv_lv_ratio_lt_1\", \"rv_lv_ratio_gte_1\", \"central_pe\",\n            \"rightsided_pe\", \"leftsided_pe\", \"acute_and_chronic_pe\", \"chronic_pe\"]\n        tmp_val = df_neg.copy().loc[indices, columns].apply(lambda row: rule2b(row), axis=1).values\n        for col, val in zip(columns, zip(*tmp_val)):\n            df.loc[(neg_indices & indices), col] = val\n\npostprocess_rule2b(df, df_neg, neg_indices)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Restore the submission.csv format"},{"metadata":{"trusted":true},"cell_type":"code","source":"remove_columns = [\"SeriesInstanceUID\", \"SOPInstanceUID\", \"positive_images_in_exam\", \"pe_present_on_image\"]\ndf_columns = [c for c in df.columns if c not in remove_columns]\ndf = df[df_columns]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.melt(id_vars=[\"StudyInstanceUID\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.rename(columns={\"variable\": \"id\", \"value\": \"label\"}, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df.shape)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[\"id\"] = df.apply(lambda x: x[\"StudyInstanceUID\"] + \"_\" + x[\"id\"], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.drop(\"StudyInstanceUID\", inplace=True, axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df.drop_duplicates().reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df.shape)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_image = sub.loc[sub.id.isin(test.SOPInstanceUID)].reset_index(drop = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df_image.shape)\ndf_image.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.concat([df, df_image])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"assert sub.shape == df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\n\nerrors = check_consistency(df, test)\nerrors['broken_rule'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}