{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!conda install ../input/how-to-use-pyvips-offline/*.tar.bz2  # without Internet","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-09-28T15:18:30.696067Z","iopub.execute_input":"2022-09-28T15:18:30.696544Z","iopub.status.idle":"2022-09-28T15:19:18.91682Z","shell.execute_reply.started":"2022-09-28T15:18:30.696443Z","shell.execute_reply":"2022-09-28T15:19:18.91569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport sys\nfrom glob import glob\nfrom tqdm import tqdm\nimport pickle\n\nimport numpy as np\nimport pandas as pd\nimport math\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image, ImageOps\nimport PIL.Image\nimport pyvips\n\nimport keras\nfrom keras.models import load_model\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:18.918889Z","iopub.execute_input":"2022-09-28T15:19:18.919298Z","iopub.status.idle":"2022-09-28T15:19:24.418233Z","shell.execute_reply.started":"2022-09-28T15:19:18.919264Z","shell.execute_reply":"2022-09-28T15:19:24.417085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def imageSliceArray(img_path):\n    img = pyvips.Image.new_from_file(img_path, access='sequential')\n    \n    for i in range(1,6):\n        img_crop = img.crop(round(img.width/6*i), round(img.height/15*1), 512, round(img.height/15*13))\n        if i==1:\n            img_concat = img_crop\n        else:\n            img_concat = np.concatenate([img_concat, img_crop.numpy()], axis=1)\n        \n    return img_concat\n\n\ndef makeTileArray(img_concat, height, width):\n    \n    im_height, im_width = img_concat.shape[0], img_concat.shape[1]\n\n    a = im_height//height\n    b = im_width//width\n        \n    min_mean0 = 999\n    min_mean1 = 999\n    min_mean2 = 999\n    min_mean3 = 999\n    min_mean4 = 999\n    min_mean5 = 999\n    min_mean6 = 999\n    min_mean7 = 999\n    \n    min_arrs = []\n    for h1 in range(a):\n        for w1 in range(b):\n            w2 = w1 * width\n            h2 = h1 * height\n            temp = img_concat[h2:h2+height,w2:w2+width] \n            temp_mean = temp.mean()\n            \n            if min_mean0 > temp_mean:\n                min_mean0 = temp_mean\n                min_arr0 = temp\n            elif min_mean1 > temp_mean:\n                min_mean1 = temp_mean\n                min_arr1 = temp\n            elif min_mean2 > temp_mean:\n                min_mean2 = temp_mean\n                min_arr2 = temp\n            elif min_mean3 > temp_mean:\n                min_mean3 = temp_mean\n                min_arr3 = temp\n            elif min_mean4 > temp_mean:\n                min_mean4 = temp_mean\n                min_arr4 = temp\n            elif min_mean5 > temp_mean:\n                min_mean5 = temp_mean\n                min_arr5 = temp\n            elif min_mean6 > temp_mean:\n                min_mean6 = temp_mean\n                min_arr6 = temp\n            elif min_mean7 > temp_mean:\n                min_mean7 = temp_mean\n                min_arr7 = temp\n    \n    min_arrs.append(min_arr0)\n    min_arrs.append(min_arr1)\n    min_arrs.append(min_arr2)\n    min_arrs.append(min_arr3)\n    min_arrs.append(min_arr4)\n    min_arrs.append(min_arr5)\n    min_arrs.append(min_arr6)\n    min_arrs.append(min_arr7)\n    \n    return(min_arrs)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:24.420899Z","iopub.execute_input":"2022-09-28T15:19:24.421535Z","iopub.status.idle":"2022-09-28T15:19:24.437806Z","shell.execute_reply.started":"2022-09-28T15:19:24.4215Z","shell.execute_reply":"2022-09-28T15:19:24.4358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make dataset\ndef makeFeatures(df, size, n, nOfTiles, method, meanmin, meanmax):\n    \n    x = []\n    imgId = []\n    ptId = []\n    \n    range_max = int((math.sqrt(n)-1)*size+1)\n    \n    for i in tqdm(range(df.shape[0])):\n        \n        samplesize = 0\n        for j in range(nOfTiles):\n            tile = Image.open(outdir+df.loc[df.index[i],\"image_id\"]+\"_\"+str(j)+\".jpg\")\n            tile_arr = np.asarray(tile)\n            \n            # make small tiles\n            for k in range(0, range_max, size):\n                for l in range(0, range_max, size):\n                    temp_tile_arr = tile_arr[k:k+size, l:l+size]\n                    \n                    if method==\"lgb\":\n                        temp_tile_arr = temp_tile_arr.flatten()\n                        # delete outlier\n                        if temp_tile_arr.mean()<=meanmax and temp_tile_arr.mean()>=meanmin:\n                            # add new features\n                            for t in range(0, 201, 50):\n                                cnt = np.count_nonzero((temp_tile_arr >= t) & (temp_tile_arr < t+50))\n                                temp_tile_arr = np.insert(temp_tile_arr, 0, cnt)\n\n                            x.append(temp_tile_arr)\n                            samplesize += 1\n                            \n                    elif method==\"cnn\":\n                        # delete outlier\n                        if temp_tile_arr.mean()<=meanmax and temp_tile_arr.mean()>=meanmin:\n                            x.append(temp_tile_arr)\n                            samplesize += 1\n                                   \n        temp_imgId = np.array(df.loc[df.index[i],\"image_id\"]).repeat(samplesize)\n        imgId = np.hstack([imgId, temp_imgId])\n        \n        temp_ptId = np.array(df.loc[df.index[i],\"patient_id\"]).repeat(samplesize)\n        ptId = np.hstack([ptId, temp_ptId])\n    \n    x = np.array(x)\n    x = x.astype('float32') / 255\n    \n    return(x, imgId, ptId)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:24.440282Z","iopub.execute_input":"2022-09-28T15:19:24.44065Z","iopub.status.idle":"2022-09-28T15:19:24.457146Z","shell.execute_reply.started":"2022-09-28T15:19:24.44062Z","shell.execute_reply":"2022-09-28T15:19:24.456011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predictByCenterId(df, nOfTiles, input_size, number_of_split, method, model):\n        \n    # make features\n    input_size = input_size #32\n    number_of_split = number_of_split #49\n    \n    x, imgId, ptId = makeFeatures(\n        df=df, \n        size=input_size, \n        n=number_of_split,\n        nOfTiles=nOfTiles,\n        method=method,\n        meanmin=0,\n        meanmax=255\n    )\n    \n    # predict\n    if method==\"cnn\":\n        pred = model.predict(x, batch_size=8, verbose=1)\n    elif method==\"lgb\":\n        pred = model.predict(x)\n    \n    return ptId, pred","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:24.458823Z","iopub.execute_input":"2022-09-28T15:19:24.459423Z","iopub.status.idle":"2022-09-28T15:19:24.472441Z","shell.execute_reply.started":"2022-09-28T15:19:24.459391Z","shell.execute_reply":"2022-09-28T15:19:24.471232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make directories\nif not os.path.exists('./tiles'):\n    os.mkdir(\"./tiles\")\n    \n# path\ntest_images = \"../input/mayo-clinic-strip-ai/test/\"\noutdir = \"./tiles/\"\n\n# read metadata\ndf_test = pd.read_csv(\"../input/mayo-clinic-strip-ai/test.csv\")\n\n#df_test.loc[0,\"center_id\"] = \"a\"  # error test\n\n# One-hot Encoder for center_id\ndf_test[[\"center_id\"]] = df_test[[\"center_id\"]].astype('object')\nlabel_ohe = pd.get_dummies(df_test[[\"center_id\"]], dummy_na=False, drop_first=False)\ndf_test = pd.concat([df_test, label_ohe], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:24.473982Z","iopub.execute_input":"2022-09-28T15:19:24.474312Z","iopub.status.idle":"2022-09-28T15:19:24.515786Z","shell.execute_reply.started":"2022-09-28T15:19:24.474282Z","shell.execute_reply":"2022-09-28T15:19:24.514551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# make tiles\nfor i in tqdm(range(df_test.shape[0])): \n    img_path = test_images+df_test.loc[i,\"image_id\"]+\".tif\"\n    img_concat = imageSliceArray(img_path)\n    min_arrs = makeTileArray(img_concat, 512, 512)\n    \n    for j in range(8):\n        temp_tile = Image.fromarray(min_arrs[j])\n        temp_tile.save(outdir+df_test.loc[i,\"image_id\"]+\"_\"+str(j)+\".jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:24.517202Z","iopub.execute_input":"2022-09-28T15:19:24.517536Z","iopub.status.idle":"2022-09-28T15:19:58.534815Z","shell.execute_reply.started":"2022-09-28T15:19:24.517507Z","shell.execute_reply":"2022-09-28T15:19:58.533691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn_ptId = []\ncnn_pred = []\nlgb_ptId = []\nlgb_pred = []\n\nfor m in (\"cnn\",\"lgb\"):\n    for i in range(1,12):\n        if \"center_id_\"+str(i) in df_test.columns:\n            \n            print(\"center_id: \", i)\n\n            # select center_id\n            temp_df = df_test[df_test.center_id==i]\n\n            # load model \n            if m==\"cnn\":\n                model = load_model('../input/mayo01-model-training-with-cnn/model_'+str(i)+'.h5')\n            \n                if i==1:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=4,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==2:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==3:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=2,\n                        input_size=64,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==4:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=64,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==5:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==6:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=48,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==7:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=2,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==8:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==9:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==10:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                elif i==11:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=5,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                else:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n            \n                # merge result\n                cnn_ptId = np.hstack([cnn_ptId, temp_ptId])\n                cnn_pred = np.hstack([cnn_pred, temp_pred.flatten()])\n            \n            elif m==\"lgb\":\n                model = pickle.load(open('../input/mayo01-model-training-with-lgb/model_'+str(i)+'.pkl', 'rb'))\n        \n                if i==1:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=169,\n                        method=m,\n                        model=model\n                    )\n                else:\n                    temp_ptId, temp_pred = predictByCenterId(\n                        df=temp_df,\n                        nOfTiles=8,\n                        input_size=32,\n                        number_of_split=49,\n                        method=m,\n                        model=model\n                    )\n                \n                lgb_ptId = np.hstack([lgb_ptId, temp_ptId])\n                lgb_pred = np.hstack([lgb_pred, temp_pred])\n    \n    # when center_id is not 1~11\n    # select data\n    temp_df = df_test[~df_test.center_id.isin([1,2,3,4,5,6,7,8,9,10,11])]\n    \n    if temp_df.shape[0]!=0:\n        \n        print(\"center_id: other\")\n        \n        # load model \n        if m==\"cnn\":\n            model = load_model('../input/mayo01-model-training-with-cnn/model_'+str(11)+'.h5')\n        elif m==\"lgb\":\n            model = pickle.load(open('../input/mayo01-model-training-with-lgb/model_'+str(11)+'.pkl', 'rb'))\n\n        # predict\n        temp_ptId, temp_pred = predictByCenterId(\n            df=temp_df,\n            nOfTiles=8,\n            input_size=32,\n            number_of_split=49,\n            method=m,\n            model=model\n        )\n\n        # merge result\n        if m==\"cnn\":\n            cnn_ptId = np.hstack([cnn_ptId, temp_ptId])\n            cnn_pred = np.hstack([cnn_pred, temp_pred.flatten()])\n        if m==\"lgb\":\n            lgb_ptId = np.hstack([lgb_ptId, temp_ptId])\n            lgb_pred = np.hstack([lgb_pred, temp_pred])","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:19:58.536435Z","iopub.execute_input":"2022-09-28T15:19:58.536796Z","iopub.status.idle":"2022-09-28T15:20:03.561303Z","shell.execute_reply.started":"2022-09-28T15:19:58.536762Z","shell.execute_reply":"2022-09-28T15:20:03.560377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result of cnn\ndf_pred_cnn = pd.DataFrame({\n    \"CE_prob_cnn\": cnn_pred.flatten(),\n    \"patient_id\": cnn_ptId\n}).groupby(\"patient_id\").median().reset_index()\n\n# result of lgb\ndf_pred_lgb = pd.DataFrame({\n    \"CE_prob_lgb\": lgb_pred.flatten(),\n    \"patient_id\": lgb_ptId\n}).groupby(\"patient_id\").median().reset_index()\n\n# merge\ndf_pred = pd.merge(df_pred_cnn, df_pred_lgb, on='patient_id')\n\n# merge center_id\ndf_pt = df_test[[\"patient_id\",\"center_id\"]].drop_duplicates()\ndf_pred = pd.merge(df_pred, df_pt, on='patient_id')\n\n# ensemble\nfor i in range(df_pred.shape[0]):\n    if df_pred.loc[i,\"center_id\"] in [1,4,5,6,9,11]:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,[\"CE_prob_cnn\",\"CE_prob_lgb\"]].max()\n    elif df_pred.loc[i,\"center_id\"] in [3]:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,[\"CE_prob_cnn\",\"CE_prob_lgb\"]].min()\n    elif df_pred.loc[i,\"center_id\"] in [10]:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,[\"CE_prob_cnn\",\"CE_prob_lgb\"]].mean()\n    elif df_pred.loc[i,\"center_id\"] in [2]:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,\"CE_prob_cnn\"]\n    elif df_pred.loc[i,\"center_id\"] in [7,8]:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,\"CE_prob_lgb\"]\n    else:\n        df_pred.loc[i,\"CE_prob\"] = df_pred.loc[i,[\"CE_prob_cnn\",\"CE_prob_lgb\"]].max()","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:20:03.562674Z","iopub.execute_input":"2022-09-28T15:20:03.566251Z","iopub.status.idle":"2022-09-28T15:20:03.608422Z","shell.execute_reply.started":"2022-09-28T15:20:03.566207Z","shell.execute_reply":"2022-09-28T15:20:03.607377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(\n    {\n        \"patient_id\": df_pred.patient_id,\n        \"CE\": df_pred.CE_prob,\n        \"LAA\": 1-df_pred.CE_prob,\n    }\n)\n\nsubmission = submission.groupby(\"patient_id\").mean()\n\nsubmission = submission[[\"CE\", \"LAA\"]].round(6).reset_index()\n\ndisplay(submission)\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-09-28T15:20:03.610648Z","iopub.execute_input":"2022-09-28T15:20:03.610971Z","iopub.status.idle":"2022-09-28T15:20:03.635273Z","shell.execute_reply.started":"2022-09-28T15:20:03.610942Z","shell.execute_reply":"2022-09-28T15:20:03.634473Z"},"trusted":true},"execution_count":null,"outputs":[]}]}