{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Separated Normalization:Make image clearer\nHere, we normalize the data greater than 0.02 and put it into the original place to achieve the purpose of making the image clearer","metadata":{}},{"cell_type":"code","source":"!pip install dicomsdl","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-02-13T06:19:45.615171Z","iopub.execute_input":"2023-02-13T06:19:45.61578Z","iopub.status.idle":"2023-02-13T06:20:00.56698Z","shell.execute_reply.started":"2023-02-13T06:19:45.615658Z","shell.execute_reply":"2023-02-13T06:20:00.565709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Requirements","metadata":{}},{"cell_type":"code","source":"import dicomsdl\nimport matplotlib.pyplot as plt\nfrom matplotlib.pyplot import figure\nimport numpy as np\nimport torch as tc\nimport pandas as pd\nimport warnings,cv2\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-02-13T06:20:00.569816Z","iopub.execute_input":"2023-02-13T06:20:00.570282Z","iopub.status.idle":"2023-02-13T06:20:02.901879Z","shell.execute_reply.started":"2023-02-13T06:20:00.570225Z","shell.execute_reply":"2023-02-13T06:20:02.900679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def is_not_None(x)->bool:\n    \"\"\"return not (x is None or x is False)\"\"\"\n    return not (x is None or x is False)\n\nclass show_distribution:\n    def __init__(self,Graininess=0.01,lines=[]) -> None:\n        self.data=[]\n        self.name=[]\n        self.Graininess=Graininess\n        if len(lines):\n            for x in lines:\n                self.add_line(x)\n            self.show()\n    \n    def add_line(self,x_data,name=None) -> None:\n        \"\"\"x shout in [0,1]\"\"\"\n        x_rate=int(1/self.Graininess)\n        data=tc.arange(x_rate+1)[None,:]/x_rate\n        data=(x_data[:,None]<data).to(tc.float32).sum(0)\n        data=data[1:]-data[:-1]\n        data/=data.sum()\n        l1,=plt.plot(tc.arange(x_rate)/x_rate,data)\n        \n        self.data.append(l1)\n        if is_not_None(name):\n            self.name.append(name)\n        else :\n            self.name.append(str(len(self.data)))\n    \n    def show(self)->None:\n        plt.legend(handles=self.data,labels=self.name,loc='best')\n        plt.show()\n        ","metadata":{"execution":{"iopub.status.busy":"2023-02-13T06:20:02.903249Z","iopub.execute_input":"2023-02-13T06:20:02.903956Z","iopub.status.idle":"2023-02-13T06:20:02.918742Z","shell.execute_reply.started":"2023-02-13T06:20:02.90392Z","shell.execute_reply":"2023-02-13T06:20:02.917074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load dataset","metadata":{}},{"cell_type":"code","source":"def load_data(path):\n    #path~~-> /kaggle/input/rsna-breast-cancer-detection/test_images/10008/361203119.dcm\n    dicom = dicomsdl.open(path)\n    img = dicom.pixelData(storedvalue=False)\n\n    img:np.ndarray = (img - img.min()) / (img.max() - img.min())\n    if dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        img = 1 - img\n    return img\n\nclass Load_png(tc.utils.data.Dataset):\n    def __init__(self,mode=\"test\",\n                size=512):\n        super().__init__()\n        self.data=pd.read_csv(f\"/kaggle/input/rsna-breast-cancer-detection/{mode}.csv\")\n        self.root_path=f\"/kaggle/input/rsna-breast-cancer-detection/{mode}_images/\",\n        self.size=size\n        self.data.sort_values(['patient_id','image_id']).reset_index(drop=True)\n        \n        self.x=self.data[\"patient_id\"].astype(\"str\")+\"/\"+self.data[\"image_id\"].astype(\"str\")\n        self.x=self.root_path+self.x+\".dcm\"\n        self.x=np.array(self.x)\n        self.labels=self.data[\"patient_id\"].astype(\"str\")+\"_\"+self.data[\"laterality\"].astype(\"str\")\n    \n    def __len__(self):\n        return len(self.data)\n        \n    def __getitem__(self,index):\n        path=self.x[index]\n        label=self.labels[index]\n\n        data=[]\n        img=load_data(path)\n        \n        img=img[:,np.any(img>0.07,axis=0)]\n        img=img[img.mean(axis=1)>0.1,:]\n        \n        h0, w0 = img.shape#flip\n        if img[:,int(-w0*0.10):].sum()>img[:,:int(w0*0.10)].sum():\n            img = np.flip(img,axis=1)\n        \n        if img.shape[0]//2-img.shape[1]>0:\n            img=np.concatenate((img,np.zeros((img.shape[0],img.shape[0]//2-img.shape[1]))),axis=1)\n        elif img.shape[1]//2-img.shape[0]>0:\n            img=np.concatenate((img,np.zeros((img.shape[1]//2-img.shape[0],img.shape[1]))),axis=0)\n            \n        img=cv2.resize(img,(self.size,self.size*2),interpolation=cv2.INTER_NEAREST)\n        img = (img - img.min()) / (img.max() - img.min())\n        \n        img=tc.from_numpy(img.copy()).to(tc.float32)[None]\n        return img,label","metadata":{"execution":{"iopub.status.busy":"2023-02-13T06:20:02.92184Z","iopub.execute_input":"2023-02-13T06:20:02.922398Z","iopub.status.idle":"2023-02-13T06:20:02.943004Z","shell.execute_reply.started":"2023-02-13T06:20:02.922347Z","shell.execute_reply":"2023-02-13T06:20:02.941504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Img_normalization","metadata":{}},{"cell_type":"code","source":"def img_normalization(x,std_target=0.1,mean_target=0.5)->tc.Tensor:\n    #x.shape=(batch,1,f1,f2)\n    x=(x-x.min())/(x.max()+x.min()+1e-9)\n\n    cache=x[x>0.02]\n    std_=cache.std()\n    for i in range(2):\n        cache=(cache-cache.mean())/std_*std_target+mean_target\n        cache[cache<0.02]=0\n        cache[cache>1]=1\n        x[x>0.02]=cache#This will cause the std to change\n\n        cache=x[x>0.02]\n        std_=cache.std()\n        if std_>std_target-0.02 and std_target+0.02:\n            break\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-02-13T06:20:02.945145Z","iopub.execute_input":"2023-02-13T06:20:02.94577Z","iopub.status.idle":"2023-02-13T06:20:02.963758Z","shell.execute_reply.started":"2023-02-13T06:20:02.94572Z","shell.execute_reply":"2023-02-13T06:20:02.962364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Effect","metadata":{}},{"cell_type":"code","source":"dataset=Load_png(size=728,mode=\"train\")\na=show_distribution(0.015)\nindex=[8107,31701,1294,24253,5786,4504,6917,1963,7219]\ndata=[]\nfor i in index:\n    print(i)    \n    figure(figsize=(8,6),dpi=80)\n    x,y=dataset[i]\n    \n    plt.subplot(141)\n    plt.title(\"original\")\n    plt.imshow(x.numpy()[0])\n    data.append(x.clone().ravel())\n    \n    new_x=img_normalization(x[None],0.2,0.4)[0]#more clearer for human\n    plt.subplot(142)\n    plt.title(\"normalized\")\n    plt.imshow(new_x.numpy()[0])\n    \n    data.append(new_x.clone().ravel())\n    plt.subplot(122)\n    plt.title(\"distribution\")\n\n    a.add_line(x[x>0.02],f\"original{i//2}\")\n    a.add_line(new_x[new_x>0.02],f\"normalized{i//2}\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-02-13T06:21:53.216807Z","iopub.execute_input":"2023-02-13T06:21:53.217325Z","iopub.status.idle":"2023-02-13T06:22:14.387674Z","shell.execute_reply.started":"2023-02-13T06:21:53.217259Z","shell.execute_reply":"2023-02-13T06:22:14.386433Z"},"trusted":true},"execution_count":null,"outputs":[]}]}