{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![Torch-Tensorrt](https://developer-blogs.nvidia.com/wp-content/uploads/2021/12/pytorch-torch-tensorrt.png)","metadata":{}},{"cell_type":"markdown","source":"[Torch-TensorRT](http://https://developer.nvidia.com/blog/accelerating-inference-up-to-6x-faster-in-pytorch-with-torch-tensorrt/) is an integration for PyTorch that leverages inference optimizations of TensorRT on NVIDIA GPUs.  This notebook demonstrates how to install the necessary libraries for torch_tensorrt and how to convert models for speedup.","metadata":{}},{"cell_type":"markdown","source":"Plenty of great torch_tensorrt example notebooks reside [here](https://github.com/pytorch/TensorRT/tree/master/notebooks).  This notebook was adapted from the efn notebook in that folder.","metadata":{}},{"cell_type":"markdown","source":"# Install Dependencies","metadata":{}},{"cell_type":"code","source":"#upgrade pytorch to 1.12\n!pip install /kaggle/input/pytorch112-cu113/{torch-1.12.1+cu113-cp37-cp37m-linux_x86_64.whl,torchvision-0.13.1+cu113-cp37-cp37m-linux_x86_64.whl}\n#install nvidia-pyindex\n!pip install /kaggle/input/torch-tensorrt-pkg/nvidia_pyindex-1.0.9-py3-none-any.whl\n#install nvidia_tensorrt\n!mkdir -p /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cublas-cu11-2022.4.8.xyz /tmp/pip/cache/nvidia-cublas-cu11-2022.4.8.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cuda-runtime-cu11-2022.4.25.xyz /tmp/pip/cache/nvidia-cuda-runtime-cu11-2022.4.25.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cudnn-cu11-2022.5.19.xyz /tmp/pip/cache/nvidia-cudnn-cu11-2022.5.19.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cublas_cu117-11.10.1.25-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cuda_runtime_cu117-11.7.60-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cudnn_cu116-8.4.0.27-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_tensorrt-8.4.3.1-cp37-none-linux_x86_64.whl /tmp/pip/cache/\n!pip install --no-index --find-links /tmp/pip/cache/ nvidia_tensorrt\n#install torch_tensorrt\n!pip install /kaggle/input/torch-tensorrt-pkg/torch_tensorrt-1.2.0-cp37-cp37m-linux_x86_64.whl\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-03T20:59:31.482275Z","iopub.execute_input":"2023-01-03T20:59:31.483318Z","iopub.status.idle":"2023-01-03T21:03:11.784834Z","shell.execute_reply.started":"2023-01-03T20:59:31.48319Z","shell.execute_reply":"2023-01-03T21:03:11.783671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries and benchmarking functions","metadata":{}},{"cell_type":"code","source":"import torch\nimport tensorrt\nimport torch_tensorrt\nimport sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\nimport timm\n\n!mkdir -p /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/few-imagenet-examples/efficientnet_b0_ra-3dd342df.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/few-imagenet-examples/efficientnet_b2_ra-bcdf34b7.pth /root/.cache/torch/hub/checkpoints/\n\nimport time\nimport numpy as np\nimport torch.backends.cudnn as cudnn\nfrom timm.data import resolve_data_config\nfrom timm.data.transforms_factory import create_transform\nfrom PIL import Image\nfrom torchvision import transforms\nimport matplotlib.pyplot as plt\nimport json\n\n# loading labels\nwith open(\"/kaggle/input/few-imagenet-examples/data/imagenet_class_index.json\") as json_file: \n    d = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:11.788898Z","iopub.execute_input":"2023-01-03T21:03:11.789205Z","iopub.status.idle":"2023-01-03T21:03:18.032463Z","shell.execute_reply.started":"2023-01-03T21:03:11.789176Z","shell.execute_reply":"2023-01-03T21:03:18.031101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE=1024\nN_CHANNELS=3\nBATCH_SIZE=8","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:18.03447Z","iopub.execute_input":"2023-01-03T21:03:18.035437Z","iopub.status.idle":"2023-01-03T21:03:18.041639Z","shell.execute_reply.started":"2023-01-03T21:03:18.035395Z","shell.execute_reply":"2023-01-03T21:03:18.040481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create model\nefn_b2_model = timm.create_model('efficientnet_b2',pretrained=True)\nmodel = efn_b2_model.eval().to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:18.045259Z","iopub.execute_input":"2023-01-03T21:03:18.045883Z","iopub.status.idle":"2023-01-03T21:03:22.226805Z","shell.execute_reply.started":"2023-01-03T21:03:18.045847Z","shell.execute_reply":"2023-01-03T21:03:22.225795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cudnn.benchmark = True\n\ndef efficientnet_preprocess():\n    config = resolve_data_config({}, model=model)\n    transform = create_transform(**config)\n    return transform\n\n# decode the results into ([predicted class, description], probability)\ndef predict(img_path, model, dtype):\n    img = Image.open(img_path)\n    preprocess = efficientnet_preprocess()\n    input_tensor = preprocess(img)\n    resize_tform = transforms.Compose([transforms.Resize((IMG_SIZE))])\n    input_tensor = resize_tform(input_tensor)\n    input_batch = input_tensor.unsqueeze(0) # create a mini-batch as expected by the model\n    if dtype=='fp16':\n        input_batch = input_batch.half()\n    \n    # move the input and model to GPU for speed if available\n    if torch.cuda.is_available():\n        input_batch = input_batch.to('cuda')\n        model.to('cuda')\n\n    with torch.no_grad():\n        output = model(input_batch)\n        # Tensor of shape 1000, with confidence scores over Imagenet's 1000 classes\n        sm_output = torch.nn.functional.softmax(output[0], dim=0)\n        \n    ind = torch.argmax(sm_output)\n    return d[str(ind.item())], sm_output[ind] #([predicted class, description], probability)\n\ndef benchmark(model, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp32', nwarmup=50, nruns=10000):\n    input_data = torch.randn(input_shape)\n    input_data = input_data.to(\"cuda\")\n    if dtype=='fp16':\n        input_data = input_data.half()\n        \n    print(\"Warm up ...\")\n    with torch.no_grad():\n        for _ in range(nwarmup):\n            features = model(input_data)\n    torch.cuda.synchronize()\n    print(\"Start timing ...\")\n    timings = []\n    with torch.no_grad():\n        for i in range(1, nruns+1):\n            start_time = time.time()\n            features = model(input_data)\n            torch.cuda.synchronize()\n            end_time = time.time()\n            timings.append(end_time - start_time)\n            if i%10==0:\n                print('Iteration %d/%d, avg batch time %.2f ms'%(i, nruns, np.mean(timings)*1000))\n\n    print(\"Input shape:\", input_data.size())\n    print(\"Output features size:\", features.size())\n    print('Average throughput: %.2f images/second'%(input_shape[0]/np.mean(timings)))","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:22.228381Z","iopub.execute_input":"2023-01-03T21:03:22.228732Z","iopub.status.idle":"2023-01-03T21:03:22.244742Z","shell.execute_reply.started":"2023-01-03T21:03:22.228696Z","shell.execute_reply":"2023-01-03T21:03:22.243812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run regular torch model (fp32), no torch_tensorrt","metadata":{}},{"cell_type":"code","source":"# Model benchmark in Pytorch fp32 without Torch-TensorRT\nbenchmark(model, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:22.246414Z","iopub.execute_input":"2023-01-03T21:03:22.246826Z","iopub.status.idle":"2023-01-03T21:03:43.258784Z","shell.execute_reply.started":"2023-01-03T21:03:22.246791Z","shell.execute_reply":"2023-01-03T21:03:43.257804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and infer torch_tensorrt model (fp32)","metadata":{}},{"cell_type":"code","source":"trt_model_fp32 = torch_tensorrt.compile(model, inputs = [torch_tensorrt.Input(min_shape=[1, N_CHANNELS, IMG_SIZE, IMG_SIZE],opt_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],max_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],dtype=torch.float32)],\n    enabled_precisions = torch.float32, # Run with FP32\n    workspace_size = 1 << 32\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:03:43.260022Z","iopub.execute_input":"2023-01-03T21:03:43.261022Z","iopub.status.idle":"2023-01-03T21:05:12.278835Z","shell.execute_reply.started":"2023-01-03T21:03:43.260983Z","shell.execute_reply":"2023-01-03T21:05:12.277794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the average time taken by a batch of input\nbenchmark(trt_model_fp32, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:05:12.280994Z","iopub.execute_input":"2023-01-03T21:05:12.281621Z","iopub.status.idle":"2023-01-03T21:05:24.412071Z","shell.execute_reply.started":"2023-01-03T21:05:12.281584Z","shell.execute_reply":"2023-01-03T21:05:24.410962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matched results between pytorch and torch_tensorrt models.","metadata":{}},{"cell_type":"code","source":"#FP32 prediction\n\nfor i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model, dtype='fp32')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:05:24.414968Z","iopub.execute_input":"2023-01-03T21:05:24.415689Z","iopub.status.idle":"2023-01-03T21:05:26.642169Z","shell.execute_reply.started":"2023-01-03T21:05:24.415657Z","shell.execute_reply":"2023-01-03T21:05:26.641192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FP32 prediction torch_tensorrt\nfor i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, trt_model_fp32, dtype='fp32')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:05:26.645006Z","iopub.execute_input":"2023-01-03T21:05:26.646376Z","iopub.status.idle":"2023-01-03T21:05:28.048117Z","shell.execute_reply.started":"2023-01-03T21:05:26.646336Z","shell.execute_reply":"2023-01-03T21:05:28.047179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run regular torch model (fp16), no torch_tensorrt","metadata":{}},{"cell_type":"code","source":"# Model benchmark in Pytorch fp32 without Torch-TensorRT\nbenchmark(model.half().eval(), input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp16', nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:05:28.0498Z","iopub.execute_input":"2023-01-03T21:05:28.050489Z","iopub.status.idle":"2023-01-03T21:05:42.592562Z","shell.execute_reply.started":"2023-01-03T21:05:28.050448Z","shell.execute_reply":"2023-01-03T21:05:42.591502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and infer torch_tensorrt model (fp16)","metadata":{}},{"cell_type":"code","source":"trt_model_fp16 = torch_tensorrt.compile(model.half().eval(), inputs = [torch_tensorrt.Input(min_shape=[1, N_CHANNELS, IMG_SIZE, IMG_SIZE],opt_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],max_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],dtype=torch.half)],\n    enabled_precisions = {torch.half}, # Run with FP16\n    workspace_size = 1 << 32\n)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:05:42.594286Z","iopub.execute_input":"2023-01-03T21:05:42.594983Z","iopub.status.idle":"2023-01-03T21:07:53.911028Z","shell.execute_reply.started":"2023-01-03T21:05:42.594943Z","shell.execute_reply":"2023-01-03T21:07:53.909942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the average time taken by a batch of input\nbenchmark(trt_model_fp16, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp16', nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:07:53.913088Z","iopub.execute_input":"2023-01-03T21:07:53.913871Z","iopub.status.idle":"2023-01-03T21:08:04.354223Z","shell.execute_reply.started":"2023-01-03T21:07:53.91383Z","shell.execute_reply":"2023-01-03T21:08:04.353104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matched results between torch and torch_tensorrt models (FP16)","metadata":{}},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model.half().eval(), dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:08:04.355749Z","iopub.execute_input":"2023-01-03T21:08:04.356331Z","iopub.status.idle":"2023-01-03T21:08:06.102787Z","shell.execute_reply.started":"2023-01-03T21:08:04.356291Z","shell.execute_reply":"2023-01-03T21:08:06.099394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, trt_model_fp16, dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:08:06.105085Z","iopub.execute_input":"2023-01-03T21:08:06.106031Z","iopub.status.idle":"2023-01-03T21:08:07.879961Z","shell.execute_reply.started":"2023-01-03T21:08:06.105992Z","shell.execute_reply":"2023-01-03T21:08:07.879051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save and load torch_tensorrt model for inference.","metadata":{}},{"cell_type":"code","source":"#save and reload trt model for inference:\ntorch.jit.save(trt_model_fp16, f\"/kaggle/working/trt_model_fp16.ts\")\n\nprint('Loading TensorRT model ...')\nmodel_loaded = torch.jit.load(f\"/kaggle/working/trt_model_fp16.ts\")","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:11:46.285608Z","iopub.execute_input":"2023-01-03T21:11:46.286135Z","iopub.status.idle":"2023-01-03T21:11:46.564513Z","shell.execute_reply.started":"2023-01-03T21:11:46.28609Z","shell.execute_reply":"2023-01-03T21:11:46.563471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model_loaded, dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-01-03T21:12:39.970626Z","iopub.execute_input":"2023-01-03T21:12:39.971001Z","iopub.status.idle":"2023-01-03T21:12:41.466738Z","shell.execute_reply.started":"2023-01-03T21:12:39.970969Z","shell.execute_reply":"2023-01-03T21:12:41.465557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}