{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![Torch-Tensorrt](https://developer-blogs.nvidia.com/wp-content/uploads/2021/12/pytorch-torch-tensorrt.png)","metadata":{}},{"cell_type":"markdown","source":"[Torch-TensorRT](http://https://developer.nvidia.com/blog/accelerating-inference-up-to-6x-faster-in-pytorch-with-torch-tensorrt/) is an integration for PyTorch that leverages inference optimizations of TensorRT on NVIDIA GPUs.  This notebook demonstrates how to install the necessary libraries for torch_tensorrt and how to convert models for speedup.","metadata":{}},{"cell_type":"markdown","source":"Plenty of great torch_tensorrt example notebooks reside [here](https://github.com/pytorch/TensorRT/tree/master/notebooks).  This notebook was adapted from the efn notebook in that folder.","metadata":{}},{"cell_type":"markdown","source":"# Install Dependencies","metadata":{}},{"cell_type":"code","source":"#upgrade pytorch to 1.12\n!pip install /kaggle/input/pytorch112-cu113/{torch-1.12.1+cu113-cp37-cp37m-linux_x86_64.whl,torchvision-0.13.1+cu113-cp37-cp37m-linux_x86_64.whl}\n#install nvidia-pyindex\n!pip install /kaggle/input/torch-tensorrt-pkg/nvidia_pyindex-1.0.9-py3-none-any.whl\n#install nvidia_tensorrt\n!mkdir -p /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cublas-cu11-2022.4.8.xyz /tmp/pip/cache/nvidia-cublas-cu11-2022.4.8.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cuda-runtime-cu11-2022.4.25.xyz /tmp/pip/cache/nvidia-cuda-runtime-cu11-2022.4.25.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia-cudnn-cu11-2022.5.19.xyz /tmp/pip/cache/nvidia-cudnn-cu11-2022.5.19.tar.gz\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cublas_cu117-11.10.1.25-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cuda_runtime_cu117-11.7.60-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_cudnn_cu116-8.4.0.27-py3-none-manylinux1_x86_64.whl /tmp/pip/cache/\n!cp /kaggle/input/torch-tensorrt-pkg/nvidia_tensorrt-8.4.3.1-cp37-none-linux_x86_64.whl /tmp/pip/cache/\n!pip install --no-index --find-links /tmp/pip/cache/ nvidia_tensorrt\n#install torch_tensorrt\n!pip install /kaggle/input/torch-tensorrt-pkg/torch_tensorrt-1.2.0-cp37-cp37m-linux_x86_64.whl\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-10T08:25:21.153902Z","iopub.execute_input":"2023-03-10T08:25:21.154332Z","iopub.status.idle":"2023-03-10T08:29:13.147664Z","shell.execute_reply.started":"2023-03-10T08:25:21.154241Z","shell.execute_reply":"2023-03-10T08:29:13.146506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries and benchmarking functions","metadata":{}},{"cell_type":"code","source":"import torch\nimport tensorrt\nimport torch_tensorrt\nimport sys\nsys.path.append('../input/timm-pytorch-image-models/pytorch-image-models-master')\nimport timm\n\n!mkdir -p /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/few-imagenet-examples/efficientnet_b0_ra-3dd342df.pth /root/.cache/torch/hub/checkpoints/\n!cp /kaggle/input/few-imagenet-examples/efficientnet_b2_ra-bcdf34b7.pth /root/.cache/torch/hub/checkpoints/\n\nimport time\nimport numpy as np\nimport torch.backends.cudnn as cudnn\nfrom timm.data import resolve_data_config\nfrom timm.data.transforms_factory import create_transform\nfrom PIL import Image\nfrom torchvision import transforms\nimport matplotlib.pyplot as plt\nimport json\n\n# loading labels\nwith open(\"/kaggle/input/few-imagenet-examples/data/imagenet_class_index.json\") as json_file: \n    d = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:37:04.472465Z","iopub.execute_input":"2023-03-10T08:37:04.473201Z","iopub.status.idle":"2023-03-10T08:37:14.694647Z","shell.execute_reply.started":"2023-03-10T08:37:04.473162Z","shell.execute_reply":"2023-03-10T08:37:14.693269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE=1024\nN_CHANNELS=3\nBATCH_SIZE=8","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:37:22.57608Z","iopub.execute_input":"2023-03-10T08:37:22.577841Z","iopub.status.idle":"2023-03-10T08:37:22.582891Z","shell.execute_reply.started":"2023-03-10T08:37:22.577793Z","shell.execute_reply":"2023-03-10T08:37:22.581896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create model\nefn_b2_model = timm.create_model('efficientnet_b2',pretrained=True)\nmodel = efn_b2_model.eval().to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:37:39.907325Z","iopub.execute_input":"2023-03-10T08:37:39.907947Z","iopub.status.idle":"2023-03-10T08:37:44.648644Z","shell.execute_reply.started":"2023-03-10T08:37:39.90791Z","shell.execute_reply":"2023-03-10T08:37:44.647337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cudnn.benchmark = True\n\ndef efficientnet_preprocess():\n    config = resolve_data_config({}, model=model)\n    transform = create_transform(**config)\n    return transform\n\n# decode the results into ([predicted class, description], probability)\ndef predict(img_path, model, dtype):\n    img = Image.open(img_path)\n    preprocess = efficientnet_preprocess()\n    input_tensor = preprocess(img)\n    resize_tform = transforms.Compose([transforms.Resize((IMG_SIZE))])\n    input_tensor = resize_tform(input_tensor)\n    input_batch = input_tensor.unsqueeze(0) # create a mini-batch as expected by the model\n    if dtype=='fp16':\n        input_batch = input_batch.half()\n    \n    # move the input and model to GPU for speed if available\n    if torch.cuda.is_available():\n        input_batch = input_batch.to('cuda')\n        model.to('cuda')\n\n    with torch.no_grad():\n        output = model(input_batch)\n        # Tensor of shape 1000, with confidence scores over Imagenet's 1000 classes\n        sm_output = torch.nn.functional.softmax(output[0], dim=0)\n        \n    ind = torch.argmax(sm_output)\n    return d[str(ind.item())], sm_output[ind] #([predicted class, description], probability)\n\ndef benchmark(model, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp32', nwarmup=50, nruns=10000):\n    input_data = torch.randn(input_shape)\n    input_data = input_data.to(\"cuda\")\n    if dtype=='fp16':\n        input_data = input_data.half()\n        \n    print(\"Warm up ...\")\n    with torch.no_grad():\n        for _ in range(nwarmup):\n            features = model(input_data)\n    torch.cuda.synchronize()\n    print(\"Start timing ...\")\n    timings = []\n    with torch.no_grad():\n        for i in range(1, nruns+1):\n            start_time = time.time()\n            features = model(input_data)\n            torch.cuda.synchronize()\n            end_time = time.time()\n            timings.append(end_time - start_time)\n            if i%10==0:\n                print('Iteration %d/%d, avg batch time %.2f ms'%(i, nruns, np.mean(timings)*1000))\n\n    print(\"Input shape:\", input_data.size())\n    print(\"Output features size:\", features.size())\n    print('Average throughput: %.2f images/second'%(input_shape[0]/np.mean(timings)))","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:43:08.654997Z","iopub.execute_input":"2023-03-10T08:43:08.656086Z","iopub.status.idle":"2023-03-10T08:43:08.669407Z","shell.execute_reply.started":"2023-03-10T08:43:08.656038Z","shell.execute_reply":"2023-03-10T08:43:08.66829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run regular torch model (fp32), no torch_tensorrt","metadata":{}},{"cell_type":"code","source":"# Model benchmark in Pytorch fp32 without Torch-TensorRT\nbenchmark(model, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:43:25.200323Z","iopub.execute_input":"2023-03-10T08:43:25.201427Z","iopub.status.idle":"2023-03-10T08:43:56.273412Z","shell.execute_reply.started":"2023-03-10T08:43:25.201388Z","shell.execute_reply":"2023-03-10T08:43:56.272281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and infer torch_tensorrt model (fp32)","metadata":{}},{"cell_type":"code","source":"trt_model_fp32 = torch_tensorrt.compile(model, inputs = [torch_tensorrt.Input(min_shape=[1, N_CHANNELS, IMG_SIZE, IMG_SIZE],opt_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],max_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],dtype=torch.float32)],\n    enabled_precisions = torch.float32, # Run with FP32\n    workspace_size = 1 << 32\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:45:42.7169Z","iopub.execute_input":"2023-03-10T08:45:42.717293Z","iopub.status.idle":"2023-03-10T08:47:30.168352Z","shell.execute_reply.started":"2023-03-10T08:45:42.717251Z","shell.execute_reply":"2023-03-10T08:47:30.167307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"help(torch_tensorrt.compile)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:47:55.585907Z","iopub.execute_input":"2023-03-10T08:47:55.586263Z","iopub.status.idle":"2023-03-10T08:47:55.594211Z","shell.execute_reply.started":"2023-03-10T08:47:55.586233Z","shell.execute_reply":"2023-03-10T08:47:55.593133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the average time taken by a batch of input\nbenchmark(trt_model_fp32, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:48:35.803553Z","iopub.execute_input":"2023-03-10T08:48:35.804571Z","iopub.status.idle":"2023-03-10T08:48:54.250297Z","shell.execute_reply.started":"2023-03-10T08:48:35.804533Z","shell.execute_reply":"2023-03-10T08:48:54.249126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matched results between pytorch and torch_tensorrt models.","metadata":{}},{"cell_type":"code","source":"#FP32 prediction\n\nfor i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model, dtype='fp32')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:49:23.700138Z","iopub.execute_input":"2023-03-10T08:49:23.700518Z","iopub.status.idle":"2023-03-10T08:49:26.523191Z","shell.execute_reply.started":"2023-03-10T08:49:23.700484Z","shell.execute_reply":"2023-03-10T08:49:26.522206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FP32 prediction torch_tensorrt\nfor i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, trt_model_fp32, dtype='fp32')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:49:48.471774Z","iopub.execute_input":"2023-03-10T08:49:48.472149Z","iopub.status.idle":"2023-03-10T08:49:49.897242Z","shell.execute_reply.started":"2023-03-10T08:49:48.472118Z","shell.execute_reply":"2023-03-10T08:49:49.896331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run regular torch model (fp16), no torch_tensorrt","metadata":{}},{"cell_type":"code","source":"# Model benchmark in Pytorch fp32 without Torch-TensorRT\nbenchmark(model.half().eval(), input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp16', nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:51:09.249413Z","iopub.execute_input":"2023-03-10T08:51:09.250412Z","iopub.status.idle":"2023-03-10T08:51:32.167424Z","shell.execute_reply.started":"2023-03-10T08:51:09.250375Z","shell.execute_reply":"2023-03-10T08:51:32.165995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compile and infer torch_tensorrt model (fp16)","metadata":{}},{"cell_type":"code","source":"trt_model_fp16 = torch_tensorrt.compile(model.half().eval(), inputs = [torch_tensorrt.Input(min_shape=[1, N_CHANNELS, IMG_SIZE, IMG_SIZE],opt_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],max_shape=[BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE],dtype=torch.half)],\n    enabled_precisions = {torch.half}, # Run with FP16\n    workspace_size = 1 << 32\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:53:24.861504Z","iopub.execute_input":"2023-03-10T08:53:24.862238Z","iopub.status.idle":"2023-03-10T08:56:01.715404Z","shell.execute_reply.started":"2023-03-10T08:53:24.862201Z","shell.execute_reply":"2023-03-10T08:56:01.714483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Obtain the average time taken by a batch of input\nbenchmark(trt_model_fp16, input_shape=(BATCH_SIZE, N_CHANNELS, IMG_SIZE, IMG_SIZE), dtype='fp16', nruns=100)","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:56:39.347727Z","iopub.execute_input":"2023-03-10T08:56:39.34809Z","iopub.status.idle":"2023-03-10T08:56:55.002493Z","shell.execute_reply.started":"2023-03-10T08:56:39.34806Z","shell.execute_reply":"2023-03-10T08:56:55.00134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matched results between torch and torch_tensorrt models (FP16)","metadata":{}},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model.half().eval(), dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:57:25.347864Z","iopub.execute_input":"2023-03-10T08:57:25.348223Z","iopub.status.idle":"2023-03-10T08:57:27.888021Z","shell.execute_reply.started":"2023-03-10T08:57:25.348193Z","shell.execute_reply":"2023-03-10T08:57:27.886947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, trt_model_fp16, dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-10T08:57:31.533978Z","iopub.execute_input":"2023-03-10T08:57:31.534336Z","iopub.status.idle":"2023-03-10T08:57:32.935724Z","shell.execute_reply.started":"2023-03-10T08:57:31.534306Z","shell.execute_reply":"2023-03-10T08:57:32.93442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save and load torch_tensorrt model for inference.","metadata":{}},{"cell_type":"code","source":"#save and reload trt model for inference:\ntorch.jit.save(trt_model_fp16, f\"/kaggle/working/trt_model_fp16.ts\")\n\nprint('Loading TensorRT model ...')\nmodel_loaded = torch.jit.load(f\"/kaggle/working/trt_model_fp16.ts\")","metadata":{"execution":{"iopub.status.busy":"2023-03-10T09:01:41.930563Z","iopub.execute_input":"2023-03-10T09:01:41.93121Z","iopub.status.idle":"2023-03-10T09:01:42.183159Z","shell.execute_reply.started":"2023-03-10T09:01:41.931174Z","shell.execute_reply":"2023-03-10T09:01:42.182113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(4):\n    img_path = '/kaggle/input/few-imagenet-examples/data/img%d.JPG'%i\n    img = Image.open(img_path)\n    \n    pred, prob = predict(img_path, model_loaded, dtype='fp16')\n    print('{} - Predicted: {}, Probablility: {}'.format(img_path, pred, prob))\n\n    plt.subplot(2,2,i+1)\n    plt.imshow(img);\n    plt.axis('off');\n    plt.title(pred[1])","metadata":{"execution":{"iopub.status.busy":"2023-03-10T09:01:44.876025Z","iopub.execute_input":"2023-03-10T09:01:44.876383Z","iopub.status.idle":"2023-03-10T09:01:46.342154Z","shell.execute_reply.started":"2023-03-10T09:01:44.876347Z","shell.execute_reply":"2023-03-10T09:01:46.34105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}