From b8d8c8ab922e731e0396ed89cd29a7ee9b37d0cc Mon Sep 17 00:00:00 2001 From: wang-xinyu Date: Wed, 7 Sep 2022 14:58:12 +0800 Subject: [PATCH] yolov5 cls cpp postprocessing --- yolov5/README.md | 7 +++-- yolov5/yolov5_cls.cpp | 67 ++++++++++++++++++++++++++++++++++------ yolov5/yolov5_cls_trt.py | 5 ++- 3 files changed, 65 insertions(+), 14 deletions(-) diff --git a/yolov5/README.md b/yolov5/README.md index 216c558..0755d4e 100644 --- a/yolov5/README.md +++ b/yolov5/README.md @@ -4,9 +4,10 @@ The Pytorch implementation is [ultralytics/yolov5](https://github.com/ultralytic ## Different versions of yolov5 -Currently, we support yolov5 v1.0, v2.0, v3.0, v3.1, v4.0, v5.0 and v6.0. +Currently, we support yolov5 v1.0, v2.0, v3.0, v3.1, v4.0, v5.0, v6.0, v6.2. -- For yolov5 v6.0, download .pt from [yolov5 release v6.0](https://github.com/ultralytics/yolov5/releases/tag/v6.0), `git clone -b v6.0 https://github.com/ultralytics/yolov5.git` and `git clone https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in current page. +- For yolov5 v6.2, download .pt from [yolov5 release v6.2](https://github.com/ultralytics/yolov5/releases/tag/v6.2), `git clone -b v6.2 https://github.com/ultralytics/yolov5.git` and `git clone -b yolov5-v6.2 https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in current page. +- For yolov5 v6.0, download .pt from [yolov5 release v6.0](https://github.com/ultralytics/yolov5/releases/tag/v6.0), `git clone -b v6.0 https://github.com/ultralytics/yolov5.git` and `git clone -b yolov5-v6.0 https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in [tensorrtx/yolov5-v6.0](https://github.com/wang-xinyu/tensorrtx/tree/yolov5-v6.0/yolov5). - For yolov5 v5.0, download .pt from [yolov5 release v5.0](https://github.com/ultralytics/yolov5/releases/tag/v5.0), `git clone -b v5.0 https://github.com/ultralytics/yolov5.git` and `git clone -b yolov5-v5.0 https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in [tensorrtx/yolov5-v5.0](https://github.com/wang-xinyu/tensorrtx/tree/yolov5-v5.0/yolov5). - For yolov5 v4.0, download .pt from [yolov5 release v4.0](https://github.com/ultralytics/yolov5/releases/tag/v4.0), `git clone -b v4.0 https://github.com/ultralytics/yolov5.git` and `git clone -b yolov5-v4.0 https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in [tensorrtx/yolov5-v4.0](https://github.com/wang-xinyu/tensorrtx/tree/yolov5-v4.0/yolov5). - For yolov5 v3.1, download .pt from [yolov5 release v3.1](https://github.com/ultralytics/yolov5/releases/tag/v3.1), `git clone -b v3.1 https://github.com/ultralytics/yolov5.git` and `git clone -b yolov5-v3.1 https://github.com/wang-xinyu/tensorrtx.git`, then follow how-to-run in [tensorrtx/yolov5-v3.1](https://github.com/wang-xinyu/tensorrtx/tree/yolov5-v3.1/yolov5). @@ -71,6 +72,8 @@ python yolov5_trt.py python yolov5_trt_cuda_python.py ``` +5. optional, run yolov5 classification models with similar steps + # INT8 Quantization 1. Prepare calibration images, you can randomly select 1000s images from your train set. For coco, you can also download my calibration images `coco_calib` from [GoogleDrive](https://drive.google.com/drive/folders/1s7jE9DtOngZMzJC1uL307J2MiaGwdRSI?usp=sharing) or [BaiduPan](https://pan.baidu.com/s/1GOm_-JobpyLMAqZWCDUhKg) pwd: a9wh diff --git a/yolov5/yolov5_cls.cpp b/yolov5/yolov5_cls.cpp index 3d89dc4..6ea2504 100644 --- a/yolov5/yolov5_cls.cpp +++ b/yolov5/yolov5_cls.cpp @@ -1,6 +1,7 @@ #include #include #include +#include #include "cuda_utils.h" #include "logging.h" #include "common.hpp" @@ -9,17 +10,14 @@ #define USE_FP32 // set USE_INT8 or USE_FP16 or USE_FP32 #define DEVICE 0 // GPU id -#define NMS_THRESH 0.4 -#define CONF_THRESH 0.5 #define BATCH_SIZE 1 -#define MAX_IMAGE_INPUT_SIZE_THRESH 3000 * 3000 // ensure it exceed the maximum size in the input images ! // stuff we know about the network and the input/output blobs static const int INPUT_H = 224; static const int INPUT_W = 224; static const int CLASS_NUM = 1000; -static const int OUTPUT_SIZE = Yolo::MAX_OUTPUT_BBOX_COUNT * sizeof(Yolo::Detection) / sizeof(float) + 1; // we assume the yololayer outputs no more than MAX_OUTPUT_BBOX_COUNT boxes that conf >= 0.1 +static const int OUTPUT_SIZE = CLASS_NUM; const char* INPUT_BLOB_NAME = "data"; const char* OUTPUT_BLOB_NAME = "prob"; static Logger gLogger; @@ -37,6 +35,49 @@ static int get_depth(int x, float gd) { return std::max(r, 1); } +std::vector softmax(float *prob, int n) { + std::vector res; + float sum = 0.0f; + float t; + for (int i = 0; i < n; i++) { + t = expf(prob[i]); + res.push_back(t); + sum += t; + } + for (int i = 0; i < n; i++) { + res[i] /= sum; + } + return res; +} + +std::vector topk(const std::vector& vec, int k) { + std::vector topk_index; + std::vector vec_index(vec.size()); + std::iota(vec_index.begin(), vec_index.end(), 0); + + std::sort(vec_index.begin(), vec_index.end(), [&vec](size_t index_1, size_t index_2) { return vec[index_1] > vec[index_2]; }); + + int k_num = std::min(vec.size(), k); + + for (int i = 0; i < k_num; ++i) { + topk_index.push_back(vec_index[i]); + } + + return topk_index; +} + +std::vector read_classes(std::string file_name) { + std::vector classes; + std::ifstream ifs(file_name, std::ios::in); + assert(ifs.is_open()); + std::string s; + while (std::getline(ifs, s)) { + classes.push_back(s); + } + ifs.close(); + return classes; +} + ICudaEngine* build_engine(unsigned int maxBatchSize, IBuilder* builder, IBuilderConfig* config, DataType dt, float& gd, float& gw, std::string& wts_name) { INetworkDefinition* network = builder->createNetworkV2(0U); @@ -87,8 +128,7 @@ ICudaEngine* build_engine(unsigned int maxBatchSize, IBuilder* builder, IBuilder network->destroy(); // Release host memory - for (auto& mem : weightMap) - { + for (auto& mem : weightMap) { free((void*)(mem.second.values)); } @@ -210,6 +250,7 @@ int main(int argc, char** argv) { std::cerr << "read_files_in_dir failed." << std::endl; return -1; } + auto classes = read_classes("../imagenet_classes.txt"); static float data[BATCH_SIZE * 3 * INPUT_H * INPUT_W]; static float prob[BATCH_SIZE * OUTPUT_SIZE]; @@ -240,8 +281,6 @@ int main(int argc, char** argv) { for (int f = 0; f < (int)file_names.size(); f++) { fcount++; if (fcount < BATCH_SIZE && f + 1 != (int)file_names.size()) continue; - //auto start = std::chrono::system_clock::now(); - float* buffer_idx = (float*)buffers[inputIndex]; for (int b = 0; b < fcount; b++) { cv::Mat img = cv::imread(img_dir + "/" + file_names[f - fcount + 1 + b]); if (img.empty()) continue; @@ -251,7 +290,7 @@ int main(int argc, char** argv) { for (int row = 0; row < INPUT_H; ++row) { uchar* uc_pixel = pr_img.data + row * pr_img.step; for (int col = 0; col < INPUT_W; ++col) { - data[b * 3 * INPUT_H * INPUT_W + i] = ((float)uc_pixel[2] / 255.0 - 0.485) / 0.229; // R-0.485 + data[b * 3 * INPUT_H * INPUT_W + i] = ((float)uc_pixel[2] / 255.0 - 0.485) / 0.229; // R - 0.485 data[b * 3 * INPUT_H * INPUT_W + i + INPUT_H * INPUT_W] = ((float)uc_pixel[1] / 255.0 - 0.456) / 0.224; data[b * 3 * INPUT_H * INPUT_W + i + 2 * INPUT_H * INPUT_W] = ((float)uc_pixel[0] / 255.0 - 0.406) / 0.225; uc_pixel += 3; @@ -264,6 +303,15 @@ int main(int argc, char** argv) { doInference(*context, stream, buffers, data, prob, BATCH_SIZE); auto end = std::chrono::system_clock::now(); std::cout << "inference time: " << std::chrono::duration_cast(end - start).count() << "ms" << std::endl; + for (int b = 0; b < fcount; b++) { + float *p = &prob[b * OUTPUT_SIZE]; + auto res = softmax(p, OUTPUT_SIZE); + auto topk_idx = topk(res, 3); + std::cout << file_names[f - fcount + 1 + b] << std::endl; + for (auto idx: topk_idx) { + std::cout << " " << classes[idx] << " " << res[idx] << std::endl; + } + } fcount = 0; } @@ -279,3 +327,4 @@ int main(int argc, char** argv) { return 0; } + diff --git a/yolov5/yolov5_cls_trt.py b/yolov5/yolov5_cls_trt.py index 6f5507b..510c5da 100644 --- a/yolov5/yolov5_cls_trt.py +++ b/yolov5/yolov5_cls_trt.py @@ -134,6 +134,7 @@ class YoLov5TRT(object): output) cv2.putText(batch_image_raw[i], str( classes_ls), (10, 50), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 1, cv2.LINE_AA) + print(classes_ls, predicted_conf_ls) return batch_image_raw, end - start def destroy(self): @@ -173,7 +174,7 @@ class YoLov5TRT(object): output_data = output_data.reshape(self.batch_size, -1) output_data = torch.Tensor(output_data) p = torch.nn.functional.softmax(output_data, dim=1) - score, index = torch.topk(output_data, 3) + score, index = torch.topk(p, 3) for ind in range(index.shape[0]): input_category_id = index[ind][0].item() # 716 category_id_ls.append(input_category_id) @@ -219,8 +220,6 @@ if __name__ == "__main__": if len(sys.argv) > 1: engine_file_path = sys.argv[1] - if len(sys.argv) > 2: - PLUGIN_LIBRARY = sys.argv[2] if os.path.exists('output/'): shutil.rmtree('output/')