microsoft / microsoft/onnxruntime

[CPU] quant op input type error

Open
#13,938 3 comments 0 reactions 1 assignee View on GitHub

@shalvamist is already working on this.

Since Dec 13, 2022.

platform:web
Dominant language
C++
Stars
21.9k
Forks
4.2k
Avg merge
4d 11h
Merged PRs (30d)
184

Description

Describe the issue

I have a quanted model(int8),

  • when I use cpu or wasm backend, from profiling file, I saw the quant op input is uint8 and the performance is bad
  • when I use wasm-xnnpack backend, from profiling file, I saw the quant op input is int8
    image
To reproduce

convnets_modified_v1.zip

import cv2
import time
import torch
import numpy as np
import onnx
import onnxruntime as ort
import multiprocessing

class OnnxModel():
    def __init__(self, onnx_path):
        self.onnx_session = onnxruntime.InferenceSession(onnx_path)
        self.input_name = self.get_input_name(self.onnx_session)
        self.output_name = self.get_output_name(self.onnx_session)
        print("input_name:{}".format(self.input_name))
        print("output_name:{}".format(self.output_name))

    def get_output_name(self, onnx_session):
        output_name = []
        for node in onnx_session.get_outputs():
            output_name.append(node.name)
        return output_name

    def get_input_name(self, onnx_session):
        input_name = []
        for node in onnx_session.get_inputs():
            input_name.append(node.name)
        return input_name

    def get_input_feed(self, input_name, image_numpy):
        input_feed = {}
        for name in input_name:
            input_feed[name] = image_numpy
        return input_feed

    def forward(self, image_numpy):
        # scores, boxes = self.onnx_session.run(None, {self.input_name: image_numpy})
        # scores, boxes = self.onnx_session.run(self.output_name, input_feed={self.input_name: iimage_numpy})
        input_feed = self.get_input_feed(self.input_name, image_numpy)
        scores, boxes = self.onnx_session.run(self.output_name, input_feed=input_feed)
        return scores, boxes

def sleepTime(hour, min, sec):
    return hour * 3600 + min * 60 + sec

def get_tensor_from_videoStream(frame):
    frame_list = []
    frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
    frame_list.append(frame)
    # cap.release()
    result_frame = torch.as_tensor(np.stack(frame_list))
    # print(np.stack(frame_list).shape)
    src = np.stack(frame_list)
    src = src.transpose(0,3,1,2)
    [a,b,c,d] = src.shape
    float_list = np.empty(src.shape, dtype = np.float32)
    for i in range(a):
        for j in range(b):
            for m in range(c):
                for n in range(d):
                    float_list[i,j,m,n] = src[i,j,m,n] / 255.0
    return float_list

if __name__=='__main__':
    A = np.array([0.0], dtype = np.float32)
    B = np.array([0.25], dtype = np.float32)
    A1 = np.resize(A, (1,1,1,1))
    r1i = np.zeros((1, 16, 80, 90), dtype=np.float32)
    r2i = np.zeros((1, 20, 40, 45), dtype=np.float32)
    r3i = np.zeros((1, 40, 20, 23), dtype=np.float32)
    r4i = np.zeros((1, 64, 10, 12), dtype=np.float32)
    downsample_ratio = torch.as_tensor(B)

    second = sleepTime(0, 0, 1)
    sess_options = ort.SessionOptions()
    sess_options.intra_op_num_threads = 1
    # sess_options.inter_op_num_threads = 0

    sess_options.enable_profiling = True

    # sess_options.intra_op_num_threads = 2
    # sess_options.execution_mode = ort.ExecutionMode.ORT_PARALLEL
    # sess_options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL

    model_name = "ort_web/convnets_modified_v1.onnx"
    video_path = 0
    video_path = "/data/meeting_02_720x640.mp4"
    print(multiprocessing.cpu_count())
    ort_session = ort.InferenceSession(model_name, sess_options, providers=['CPUExecutionProvider'])
    # ort_session.set_providers(['CUDAExecutionProvider'])
    options = ort_session.get_session_options()

    cap = cv2.VideoCapture(video_path)

    nums = 0
    avg = 0
    while nums <= 10:
        nums += 1
        ret, frame = cap.read()
        dim = (180, 160)
        resized = cv2.resize(frame, dim, interpolation = cv2.INTER_AREA)
        tensor = get_tensor_from_videoStream(resized)
        print(tensor.shape)

        start_time = time.time()
        pha, r1o, r2o, r3o, r4o = ort_session.run(None, {'src':tensor, 'r1i':r1i, 'r2i':r2i, 'r3i':r3i, 'r4i':r4i})
        # outputs= ort_session.run(None, {'src':tensor, 'r1i':A1, 'r2i':A1, 'r3i':A1, 'r4i':A1, 'downsample_ratio':B})
        end_time = time.time()

        bgr = np.array([0.47, 1., 0.6]).reshape((3, 1, 1))

        duration = end_time - start_time
        if avg == 0:
            avg = duration
        else:
            exp = 0.8
            avg = avg * exp + duration * (1-exp)
        print(f"the running time is : {avg} s")
        # print(outputs)
Urgency

urgent

Platform

Linux

OS Version

cenos-7.9

ONNX Runtime Installation

Built from Source

ONNX Runtime Version or Commit ID

release version

ONNX Runtime API

Python

Architecture

X64

Execution Provider

Default CPU

Execution Provider Library Version

No response

Contributor guide

Open the contributing guide

First steps

  1. Read the whole issue, then the project's contributing guide.
  2. Comment on the issue to say you are picking it up — it saves two people doing the same work.
  3. Fork the repository and make your change on a branch.
  4. Open a pull request that references the issue number.

Assessment

This issue has not been assessed yet.

Get new issues in your inbox

A short digest of beginner-friendly GitHub issues.