[IE Python Speech Sample] Migrate to OV 2.0 API (#9348)

* Create mvp

* Implement new API & Refactoring

* Fix -oname for models whose name of output layer contains a port number

* Fix step numbers

* Create utils.py

* Remove shebang from utils.py

* Fix `-iname` option
This commit is contained in:
Dmitry Pigasin
2021-12-27 09:22:15 +03:00
committed by GitHub
parent a9cee5f101
commit d6dcf58846
2 changed files with 204 additions and 183 deletions
+124 -183
View File
@@ -2,179 +2,106 @@
# -*- coding: utf-8 -*-
# Copyright (C) 2018-2021 Intel Corporation
# SPDX-License-Identifier: Apache-2.0
import argparse
import logging as log
import re
import sys
from timeit import default_timer
from typing import Union
from typing import Dict
import numpy as np
from openvino.preprocess import PrePostProcessor
from openvino.runtime import Core, InferRequest, Layout, Type, set_batch
from arg_parser import parse_args
from file_options import read_utterance_file, write_utterance_file
from openvino.inference_engine import ExecutableNetwork, IECore, IENetwork
# Operating Frequency for GNA HW devices for Core and Atom architecture
GNA_CORE_FREQUENCY = 400
GNA_ATOM_FREQUENCY = 200
from utils import (GNA_ATOM_FREQUENCY, GNA_CORE_FREQUENCY,
compare_with_reference, get_scale_factor, log,
parse_outputs_from_args, parse_scale_factors,
set_scale_factors)
def get_scale_factor(matrix: np.ndarray) -> float:
"""Get scale factor for quantization using utterance matrix"""
# Max to find scale factor
target_max = 16384
max_val = np.max(matrix)
if max_val == 0:
return 1.0
else:
return target_max / max_val
def infer_data(
data: dict, exec_net: ExecutableNetwork, input_blobs: list, output_blobs: list, cw_l: int = 0, cw_r: int = 0,
) -> np.ndarray:
def infer_data(data: Dict[str, np.ndarray], infer_request: InferRequest, cw_l: int = 0, cw_r: int = 0) -> np.ndarray:
"""Do a synchronous matrix inference"""
matrix_shape = next(iter(data.values())).shape
frames_to_infer = {}
result = {}
for blob_name in output_blobs:
output_shape = exec_net.outputs[blob_name].shape
batch_size = output_shape[0]
result[blob_name] = np.ndarray((matrix_shape[0], np.prod(output_shape[1:])))
batch_size = infer_request.get_input_tensor(0).shape[0]
num_of_frames = next(iter(data.values())).shape[0]
for i in range(-cw_l, matrix_shape[0] + cw_r, batch_size):
for output in infer_request.outputs:
result[output.any_name] = np.ndarray((num_of_frames, np.prod(tuple(output.shape)[1:])))
for i in range(-cw_l, num_of_frames + cw_r, batch_size):
if i < 0:
index = 0
elif i >= matrix_shape[0]:
index = matrix_shape[0] - 1
elif i >= num_of_frames:
index = num_of_frames - 1
else:
index = i
vectors = {blob_name: data[blob_name][index:index + batch_size] for blob_name in input_blobs}
for _input in infer_request.inputs:
frames_to_infer[_input.any_name] = data[_input.any_name][index:index + batch_size]
num_of_frames_to_infer = len(frames_to_infer[_input.any_name])
num_of_vectors = next(iter(vectors.values())).shape[0]
# Add [batch_size - num_of_frames_to_infer] zero rows to 2d numpy array
# Used to infer fewer frames than the batch size
frames_to_infer[_input.any_name] = np.pad(
frames_to_infer[_input.any_name],
[(0, batch_size - num_of_frames_to_infer), (0, 0)],
)
if num_of_vectors < batch_size:
temp = {blob_name: np.zeros((batch_size, vectors[blob_name].shape[1])) for blob_name in input_blobs}
frames_to_infer[_input.any_name] = frames_to_infer[_input.any_name].reshape(_input.tensor.shape)
for blob_name in input_blobs:
temp[blob_name][:num_of_vectors] = vectors[blob_name]
vectors = temp
for blob_name in input_blobs:
vectors[blob_name] = vectors[blob_name].reshape(exec_net.input_info[blob_name].input_data.shape)
vector_results = exec_net.infer(vectors)
frame_results = infer_request.infer(frames_to_infer)
if i - cw_r < 0:
continue
for blob_name in output_blobs:
vector_result = vector_results[blob_name].reshape((batch_size, result[blob_name].shape[1]))
result[blob_name][i - cw_r:i - cw_r + batch_size] = vector_result[:num_of_vectors]
for output in frame_results.keys():
vector_result = frame_results[output].reshape((batch_size, result[output.any_name].shape[1]))
result[output.any_name][i - cw_r:i - cw_r + batch_size] = vector_result[:num_of_frames_to_infer]
return result
def compare_with_reference(result: np.ndarray, reference: np.ndarray):
error_matrix = np.absolute(result - reference)
max_error = np.max(error_matrix)
sum_error = np.sum(error_matrix)
avg_error = sum_error / error_matrix.size
sum_square_error = np.sum(np.square(error_matrix))
avg_rms_error = np.sqrt(sum_square_error / error_matrix.size)
stdev_error = np.sqrt(sum_square_error / error_matrix.size - avg_error * avg_error)
log.info(f'max error: {max_error:.7f}')
log.info(f'avg error: {avg_error:.7f}')
log.info(f'avg rms error: {avg_rms_error:.7f}')
log.info(f'stdev error: {stdev_error:.7f}')
def get_input_layer_list(net: Union[IENetwork, ExecutableNetwork], args: argparse.Namespace) -> list:
"""Get a list of input layer names"""
return re.split(', |,', args.input_layers) if args.input_layers else [next(iter(net.input_info))]
def get_output_layer_list(net: Union[IENetwork, ExecutableNetwork],
args: argparse.Namespace, with_ports: bool) -> list:
"""Get a list of output layer names"""
if args.output_layers:
output_name_port = [output.split(':') for output in re.split(', |,', args.output_layers)]
if with_ports:
try:
return [(blob_name, int(port)) for blob_name, port in output_name_port]
except ValueError:
log.error('Incorrect value for -oname/--output_layers option, please specify a port for output layer.')
sys.exit(-4)
else:
return [blob_name for blob_name, _ in output_name_port]
else:
return [list(net.outputs.keys())[-1]]
def parse_scale_factors(args: argparse.Namespace) -> list:
"""Get a list of scale factors for input files"""
input_files = re.split(', |,', args.input)
scale_factors = re.split(', |,', str(args.scale_factor))
scale_factors = list(map(float, scale_factors))
if len(input_files) != len(scale_factors):
log.error(f'Incorrect command line for multiple inputs: {len(scale_factors)} scale factors provided for '
f'{len(input_files)} input files.')
sys.exit(-7)
for i, scale_factor in enumerate(scale_factors):
if float(scale_factor) < 0:
log.error(f'Scale factor for input #{i} (counting from zero) is out of range (must be positive).')
sys.exit(-8)
return scale_factors
def set_scale_factors(plugin_config: dict, scale_factors: list):
"""Set a scale factor provided for each input"""
for i, scale_factor in enumerate(scale_factors):
log.info(f'For input {i} using scale factor of {scale_factor:.7f}')
plugin_config[f'GNA_SCALE_FACTOR_{i}'] = str(scale_factor)
def main():
log.basicConfig(format='[ %(levelname)s ] %(message)s', level=log.INFO, stream=sys.stdout)
args = parse_args()
# ---------------------------Step 1. Initialize inference engine core--------------------------------------------------
log.info('Creating Inference Engine')
ie = IECore()
# --------------------------- Step 1. Initialize OpenVINO Runtime Core ------------------------------------------------
log.info('Creating OpenVINO Runtime Core')
core = Core()
# ---------------------------Step 2. Read a model in OpenVINO Intermediate Representation---------------
# --------------------------- Step 2. Read a model --------------------------------------------------------------------
if args.model:
log.info(f'Reading the network: {args.model}')
# .xml and .bin files
net = ie.read_network(model=args.model)
log.info(f'Reading the model: {args.model}')
# (.xml and .bin files) or (.onnx file)
model = core.read_model(args.model)
# ---------------------------Step 3. Configure input & output----------------------------------------------------------
log.info('Configuring input and output blobs')
# Mark layers from args.output_layers as outputs
# --------------------------- Step 3. Apply preprocessing -------------------------------------------------------------
if args.output_layers:
net.add_outputs(get_output_layer_list(net, args, with_ports=True))
output_names, output_ports = parse_outputs_from_args(args)
model.add_outputs(list(zip(output_names, output_ports)))
# Get names of input and output blobs
input_blobs = get_input_layer_list(net, args)
output_blobs = get_output_layer_list(net, args, with_ports=False)
ppp = PrePostProcessor(model)
# Set input and output precision manually
for blob_name in input_blobs:
net.input_info[blob_name].precision = 'FP32'
for i in range(len(model.inputs)):
ppp.input(i).tensor() \
.set_element_type(Type.f32) \
.set_layout(Layout('NC')) # noqa: N400
for blob_name in output_blobs:
net.outputs[blob_name].precision = 'FP32'
ppp.input(i).model().set_layout(Layout('NC'))
net.batch_size = args.batch_size if args.context_window_left + args.context_window_right == 0 else 1
for i in range(len(model.outputs)):
ppp.output(i).tensor().set_element_type(Type.f32)
# ---------------------------Step 4. Loading model to the device-------------------------------------------------------
model = ppp.build()
if args.context_window_left == args.context_window_right == 0:
set_batch(model, args.batch_size)
else:
set_batch(model, 1)
# ---------------------------Step 4. Configure plugin ---------------------------------------------------------
devices = args.device.replace('HETERO:', '').split(',')
plugin_config = {}
@@ -215,39 +142,17 @@ def main():
device_str = f'HETERO:{",".join(devices)}' if 'HETERO' in args.device else devices[0]
# --------------------------- Step 5. Loading model to the device -----------------------------------------------------
log.info('Loading the model to the plugin')
if args.model:
exec_net = ie.load_network(net, device_str, plugin_config)
compiled_model = core.compile_model(model, device_str, plugin_config)
else:
exec_net = ie.import_network(args.import_gna_model, device_str, plugin_config)
input_blobs = get_input_layer_list(exec_net, args)
output_blobs = get_output_layer_list(exec_net, args, with_ports=False)
if args.input:
input_files = re.split(', |,', args.input)
if len(input_blobs) != len(input_files):
log.error(f'Number of network inputs ({len(input_blobs)}) is not equal '
f'to number of ark files ({len(input_files)})')
sys.exit(-3)
if args.reference:
reference_files = re.split(', |,', args.reference)
if len(output_blobs) != len(reference_files):
log.error('The number of reference files is not equal to the number of network outputs.')
sys.exit(-5)
if args.output:
output_files = re.split(', |,', args.output)
if len(output_blobs) != len(output_files):
log.error('The number of output files is not equal to the number of network outputs.')
sys.exit(-6)
compiled_model = core.import_model(args.import_gna_model, device_str, plugin_config)
# --------------------------- Exporting GNA model using InferenceEngine AOT API ---------------------------------------
if args.export_gna_model:
log.info(f'Writing GNA Model to {args.export_gna_model}')
exec_net.export(args.export_gna_model)
compiled_model.export_model(args.export_gna_model)
return 0
if args.export_embedded_gna_model:
@@ -255,68 +160,104 @@ def main():
log.info(f'GNA embedded model export done for GNA generation {args.embedded_gna_configuration}')
return 0
# ---------------------------Step 5. Create infer request--------------------------------------------------------------
# load_network() method of the IECore class with a specified number of requests (default 1) returns an ExecutableNetwork
# instance which stores infer requests. So you already created Infer requests in the previous step.
# --------------------------- Step 6. Set up input --------------------------------------------------------------------
if args.input_layers:
input_names = re.split(', |,', args.input_layers)
else:
input_names = [_input.any_name for _input in compiled_model.inputs]
if args.output_layers:
output_names, output_ports = parse_outputs_from_args(args)
# If a name of output layer contains a port number then concatenate output_names and output_ports
if ':' in compiled_model.outputs[0].any_name:
output_names = [f'{output_names[i]}:{output_ports[i]}' for i in range(len(output_names))]
else:
output_names = [compiled_model.outputs[0].any_name]
if args.input:
input_files = re.split(', |,', args.input)
if len(input_names) != len(input_files):
log.error(f'Number of network inputs ({len(compiled_model.inputs)}) is not equal '
f'to number of ark files ({len(input_files)})')
sys.exit(-3)
if args.reference:
reference_files = re.split(', |,', args.reference)
if len(output_names) != len(reference_files):
log.error('The number of reference files is not equal to the number of network outputs.')
sys.exit(-5)
if args.output:
output_files = re.split(', |,', args.output)
if len(output_names) != len(output_files):
log.error('The number of output files is not equal to the number of network outputs.')
sys.exit(-6)
# ---------------------------Step 6. Prepare input---------------------------------------------------------------------
file_data = [read_utterance_file(file_name) for file_name in input_files]
input_data = {
utterance_name: {
input_blobs[i]: file_data[i][utterance_name] for i in range(len(input_blobs))
input_names[i]: file_data[i][utterance_name] for i in range(len(input_names))
}
for utterance_name in file_data[0].keys()
}
if args.reference:
references = {output_blobs[i]: read_utterance_file(reference_files[i]) for i in range(len(output_blobs))}
references = {output_names[i]: read_utterance_file(reference_files[i]) for i in range(len(output_names))}
# ---------------------------Step 7. Do inference----------------------------------------------------------------------
# --------------------------- Step 7. Create infer request ------------------------------------------------------------
infer_request = compiled_model.create_infer_request()
# --------------------------- Step 8. Do inference --------------------------------------------------------------------
log.info('Starting inference in synchronous mode')
results = {blob_name: {} for blob_name in output_blobs}
results = {name: {} for name in output_names}
total_infer_time = 0
for i, key in enumerate(sorted(input_data)):
start_infer_time = default_timer()
# Reset states between utterance inferences to remove a memory impact
for request in exec_net.requests:
for state in request.query_state():
state.reset()
for state in infer_request.query_state():
state.reset()
result = infer_data(
input_data[key], exec_net, input_blobs, output_blobs, args.context_window_left, args.context_window_right,
input_data[key],
infer_request,
args.context_window_left,
args.context_window_right,
)
for blob_name in result.keys():
results[blob_name][key] = result[blob_name]
for name in output_names:
results[name][key] = result[name]
infer_time = default_timer() - start_infer_time
total_infer_time += infer_time
num_of_frames = file_data[0][key].shape[0]
avg_infer_time_per_frame = infer_time / num_of_frames
# ---------------------------Step 8. Process output--------------------------------------------------------------------
# --------------------------- Step 9. Process output ------------------------------------------------------------------
log.info('')
log.info(f'Utterance {i} ({key}):')
log.info(f'Total time in Infer (HW and SW): {infer_time * 1000:.2f}ms')
log.info(f'Frames in utterance: {num_of_frames}')
log.info(f'Average Infer time per frame: {avg_infer_time_per_frame * 1000:.2f}ms')
for blob_name in output_blobs:
for name in output_names:
log.info('')
log.info(f'Output blob name: {blob_name}')
log.info(f'Number scores per frame: {results[blob_name][key].shape[1]}')
log.info(f'Output blob name: {name}')
log.info(f'Number scores per frame: {results[name][key].shape[1]}')
if args.reference:
log.info('')
compare_with_reference(results[blob_name][key], references[blob_name][key])
compare_with_reference(results[name][key], references[name][key])
if args.performance_counter:
if 'GNA' in args.device:
pc = exec_net.requests[0].get_perf_counts()
total_cycles = int(pc['1.1 Total scoring time in HW']['real_time'])
stall_cycles = int(pc['1.2 Stall scoring time in HW']['real_time'])
total_cycles = infer_request.profiling_info[0].real_time.total_seconds()
stall_cycles = infer_request.profiling_info[1].real_time.total_seconds()
active_cycles = total_cycles - stall_cycles
frequency = 10**6
if args.arch == 'CORE':
@@ -336,8 +277,8 @@ def main():
log.info(f'Total sample time: {total_infer_time * 1000:.2f}ms')
if args.output:
for i, blob_name in enumerate(results):
write_utterance_file(output_files[i], results[blob_name])
for i, name in enumerate(results):
write_utterance_file(output_files[i], results[name])
log.info(f'File {output_files[i]} was created!')
# ----------------------------------------------------------------------------------------------------------------------
+80
View File
@@ -0,0 +1,80 @@
# -*- coding: utf-8 -*-
# Copyright (C) 2018-2021 Intel Corporation
# SPDX-License-Identifier: Apache-2.0
import argparse
import logging as log
import re
import sys
from typing import List, Tuple
import numpy as np
# Operating Frequency for GNA HW devices for Core and Atom architecture
GNA_CORE_FREQUENCY = 400
GNA_ATOM_FREQUENCY = 200
log.basicConfig(format='[ %(levelname)s ] %(message)s', level=log.INFO, stream=sys.stdout)
def compare_with_reference(result: np.ndarray, reference: np.ndarray):
error_matrix = np.absolute(result - reference)
max_error = np.max(error_matrix)
sum_error = np.sum(error_matrix)
avg_error = sum_error / error_matrix.size
sum_square_error = np.sum(np.square(error_matrix))
avg_rms_error = np.sqrt(sum_square_error / error_matrix.size)
stdev_error = np.sqrt(sum_square_error / error_matrix.size - avg_error * avg_error)
log.info(f'max error: {max_error:.7f}')
log.info(f'avg error: {avg_error:.7f}')
log.info(f'avg rms error: {avg_rms_error:.7f}')
log.info(f'stdev error: {stdev_error:.7f}')
def get_scale_factor(matrix: np.ndarray) -> float:
"""Get scale factor for quantization using utterance matrix"""
# Max to find scale factor
target_max = 16384
max_val = np.max(matrix)
if max_val == 0:
return 1.0
else:
return target_max / max_val
def set_scale_factors(plugin_config: dict, scale_factors: list):
"""Set a scale factor provided for each input"""
for i, scale_factor in enumerate(scale_factors):
log.info(f'For input {i} using scale factor of {scale_factor:.7f}')
plugin_config[f'GNA_SCALE_FACTOR_{i}'] = str(scale_factor)
def parse_scale_factors(args: argparse.Namespace) -> list:
"""Get a list of scale factors for input files"""
input_files = re.split(', |,', args.input)
scale_factors = re.split(', |,', str(args.scale_factor))
scale_factors = list(map(float, scale_factors))
if len(input_files) != len(scale_factors):
log.error(f'Incorrect command line for multiple inputs: {len(scale_factors)} scale factors provided for '
f'{len(input_files)} input files.')
sys.exit(-7)
for i, scale_factor in enumerate(scale_factors):
if float(scale_factor) < 0:
log.error(f'Scale factor for input #{i} (counting from zero) is out of range (must be positive).')
sys.exit(-8)
return scale_factors
def parse_outputs_from_args(args: argparse.Namespace) -> Tuple[List[str], List[int]]:
"""Get a list of outputs specified in the args"""
name_and_port = [output.split(':') for output in re.split(', |,', args.output_layers)]
try:
return [name for name, _ in name_and_port], [int(port) for _, port in name_and_port]
except ValueError:
log.error('Incorrect value for -oname/--output_layers option, please specify a port for each output layer.')
sys.exit(-4)