# The GPU path runs the upscaler in half precision. The chip is given 2.3MB of # weights instead of 4.9MB, and where its shaders can multiply in fp16 (the # adapter's `shader-f16`) it does twice the work per pass. # # The export itself is unchanged: the tensor the app hands over is still float32, # and the two Cast nodes below are the model's own edge, so nothing in # superRes.ts or App.tsx has to know which copy it got. # # python3 realesr-fp16.py ../public/models/realesr-general-x4v3.onnx \ # ../public/models/realesr-general-x4v3-f16.onnx # # The source is the PReLU-rewritten model realesr-gpu.py writes — this runs # after that one, never instead of it. # # onnxconverter-common's `keep_io_types` rewrites the input's own consumers and # misses the one that does not go through the network: this model adds a Resize # of the ORIGINAL photo to the upsampler's output, that Resize reads the graph # input directly, and the runtime refuses a graph whose final Add mixes the two. # The consumer is rewired onto the cast here, which is all `keep_io_types` should # have done. # # Its second miss is Resize's `scales`, which ONNX defines as a float input # whatever the graph around it is doing, and which came out half precision. A # runtime that loads the file at all rejects the whole graph for it — "Type # 'tensor(float16)' of input parameter (/Constant_output_0) of operator (Resize) # is invalid" — on the GPU as much as on the processor. Both are put right here. import sys import numpy as np import onnx from onnx import numpy_helper from onnxconverter_common import float16 # Resize's own float inputs: the region of interest and the scales. Neither is # part of the picture, so neither has any business in half precision. FLOAT_OPERAND = {("Resize", 1), ("Resize", 2)} def convert(src, dst): model = onnx.load(src) half = float16.convert_float_to_float16(model, keep_io_types=True, disable_shape_infer=True) graph = half.graph io = graph.input[0].name cast = next(n for n in graph.node if n.op_type == "Cast" and list(n.input) == [io]) rewired = 0 for node in graph.node: for i, name in enumerate(node.input): if name == io and node is not cast: node.input[i] = cast.output[0] rewired += 1 widened = 0 producer = {o: n for n in graph.node for o in n.output} for node in graph.node: for i, name in enumerate(node.input): if (node.op_type, i) not in FLOAT_OPERAND or not name: continue src_node = producer.get(name) if src_node is None or src_node.op_type != "Constant": continue t = src_node.attribute[0].t if t.data_type != onnx.TensorProto.FLOAT16: continue arr = numpy_helper.to_array(t).astype(np.float32) src_node.attribute[0].t.CopyFrom(numpy_helper.from_array(arr, t.name)) widened += 1 # The tensor the app builds is float32 and has to stay that way: a model whose # input is fp16 would need the pixels converted in JavaScript, where there is # no half-float array to convert them into. assert graph.input[0].type.tensor_type.elem_type == onnx.TensorProto.FLOAT assert graph.output[0].type.tensor_type.elem_type == onnx.TensorProto.FLOAT assert rewired >= 1, "the input's direct consumer is gone: is this still the same model?" assert widened >= 1, "Resize's scales are still fp16: the runtime will reject the graph" ons = sum(1 for i in graph.initializer if i.data_type == onnx.TensorProto.FLOAT16) assert ons == len(graph.initializer), f"{ons}/{len(graph.initializer)} weights are fp16" onnx.save(half, dst) print(f"{len(graph.node)} nodes, {rewired} graph input consumer(s) rewired, {widened} Resize operand(s) widened back to float, {ons} fp16 weights -> {dst}") if __name__ == "__main__": convert(sys.argv[1], sys.argv[2])