added lapsrn an macbeth onnx
#16
by SavyaSanchi - opened
- human_parsing_jppnet/LICENSE +21 -0
- human_parsing_jppnet/README.md +127 -0
- human_parsing_jppnet/convert_to_onnx.py +131 -0
- human_parsing_jppnet/demo.cpp +143 -0
- human_parsing_jppnet/demo.py +99 -0
- human_parsing_jppnet/example_outputs/input_image.png +3 -0
- human_parsing_jppnet/example_outputs/output_image.png +3 -0
- human_parsing_jppnet/human_parsing_jppnet_2026sep.onnx +3 -0
- lapsrn/LICENSE +202 -0
- lapsrn/README.md +128 -0
- lapsrn/convert_to_onnx.py +134 -0
- lapsrn/demo.cpp +83 -0
- lapsrn/demo.py +60 -0
- lapsrn/example_outputs/input_image.png +3 -0
- lapsrn/example_outputs/output_image_2x.png +3 -0
- lapsrn/example_outputs/output_image_4x.png +3 -0
- lapsrn/lapsrn_x4_2026sep.onnx +3 -0
- macbeth_chart_detector/LICENSE +201 -0
- macbeth_chart_detector/README.md +97 -0
- macbeth_chart_detector/convert_to_onnx.py +40 -0
- macbeth_chart_detector/demo.py +59 -0
- macbeth_chart_detector/example_outputs/input_image.png +3 -0
- macbeth_chart_detector/example_outputs/output_image.png +3 -0
- macbeth_chart_detector/macbeth_chart_detector_2026sep.onnx +3 -0
human_parsing_jppnet/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2016 Vladimir Nekrasov
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
human_parsing_jppnet/README.md
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Human Parsing (JPPNet)
|
| 2 |
+
|
| 3 |
+
Single-person human parsing with JPPNet — it labels every pixel of a person as one of
|
| 4 |
+
20 LIP classes (hair, face, upper clothes, pants, left/right arm, left/right shoe, and
|
| 5 |
+
so on). The model was originally distributed as a frozen TensorFlow graph
|
| 6 |
+
(`lip_jppnet_384.pb`, the one referenced by OpenCV's `samples/dnn/human_parsing.py`)
|
| 7 |
+
and converted to ONNX for use with OpenCV's DNN module.
|
| 8 |
+
|
| 9 |
+
## Model Details
|
| 10 |
+
- **Architecture**: JPPNet — a ResNet-101 backbone with atrous convolutions, three
|
| 11 |
+
parsing heads (`fc1_human`, `fc2_parsing`, `fc3_parsing`) built from four-branch ASPP
|
| 12 |
+
modules, run at three input scales (0.75x, 1.0x, 1.25x) whose predictions are averaged
|
| 13 |
+
- **Input**: BGR image, 384×384, raw 0–255 float, mean `(104.00698793, 116.66876762,
|
| 14 |
+
122.67891434)`, no scaling, no swapRB, NCHW layout (`input:0`, shape `[2, 3, 384, 384]`)
|
| 15 |
+
- **Output**: `Mean_3:0` — class scores, shape `[2, 20, 384, 384]`
|
| 16 |
+
- **Framework**: ONNX opset 15 (converted from the TensorFlow frozen graph via tf2onnx)
|
| 17 |
+
- **Original weights**: https://github.com/Engineering-Course/LIP_JPPNet
|
| 18 |
+
- **Paper**: Liang, Gong, Shen and Lin, *Look into Person: Joint Body Parsing & Pose
|
| 19 |
+
Estimation Network and a New Benchmark*, T-PAMI 2018
|
| 20 |
+
|
| 21 |
+
The batch dimension is fixed at **2** and is not a batch of two different pictures. The
|
| 22 |
+
network expects the image *and its horizontal mirror*: the two predictions get averaged,
|
| 23 |
+
which is what the original evaluation script does to stabilise the result. Before
|
| 24 |
+
averaging, the mirrored half has its left/right channel pairs swapped back
|
| 25 |
+
(LeftArm↔RightArm, LeftLeg↔RightLeg, LeftShoe↔RightShoe) and is flipped horizontally.
|
| 26 |
+
`demo.py` and `demo.cpp` both do this; skipping it costs accuracy on the limbs.
|
| 27 |
+
|
| 28 |
+
The 20 classes, in channel order:
|
| 29 |
+
|
| 30 |
+
| | | | |
|
| 31 |
+
|---|---|---|---|
|
| 32 |
+
| 0 Background | 5 UpperClothes | 10 Jumpsuits | 15 RightArm |
|
| 33 |
+
| 1 Hat | 6 Dress | 11 Scarf | 16 LeftLeg |
|
| 34 |
+
| 2 Hair | 7 Coat | 12 Skirt | 17 RightLeg |
|
| 35 |
+
| 3 Glove | 8 Socks | 13 Face | 18 LeftShoe |
|
| 36 |
+
| 4 Sunglasses | 9 Pants | 14 LeftArm | 19 RightShoe |
|
| 37 |
+
|
| 38 |
+
The spatial size is frozen at 384×384 — the graph resizes its own heads back to that
|
| 39 |
+
resolution with hardcoded constants, so it always predicts at 384×384 no matter what it
|
| 40 |
+
is fed. The demos resize the picture to 384×384 on the way in and scale the score maps
|
| 41 |
+
back to the original resolution before taking the argmax.
|
| 42 |
+
|
| 43 |
+
## Usage
|
| 44 |
+
|
| 45 |
+
### Python
|
| 46 |
+
```bash
|
| 47 |
+
python demo.py --model human_parsing_jppnet_2026sep.onnx \
|
| 48 |
+
--image example_outputs/input_image.png \
|
| 49 |
+
--output example_outputs/output_image.png
|
| 50 |
+
```
|
| 51 |
+
|
| 52 |
+
Or import directly:
|
| 53 |
+
```python
|
| 54 |
+
import cv2
|
| 55 |
+
|
| 56 |
+
net = cv2.dnn.readNetFromONNX("human_parsing_jppnet_2026sep.onnx")
|
| 57 |
+
# see demo.py for preprocessing, the mirror merge and the LIP palette
|
| 58 |
+
```
|
| 59 |
+
|
| 60 |
+
### C++
|
| 61 |
+
The C++ demo runs inference with OpenCV's DNN module (default engine — no ONNX Runtime
|
| 62 |
+
needed). Adjust the OpenCV paths to your setup:
|
| 63 |
+
```bash
|
| 64 |
+
OCV=/path/to/opencv # OpenCV source tree
|
| 65 |
+
OCVBUILD=/path/to/opencv/build # OpenCV build directory (generated headers + libs)
|
| 66 |
+
g++ -std=c++17 demo.cpp -o demo \
|
| 67 |
+
-I$OCV/include \
|
| 68 |
+
-I$OCV/modules/core/include \
|
| 69 |
+
-I$OCV/modules/dnn/include \
|
| 70 |
+
-I$OCV/modules/imgproc/include \
|
| 71 |
+
-I$OCV/modules/imgcodecs/include \
|
| 72 |
+
-I$OCVBUILD \
|
| 73 |
+
-L$OCVBUILD/lib -Wl,-rpath,$OCVBUILD/lib -lopencv_dnn -lopencv_imgcodecs -lopencv_imgproc -lopencv_core
|
| 74 |
+
./demo --model human_parsing_jppnet_2026sep.onnx --image example_outputs/input_image.png
|
| 75 |
+
```
|
| 76 |
+
|
| 77 |
+
Both demos produce the same label map, pixel for pixel.
|
| 78 |
+
|
| 79 |
+
## Conversion
|
| 80 |
+
The ONNX model was exported from the frozen TensorFlow graph with tf2onnx (opset 15) via
|
| 81 |
+
[convert_to_onnx.py](./convert_to_onnx.py) — input `input:0`, output `Mean_3:0`, both
|
| 82 |
+
forced to NCHW (`inputs_as_nchw` / `outputs_as_nchw`) so the tensors match OpenCV's
|
| 83 |
+
layout, with the input shape overridden to `[2, 384, 384, 3]`. Requires `tensorflow`,
|
| 84 |
+
`tf2onnx`, `onnx` and `opencv-python` (the last only for `--verify`).
|
| 85 |
+
|
| 86 |
+
```bash
|
| 87 |
+
python convert_to_onnx.py --pb ../pb/lip_jppnet_384.pb --verify example_outputs/input_image.png
|
| 88 |
+
```
|
| 89 |
+
|
| 90 |
+
Fixing the input shape lets tf2onnx fold the whole graph down to nine op types — `Conv`,
|
| 91 |
+
`Relu`, `Sum`, `Add`, `Concat`, `Resize`, `Unsqueeze`, `ReduceMean` and `MaxPool`. The
|
| 92 |
+
batch-norms fuse into the convolutions, and the `SpaceToBatchND`/`BatchToSpaceND` pairs
|
| 93 |
+
that implement atrous convolution collapse into plain `Conv` nodes with `dilations`.
|
| 94 |
+
|
| 95 |
+
One post-processing step is applied to the exported graph. The three ASPP heads sum four
|
| 96 |
+
branches each, which tf2onnx emits as nine `Sum` nodes with four inputs; OpenCV 4.x's
|
| 97 |
+
ONNX importer mis-computes those (about 14% of the pixels came out with the wrong label).
|
| 98 |
+
`split_multi_input_sums` rewrites them as chains of two-input `Add`, which is numerically
|
| 99 |
+
identical and imports correctly everywhere.
|
| 100 |
+
|
| 101 |
+
The 412 MB file size is expected: the weights are shared across the three scale branches,
|
| 102 |
+
so it is close to a single ResNet-101 plus the parsing heads at float32.
|
| 103 |
+
|
| 104 |
+
### Verification
|
| 105 |
+
The exported model was checked against the original TensorFlow graph on the example
|
| 106 |
+
image, over all 2 × 20 × 384 × 384 output values:
|
| 107 |
+
|
| 108 |
+
| runtime | max abs diff | mean abs diff | pixels with the same label |
|
| 109 |
+
|---|---|---|---|
|
| 110 |
+
| ONNX Runtime 1.23 | 3.05e-05 | 2.33e-06 | 100.0000% |
|
| 111 |
+
| OpenCV 4.11 DNN | 3.34e-05 | 2.60e-06 | 100.0000% |
|
| 112 |
+
| OpenCV 5.0 DNN | 3.62e-05 | 2.58e-06 | 100.0000% |
|
| 113 |
+
|
| 114 |
+
## Example
|
| 115 |
+
`example_outputs/input_image.png` is the standing-person image from OpenCV's own test
|
| 116 |
+
data (`opencv_extra/testdata/dnn/pose.png`); `example_outputs/output_image.png` is the
|
| 117 |
+
parsing result painted with the LIP palette.
|
| 118 |
+
|
| 119 |
+
## License
|
| 120 |
+
See [LICENSE](./LICENSE) — the MIT license file shipped in
|
| 121 |
+
[Engineering-Course/LIP_JPPNet](https://github.com/Engineering-Course/LIP_JPPNet),
|
| 122 |
+
carried over from the DeepLab-ResNet TensorFlow implementation the code is built on
|
| 123 |
+
(hence the `Copyright (c) 2016 Vladimir Nekrasov` line).
|
| 124 |
+
|
| 125 |
+
Note that the license covers the code. The released weights are trained on the
|
| 126 |
+
[LIP dataset](https://lip.sysuhcp.com/), which the authors distribute for non-commercial
|
| 127 |
+
research use; check the dataset's own terms before using the model in a product.
|
human_parsing_jppnet/convert_to_onnx.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import datetime
|
| 3 |
+
|
| 4 |
+
import numpy as np
|
| 5 |
+
import onnx
|
| 6 |
+
from onnx import helper
|
| 7 |
+
import tensorflow as tf
|
| 8 |
+
import tf2onnx
|
| 9 |
+
|
| 10 |
+
# The frozen graph takes a batch of exactly two images (the picture and its
|
| 11 |
+
# horizontal mirror) and always emits a 384x384 label map, so the spatial size
|
| 12 |
+
# is the only thing worth parameterising.
|
| 13 |
+
INPUT_NAME = "input:0"
|
| 14 |
+
OUTPUT_NAME = "Mean_3:0"
|
| 15 |
+
BATCH = 2
|
| 16 |
+
MEAN = (104.00698793, 116.66876762, 122.67891434)
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def load_graph_def(pb_path):
|
| 20 |
+
with tf.io.gfile.GFile(pb_path, "rb") as f:
|
| 21 |
+
graph_def = tf.compat.v1.GraphDef()
|
| 22 |
+
graph_def.ParseFromString(f.read())
|
| 23 |
+
return graph_def
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def split_multi_input_sums(model):
|
| 27 |
+
"""Rewrite Sum nodes that take more than two inputs into chains of Add.
|
| 28 |
+
|
| 29 |
+
The three ASPP heads are exported as four-input Sum nodes. OpenCV 4.x's
|
| 30 |
+
ONNX importer gets those wrong (roughly 14% of the pixels end up with the
|
| 31 |
+
wrong label) while two-input adds import correctly on every version we
|
| 32 |
+
tested, and the rewrite is numerically identical.
|
| 33 |
+
"""
|
| 34 |
+
graph = model.graph
|
| 35 |
+
rewritten = []
|
| 36 |
+
count = 0
|
| 37 |
+
for node in graph.node:
|
| 38 |
+
if node.op_type != "Sum" or len(node.input) <= 2:
|
| 39 |
+
rewritten.append(node)
|
| 40 |
+
continue
|
| 41 |
+
count += 1
|
| 42 |
+
acc = node.input[0]
|
| 43 |
+
for i, operand in enumerate(node.input[1:]):
|
| 44 |
+
name = "%s__add%d" % (node.name, i)
|
| 45 |
+
last = i == len(node.input) - 2
|
| 46 |
+
out = node.output[0] if last else name
|
| 47 |
+
rewritten.append(helper.make_node("Add", [acc, operand], [out], name=name))
|
| 48 |
+
acc = out
|
| 49 |
+
del graph.node[:]
|
| 50 |
+
graph.node.extend(rewritten)
|
| 51 |
+
return count
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
def make_input(image_path, size):
|
| 55 |
+
"""Build the [2, 3, size, size] NCHW blob the exported model expects."""
|
| 56 |
+
import cv2
|
| 57 |
+
|
| 58 |
+
image = cv2.imread(image_path)
|
| 59 |
+
if image is None:
|
| 60 |
+
raise OSError("cannot read %s" % image_path)
|
| 61 |
+
image = cv2.resize(image, (size, size))
|
| 62 |
+
mirror = np.flip(image, axis=1)
|
| 63 |
+
return cv2.dnn.blobFromImages([image, mirror], mean=MEAN)
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def verify(graph_def, onnx_path, blob, size):
|
| 67 |
+
"""Compare the exported model against the original TensorFlow graph."""
|
| 68 |
+
import cv2
|
| 69 |
+
|
| 70 |
+
graph = tf.Graph()
|
| 71 |
+
with graph.as_default():
|
| 72 |
+
tf.import_graph_def(graph_def, name="")
|
| 73 |
+
with tf.compat.v1.Session(graph=graph) as sess:
|
| 74 |
+
nhwc = blob.transpose(0, 2, 3, 1).copy()
|
| 75 |
+
tf_out = sess.run(OUTPUT_NAME, feed_dict={INPUT_NAME: nhwc})
|
| 76 |
+
tf_out = tf_out.transpose(0, 3, 1, 2)
|
| 77 |
+
|
| 78 |
+
net = cv2.dnn.readNetFromONNX(onnx_path)
|
| 79 |
+
net.setInput(blob)
|
| 80 |
+
onnx_out = net.forward()
|
| 81 |
+
|
| 82 |
+
diff = np.abs(tf_out.astype(np.float64) - onnx_out.astype(np.float64))
|
| 83 |
+
agree = 100.0 * (tf_out.argmax(1) == onnx_out.argmax(1)).mean()
|
| 84 |
+
print("verify: tensorflow %s vs opencv %s" % (tf_out.shape, onnx_out.shape))
|
| 85 |
+
print("verify: max abs diff %.3e, mean abs diff %.3e" % (diff.max(), diff.mean()))
|
| 86 |
+
print("verify: identical labels on %.4f%% of the pixels" % agree)
|
| 87 |
+
return diff.max()
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def main():
|
| 91 |
+
parser = argparse.ArgumentParser(
|
| 92 |
+
description="Export the frozen LIP_JPPNet human parsing graph to ONNX")
|
| 93 |
+
parser.add_argument("--pb", default="../pb/lip_jppnet_384.pb",
|
| 94 |
+
help="frozen TensorFlow graph (lip_jppnet_384.pb)")
|
| 95 |
+
parser.add_argument("--size", type=int, default=384, help="input resolution")
|
| 96 |
+
parser.add_argument("--opset", type=int, default=15)
|
| 97 |
+
parser.add_argument("--output", default=None, help="output .onnx path")
|
| 98 |
+
parser.add_argument("--verify", metavar="IMAGE", default=None,
|
| 99 |
+
help="after exporting, compare the ONNX model against "
|
| 100 |
+
"the TensorFlow graph on this image")
|
| 101 |
+
args = parser.parse_args()
|
| 102 |
+
|
| 103 |
+
graph_def = load_graph_def(args.pb)
|
| 104 |
+
|
| 105 |
+
model_proto, _ = tf2onnx.convert.from_graph_def(
|
| 106 |
+
graph_def,
|
| 107 |
+
input_names=[INPUT_NAME],
|
| 108 |
+
output_names=[OUTPUT_NAME],
|
| 109 |
+
inputs_as_nchw=[INPUT_NAME],
|
| 110 |
+
outputs_as_nchw=[OUTPUT_NAME],
|
| 111 |
+
opset=args.opset,
|
| 112 |
+
shape_override={INPUT_NAME: [BATCH, args.size, args.size, 3]},
|
| 113 |
+
)
|
| 114 |
+
n = split_multi_input_sums(model_proto)
|
| 115 |
+
print("rewrote %d multi-input Sum nodes as Add chains" % n)
|
| 116 |
+
onnx.checker.check_model(model_proto)
|
| 117 |
+
|
| 118 |
+
onnx_path = args.output
|
| 119 |
+
if onnx_path is None:
|
| 120 |
+
stamp = datetime.datetime.now().strftime("%Y%b").lower()
|
| 121 |
+
onnx_path = "human_parsing_jppnet_%s.onnx" % stamp
|
| 122 |
+
with open(onnx_path, "wb") as f:
|
| 123 |
+
f.write(model_proto.SerializeToString())
|
| 124 |
+
print("wrote", onnx_path)
|
| 125 |
+
|
| 126 |
+
if args.verify:
|
| 127 |
+
verify(graph_def, onnx_path, make_input(args.verify, args.size), args.size)
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
if __name__ == "__main__":
|
| 131 |
+
main()
|
human_parsing_jppnet/demo.cpp
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#include <opencv2/dnn.hpp>
|
| 2 |
+
#include <opencv2/imgproc.hpp>
|
| 3 |
+
#include <opencv2/imgcodecs.hpp>
|
| 4 |
+
#include <iostream>
|
| 5 |
+
#include <string>
|
| 6 |
+
#include <vector>
|
| 7 |
+
|
| 8 |
+
// The exported graph is frozen at 384x384 and always takes two images: the
|
| 9 |
+
// picture and its horizontal mirror.
|
| 10 |
+
static const int INPUT_SIZE = 384;
|
| 11 |
+
static const cv::Scalar MEAN(104.00698793, 116.66876762, 122.67891434);
|
| 12 |
+
|
| 13 |
+
// The 20 LIP classes, in channel order.
|
| 14 |
+
static const char* CLASSES[] = {
|
| 15 |
+
"Background", "Hat", "Hair", "Glove", "Sunglasses", "UpperClothes",
|
| 16 |
+
"Dress", "Coat", "Socks", "Pants", "Jumpsuits", "Scarf", "Skirt",
|
| 17 |
+
"Face", "LeftArm", "RightArm", "LeftLeg", "RightLeg", "LeftShoe",
|
| 18 |
+
"RightShoe"
|
| 19 |
+
};
|
| 20 |
+
static const int NUM_CLASSES = 20;
|
| 21 |
+
|
| 22 |
+
// LIP palette, RGB.
|
| 23 |
+
static const unsigned char COLORS[NUM_CLASSES][3] = {
|
| 24 |
+
{0, 0, 0}, {128, 0, 0}, {255, 0, 0}, {0, 85, 0}, {170, 0, 51},
|
| 25 |
+
{255, 85, 0}, {0, 0, 85}, {0, 119, 221}, {85, 85, 0}, {0, 85, 85},
|
| 26 |
+
{85, 51, 0}, {52, 86, 128}, {0, 128, 0}, {0, 0, 255}, {51, 170, 221},
|
| 27 |
+
{0, 255, 255}, {85, 255, 170}, {170, 255, 85}, {255, 255, 0}, {255, 170, 0}
|
| 28 |
+
};
|
| 29 |
+
|
| 30 |
+
static std::string argVal(int argc, char** argv, const std::string& key, const std::string& def)
|
| 31 |
+
{
|
| 32 |
+
for (int i = 1; i + 1 < argc; ++i)
|
| 33 |
+
if (key == argv[i]) return argv[i + 1];
|
| 34 |
+
return def;
|
| 35 |
+
}
|
| 36 |
+
|
| 37 |
+
// Mirroring an image swaps the left and right body parts, so the channels of
|
| 38 |
+
// the mirrored prediction have to be swapped back before the two are averaged.
|
| 39 |
+
static int mirrorChannel(int c)
|
| 40 |
+
{
|
| 41 |
+
switch (c)
|
| 42 |
+
{
|
| 43 |
+
case 14: return 15; // LeftArm <-> RightArm
|
| 44 |
+
case 15: return 14;
|
| 45 |
+
case 16: return 17; // LeftLeg <-> RightLeg
|
| 46 |
+
case 17: return 16;
|
| 47 |
+
case 18: return 19; // LeftShoe <-> RightShoe
|
| 48 |
+
case 19: return 18;
|
| 49 |
+
default: return c;
|
| 50 |
+
}
|
| 51 |
+
}
|
| 52 |
+
|
| 53 |
+
// Build the [2, 3, 384, 384] blob: the image and its mirror, mean-subtracted.
|
| 54 |
+
static cv::Mat preprocess(const cv::Mat& image)
|
| 55 |
+
{
|
| 56 |
+
cv::Mat resized, mirror;
|
| 57 |
+
cv::resize(image, resized, cv::Size(INPUT_SIZE, INPUT_SIZE));
|
| 58 |
+
cv::flip(resized, mirror, 1);
|
| 59 |
+
|
| 60 |
+
std::vector<cv::Mat> images;
|
| 61 |
+
images.push_back(resized);
|
| 62 |
+
images.push_back(mirror);
|
| 63 |
+
|
| 64 |
+
return cv::dnn::blobFromImages(images, 1.0, cv::Size(), MEAN);
|
| 65 |
+
}
|
| 66 |
+
|
| 67 |
+
// Average the two predictions and turn them into a label map of `size`.
|
| 68 |
+
// `out` is [2, 20, 384, 384].
|
| 69 |
+
static cv::Mat postprocess(const cv::Mat& out, const cv::Size& size)
|
| 70 |
+
{
|
| 71 |
+
cv::Mat labels(size, CV_8U, cv::Scalar(0));
|
| 72 |
+
cv::Mat best(size, CV_32F, cv::Scalar(-FLT_MAX));
|
| 73 |
+
|
| 74 |
+
for (int c = 0; c < NUM_CLASSES; ++c)
|
| 75 |
+
{
|
| 76 |
+
cv::Mat direct(INPUT_SIZE, INPUT_SIZE, CV_32F,
|
| 77 |
+
const_cast<float*>(out.ptr<float>(0, c)));
|
| 78 |
+
cv::Mat mirrored(INPUT_SIZE, INPUT_SIZE, CV_32F,
|
| 79 |
+
const_cast<float*>(out.ptr<float>(1, mirrorChannel(c))));
|
| 80 |
+
|
| 81 |
+
cv::Mat unmirrored;
|
| 82 |
+
cv::flip(mirrored, unmirrored, 1);
|
| 83 |
+
|
| 84 |
+
cv::Mat score = 0.5f * (direct + unmirrored);
|
| 85 |
+
cv::resize(score, score, size, 0, 0, cv::INTER_LINEAR);
|
| 86 |
+
|
| 87 |
+
cv::Mat better = score > best;
|
| 88 |
+
score.copyTo(best, better);
|
| 89 |
+
labels.setTo(c, better);
|
| 90 |
+
}
|
| 91 |
+
return labels;
|
| 92 |
+
}
|
| 93 |
+
|
| 94 |
+
// Paint a label map with the LIP palette, as a BGR image.
|
| 95 |
+
static cv::Mat colorize(const cv::Mat& labels)
|
| 96 |
+
{
|
| 97 |
+
cv::Mat lut(1, NUM_CLASSES, CV_8UC3);
|
| 98 |
+
for (int i = 0; i < NUM_CLASSES; ++i)
|
| 99 |
+
lut.at<cv::Vec3b>(0, i) = cv::Vec3b(COLORS[i][2], COLORS[i][1], COLORS[i][0]);
|
| 100 |
+
|
| 101 |
+
cv::Mat segmentation(labels.size(), CV_8UC3);
|
| 102 |
+
for (int y = 0; y < labels.rows; ++y)
|
| 103 |
+
{
|
| 104 |
+
const unsigned char* src = labels.ptr<unsigned char>(y);
|
| 105 |
+
cv::Vec3b* dst = segmentation.ptr<cv::Vec3b>(y);
|
| 106 |
+
for (int x = 0; x < labels.cols; ++x)
|
| 107 |
+
dst[x] = lut.at<cv::Vec3b>(0, src[x]);
|
| 108 |
+
}
|
| 109 |
+
return segmentation;
|
| 110 |
+
}
|
| 111 |
+
|
| 112 |
+
int main(int argc, char** argv)
|
| 113 |
+
{
|
| 114 |
+
std::string model = argVal(argc, argv, "--model", "human_parsing_jppnet_2026sep.onnx");
|
| 115 |
+
std::string image = argVal(argc, argv, "--image", "example_outputs/input_image.png");
|
| 116 |
+
std::string output = argVal(argc, argv, "--output", "example_outputs/output_image.png");
|
| 117 |
+
|
| 118 |
+
cv::Mat img = cv::imread(image);
|
| 119 |
+
if (img.empty())
|
| 120 |
+
{
|
| 121 |
+
std::cerr << "could not read image: " << image << std::endl;
|
| 122 |
+
return 1;
|
| 123 |
+
}
|
| 124 |
+
|
| 125 |
+
cv::dnn::Net net = cv::dnn::readNetFromONNX(model);
|
| 126 |
+
net.setInput(preprocess(img));
|
| 127 |
+
cv::Mat out = net.forward();
|
| 128 |
+
|
| 129 |
+
cv::Mat labels = postprocess(out, img.size());
|
| 130 |
+
cv::Mat segmentation = colorize(labels);
|
| 131 |
+
|
| 132 |
+
std::cout << "human_parsing_jppnet input " << img.cols << "x" << img.rows << std::endl;
|
| 133 |
+
for (int c = 0; c < NUM_CLASSES; ++c)
|
| 134 |
+
{
|
| 135 |
+
int pixels = cv::countNonZero(labels == c);
|
| 136 |
+
if (pixels > 0)
|
| 137 |
+
std::cout << " " << CLASSES[c] << " " << pixels << " px" << std::endl;
|
| 138 |
+
}
|
| 139 |
+
|
| 140 |
+
cv::imwrite(output, segmentation);
|
| 141 |
+
std::cout << "wrote " << output << std::endl;
|
| 142 |
+
return 0;
|
| 143 |
+
}
|
human_parsing_jppnet/demo.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import os
|
| 3 |
+
|
| 4 |
+
import cv2 as cv
|
| 5 |
+
import numpy as np
|
| 6 |
+
|
| 7 |
+
here = os.path.dirname(os.path.abspath(__file__))
|
| 8 |
+
|
| 9 |
+
# The exported graph is frozen at 384x384 and always takes two images: the
|
| 10 |
+
# picture and its horizontal mirror.
|
| 11 |
+
INPUT_SIZE = 384
|
| 12 |
+
MEAN = (104.00698793, 116.66876762, 122.67891434)
|
| 13 |
+
|
| 14 |
+
# The 20 LIP classes, in channel order.
|
| 15 |
+
CLASSES = ["Background", "Hat", "Hair", "Glove", "Sunglasses", "UpperClothes",
|
| 16 |
+
"Dress", "Coat", "Socks", "Pants", "Jumpsuits", "Scarf", "Skirt",
|
| 17 |
+
"Face", "LeftArm", "RightArm", "LeftLeg", "RightLeg", "LeftShoe",
|
| 18 |
+
"RightShoe"]
|
| 19 |
+
|
| 20 |
+
# LIP palette, RGB.
|
| 21 |
+
COLORS = [(0, 0, 0), (128, 0, 0), (255, 0, 0), (0, 85, 0), (170, 0, 51),
|
| 22 |
+
(255, 85, 0), (0, 0, 85), (0, 119, 221), (85, 85, 0), (0, 85, 85),
|
| 23 |
+
(85, 51, 0), (52, 86, 128), (0, 128, 0), (0, 0, 255), (51, 170, 221),
|
| 24 |
+
(0, 255, 255), (85, 255, 170), (170, 255, 85), (255, 255, 0),
|
| 25 |
+
(255, 170, 0)]
|
| 26 |
+
|
| 27 |
+
# Mirroring an image swaps the left and right body parts, so the channels of
|
| 28 |
+
# the mirrored prediction have to be swapped back before the two are averaged.
|
| 29 |
+
FLIP_PAIRS = [(14, 15), (16, 17), (18, 19)]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def preprocess(image):
|
| 33 |
+
"""Build the [2, 3, 384, 384] blob: the image and its mirror, mean-subtracted."""
|
| 34 |
+
resized = cv.resize(image, (INPUT_SIZE, INPUT_SIZE))
|
| 35 |
+
mirror = np.flip(resized, axis=1)
|
| 36 |
+
return cv.dnn.blobFromImages([resized, mirror], mean=MEAN)
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def postprocess(out, size):
|
| 40 |
+
"""Average the two predictions and turn them into a label map of `size`.
|
| 41 |
+
|
| 42 |
+
`out` is [2, 20, 384, 384]; `size` is the (width, height) to scale back to.
|
| 43 |
+
"""
|
| 44 |
+
direct, mirrored = out[0], out[1]
|
| 45 |
+
|
| 46 |
+
order = list(range(len(CLASSES)))
|
| 47 |
+
for left, right in FLIP_PAIRS:
|
| 48 |
+
order[left], order[right] = right, left
|
| 49 |
+
mirrored = np.flip(mirrored[order], axis=2)
|
| 50 |
+
|
| 51 |
+
scores = 0.5 * (direct + mirrored)
|
| 52 |
+
scores = np.stack([cv.resize(c, size, interpolation=cv.INTER_LINEAR) for c in scores])
|
| 53 |
+
return scores.argmax(axis=0).astype(np.uint8)
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def colorize(labels):
|
| 57 |
+
"""Paint a label map with the LIP palette, as a BGR image."""
|
| 58 |
+
palette = np.array(COLORS, dtype=np.uint8)[:, ::-1] # RGB -> BGR
|
| 59 |
+
return palette[labels]
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def parse_human(net, image):
|
| 63 |
+
"""Run the network on one image and return its label map."""
|
| 64 |
+
net.setInput(preprocess(image))
|
| 65 |
+
out = net.forward()
|
| 66 |
+
return postprocess(out, (image.shape[1], image.shape[0]))
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def main():
|
| 70 |
+
parser = argparse.ArgumentParser(description="JPPNet human parsing (ONNX) demo")
|
| 71 |
+
parser.add_argument("--model", default=os.path.join(here, "human_parsing_jppnet_2026sep.onnx"))
|
| 72 |
+
parser.add_argument("--image", default=os.path.join(here, "example_outputs", "input_image.png"))
|
| 73 |
+
parser.add_argument("--output", default=os.path.join(here, "example_outputs", "output_image.png"))
|
| 74 |
+
parser.add_argument("--show", action="store_true", help="display the result in a window")
|
| 75 |
+
args = parser.parse_args()
|
| 76 |
+
|
| 77 |
+
image = cv.imread(args.image)
|
| 78 |
+
if image is None:
|
| 79 |
+
raise SystemExit("could not read image: %s" % args.image)
|
| 80 |
+
|
| 81 |
+
net = cv.dnn.readNetFromONNX(args.model)
|
| 82 |
+
labels = parse_human(net, image)
|
| 83 |
+
segmentation = colorize(labels)
|
| 84 |
+
|
| 85 |
+
present = [(CLASSES[i], int((labels == i).sum())) for i in np.unique(labels)]
|
| 86 |
+
print("human_parsing_jppnet input %dx%d" % (image.shape[1], image.shape[0]))
|
| 87 |
+
for name, pixels in sorted(present, key=lambda p: -p[1]):
|
| 88 |
+
print(" %-14s %7d px" % (name, pixels))
|
| 89 |
+
|
| 90 |
+
cv.imwrite(args.output, segmentation)
|
| 91 |
+
print("wrote", args.output)
|
| 92 |
+
|
| 93 |
+
if args.show:
|
| 94 |
+
cv.imshow("Deep learning human parsing in OpenCV", segmentation)
|
| 95 |
+
cv.waitKey()
|
| 96 |
+
|
| 97 |
+
|
| 98 |
+
if __name__ == "__main__":
|
| 99 |
+
main()
|
human_parsing_jppnet/example_outputs/input_image.png
ADDED
|
Git LFS Details
|
human_parsing_jppnet/example_outputs/output_image.png
ADDED
|
Git LFS Details
|
human_parsing_jppnet/human_parsing_jppnet_2026sep.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:eeace33fc98ed398bb33a3df3dca20839d9c7bcae8cfd3984409b4ddf8c426b4
|
| 3 |
+
size 431265151
|
lapsrn/LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright [yyyy] [name of copyright owner]
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
lapsrn/README.md
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# LapSRN x4
|
| 2 |
+
|
| 3 |
+
Single-image super-resolution with a Laplacian Pyramid Super-Resolution Network. One forward
|
| 4 |
+
pass produces both a 2x and a 4x image, because the pyramid reconstructs each scale from the
|
| 5 |
+
previous one and both intermediate results are exposed as outputs.
|
| 6 |
+
|
| 7 |
+
The model was originally distributed as a frozen TensorFlow graph (`LapSRN_x4.pb`, the model
|
| 8 |
+
used by opencv_contrib's `dnn_superres` module) and converted to ONNX for use with OpenCV's
|
| 9 |
+
DNN module.
|
| 10 |
+
|
| 11 |
+
## Model Details
|
| 12 |
+
- **Architecture**: LapSRN — Laplacian pyramid, progressive upsampling with `DepthToSpace`
|
| 13 |
+
and recursive residual blocks (24 `Conv`, 4 `DepthToSpace`)
|
| 14 |
+
- **Input**: `IteratorGetNext:0`, **NHWC** `[1, H, W, 1]` — a single **luma (Y)** channel,
|
| 15 |
+
float32 normalized to `[0, 1]`. `H`/`W` are dynamic; the network is fully convolutional.
|
| 16 |
+
- **Outputs** (both produced in one pass, NCHW):
|
| 17 |
+
- `NCHW_output_2x` — `[1, 1, H*2, W*2]`
|
| 18 |
+
- `NCHW_output_4x` — `[1, 1, H*4, W*4]`
|
| 19 |
+
- **Framework**: ONNX (converted from the TensorFlow frozen graph via tf2onnx, opset 13)
|
| 20 |
+
- **Original weights**: https://github.com/fannymonori/TF-LapSRN/tree/master/export
|
| 21 |
+
- **Paper**: Lai, Huang, Ahuja, Yang, *Deep Laplacian Pyramid Networks for Fast and Accurate
|
| 22 |
+
Super-Resolution*, CVPR 2017 — https://arxiv.org/abs/1704.03915
|
| 23 |
+
|
| 24 |
+
> **Note on layout.** The input is NHWC while the outputs are NCHW (hence their names). This
|
| 25 |
+
> is not a mistake in the conversion — it is what `dnn_superres` feeds and reads, so the
|
| 26 |
+
> conversion preserves it. Do not use `blobFromImage` here; it would produce NCHW input.
|
| 27 |
+
|
| 28 |
+
### Colour handling
|
| 29 |
+
|
| 30 |
+
The network only sees the Y channel. To get a colour image, the demos follow exactly what
|
| 31 |
+
`dnn_superres` does (`preprocess_YCrCb` / `reconstruct_YCrCb`):
|
| 32 |
+
|
| 33 |
+
1. Convert BGR to YCrCb, scale to float `[0, 1]`.
|
| 34 |
+
2. Feed **only Y**, reshaped to `[1, H, W, 1]`.
|
| 35 |
+
3. Bicubically upscale Cr and Cb by the same factor.
|
| 36 |
+
4. Merge `(Y_hr, Cr_hr, Cb_hr)`, scale back to 8-bit, convert YCrCb to BGR.
|
| 37 |
+
|
| 38 |
+
## Usage
|
| 39 |
+
|
| 40 |
+
### Python
|
| 41 |
+
```bash
|
| 42 |
+
python demo.py --model lapsrn_x4_2026sep.onnx --image example_outputs/input_image.png --output-dir example_outputs
|
| 43 |
+
```
|
| 44 |
+
Writes `output_image_2x.png` and `output_image_4x.png`. Pass `--scale 2` or `--scale 4` to
|
| 45 |
+
run just one head.
|
| 46 |
+
|
| 47 |
+
### C++
|
| 48 |
+
The C++ demo runs inference with OpenCV's DNN module (default engine — no ONNX Runtime
|
| 49 |
+
needed). Adjust the OpenCV paths to your setup:
|
| 50 |
+
```bash
|
| 51 |
+
OCV=/path/to/opencv # OpenCV source tree
|
| 52 |
+
OCVBUILD=/path/to/opencv/build # OpenCV build directory (generated headers + libs)
|
| 53 |
+
g++ -std=c++17 demo.cpp -o demo \
|
| 54 |
+
-I$OCV/include \
|
| 55 |
+
-I$OCV/modules/core/include \
|
| 56 |
+
-I$OCV/modules/dnn/include \
|
| 57 |
+
-I$OCV/modules/imgproc/include \
|
| 58 |
+
-I$OCV/modules/imgcodecs/include \
|
| 59 |
+
-I$OCVBUILD \
|
| 60 |
+
-L$OCVBUILD/lib -Wl,-rpath,$OCVBUILD/lib -lopencv_dnn -lopencv_imgcodecs -lopencv_imgproc -lopencv_core
|
| 61 |
+
./demo --model lapsrn_x4_2026sep.onnx --image example_outputs/input_image.png --output-dir example_outputs
|
| 62 |
+
```
|
| 63 |
+
|
| 64 |
+
The Python and C++ demos agree to within 1–2 intensity levels on under 1% of pixels — NumPy
|
| 65 |
+
and OpenCV break rounding ties differently, which the YCrCb to BGR conversion mildly amplifies.
|
| 66 |
+
|
| 67 |
+
### With the dnn_superres module
|
| 68 |
+
|
| 69 |
+
This is the same file the module's own test loads, so it also works through the module API:
|
| 70 |
+
|
| 71 |
+
```cpp
|
| 72 |
+
DnnSuperResImpl sr;
|
| 73 |
+
sr.readModel("lapsrn_x4_2026sep.onnx");
|
| 74 |
+
sr.setModel("lapsrn", 4);
|
| 75 |
+
|
| 76 |
+
std::vector<Mat> outputs;
|
| 77 |
+
sr.upsampleMultioutput(img, outputs, {2, 4}, {"NCHW_output_2x", "NCHW_output_4x"});
|
| 78 |
+
```
|
| 79 |
+
|
| 80 |
+
## Conversion
|
| 81 |
+
|
| 82 |
+
Exported from the frozen TensorFlow graph with tf2onnx (opset 13) via
|
| 83 |
+
[convert_to_onnx.py](./convert_to_onnx.py). Requires `tensorflow`, `tf2onnx` and `onnx`.
|
| 84 |
+
|
| 85 |
+
```bash
|
| 86 |
+
python convert_to_onnx.py --pb LapSRN_x4.pb
|
| 87 |
+
```
|
| 88 |
+
|
| 89 |
+
Three details are specific to this model and are handled by the script:
|
| 90 |
+
|
| 91 |
+
- **The weights are stored quantized and have to be folded back to float.** The frozen graph
|
| 92 |
+
went through TensorFlow's `quantize_weights` transform, so each of the 20 weight tensors is
|
| 93 |
+
a `quint8` const plus a min/max pair, reconstituted at runtime by a `Dequantize` node in
|
| 94 |
+
`MIN_FIRST` mode. tf2onnx has no mapping for `Dequantize`, so the script evaluates those
|
| 95 |
+
nodes in a TF session and rewrites each one as a plain float32 const. This is why the ONNX
|
| 96 |
+
(2.7 MB) is about 4x the size of the frozen graph (703 KB).
|
| 97 |
+
- **The graph is pruned to the two heads that are used.** The frozen graph carries a third
|
| 98 |
+
head, `NCHW_output`, that nothing reads. `extract_sub_graph` on `NCHW_output_2x` /
|
| 99 |
+
`NCHW_output_4x` drops it — that head and its `perm` const are the only two nodes removed,
|
| 100 |
+
so no compute changes (329 to 327 nodes).
|
| 101 |
+
- **The output names are stripped of their `:0` suffix.** tf2onnx names ONNX outputs after TF
|
| 102 |
+
tensors, so the heads come out as `NCHW_output_2x:0`, and
|
| 103 |
+
`net.forward("NCHW_output_2x")` cannot resolve that. The script renames them after
|
| 104 |
+
conversion.
|
| 105 |
+
|
| 106 |
+
Note that `shape_override` uses `None`, not `-1`, for the dynamic H/W dims: the frozen graph's
|
| 107 |
+
placeholder has unknown rank, and tf2onnx's shape inference passes the override straight to
|
| 108 |
+
`set_shape`, which rejects `-1`.
|
| 109 |
+
|
| 110 |
+
Re-running the script on `LapSRN_x4.pb` reproduces the shipped model: same opset, node count
|
| 111 |
+
and initializers, and **bit-identical inference output**. The only differences are tf2onnx's
|
| 112 |
+
auto-generated constant names (`const_fold_opt__611` and friends), whose numbering depends on
|
| 113 |
+
graph traversal order.
|
| 114 |
+
|
| 115 |
+
The input is deliberately left NHWC — `--inputs-as-nchw` is *not* used here, unlike the other
|
| 116 |
+
`dnn_superres` conversions.
|
| 117 |
+
|
| 118 |
+
## Example
|
| 119 |
+
|
| 120 |
+
| | |
|
| 121 |
+
|---|---|
|
| 122 |
+
| `example_outputs/input_image.png` | 256x256 (`butterfly.png`, the `dnn_superres` test image) |
|
| 123 |
+
| `example_outputs/output_image_2x.png` | 512x512, from `NCHW_output_2x` |
|
| 124 |
+
| `example_outputs/output_image_4x.png` | 1024x1024, from `NCHW_output_4x` |
|
| 125 |
+
|
| 126 |
+
## License
|
| 127 |
+
See [LICENSE](./LICENSE) — the model is released by the TF-LapSRN author under the
|
| 128 |
+
Apache License 2.0.
|
lapsrn/convert_to_onnx.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Export the dnn_superres LapSRN_x4 TensorFlow frozen graph to ONNX.
|
| 2 |
+
|
| 3 |
+
Three things make this model different from the other dnn_superres conversions:
|
| 4 |
+
|
| 5 |
+
1. Its weights are stored quantized. The graph was run through TensorFlow's
|
| 6 |
+
`quantize_weights` transform, so every weight is a `quint8` const plus a
|
| 7 |
+
min/max pair, reconstituted at runtime by a `Dequantize` node (MIN_FIRST).
|
| 8 |
+
tf2onnx has no mapping for `Dequantize`, so the 20 of them are constant-folded
|
| 9 |
+
back to float32 consts before conversion. This is why the ONNX (2.7 MB) is
|
| 10 |
+
roughly 4x the size of the frozen graph (703 KB).
|
| 11 |
+
|
| 12 |
+
2. The graph carries a third output head, `NCHW_output`, that nothing uses.
|
| 13 |
+
Pruning to the 2x and 4x heads drops it (it and its `perm` const are the only
|
| 14 |
+
two nodes removed - no compute changes).
|
| 15 |
+
|
| 16 |
+
3. tf2onnx names ONNX outputs after TF tensors, so the heads come out as
|
| 17 |
+
`NCHW_output_2x:0`. OpenCV's `net.forward("NCHW_output_2x")` cannot resolve a
|
| 18 |
+
name with the trailing `:0`, so the graph outputs are renamed afterwards.
|
| 19 |
+
|
| 20 |
+
The input is deliberately left in NHWC `[1, H, W, 1]` (no `--inputs-as-nchw`):
|
| 21 |
+
that is what dnn_superres feeds, and the outputs are NCHW regardless.
|
| 22 |
+
|
| 23 |
+
Requires `tensorflow`, `tf2onnx` and `onnx`.
|
| 24 |
+
"""
|
| 25 |
+
|
| 26 |
+
import argparse
|
| 27 |
+
import datetime
|
| 28 |
+
|
| 29 |
+
import onnx
|
| 30 |
+
import tensorflow as tf
|
| 31 |
+
import tf2onnx
|
| 32 |
+
from tensorflow.core.framework import attr_value_pb2, graph_pb2, node_def_pb2
|
| 33 |
+
|
| 34 |
+
INPUT_TENSOR = "IteratorGetNext:0"
|
| 35 |
+
OUTPUT_NODES = ["NCHW_output_2x", "NCHW_output_4x"]
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
def load_graph_def(pb_path):
|
| 39 |
+
with tf.io.gfile.GFile(pb_path, "rb") as f:
|
| 40 |
+
graph_def = tf.compat.v1.GraphDef()
|
| 41 |
+
graph_def.ParseFromString(f.read())
|
| 42 |
+
return graph_def
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def fold_dequantize(graph_def):
|
| 46 |
+
"""Replace every Dequantize node with a plain float32 Const of its value.
|
| 47 |
+
|
| 48 |
+
The quint8 weight consts and their min/max pairs become unreferenced and are
|
| 49 |
+
dropped, so the returned graph is one tf2onnx can handle.
|
| 50 |
+
"""
|
| 51 |
+
names = [n.name for n in graph_def.node if n.op == "Dequantize"]
|
| 52 |
+
if not names:
|
| 53 |
+
return graph_def
|
| 54 |
+
|
| 55 |
+
graph = tf.Graph()
|
| 56 |
+
with graph.as_default():
|
| 57 |
+
tf.import_graph_def(graph_def, name="")
|
| 58 |
+
with tf.compat.v1.Session(graph=graph) as sess:
|
| 59 |
+
values = sess.run(["%s:0" % n for n in names])
|
| 60 |
+
folded = dict(zip(names, values))
|
| 61 |
+
|
| 62 |
+
out = graph_pb2.GraphDef()
|
| 63 |
+
out.versions.CopyFrom(graph_def.versions)
|
| 64 |
+
for node in graph_def.node:
|
| 65 |
+
if node.name not in folded:
|
| 66 |
+
out.node.append(node)
|
| 67 |
+
continue
|
| 68 |
+
value = folded[node.name]
|
| 69 |
+
const = node_def_pb2.NodeDef()
|
| 70 |
+
const.name = node.name
|
| 71 |
+
const.op = "Const"
|
| 72 |
+
const.attr["dtype"].CopyFrom(attr_value_pb2.AttrValue(type=tf.float32.as_datatype_enum))
|
| 73 |
+
const.attr["value"].CopyFrom(attr_value_pb2.AttrValue(
|
| 74 |
+
tensor=tf.make_tensor_proto(value, dtype=tf.float32, shape=value.shape)))
|
| 75 |
+
out.node.append(const)
|
| 76 |
+
|
| 77 |
+
# Pruning again drops the now-orphaned *_quantized_const / _min / _max nodes.
|
| 78 |
+
return tf.compat.v1.graph_util.extract_sub_graph(out, OUTPUT_NODES)
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def strip_output_suffix(model_proto):
|
| 82 |
+
"""Rename `NCHW_output_2x:0` -> `NCHW_output_2x` so net.forward() can find it."""
|
| 83 |
+
renames = {o.name: o.name.rsplit(":", 1)[0] for o in model_proto.graph.output if ":" in o.name}
|
| 84 |
+
if not renames:
|
| 85 |
+
return
|
| 86 |
+
for out in model_proto.graph.output:
|
| 87 |
+
out.name = renames.get(out.name, out.name)
|
| 88 |
+
for node in model_proto.graph.node:
|
| 89 |
+
for i, name in enumerate(node.output):
|
| 90 |
+
if name in renames:
|
| 91 |
+
node.output[i] = renames[name]
|
| 92 |
+
for i, name in enumerate(node.input):
|
| 93 |
+
if name in renames:
|
| 94 |
+
node.input[i] = renames[name]
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def main():
|
| 98 |
+
parser = argparse.ArgumentParser(description="Export LapSRN_x4.pb to ONNX")
|
| 99 |
+
parser.add_argument("--pb", default="LapSRN_x4.pb")
|
| 100 |
+
parser.add_argument("--opset", type=int, default=13)
|
| 101 |
+
args = parser.parse_args()
|
| 102 |
+
|
| 103 |
+
graph_def = load_graph_def(args.pb)
|
| 104 |
+
|
| 105 |
+
# Drop the unused NCHW_output head, then de-quantize the weights.
|
| 106 |
+
graph_def = tf.compat.v1.graph_util.extract_sub_graph(graph_def, OUTPUT_NODES)
|
| 107 |
+
graph_def = fold_dequantize(graph_def)
|
| 108 |
+
|
| 109 |
+
left = [n.name for n in graph_def.node if n.op == "Dequantize"]
|
| 110 |
+
if left:
|
| 111 |
+
raise SystemExit("Dequantize nodes survived folding: %s" % ", ".join(left))
|
| 112 |
+
|
| 113 |
+
model_proto, _ = tf2onnx.convert.from_graph_def(
|
| 114 |
+
graph_def,
|
| 115 |
+
input_names=[INPUT_TENSOR],
|
| 116 |
+
output_names=["%s:0" % n for n in OUTPUT_NODES],
|
| 117 |
+
opset=args.opset,
|
| 118 |
+
# Dynamic H/W: the model is fully convolutional, any input size works.
|
| 119 |
+
shape_override={INPUT_TENSOR: [1, None, None, 1]},
|
| 120 |
+
)
|
| 121 |
+
|
| 122 |
+
strip_output_suffix(model_proto)
|
| 123 |
+
onnx.checker.check_model(model_proto)
|
| 124 |
+
|
| 125 |
+
stamp = datetime.datetime.now().strftime("%Y%b").lower()
|
| 126 |
+
onnx_path = "lapsrn_x4_%s.onnx" % stamp
|
| 127 |
+
with open(onnx_path, "wb") as f:
|
| 128 |
+
f.write(model_proto.SerializeToString())
|
| 129 |
+
print("wrote", onnx_path)
|
| 130 |
+
print("outputs", [o.name for o in model_proto.graph.output])
|
| 131 |
+
|
| 132 |
+
|
| 133 |
+
if __name__ == "__main__":
|
| 134 |
+
main()
|
lapsrn/demo.cpp
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#include <opencv2/dnn.hpp>
|
| 2 |
+
#include <opencv2/imgproc.hpp>
|
| 3 |
+
#include <opencv2/imgcodecs.hpp>
|
| 4 |
+
#include <iostream>
|
| 5 |
+
#include <string>
|
| 6 |
+
#include <vector>
|
| 7 |
+
|
| 8 |
+
static std::string argVal(int argc, char** argv, const std::string& key, const std::string& def)
|
| 9 |
+
{
|
| 10 |
+
for (int i = 1; i + 1 < argc; ++i)
|
| 11 |
+
if (key == argv[i]) return argv[i + 1];
|
| 12 |
+
return def;
|
| 13 |
+
}
|
| 14 |
+
|
| 15 |
+
// Run one pyramid head and rebuild a colour image from its luma output.
|
| 16 |
+
// Mirrors dnn_superres' preprocess_YCrCb / reconstruct_YCrCb: the network only
|
| 17 |
+
// ever sees the Y channel, and Cr/Cb are bicubically upscaled and merged back.
|
| 18 |
+
static cv::Mat upsample(cv::dnn::Net& net, const cv::Mat& img, const std::string& nodeName, int scale)
|
| 19 |
+
{
|
| 20 |
+
cv::Mat ycrcb;
|
| 21 |
+
cv::cvtColor(img, ycrcb, cv::COLOR_BGR2YCrCb);
|
| 22 |
+
ycrcb.convertTo(ycrcb, CV_32F, 1.0 / 255.0);
|
| 23 |
+
|
| 24 |
+
cv::Mat ch[3];
|
| 25 |
+
cv::split(ycrcb, ch);
|
| 26 |
+
cv::Mat Y = ch[0];
|
| 27 |
+
if (!Y.isContinuous()) Y = Y.clone();
|
| 28 |
+
|
| 29 |
+
// This model takes NHWC [1, H, W, 1] even though its outputs are NCHW.
|
| 30 |
+
int blobShape[] = {1, Y.rows, Y.cols, 1};
|
| 31 |
+
cv::Mat blob(4, blobShape, CV_32F, Y.data);
|
| 32 |
+
|
| 33 |
+
net.setInput(blob);
|
| 34 |
+
cv::Mat outBlob = net.forward(nodeName);
|
| 35 |
+
|
| 36 |
+
// Output blob is NCHW [1, 1, H*scale, W*scale] -> wrap as a single-channel Mat.
|
| 37 |
+
cv::Mat yHr(outBlob.size[2], outBlob.size[3], CV_32F, outBlob.ptr<float>());
|
| 38 |
+
|
| 39 |
+
cv::Mat crHr, cbHr;
|
| 40 |
+
cv::resize(ch[1], crHr, cv::Size(), scale, scale);
|
| 41 |
+
cv::resize(ch[2], cbHr, cv::Size(), scale, scale);
|
| 42 |
+
|
| 43 |
+
std::vector<cv::Mat> merged = {yHr, crHr, cbHr};
|
| 44 |
+
cv::Mat hr;
|
| 45 |
+
cv::merge(merged, hr);
|
| 46 |
+
hr.convertTo(hr, CV_8U, 255.0);
|
| 47 |
+
cv::cvtColor(hr, hr, cv::COLOR_YCrCb2BGR);
|
| 48 |
+
return hr;
|
| 49 |
+
}
|
| 50 |
+
|
| 51 |
+
int main(int argc, char** argv)
|
| 52 |
+
{
|
| 53 |
+
std::string model = argVal(argc, argv, "--model", "lapsrn_x4_2026sep.onnx");
|
| 54 |
+
std::string image = argVal(argc, argv, "--image", "example_outputs/input_image.png");
|
| 55 |
+
std::string outputDir = argVal(argc, argv, "--output-dir", "example_outputs");
|
| 56 |
+
|
| 57 |
+
cv::Mat img = cv::imread(image);
|
| 58 |
+
if (img.empty())
|
| 59 |
+
{
|
| 60 |
+
std::cerr << "could not read image: " << image << std::endl;
|
| 61 |
+
return 1;
|
| 62 |
+
}
|
| 63 |
+
|
| 64 |
+
cv::dnn::Net net = cv::dnn::readNetFromONNX(model);
|
| 65 |
+
|
| 66 |
+
// The two heads of the Laplacian pyramid, and the scale each one produces.
|
| 67 |
+
const std::vector<std::pair<std::string, int> > outputs =
|
| 68 |
+
{{"NCHW_output_2x", 2}, {"NCHW_output_4x", 4}};
|
| 69 |
+
|
| 70 |
+
std::cout << "lapsrn_x4 input " << img.cols << "x" << img.rows << std::endl;
|
| 71 |
+
for (size_t i = 0; i < outputs.size(); ++i)
|
| 72 |
+
{
|
| 73 |
+
const std::string& nodeName = outputs[i].first;
|
| 74 |
+
int scale = outputs[i].second;
|
| 75 |
+
|
| 76 |
+
cv::Mat hr = upsample(net, img, nodeName, scale);
|
| 77 |
+
std::string path = cv::format("%s/output_image_%dx.png", outputDir.c_str(), scale);
|
| 78 |
+
cv::imwrite(path, hr);
|
| 79 |
+
std::cout << " " << nodeName << " x" << scale
|
| 80 |
+
<< " -> " << hr.cols << "x" << hr.rows << " " << path << std::endl;
|
| 81 |
+
}
|
| 82 |
+
return 0;
|
| 83 |
+
}
|
lapsrn/demo.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import os
|
| 3 |
+
|
| 4 |
+
import cv2 as cv
|
| 5 |
+
import numpy as np
|
| 6 |
+
|
| 7 |
+
here = os.path.dirname(os.path.abspath(__file__))
|
| 8 |
+
|
| 9 |
+
# The two heads of the Laplacian pyramid, and the scale each one produces.
|
| 10 |
+
OUTPUTS = [("NCHW_output_2x", 2), ("NCHW_output_4x", 4)]
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def upsample(net, img, node_name, scale):
|
| 14 |
+
"""Run one pyramid head and rebuild a colour image from its luma output.
|
| 15 |
+
|
| 16 |
+
Mirrors dnn_superres' preprocess_YCrCb / reconstruct_YCrCb: the network only
|
| 17 |
+
ever sees the Y channel, and Cr/Cb are bicubically upscaled and merged back.
|
| 18 |
+
"""
|
| 19 |
+
ycrcb = cv.cvtColor(img, cv.COLOR_BGR2YCrCb).astype(np.float32) / 255.0
|
| 20 |
+
y, cr, cb = cv.split(ycrcb)
|
| 21 |
+
|
| 22 |
+
# This model takes NHWC [1, H, W, 1] even though its outputs are NCHW.
|
| 23 |
+
net.setInput(y.reshape(1, y.shape[0], y.shape[1], 1))
|
| 24 |
+
y_hr = net.forward(node_name)[0, 0]
|
| 25 |
+
|
| 26 |
+
cr_hr = cv.resize(cr, None, fx=scale, fy=scale)
|
| 27 |
+
cb_hr = cv.resize(cb, None, fx=scale, fy=scale)
|
| 28 |
+
|
| 29 |
+
merged = cv.merge([y_hr, cr_hr, cb_hr])
|
| 30 |
+
# np.rint, not a plain cast: convertTo(CV_8U) rounds, and truncating here
|
| 31 |
+
# would drift from the C++ demo by a few levels after YCrCb -> BGR.
|
| 32 |
+
merged_8u = np.rint(merged * 255.0).clip(0, 255).astype(np.uint8)
|
| 33 |
+
return cv.cvtColor(merged_8u, cv.COLOR_YCrCb2BGR)
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def main():
|
| 37 |
+
parser = argparse.ArgumentParser(description="LapSRN x4 super-resolution (ONNX) demo")
|
| 38 |
+
parser.add_argument("--model", default=os.path.join(here, "lapsrn_x4_2026sep.onnx"))
|
| 39 |
+
parser.add_argument("--image", default=os.path.join(here, "example_outputs", "input_image.png"))
|
| 40 |
+
parser.add_argument("--output-dir", default=os.path.join(here, "example_outputs"))
|
| 41 |
+
parser.add_argument("--scale", type=int, choices=[2, 4], help="only run this scale (default: both)")
|
| 42 |
+
args = parser.parse_args()
|
| 43 |
+
|
| 44 |
+
img = cv.imread(args.image)
|
| 45 |
+
if img is None:
|
| 46 |
+
raise SystemExit("could not read image: %s" % args.image)
|
| 47 |
+
|
| 48 |
+
net = cv.dnn.readNetFromONNX(args.model)
|
| 49 |
+
|
| 50 |
+
wanted = [(n, s) for n, s in OUTPUTS if args.scale in (None, s)]
|
| 51 |
+
print("lapsrn_x4 input %dx%d" % (img.shape[1], img.shape[0]))
|
| 52 |
+
for node_name, scale in wanted:
|
| 53 |
+
out = upsample(net, img, node_name, scale)
|
| 54 |
+
path = os.path.join(args.output_dir, "output_image_%dx.png" % scale)
|
| 55 |
+
cv.imwrite(path, out)
|
| 56 |
+
print(" %-16s x%d -> %dx%d %s" % (node_name, scale, out.shape[1], out.shape[0], path))
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
if __name__ == "__main__":
|
| 60 |
+
main()
|
lapsrn/example_outputs/input_image.png
ADDED
|
Git LFS Details
|
lapsrn/example_outputs/output_image_2x.png
ADDED
|
Git LFS Details
|
lapsrn/example_outputs/output_image_4x.png
ADDED
|
Git LFS Details
|
lapsrn/lapsrn_x4_2026sep.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:70474758dab65ae3f488bed94b11b60fb77578f9de3d45a66d72263dcfefadb7
|
| 3 |
+
size 2708806
|
macbeth_chart_detector/LICENSE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
macbeth_chart_detector/README.md
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Macbeth Chart Detector
|
| 2 |
+
|
| 3 |
+
Detects a Macbeth / ColorChecker Classic colour chart in a photograph and returns its bounding
|
| 4 |
+
box. This is the detection model used by OpenCV's `mcc` colour-correction code
|
| 5 |
+
(`objdetect`, `checker_detector.cpp`) to locate a chart before reading its patches.
|
| 6 |
+
|
| 7 |
+
The model is a Faster R-CNN with an Inception V2 backbone, trained as a single-class detector
|
| 8 |
+
with the TensorFlow Object Detection API. It was originally distributed as a frozen TensorFlow
|
| 9 |
+
graph (`frozen_inference_graph.pb`) and converted to ONNX for use with OpenCV's DNN module.
|
| 10 |
+
|
| 11 |
+
## Model Details
|
| 12 |
+
- **Architecture**: Faster R-CNN, Inception V2 backbone, one class (the chart)
|
| 13 |
+
- **Input**: `image_tensor:0`, **NHWC** `[1, H, W, 3]`, **raw uint8 RGB** (0-255, no scaling and
|
| 14 |
+
no mean subtraction). `H`/`W` are dynamic — feed the image at its own resolution. The graph
|
| 15 |
+
resizes internally (keep aspect ratio, min side 768, max side 1024), so do not pre-resize,
|
| 16 |
+
and do not use `blobFromImage`: that produces NCHW.
|
| 17 |
+
- **Outputs** (all four must be requested together):
|
| 18 |
+
- `detection_boxes:0` — `[1, 300, 4]`, normalized, **`[ymin, xmin, ymax, xmax]`** (y first)
|
| 19 |
+
- `detection_scores:0` — `[1, 300]`, descending
|
| 20 |
+
- `detection_classes:0` — `[1, 300]`, always `1.0` (single class, so it carries no information)
|
| 21 |
+
- `num_detections:0` — `[1]`
|
| 22 |
+
- **Framework**: ONNX (converted from the TensorFlow frozen graph via tf2onnx, opset 18)
|
| 23 |
+
- **Original weights**: [gursimarsingh/opencv_zoo @ `mcc_model`](https://github.com/gursimarsingh/opencv_zoo/tree/mcc_model/models/macbeth_chart_detector)
|
| 24 |
+
— `frozen_inference_graph.pb`, sha1 `fae7dbef14c4ae1fca76f3662220fbd460ed5ed6`
|
| 25 |
+
|
| 26 |
+
> **`num_detections` is not a detection count.** It reports the post-NMS slot count — 100 on the
|
| 27 |
+
> example image, where exactly one chart is present. Threshold on `detection_scores` instead;
|
| 28 |
+
> iterating `num_detections` yields 99 phantom boxes.
|
| 29 |
+
|
| 30 |
+
## Usage
|
| 31 |
+
|
| 32 |
+
### Python
|
| 33 |
+
```bash
|
| 34 |
+
python demo.py --model macbeth_chart_detector_2026sep.onnx --image example_outputs/input_image.png --output example_outputs/output_image.png --conf 0.3
|
| 35 |
+
```
|
| 36 |
+
|
| 37 |
+
On the provided `example_outputs/input_image.png` (900x463) it reports one detection:
|
| 38 |
+
|
| 39 |
+
```
|
| 40 |
+
macbeth_chart_detector 1 detections
|
| 41 |
+
0.582 x1=237 y1=79 x2=650 y2=368
|
| 42 |
+
```
|
| 43 |
+
|
| 44 |
+
`demo.py` requires OpenCV 5.1 or newer, for `cv.dnn.ENGINE_OPENCV`. On OpenCV 5.0 that constant
|
| 45 |
+
does not exist (it has `ENGINE_CLASSIC` / `ENGINE_NEW`); drop the second argument to
|
| 46 |
+
`readNetFromONNX` to use the default engine instead.
|
| 47 |
+
|
| 48 |
+
## Conversion
|
| 49 |
+
|
| 50 |
+
Exported from the frozen TensorFlow graph with tf2onnx (opset 18) via
|
| 51 |
+
[convert_to_onnx.py](./convert_to_onnx.py) — input `image_tensor:0`, outputs
|
| 52 |
+
`detection_boxes:0`, `detection_scores:0`, `detection_classes:0`, `num_detections:0`.
|
| 53 |
+
Requires `tensorflow`, `tf2onnx` and `onnx`.
|
| 54 |
+
|
| 55 |
+
```bash
|
| 56 |
+
python convert_to_onnx.py --pb frozen_inference_graph.pb
|
| 57 |
+
```
|
| 58 |
+
|
| 59 |
+
The graph needs no special handling: its weights are plain float32 (no `quantize_weights` /
|
| 60 |
+
`Dequantize` nodes) and all four detection heads are present in the frozen graph.
|
| 61 |
+
|
| 62 |
+
### Conversion accuracy
|
| 63 |
+
|
| 64 |
+
Verified against the original `.pb` under TensorFlow 2.15.1, feeding both the identical uint8
|
| 65 |
+
buffer:
|
| 66 |
+
|
| 67 |
+
| | top score | box |
|
| 68 |
+
|---|---|---|
|
| 69 |
+
| TensorFlow `.pb` | 0.582205 | `0.171078 0.263475 0.794357 0.721806` |
|
| 70 |
+
| ONNX (onnxruntime) | 0.582205 | `0.171078 0.263475 0.794356 0.721806` |
|
| 71 |
+
|
| 72 |
+
The conversion is faithful to within one float32 ULP.
|
| 73 |
+
|
| 74 |
+
## Known issue: OpenCV's DNN engine
|
| 75 |
+
|
| 76 |
+
The conversion is correct, but **OpenCV's built-in DNN engine cannot currently run this model.**
|
| 77 |
+
On OpenCV 5.0.0 it aborts before producing output:
|
| 78 |
+
|
| 79 |
+
```
|
| 80 |
+
net_impl2.cpp: (-215:Assertion failed) buf.shape() == m.shape() in function 'forwardGraph'
|
| 81 |
+
```
|
| 82 |
+
|
| 83 |
+
The same failure affects the other converted Faster R-CNN / Mask R-CNN models in this
|
| 84 |
+
repository, so it is a limitation of the engine's handling of these graphs rather than
|
| 85 |
+
something specific to this model. On builds where the graph does run, the score comes out
|
| 86 |
+
around 0.02 instead of 0.582, because `Resize` with
|
| 87 |
+
`coordinate_transformation_mode="tf_crop_and_resize"` and a dynamic ROI — the ROI-pooling step
|
| 88 |
+
of the second stage — is not evaluated correctly, so every region proposal is classified from
|
| 89 |
+
the wrong features.
|
| 90 |
+
|
| 91 |
+
Until that is fixed in `modules/dnn`, treat the ONNX file as correct (onnxruntime and
|
| 92 |
+
TensorFlow agree exactly) and the OpenCV-engine results as unreliable.
|
| 93 |
+
|
| 94 |
+
## License
|
| 95 |
+
See [LICENSE](./LICENSE) — Apache License 2.0, taken from the upstream
|
| 96 |
+
[opencv_zoo](https://github.com/gursimarsingh/opencv_zoo) repository that distributes the
|
| 97 |
+
original frozen graph.
|
macbeth_chart_detector/convert_to_onnx.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import datetime
|
| 3 |
+
|
| 4 |
+
import onnx
|
| 5 |
+
import tensorflow as tf
|
| 6 |
+
import tf2onnx
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def load_graph_def(pb_path):
|
| 10 |
+
with tf.io.gfile.GFile(pb_path, "rb") as f:
|
| 11 |
+
graph_def = tf.compat.v1.GraphDef()
|
| 12 |
+
graph_def.ParseFromString(f.read())
|
| 13 |
+
return graph_def
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def main():
|
| 17 |
+
parser = argparse.ArgumentParser(description="Export the Macbeth chart detector frozen graph to ONNX")
|
| 18 |
+
parser.add_argument("--pb", default="../pb/frozen_inference_graph.pb")
|
| 19 |
+
parser.add_argument("--opset", type=int, default=18)
|
| 20 |
+
args = parser.parse_args()
|
| 21 |
+
|
| 22 |
+
graph_def = load_graph_def(args.pb)
|
| 23 |
+
|
| 24 |
+
model_proto, _ = tf2onnx.convert.from_graph_def(
|
| 25 |
+
graph_def,
|
| 26 |
+
input_names=["image_tensor:0"],
|
| 27 |
+
output_names=["detection_boxes:0", "detection_scores:0", "detection_classes:0", "num_detections:0"],
|
| 28 |
+
opset=args.opset,
|
| 29 |
+
)
|
| 30 |
+
onnx.checker.check_model(model_proto)
|
| 31 |
+
|
| 32 |
+
stamp = datetime.datetime.now().strftime("%Y%b").lower()
|
| 33 |
+
onnx_path = "macbeth_chart_detector_%s.onnx" % stamp
|
| 34 |
+
with open(onnx_path, "wb") as f:
|
| 35 |
+
f.write(model_proto.SerializeToString())
|
| 36 |
+
print("wrote", onnx_path)
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
if __name__ == "__main__":
|
| 40 |
+
main()
|
macbeth_chart_detector/demo.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import os
|
| 3 |
+
|
| 4 |
+
import cv2 as cv
|
| 5 |
+
import numpy as np
|
| 6 |
+
|
| 7 |
+
here = os.path.dirname(os.path.abspath(__file__))
|
| 8 |
+
|
| 9 |
+
# TF Object Detection API heads, in the order the graph declares them.
|
| 10 |
+
OUTPUTS = ["detection_boxes:0", "detection_scores:0", "detection_classes:0", "num_detections:0"]
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def main():
|
| 14 |
+
parser = argparse.ArgumentParser(description="Macbeth colour chart detection (ONNX) demo")
|
| 15 |
+
parser.add_argument("--model", default=os.path.join(here, "macbeth_chart_detector_2026sep.onnx"))
|
| 16 |
+
parser.add_argument("--image", default=os.path.join(here, "example_outputs", "input_image.png"))
|
| 17 |
+
parser.add_argument("--output", default=os.path.join(here, "example_outputs", "output_image.png"))
|
| 18 |
+
parser.add_argument("--conf", type=float, default=0.3, help="confidence threshold")
|
| 19 |
+
args = parser.parse_args()
|
| 20 |
+
|
| 21 |
+
img = cv.imread(args.image)
|
| 22 |
+
if img is None:
|
| 23 |
+
raise SystemExit("could not read image: %s" % args.image)
|
| 24 |
+
|
| 25 |
+
# The network takes raw uint8 RGB in NHWC layout at the image's own resolution:
|
| 26 |
+
# the graph resizes internally (keep-aspect-ratio, min side 768 / max side 1024),
|
| 27 |
+
# so do not pre-resize and do not use blobFromImage (that would give NCHW).
|
| 28 |
+
blob = cv.cvtColor(img, cv.COLOR_BGR2RGB)[None]
|
| 29 |
+
|
| 30 |
+
net = cv.dnn.readNetFromONNX(args.model, cv.dnn.ENGINE_OPENCV)
|
| 31 |
+
net.setInput(blob)
|
| 32 |
+
boxes, scores, _classes, _num = net.forward(OUTPUTS)
|
| 33 |
+
|
| 34 |
+
boxes = boxes.reshape(-1, 4)
|
| 35 |
+
scores = scores.reshape(-1)
|
| 36 |
+
|
| 37 |
+
# Threshold on the scores rather than on num_detections: that head reports the
|
| 38 |
+
# post-NMS slot count (100), not the number of charts actually found. The single
|
| 39 |
+
# class is the chart itself, so detection_classes carries no information here.
|
| 40 |
+
keep = np.nonzero(scores >= args.conf)[0]
|
| 41 |
+
|
| 42 |
+
h, w = img.shape[:2]
|
| 43 |
+
out = img.copy()
|
| 44 |
+
print("macbeth_chart_detector", len(keep), "detections")
|
| 45 |
+
for i in keep:
|
| 46 |
+
ymin, xmin, ymax, xmax = boxes[i] # normalized, y first
|
| 47 |
+
p1 = (int(round(xmin * w)), int(round(ymin * h)))
|
| 48 |
+
p2 = (int(round(xmax * w)), int(round(ymax * h)))
|
| 49 |
+
cv.rectangle(out, p1, p2, (0, 255, 0), 2)
|
| 50 |
+
cv.putText(out, "chart %.2f" % scores[i], (p1[0], max(0, p1[1] - 5)),
|
| 51 |
+
cv.FONT_HERSHEY_SIMPLEX, 0.6, (0, 255, 0), 2)
|
| 52 |
+
print(" %.3f x1=%d y1=%d x2=%d y2=%d" % (scores[i], p1[0], p1[1], p2[0], p2[1]))
|
| 53 |
+
|
| 54 |
+
cv.imwrite(args.output, out)
|
| 55 |
+
print("wrote", args.output)
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
if __name__ == "__main__":
|
| 59 |
+
main()
|
macbeth_chart_detector/example_outputs/input_image.png
ADDED
|
Git LFS Details
|
macbeth_chart_detector/example_outputs/output_image.png
ADDED
|
Git LFS Details
|
macbeth_chart_detector/macbeth_chart_detector_2026sep.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:356587d28bcea41d188a743b38f86e2c744fd892e3d41512b6a42b6af038f249
|
| 3 |
+
size 51756941
|