| { |
| "schema_version": "1.0", |
| "model_id": "VC01", |
| "evaluation_name": "mlperf_tiny_vww01_official_subset_quality", |
| "random_seed": 20260806, |
| "dataset": { |
| "name": "vww_coco2014_96_archive", |
| "title": "Visual Wake Words 96x96 archive derived from MS COCO", |
| "version": "vw_coco2014_96 referenced by MLCommons Tiny commit 4addd0fa08d216e20637637874e084895f289da4", |
| "license": "COCO image-owner/Flickr terms; no standalone archive license declared", |
| "license_note": "The public archive has no separate license file. COCO states that image licenses are available from Flickr and that image owners retain copyright; MLCommons code and evaluation metadata are Apache-2.0.", |
| "canonical_url": "https://www.silabs.com/public/files/github/machine_learning/benchmarks/datasets/vw_coco2014_96.tar.gz", |
| "url": "https://www.silabs.com/public/files/github/machine_learning/benchmarks/datasets/vw_coco2014_96.tar.gz", |
| "path": "research/downloads/mlperf_tiny_vww/vw_coco2014_96.tar.gz", |
| "expected_bytes": 234810765, |
| "expected_sha256": "f8746b9e44f8a7a4293f73be9ba6e8da9239fe69798d42364aae62b915cfab58", |
| "expected_md5": null, |
| "use": "evaluation_only_no_training_or_calibration" |
| }, |
| "upstream": { |
| "repository": "https://github.com/mlcommons/tiny", |
| "commit": "4addd0fa08d216e20637637874e084895f289da4", |
| "files": [ |
| { |
| "name": "datasets_README.md", |
| "url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/evaluation/datasets/README.md", |
| "path": "models/vision_classification/VC01/quality_evaluation/sources/datasets_README.md", |
| "expected_sha256": "ed00f61828db845481209d1b3f320f5718085b5ff39ecb6a229fa5314f320ca4", |
| "license": "Apache-2.0" |
| }, |
| { |
| "name": "y_labels.csv", |
| "url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/evaluation/datasets/vww01/y_labels.csv", |
| "path": "models/vision_classification/VC01/quality_evaluation/sources/y_labels.csv", |
| "expected_sha256": "3697ca57c48b23b21602ae9bdb32b1925407a1d41d79167cdfb365054cb9c33d", |
| "license": "Apache-2.0" |
| }, |
| { |
| "name": "train_vww.py", |
| "url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/training/visual_wake_words/train_vww.py", |
| "path": "models/vision_classification/VC01/quality_evaluation/sources/train_vww.py", |
| "expected_sha256": "0dfeb1fc24d0836f86021768b86cec3f77a97f75d719d64205eafb4b5ae3c743", |
| "license": "Apache-2.0" |
| }, |
| { |
| "name": "download_and_train_vww.sh", |
| "url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/training/visual_wake_words/download_and_train_vww.sh", |
| "path": "models/vision_classification/VC01/quality_evaluation/sources/download_and_train_vww.sh", |
| "expected_sha256": "ecccf317cdd2432a0ca93859b2158e9edf36b6589b04e3495b4f0ff14068523f", |
| "license": "Apache-2.0" |
| }, |
| { |
| "name": "LICENSE.md", |
| "url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/LICENSE.md", |
| "path": "models/vision_classification/VC01/quality_evaluation/sources/LICENSE.md", |
| "expected_sha256": "0d542e0c8804e39aa7f37eb00da5a762149dc682d7829451287e11b938e94594", |
| "license": "Apache-2.0" |
| } |
| ] |
| }, |
| "models": { |
| "fp32": { |
| "path": "models/vision_classification/VC01/source/fp32/source_fp32.tflite", |
| "source_url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/training/visual_wake_words/trained_models/vww_96_float.tflite", |
| "expected_sha256": "115bbc094d2119561320a21f01b6500a18bea8cc8589282ab007097bec8af38c" |
| }, |
| "public_int8": { |
| "path": "models/vision_classification/VC01/source/quantized/source_public_quantized.tflite", |
| "source_url": "https://raw.githubusercontent.com/mlcommons/tiny/4addd0fa08d216e20637637874e084895f289da4/benchmark/training/visual_wake_words/trained_models/vww_96_int8.tflite", |
| "expected_sha256": "597a384c8c2c8a1276f04702f25013b7838f2f814f1ca7c174d295b73e3d6b7b" |
| } |
| }, |
| "protocol": { |
| "benchmark_id": "vww01", |
| "dataset_loader": "vww_coco_archive", |
| "labels_path": "models/vision_classification/VC01/quality_evaluation/sources/y_labels.csv", |
| "input_shape": [1, 96, 96, 3], |
| "fp32_input_mode": "rgb_uint8_divide_255", |
| "public_int8_input": "quantize the same RGB/255 semantic tensor with artifact scale and zero point", |
| "class_count": 2, |
| "label_space": {"0": "no person", "1": "person"}, |
| "label_counts": {"0": 500, "1": 500}, |
| "sample_order": "pinned y_labels.csv order", |
| "metric_equation": "correct argmax predictions divided by 1000 official stimuli" |
| }, |
| "metric": { |
| "name": "top1_accuracy", |
| "sample_count": 1000, |
| "acceptance_threshold": 0.8, |
| "threshold_source": "models/vision_classification/VC01/quality_evaluation/sources/datasets_README.md" |
| }, |
| "runtime": { |
| "interpreter": "ai-edge-litert", |
| "threads": 1, |
| "batch_size": 1, |
| "quantized_input_rounding": "numpy_rint_then_int8_clip", |
| "output_decision": "argmax_raw_tensor" |
| }, |
| "provenance": { |
| "fetch_report": "models/vision_classification/VC01/quality_evaluation/fetch_report.json", |
| "stdout_log": "models/vision_classification/VC01/quality_evaluation/logs/evaluation.stdout.log", |
| "stderr_log": "models/vision_classification/VC01/quality_evaluation/logs/evaluation.stderr.log", |
| "validation_report": "models/vision_classification/VC01/quality_evaluation/results/validation_report.json" |
| } |
| } |
|
|