xujunwei
Fix Hugging Face package paths and aidem SHA checks
3a1ad37
Raw
History Blame Contribute Delete
26.2 kB
from __future__ import annotations
"""BEVFormer board demo command-line entry.
This file is the orchestration layer of the demo. It does not implement the
model math itself. Its responsibilities are:
1. Parse command-line arguments.
2. Resolve model/config/data/output paths.
3. Optionally run a dry-run check without loading AidLite.
4. Load the four QNN240 context files through :class:`BevFormerModel`.
5. Run the four-frame sample4 manifest and write a summary JSON.
The real model execution lives in ``bevformer.py``. Image preprocessing and
small utility functions live in ``utils.py``.
"""
import argparse
from datetime import datetime
import json
import sys
import time
from pathlib import Path
from bevformer import DEFAULT_SHA256, BevFormerModel
from utils import EXPECTED_TENSORS, sha256_file
# ``code/python/run_demo.py`` is placed under the packaged Python demo.
# PACKAGE_DIR: .../BevFormer-Tiny-Resnet50/code/python
# CODE_ROOT : .../BevFormer-Tiny-Resnet50/code
# REPO_ROOT : .../BevFormer-Tiny-Resnet50
PACKAGE_DIR = Path(__file__).resolve().parent
CODE_ROOT = PACKAGE_DIR.parent
REPO_ROOT = CODE_ROOT.parent
DEMO_ROOT = REPO_ROOT
# Default QNN240 AidLite context files. These encrypted model files live at
# the Hugging Face repository root under ``models/QCS8550/FP16``.
MODEL_ROOT = REPO_ROOT / "models" / "QCS8550" / "FP16"
DEFAULT_BACKBONE = MODEL_ROOT / "backbone_context.bin.aidem"
DEFAULT_ENCODER_TEMPORAL = MODEL_ROOT / "temporal_encoder_context.bin.aidem"
DEFAULT_ENCODER_SCENE_START = MODEL_ROOT / "scene_start_encoder_context.bin.aidem"
DEFAULT_DECODER = MODEL_ROOT / "decoder_context.bin.aidem"
# Default config, sample manifest, postprocess contract, and output directory.
DEFAULT_CONFIG = PACKAGE_DIR / "configs" / "demo_config.json"
DEFAULT_MANIFEST = PACKAGE_DIR / "datasets" / "sample4" / "asset_manifest.json"
DEFAULT_NMS_CONTRACT = PACKAGE_DIR / "configs" / "nms_runtime_contract.json"
DEFAULT_OUTPUT = REPO_ROOT / "outputs"
# The demo always preprocesses the six camera JPGs in parallel by default.
DEFAULT_PREPROCESS_WORKERS = 6
def parse_args() -> argparse.Namespace:
"""Parse the command-line interface used by ``python/run_test.py``.
``run_test.py`` is only a thin wrapper. All actual CLI options are defined
here so the board command remains similar to a YOLOv5-style demo command.
"""
parser = argparse.ArgumentParser(description="Run BEVFormer strict board demo with AidLite QNN240.")
# Path arguments. If omitted, values are resolved from demo_config.json or
# the hard-coded defaults above.
parser.add_argument("--config", default=str(DEFAULT_CONFIG))
parser.add_argument("--backbone_model")
parser.add_argument("--encoder_model")
parser.add_argument("--scene_start_encoder_model")
parser.add_argument("--decoder_model")
parser.add_argument("--asset_manifest")
parser.add_argument("--nms_contract")
parser.add_argument("--output_dir")
# Frame range. ``--invoke_nums`` is kept as the YOLOv5-style alias for
# "how many samples to run".
parser.add_argument("--frame_start", type=int, default=0)
parser.add_argument("--frame_count", type=int, default=4)
parser.add_argument(
"--invoke_nums",
type=int,
default=None,
help="YOLOv5-style alias for how many consecutive frames to run.",
)
# Output and visualization switches.
parser.add_argument("--save_all_raw", action="store_true")
parser.add_argument("--no_visualize", action="store_true", help="Disable camera-grid visualization image output.")
parser.add_argument("--vis_score_thr", type=float, default=0.0)
parser.add_argument("--vis_max_boxes", type=int, default=80)
# SHA checking for JPG files is optional because it adds extra file I/O.
# Model context SHA checking is always performed.
parser.add_argument(
"--check_image_sha",
action="store_true",
help="Verify every camera JPG SHA during real inference. Slower; useful for audit runs.",
)
# Only QNN240 is supported by this delivery package.
parser.add_argument("--model_type", default="QNN240")
# Dry-run mode is for host/package checks. It avoids AidLite loading and
# therefore can run on a normal development machine.
parser.add_argument(
"--dry_run",
action="store_true",
help="Inspect config, model SHA, manifest, and scene/temporal routing without loading AidLite.",
)
parser.add_argument(
"--check_raw_assets",
action="store_true",
help="In dry-run mode, also check that every raw asset path referenced by the selected frames exists.",
)
return parser.parse_args()
def require_file(name: str, path: str) -> str:
"""Return an absolute file path after existence and non-empty checks."""
value = Path(path).expanduser().resolve()
if not value.is_file() or value.stat().st_size == 0:
raise FileNotFoundError(f"{name} missing or empty: {value}")
return str(value)
def load_config(path: str) -> dict:
"""Load the main JSON config used to find default model/data paths."""
value = Path(path).expanduser().resolve()
if not value.is_file():
raise FileNotFoundError(value)
return json.loads(value.read_text(encoding="utf-8"))
def demo_path(config: dict, key_path: tuple[str, ...], fallback: Path) -> str:
"""Resolve one path from ``demo_config.json``.
``key_path`` is a nested JSON key path such as ``("models", "backbone")``.
Relative paths in the config are interpreted relative to ``DEMO_ROOT``.
If the key is missing, the function returns the provided fallback path.
"""
current = config
for key in key_path:
if not isinstance(current, dict) or key not in current:
return str(fallback)
current = current[key]
path = Path(str(current))
if not path.is_absolute():
path = DEMO_ROOT / path
return str(path)
def _resolve_repo_path(path: str | Path) -> Path:
"""Resolve manifest asset paths.
The sample manifest stores board-style relative paths such as
``bevformer_delivery_demo/datasets/...``. These are resolved under
``REPO_ROOT`` so the same manifest works when the board package is located
at ``/home/aidlux/bevformer_delivery_demo``.
"""
value = Path(path)
if value.is_absolute():
return value
return REPO_ROOT / value
def _dry_run(
*,
backbone_model: str,
encoder_model: str,
scene_start_encoder_model: str,
decoder_model: str,
asset_manifest: str,
nms_contract: str,
output_dir: Path,
frame_start: int,
frame_count: int,
check_raw_assets: bool,
) -> dict:
"""Validate package integrity without invoking AidLite or DSP.
Dry-run performs three checks:
1. All four QNN240 context files exist and match the expected SHA256.
2. The selected frame range exists in the sample manifest.
3. When ``--check_raw_assets`` is enabled, every referenced raw asset exists.
This mode is useful on Windows or a normal development container where the
AidLite module is not available.
"""
# Four model contexts are required by the split BEVFormer pipeline:
# backbone, scene-start encoder, temporal encoder, and decoder.
models = {
"backbone": backbone_model,
"encoder_temporal": encoder_model,
"encoder_scene_start": scene_start_encoder_model,
"decoder": decoder_model,
}
model_records = {}
for name, model in models.items():
model_path = Path(require_file(f"{name}_model", model))
actual_sha = sha256_file(model_path)
expected_sha = DEFAULT_SHA256[name]
status = "PASS" if actual_sha == expected_sha else "FAIL"
print(f"{name.upper()}_CONTEXT_SHA_GATE={status} {model_path.name}")
if status != "PASS":
raise RuntimeError(f"{name} context SHA mismatch: expected={expected_sha} actual={actual_sha}")
# Store the expected tensor contract in the dry-run summary so the
# package can be audited without loading AidLite.
model_records[name] = {
"path": str(model_path),
"sha256": actual_sha,
"expected_tensors": EXPECTED_TENSORS[name],
}
manifest_path = Path(require_file("asset_manifest", asset_manifest))
nms_path = Path(require_file("nms_contract", nms_contract))
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
total_frames = int(manifest.get("total_frames", len(manifest["frames"])))
end = min(total_frames, frame_start + frame_count)
if frame_start < 0 or frame_start >= total_frames or end < frame_start:
raise ValueError(f"Invalid frame range: start={frame_start} count={frame_count} total={total_frames}")
frames = {}
scene_start_count = 0
temporal_count = 0
missing_assets = []
for frame_index in range(frame_start, end):
sample = f"sample_{frame_index:03d}"
frame = manifest["frames"][sample]
is_scene_start = bool(frame.get("is_scene_start", False))
encoder_name = "encoder_scene_start" if is_scene_start else "encoder_temporal"
# frame000 is a scene-start frame. Later frames use temporal encoder and
# depend on the previous frame's live bev_embed.
if is_scene_start:
scene_start_count += 1
else:
temporal_count += 1
# Raw asset checking is optional because it touches every camera JPG and
# auxiliary tensor path. It is recommended before packaging or upload.
if check_raw_assets:
for asset_name, record in frame.get("assets", {}).items():
if asset_name == "camera_images":
for image_record in record.get("images", []):
asset_path = _resolve_repo_path(image_record["path"])
if not asset_path.is_file():
missing_assets.append({
"frame": sample,
"asset": f"camera_images/{image_record.get('name', 'UNKNOWN')}",
"path": str(asset_path),
})
continue
asset_path = _resolve_repo_path(record["path"])
if not asset_path.is_file():
missing_assets.append({
"frame": sample,
"asset": asset_name,
"path": str(asset_path),
})
frames[sample] = {
"sample_token": frame.get("sample_token"),
"is_scene_start": is_scene_start,
"encoder": encoder_name,
"status": "DRY_RUN_PASS",
}
print(f"FRAME {frame_index:03d} DRY_RUN encoder={encoder_name}")
if missing_assets:
first = missing_assets[0]
raise FileNotFoundError(f"Missing raw asset: {first['frame']} {first['asset']} {first['path']}")
output_dir.mkdir(parents=True, exist_ok=True)
run_finished_at = local_timestamp()
# Write a machine-readable summary. This is useful for proving that the
# package is self-consistent before running on the board.
result = {
"status": "DRY_RUN_PASS",
"run_timestamps": {"finished_at": run_finished_at},
"note": "AidLite/DSP was not invoked. Run without --dry_run on the board for real inference.",
"manifest": str(manifest_path),
"nms_contract": str(nms_path),
"repo_root": str(REPO_ROOT),
"frame_range": [int(frame_start), int(end - 1)] if end > frame_start else [],
"completed_frames": int(end - frame_start),
"scene_start_encoder_count": scene_start_count,
"temporal_encoder_count": temporal_count,
"models": model_records,
"raw_asset_existence_checked": bool(check_raw_assets),
"frames": frames,
}
result_path = output_dir / "bevformer_demo_dry_run_summary.json"
result_path.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
print("====================================")
print("BEVFormer demo status: DRY_RUN_PASS")
print(f"frames: {result['completed_frames']}")
print(f"scene_start_encoder: {scene_start_count}")
print(f"temporal_encoder: {temporal_count}")
print(f"summary: {result_path}")
print("AidLite/DSP not invoked in dry-run mode.")
print("====================================")
return result
def _fmt_ms(value) -> str:
"""Format optional millisecond values for console output."""
if value is None:
return "N/A"
return f"{float(value):.3f}"
def local_timestamp() -> str:
"""Return a local ISO-8601 timestamp for run logs and summaries."""
return datetime.now().astimezone().isoformat(timespec="seconds")
def timestamp_for_filename(value: str) -> str:
"""Convert an ISO timestamp to a filesystem-friendly suffix."""
return value.replace("-", "").replace(":", "").replace("T", "_").split("+")[0]
def timestamp_for_output_dir(value: str) -> str:
"""Convert an ISO timestamp to outputs/YYYY_MM_DD_HH_MM style."""
dt = datetime.fromisoformat(value)
return dt.strftime("%Y_%m_%d_%H_%M")
def make_output_dir(base_dir: Path, started_at: str) -> Path:
"""Create a timestamped output directory, avoiding same-minute overwrite."""
base_dir = base_dir.expanduser().resolve()
candidate = base_dir / timestamp_for_output_dir(started_at)
if not candidate.exists():
return candidate
for index in range(2, 100):
numbered = base_dir / f"{candidate.name}_{index:02d}"
if not numbered.exists():
return numbered
raise RuntimeError(f"Too many output directories already exist for {candidate.name}")
def write_run_log(output_dir: Path, result: dict, result_path: Path, command: list[str]) -> dict[str, str]:
"""Write a compact human-readable run log with timestamps."""
timestamps = result.get("run_timestamps", {})
started_at = timestamps.get("started_at", "unknown")
finished_at = timestamps.get("finished_at", "unknown")
stamp = timestamp_for_filename(started_at) if started_at != "unknown" else "unknown"
timestamped_log = output_dir / f"run_{stamp}.log"
latest_log = output_dir / "run.log"
e2e = result.get("end_to_end_timing_ms", {})
qnn = result.get("qnn_invoke_ms", {})
per_bin = result.get("per_bin_qnn_invoke_ms", {})
app = result.get("application_timing_ms", {})
lines = [
"BEVFormer W8A8 board demo run log",
"========================================",
f"started_at : {started_at}",
f"finished_at : {finished_at}",
f"status : {result.get('status')}",
f"command : {' '.join(command)}",
f"frames : {result.get('completed_frames')}",
f"scene-start frames : {result.get('scene_start_encoder_count')}",
f"temporal frames : {result.get('temporal_encoder_count')}",
f"mean QNN execute, selected pipeline (ms): {_fmt_ms(qnn.get('mean'))}",
"per-bin QNN invoke only, mean ms :",
f" backbone_context.bin : {_fmt_ms(per_bin.get('backbone_context.bin', {}).get('mean'))}",
f" scene_start_encoder_context.bin : {_fmt_ms(per_bin.get('scene_start_encoder_context.bin', {}).get('mean'))}",
f" temporal_encoder_context.bin : {_fmt_ms(per_bin.get('temporal_encoder_context.bin', {}).get('mean'))}",
f" decoder_context.bin : {_fmt_ms(per_bin.get('decoder_context.bin', {}).get('mean'))}",
f"full inference chain, no drawing (ms): {_fmt_ms(e2e.get('complete_inference_no_visualization_ms'))}",
f"full demo chain, with drawing (ms) : {_fmt_ms(e2e.get('complete_inference_with_visualization_ms'))}",
f"whole Python run incl. load (ms) : {_fmt_ms(app.get('total_until_program_end_ms'))}",
f"summary JSON : {result_path}",
f"output directory : {output_dir}",
]
if result.get("camera_grid_gif"):
lines.append(f"camera-grid GIF : {result['camera_grid_gif']['path']}")
text = "\n".join(lines) + "\n"
timestamped_log.write_text(text, encoding="utf-8")
latest_log.write_text(text, encoding="utf-8")
return {"timestamped": str(timestamped_log), "latest": str(latest_log)}
def main() -> int:
"""Run the board demo.
The function first resolves all paths. If ``--dry_run`` is enabled, it
stops after package checks. Otherwise it loads AidLite contexts through
``BevFormerModel`` and runs the selected continuous frame span.
"""
app_start = time.perf_counter_ns()
run_started_at = local_timestamp()
# 1. Parse CLI and resolve all default paths from demo_config.json.
args = parse_args()
config = load_config(args.config)
backbone_model = args.backbone_model or demo_path(config, ("models", "backbone"), DEFAULT_BACKBONE)
encoder_model = args.encoder_model or demo_path(config, ("models", "encoder_temporal"), DEFAULT_ENCODER_TEMPORAL)
scene_start_encoder_model = args.scene_start_encoder_model or demo_path(
config,
("models", "encoder_scene_start"),
DEFAULT_ENCODER_SCENE_START,
)
decoder_model = args.decoder_model or demo_path(config, ("models", "decoder"), DEFAULT_DECODER)
asset_manifest = args.asset_manifest or demo_path(config, ("inputs", "asset_manifest"), DEFAULT_MANIFEST)
nms_contract = args.nms_contract or demo_path(config, ("postprocess", "nms_contract"), DEFAULT_NMS_CONTRACT)
if args.output_dir:
output_dir = Path(args.output_dir).expanduser().resolve()
else:
output_root = Path(demo_path(config, ("outputs", "default_dir"), DEFAULT_OUTPUT))
output_dir = make_output_dir(output_root, run_started_at)
# ``--invoke_nums`` takes precedence over ``--frame_count``.
frame_count = args.invoke_nums if args.invoke_nums is not None else args.frame_count
# 2. Host-side package inspection path. No AidLite import or DSP execution.
if args.dry_run:
_dry_run(
backbone_model=backbone_model,
encoder_model=encoder_model,
scene_start_encoder_model=scene_start_encoder_model,
decoder_model=decoder_model,
asset_manifest=asset_manifest,
nms_contract=nms_contract,
output_dir=output_dir,
frame_start=args.frame_start,
frame_count=frame_count,
check_raw_assets=args.check_raw_assets,
)
return 0
# 3. Board-side real inference path. This requires ``import aidlite`` to
# succeed inside BevFormerModel.
model_load_start = time.perf_counter_ns()
model = BevFormerModel(
backbone_model=require_file("backbone_model", backbone_model),
encoder_temporal_model=require_file("encoder_model", encoder_model),
encoder_scene_start_model=require_file("scene_start_encoder_model", scene_start_encoder_model),
decoder_model=require_file("decoder_model", decoder_model),
model_type=args.model_type,
)
model_load_wall_ms = (time.perf_counter_ns() - model_load_start) / 1.0e6
# 4. Run sample4: raw JPG preprocessing -> backbone -> encoder -> decoder
# -> NumPy NMSFreeCoder -> NPZ/PNG/GIF outputs.
inference_start = time.perf_counter_ns()
result = model.run_manifest(
manifest_path=require_file("asset_manifest", asset_manifest),
repo_root=REPO_ROOT,
output_dir=output_dir,
nms_contract_path=require_file("nms_contract", nms_contract),
frame_start=args.frame_start,
frame_count=frame_count,
save_all_raw=args.save_all_raw,
visualize=not args.no_visualize,
vis_score_thr=args.vis_score_thr,
vis_max_boxes=args.vis_max_boxes,
check_image_sha=args.check_image_sha,
preprocess_workers=DEFAULT_PREPROCESS_WORKERS,
)
inference_wall_ms = (time.perf_counter_ns() - inference_start) / 1.0e6
# 5. Add application-level timing that is not part of model-only QNN invoke.
result["application_timing_ms"] = {
"model_load_wall_ms": model_load_wall_ms,
"run_manifest_wall_ms": inference_wall_ms,
"total_until_summary_write_excluded_ms": (time.perf_counter_ns() - app_start) / 1.0e6,
}
result["run_timestamps"] = {
"started_at": run_started_at,
"summary_started_at": local_timestamp(),
}
result_path = output_dir / "bevformer_demo_summary.json"
# Write once to measure summary serialization overhead, then write again
# after adding the measured value to the summary.
summary_write_start = time.perf_counter_ns()
result_path.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
summary_write_ms = (time.perf_counter_ns() - summary_write_start) / 1.0e6
result["application_timing_ms"]["summary_write_ms"] = summary_write_ms
result["application_timing_ms"]["total_until_program_end_ms"] = (time.perf_counter_ns() - app_start) / 1.0e6
result["run_timestamps"]["finished_at"] = local_timestamp()
result_path.write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding="utf-8")
# 6. Console summary for board demo presentation.
qnn = result.get("qnn_invoke_ms", {})
components = result.get("component_invoke_ms", {})
per_bin = result.get("per_bin_qnn_invoke_ms", {})
e2e = result.get("end_to_end_timing_ms", {})
app = result.get("application_timing_ms", {})
print("========================================")
print("BEVFormer W8A8 Board Demo Result")
print("========================================")
print(f"status : {result['status']}")
print(f"started_at : {result['run_timestamps']['started_at']}")
print(f"finished_at : {result['run_timestamps']['finished_at']}")
print(f"frames : {result['completed_frames']}")
print(f"scene-start frames : {result['scene_start_encoder_count']} -> encoder_scene_start")
print(f"temporal frames : {result['temporal_encoder_count']} -> encoder_temporal")
print(f"preprocess workers : {DEFAULT_PREPROCESS_WORKERS}")
print("")
print("QNN invoke only, grouped by delivered .bin")
print(" note: excludes image preprocessing, tensor set/get, NMS, file saving, and visualization")
print(" bin file count mean(ms) min(ms) max(ms) var")
for bin_name in (
"backbone_context.bin",
"scene_start_encoder_context.bin",
"temporal_encoder_context.bin",
"decoder_context.bin",
):
item = per_bin.get(bin_name, {})
print(
f" {bin_name:<34} {int(item.get('count', 0)):>5} "
f"{_fmt_ms(item.get('mean')):>10} {_fmt_ms(item.get('min')):>9} "
f"{_fmt_ms(item.get('max')):>9} {_fmt_ms(item.get('var')):>9}"
)
print("")
print("QNN invoke only, selected pipeline per frame")
print(f" mean : {_fmt_ms(qnn.get('mean'))} ms")
print(f" max : {_fmt_ms(qnn.get('max'))} ms")
print(f" min : {_fmt_ms(qnn.get('min'))} ms")
print(f" variance : {_fmt_ms(qnn.get('var'))}")
print("")
print("Mean time per frame (ms)")
print(f" preprocess (6 JPG -> tensor) : {_fmt_ms(result.get('image_preprocess_ms', {}).get('mean'))}")
print(f" QNN execute (3 contexts) : {_fmt_ms(qnn.get('mean'))}")
print(f" backbone context execute : {_fmt_ms(components.get('backbone', {}).get('mean'))}")
print(f" encoder context execute : {_fmt_ms(components.get('encoder', {}).get('mean'))}")
print(f" scene-start encoder .bin invoke : {_fmt_ms(components.get('encoder_scene_start', {}).get('mean'))}")
print(f" temporal encoder .bin invoke : {_fmt_ms(components.get('encoder_temporal', {}).get('mean'))}")
print(f" decoder context execute : {_fmt_ms(components.get('decoder', {}).get('mean'))}")
print(f" postprocess (NMS + save boxes) : {_fmt_ms(result.get('postprocess_ms', {}).get('mean'))}")
print(f" visualization (camera-grid PNG) : {_fmt_ms(result.get('visualization_ms', {}).get('mean'))}")
print(f" model path total (no drawing) : {_fmt_ms(result['timing_ms'].get('mean'))}")
print("")
print("Complete run time (ms)")
print(f" model loading : {_fmt_ms(app.get('model_load_wall_ms'))}")
print(f" manifest + NMS contract loading : {_fmt_ms(e2e.get('manifest_and_contract_load_ms'))}")
print(f" full inference chain (no drawing) : {_fmt_ms(e2e.get('complete_inference_no_visualization_ms'))}")
print(f" all visualization rendering : {_fmt_ms(e2e.get('visualization_total_ms'))}")
print(f" camera-grid GIF rendering : {_fmt_ms(e2e.get('camera_grid_gif_ms'))}")
print(f" full demo chain (with drawing) : {_fmt_ms(e2e.get('complete_inference_with_visualization_ms'))}")
print(f" whole Python run incl. load : {_fmt_ms(app.get('total_until_program_end_ms'))}")
run_logs = write_run_log(output_dir, result, result_path, sys.argv)
print("")
print("Output files")
print(f" summary JSON : {result_path}")
print(f" timestamped run log : {run_logs['timestamped']}")
print(f" latest run log : {run_logs['latest']}")
if result.get("visualizations"):
print(f" camera-grid PNGs : {len(result['visualizations'])} image(s)")
if result.get("camera_grid_gif"):
print(f" camera-grid GIF : {result['camera_grid_gif']['path']}")
print(f" output directory : {output_dir}")
print("========================================")
return 0
if __name__ == "__main__":
raise SystemExit(main())