HOA7-Spatial-Decoder / examples /demo_e2e_testpack.py
woodfireind's picture
HOA7 Spatial Field Decoder (hoa64 v0.5.0): 7th-order Ambisonics encode/decode, Wigner-D rotation, DOA analysis, vision fuse, diffusion conditioning
570b87b verified
Raw
History Blame Contribute Delete
3.11 kB
#!/usr/bin/env python3
"""End-to-end pack for manual testing: audio scene, vision detect, fuse, condition.
Writes artifacts under /tmp/spatial_hoa_e2e/ and prints commands for Qwythos/Comfy.
"""
from __future__ import annotations
import json
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from hoa64.audio_io import write_wav
from hoa64.conditioning import build_conditioning, comfy_txt2img_payload, save_conditioning
from hoa64.detector import detections_to_sphere_boxes, write_demo_image_with_box
from hoa64.report import report_from_scene
from hoa64.stream import SourceSpec, encode_scene
from hoa64.synth import envelope_adsr, tone
from hoa64.vision import fuse_reports, report_from_boxes
def main() -> None:
out = Path("/tmp/spatial_hoa_e2e")
out.mkdir(parents=True, exist_ok=True)
sr = 48000
dur = 0.5
n = int(sr * dur)
env = envelope_adsr(n, sr)
sources = [
SourceSpec(15.0, 0.0, tone(520, dur, sr, amplitude=0.45) * env, "beep"),
SourceSpec(-80.0, 5.0, tone(780, dur, sr, amplitude=0.2) * env, "side"),
]
audio_rep = report_from_scene(sources, sr, max_order=3)
audio_rep.save(out / "audio_report.json")
hoa = encode_scene(sources, max_order=3)
write_wav(out / "scene_ambix4.wav", hoa[:4], sr)
img = out / "demo_frame.png"
det = write_demo_image_with_box(img)
boxes = detections_to_sphere_boxes([det])
(out / "boxes.json").write_text(json.dumps(boxes, indent=2) + "\n")
vision_rep = report_from_boxes(boxes, max_order=3)
vision_rep.save(out / "vision_report.json")
fused = fuse_reports(
{**audio_rep.to_dict(), "one_liner": audio_rep.one_liner()},
{**vision_rep.to_dict(), "one_liner": vision_rep.one_liner()},
)
(out / "fuse_report.json").write_text(json.dumps(fused, indent=2) + "\n")
cond = build_conditioning(
fused,
base_prompt="cinematic interior, soft window light, photoreal",
style="natural",
)
save_conditioning(cond, out / "conditioning.json")
wf = comfy_txt2img_payload(cond, width=512, height=512, steps=20)
(out / "comfy_workflow.json").write_text(json.dumps(wf, indent=2) + "\n")
print("=== E2E artifacts ===")
for p in sorted(out.iterdir()):
print(f" {p}")
print("\nAUDIO ", audio_rep.one_liner())
print("VISION", vision_rep.one_liner())
print("FUSE ", fused["one_liner"])
print("PROMPT", cond["positive_prompt"])
print(
"""
Test commands:
curl -s http://127.0.0.1:8765/health
spatial-report analyze /tmp/spatial_hoa_e2e/scene_ambix4.wav --ambix -o /tmp/a.json
spatial-report detect --demo-image /tmp/spatial_hoa_e2e/frame2.png -o /tmp/v.json
spatial-report condition /tmp/spatial_hoa_e2e/fuse_report.json -o /tmp/c.json --prompt 'moody hall'
spatial-report live --duration 1.5 -o /tmp/live.json --write-wav /tmp/live.wav
# optional Comfy (if running):
spatial-report condition /tmp/spatial_hoa_e2e/fuse_report.json --comfy --write-workflow /tmp/wf.json
"""
)
if __name__ == "__main__":
main()