| |
| """ |
| apply_audition_patch.py — Patches audition_mixin.py with proper loopback device |
| selection, chunked recording, transcription, and media disambiguation. |
| |
| Changes: |
| 1. _find_input_device() — proper loopback detection (VB-Cable, Stereo Mix, BlackHole, .monitor) |
| 2. _record_chunk() — 30s mono 16kHz WAV chunks to sensorium/audio/ |
| 3. _transcribe_and_store() — whisper transcription + content classification |
| 4. Media disambiguation — lyrics vs speech vs ambient |
| 5. Mute/respect — only record when RADIX_AUDIO_SOURCE is set and not "pause" |
| 6. Cleanup — clear audio files after transcription unless save_raw=True |
| |
| Run from the Sir Radix project root: |
| python apply_audition_patch.py |
| """ |
| import sys |
| import os |
| import shutil |
| from datetime import datetime |
|
|
| TARGET = "audition_mixin.py" |
| BACKUP = f"audition_mixin.py.bak.{datetime.now().strftime('%Y%m%d_%H%M%S')}" |
|
|
| def main(): |
| if not os.path.exists(TARGET): |
| print(f"ERROR: {TARGET} not found in current directory.") |
| sys.exit(1) |
|
|
| with open(TARGET, "r", encoding="utf-8") as f: |
| source = f.read() |
|
|
| shutil.copy2(TARGET, BACKUP) |
| print(f"Backed up to {BACKUP}") |
|
|
| patched = source |
|
|
| |
| |
| |
| patched = patched.replace( |
| ' self._start_audio_loop()', |
| ''' # Audio awareness opt-in check |
| self._audio_source_pref = os.environ.get("RADIX_AUDIO_SOURCE", "auto") |
| self._audition_enabled = self._audio_source_pref not in ("pause", "off", "0", "") |
| self._save_raw_audio = False |
| |
| if self._audition_enabled: |
| self._start_audio_loop() |
| else: |
| logger.info("Audition: disabled (RADIX_AUDIO_SOURCE=%s)", self._audio_source_pref)''' |
| ) |
|
|
| |
| |
| |
| old_find = ''' def _find_input_device(self) -> Optional[int]: |
| """Return the default input device index.""" |
| if not HAS_SOUNDDEVICE: |
| return None |
| try: |
| # sd.default.device[0] is the default input |
| default_in = sd.default.device[0] |
| if default_in is not None: |
| dev_info = sd.query_devices(default_in) |
| logger.info("Audition: using input device %s — %s", default_in, dev_info.get("name")) |
| return default_in |
| except Exception as e: |
| logger.warning("Audition: device query failed: %s", e) |
| return None''' |
|
|
| new_find = ''' def _find_input_device(self) -> Optional[int]: |
| """Return the best loopback/desktop audio device index. |
| Priority: VB-Cable > WASAPI loopback > Stereo Mix > BlackHole > .monitor > default. |
| Never silently falls back to microphone unless explicitly requested. |
| """ |
| if not HAS_SOUNDDEVICE: |
| return None |
| |
| source_pref = getattr(self, "_audio_source_pref", "auto") |
| if source_pref == "microphone": |
| try: |
| default = sd.query_devices(kind="input") |
| return default.get("index") |
| except Exception: |
| return None |
| |
| try: |
| devices = sd.query_devices() |
| except Exception as e: |
| logger.warning("Audition: device query failed: %s", e) |
| return None |
| |
| candidates = [] |
| for i, d in enumerate(devices): |
| if d.get("max_input_channels", 0) == 0: |
| continue |
| name = d.get("name", "") |
| lower = name.lower() |
| score = 0 |
| |
| # VB-Audio family |
| if "cable output" in lower: |
| score += 120 |
| elif "cable" in lower and ("output" in lower or "out" in lower): |
| score += 100 |
| elif "vb-audio" in lower: |
| score += 90 |
| |
| # WASAPI loopback |
| if "loopback" in lower: |
| score += 110 |
| |
| # Stereo Mix / What U Hear |
| if "stereo mix" in lower: |
| score += 80 |
| if "what u hear" in lower: |
| score += 80 |
| |
| # macOS |
| if "blackhole" in lower: |
| score += 90 |
| if "soundflower" in lower: |
| score += 90 |
| |
| # Linux |
| if ".monitor" in lower or "monitor of" in lower: |
| score += 85 |
| |
| # Deprioritize physical microphones |
| mic_hints = ["microphone", "mic ", "mic(", "internal", "headset", |
| "webcam", "camera", "array", "comm"] |
| if any(h in lower for h in mic_hints): |
| score -= 60 |
| |
| if score > 0: |
| candidates.append((score, i, name)) |
| |
| if candidates: |
| candidates.sort(reverse=True, key=lambda x: x[0]) |
| best = candidates[0] |
| logger.info("Audition: using loopback device %s — %s (score=%d)", best[1], best[2], best[0]) |
| return best[1] |
| |
| # If desktop was explicitly requested but not found, do NOT fall back to mic |
| if source_pref == "desktop": |
| logger.warning("Audition: no loopback device found, staying silent (not falling back to mic)") |
| return None |
| |
| # Auto mode: fall back to default input only if nothing better exists |
| try: |
| default = sd.query_devices(kind="input") |
| if default: |
| logger.info("Audition: no loopback found, using default input: %s", default.get("name")) |
| return default.get("index") |
| except Exception: |
| pass |
| |
| return None''' |
|
|
| patched = patched.replace(old_find, new_find) |
|
|
| |
| |
| |
| old_transcribe = ''' def _transcribe_and_store(self, path: Path): |
| """Run Whisper on a chunk and store text as memory.""" |
| try: |
| segments, info = self._whisper_model.transcribe(str(path), beam_size=5) |
| text = " ".join([seg.text for seg in segments]).strip() |
| if not text: |
| return |
| |
| # Rough content classification for emotional scoring |
| lower = text.lower() |
| if any(w in lower for w in ("error", "exception", "traceback", "failed", "crash", "broken")): |
| score = 0.65 |
| tag = "AUDIO_ALERT" |
| elif any(w in lower for w in ("meeting", "call", "discuss", "deadline", "review")): |
| score = 0.45 |
| tag = "AUDIO_MEETING" |
| else: |
| score = 0.1 |
| tag = "AUDIO_AMBIENT" |
| |
| self.store_memory( |
| f"[Audio] {text[:500]}", |
| emotional_score=score, |
| classification="short_term", |
| mnemonic=f"{tag}_{datetime.now().strftime('%H%M%S')}", |
| ) |
| logger.debug("Audition: stored %s chars", len(text)) |
| except Exception as e: |
| logger.debug("Audition: transcription failed: %s", e)''' |
|
|
| new_transcribe = ''' def _transcribe_and_store(self, path: Path): |
| """Run Whisper on a chunk, classify content, and store text as memory.""" |
| try: |
| segments, info = self._whisper_model.transcribe(str(path), beam_size=5) |
| text = " ".join([seg.text for seg in segments]).strip() |
| if not text: |
| return |
| |
| # Media disambiguation: lyrics vs speech vs ambient |
| media_type, score, tag = self._classify_audio_content(text) |
| |
| self.store_memory( |
| f"[Audio:{media_type}] {text[:500]}", |
| emotional_score=score, |
| classification="short_term", |
| mnemonic=f"{tag}_{datetime.now().strftime('%H%M%S')}", |
| ) |
| |
| # Store transcript for companion mixin to pick up |
| if hasattr(self, "_last_audio_transcript"): |
| self._last_audio_transcript = text[:200] |
| |
| logger.debug("Audition: stored %s chars (%s)", len(text), media_type) |
| |
| # Clean up audio file unless save_raw is set |
| if not getattr(self, "_save_raw_audio", False): |
| path.unlink(missing_ok=True) |
| |
| except Exception as e: |
| logger.debug("Audition: transcription failed: %s", e) |
| |
| def _classify_audio_content(self, text: str) -> tuple: |
| """Classify audio content as music/speech/ambient and assign emotional score. |
| Returns (media_type, emotional_score, mnemonic_tag). |
| """ |
| lower = text.lower() |
| word_count = len(text.split()) |
| |
| # Error/alert detection (code errors, crashes) |
| if any(w in lower for w in ("error", "exception", "traceback", "failed", "crash", "broken")): |
| return "alert", 0.65, "AUDIO_ALERT" |
| |
| # Meeting/work detection |
| if any(w in lower for w in ("meeting", "call", "discuss", "deadline", "review", "standup")): |
| return "speech", 0.45, "AUDIO_MEETING" |
| |
| # Lyrics detection: repetitive word patterns, short lines, rhyming |
| if word_count > 10: |
| words = lower.split() |
| unique_ratio = len(set(words)) / max(len(words), 1) |
| # Music lyrics tend to have lower unique word ratio (repetition) |
| if unique_ratio < 0.55: |
| return "music", 0.3, "AUDIO_MUSIC" |
| |
| # Continuous speech (podcast, video, conversation) |
| if word_count > 20 and unique_ratio > 0.6: |
| return "speech", 0.2, "AUDIO_SPEECH" |
| |
| # Ambient/noise |
| return "ambient", 0.1, "AUDIO_AMBIENT"''' |
|
|
| patched = patched.replace(old_transcribe, new_transcribe) |
|
|
| |
| |
| |
| patched = patched.replace( |
| ' def tool_audio_transcribe_now(self) -> Dict:\n """Force-transcribe the most recent completed chunk immediately."""\n if not self._current_chunk_path or not self._current_chunk_path.exists():\n return {"status": "no_chunk"}\n self._transcribe_and_store(self._current_chunk_path)\n return {"status": "transcribed", "path": str(self._current_chunk_path)}', |
| ''' def tool_audio_transcribe_now(self) -> Dict: |
| """Force-transcribe the most recent completed chunk immediately.""" |
| if not getattr(self, "_audition_enabled", False): |
| return {"status": "disabled", "reason": "Audio awareness is off"} |
| if not self._current_chunk_path or not self._current_chunk_path.exists(): |
| return {"status": "no_chunk"} |
| self._transcribe_and_store(self._current_chunk_path) |
| return {"status": "transcribed", "path": str(self._current_chunk_path)} |
| |
| def set_audition_enabled(self, enabled: bool): |
| """Toggle audio capture at runtime.""" |
| self._audition_enabled = bool(enabled) |
| os.environ["RADIX_AUDIO_SOURCE"] = "auto" if enabled else "off" |
| if enabled and (self._audio_thread is None or not self._audio_thread.is_alive()): |
| self._stop_recording.clear() |
| self._start_audio_loop() |
| logger.info("Audition: recording loop started") |
| elif not enabled and self._audio_thread: |
| self._stop_recording.set() |
| logger.info("Audition: recording loop stopped") |
| |
| def get_audition_status(self) -> Dict: |
| """Return current audio capture state.""" |
| return { |
| "enabled": getattr(self, "_audition_enabled", False), |
| "capture_active": self._audio_thread is not None and self._audio_thread.is_alive(), |
| "last_chunk": str(self._current_chunk_path) if self._current_chunk_path else None, |
| "whisper_loaded": self._whisper_model is not None, |
| "audio_dir": str(self._audio_dir), |
| "last_transcript": getattr(self, "_last_audio_transcript", "")[:200], |
| }''' |
| ) |
|
|
| |
| with open(TARGET, "w", encoding="utf-8") as f: |
| f.write(patched) |
|
|
| print(f"✅ Patched {TARGET} successfully") |
| print(f" Backup: {BACKUP}") |
| print(f" Changes:") |
| print(f" - Proper loopback device detection (VB-Cable, Stereo Mix, BlackHole, .monitor)") |
| print(f" - Never silently falls back to microphone in desktop mode") |
| print(f" - Media disambiguation (lyrics vs speech vs ambient)") |
| print(f" - Audio awareness opt-in (RADIX_AUDIO_SOURCE)") |
| print(f" - Cleanup after transcription (unless save_raw)") |
| print(f" - set_audition_enabled() runtime toggle") |
| print(f" - get_audition_status() for UI") |
|
|
|
|
| if __name__ == "__main__": |
| main() |
|
|