diff --git a/.gitattributes b/.gitattributes index 7b6cd9e5e1071c2c604da59f6b06be00dc31e99b..1688e6a23139bee7770725b74e1fc060e2bed1ab 100644 --- a/.gitattributes +++ b/.gitattributes @@ -34,4 +34,3 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text *.axmodel filter=lfs diff=lfs merge=lfs -text -python/sample_speech.pcm filter=lfs diff=lfs merge=lfs -text diff --git a/NPU_ONLY_SDK.md b/NPU_ONLY_SDK.md index 8bdff937d7e01d3286c59925436b3e4d2300b045..fbc63c864edb4e2be9f9cb1fe7e597520c475c1f 100644 --- a/NPU_ONLY_SDK.md +++ b/NPU_ONLY_SDK.md @@ -1 +1,3 @@ -本交付包已通过端到端 NPU 验证,Python SDK 仅依赖 pyaxengine,不含 onnxruntime/torch/transformers 等运行时回退。 +本交付包为 rnnoise 双芯(AX650 / AX620E)合并仓库。 +AX650 已通过板端端到端 NPU 验证(C++ 2.91ms/帧,输出 cosine 0.9988); +AX620E 尚未上板验证(Pulsar2 仿真 gains cosine 0.9987),交付 SDK 保留 onnxruntime CPU 回退用于本机验证。 diff --git a/README.md b/README.md index 2b8f3012878653767697d3b57f2ba68a150adb01..27108a0f6393d7cad9ad21b1a96ee3dc0517ee28 100644 --- a/README.md +++ b/README.md @@ -4,62 +4,71 @@ pipeline_tag: audio-to-audio tags: - axmodel - axera -- rnnoise-ax650 +- rnnoise +- audio-denoising +- ax650 +- ax620e --- -# RNNoise AX650 实时降噪(AXMODEL 交付包) -48kHz 单声道实时降噪:原版 RNNoise 网络编译到 AX650 NPU3,前后处理 1:1 对齐 -官方 C 管线。板端 C++ 每帧 2.85ms(10ms 帧预算内),输出与官方实现 cosine ≈ 0.98+; -模型 3.3MB,支持 U16 混合精度推理。 +# RNNoise(爱芯 NPU 版) -## 快速开始(只需两步) +Xiph RNNoise 语音降噪在爱芯 NPU 上的部署包,支持 **AX650(NPU3)/ AX620E(NPU2)** 双芯。 +官方模型仓库:https://github.com/xiph/rnnoise -### 1. 安装环境 +## 特性 -```bash -bash setup.sh -``` +- 48kHz 单声道实时语音降噪(逐帧 10ms,16-bit 等价域 ±32768) +- 原版 rnnoise C 信号处理(特征分析 / 合成)+ AX Engine NPU 推理 +- Python 与 C++ 双 SDK;官方未提供 ONNX,本仓库 ONNX 由官方 `rnnoise_data.c` + 权重 1:1 复刻导出(Tanh/Sigmoid 有理逼近对齐 C 端) -### 2. 跑推理(板端) +## 支持平台与模型 -```bash -bash run.sh -``` +| 芯片 | NPU | 模型目录 | gains cosine | 板端验证 | +|---|---|---|---|---| +| AX650 | NPU3 | `rnnoise_ax650/` | 0.9991(板端 198 帧) | ✅ C++ 2.91ms/帧,输出 cosine 0.9988 | +| AX620E | NPU2 | `rnnoise_ax620e/` | 0.9987(Pulsar2 仿真 100 帧) | 待上板 | + +每个模型目录内含 `model.axmodel` + `model_meta.json`。 +量化:U16 链路(MatMul/Conv/Add/Mul/Div/Sub/Concat/Clip/Slice,S8 权重), +MinMax,校准数据来自真实语音(speech/speech-echo/speech-reverb + 6dB 噪声)。 -`run.sh` 会处理自带的 1 秒演示音频(`python/sample_speech.pcm`), -输出去噪 PCM 到 `output/out.pcm`,并打印语音存在比例。 +## 快速开始 -## 目录说明 +### 1. 安装 Python 环境 -| 目录 | 用途 | -|------|------| -| `models/` | `model.axmodel`(AX650 NPU3)+ `model_meta.json` | -| `python/` | Python SDK(pyaxengine,NPU 专用,仅依赖 numpy + pyaxengine)| -| `cpp/` | C++ SDK(原版 C 信号处理 + AX Engine,实时路径)| -| `model_convert/` | 模型导出 & 编译脚本(可复现)| -| `reports/` | 导出/编译/仿真/上板报告 | +```bash +pip install -r python/requirements.txt +``` -## 自己调用 SDK(核心 3 行) +### 2. 板端运行降噪 -```python -from rnnoise_ax650_sdk import RNNoiseDenoiser -denoiser = RNNoiseDenoiser("models/model.axmodel") # 板端 pyaxengine -out_frame, vad = denoiser.process_frame(pcm_frame_480) # 48k float32 帧 +```bash +# AX650 +python3 python/demo.py --chip ax650 +# AX620E +python3 python/demo.py --chip ax620e ``` -输入帧为 16-bit PCM 等价 float(±32768 量级,不做归一化,与官方 demo 一致); -模型逐帧 6 输入(features + 5 个状态)/ 7 输出(gains/vad + 5 个新状态), -状态由 SDK 内部维护。 +输出 `output/out.pcm`(48k f32le)+ `output/vad.npy`;演示样本为 +`python/sample_speech.pcm`(约 4 秒语音)。 + +## C++ SDK + +| 芯片 | 目录 | 说明 | +|---|---|---| +| AX650 | `cpp/ax650/` | cmake 链接 ax_engine/ax_sys,板端实测 2.91ms/帧 | +| AX620E | `cpp/ax620e/` | 同源码树,目标 AX620E BSP 交叉编译 | -## 常见问题 +## 精度说明 -**Q: import 报错找不到 pyaxengine?** -A: 在 AX 板端运行 `bash setup.sh` 自动安装(交付版不做 CPU 回退)。 +| 张量 | AX650 板端 cosine | AX620E 仿真 cosine | +|---|---|---| +| gains | 0.9991 | 0.9987 | +| vad | 0.99996 | 0.99997 | +| GRU 状态 | 0.992–0.996 | 0.996+ | -**Q: Python 每帧要 60ms,能实时吗?** -A: Python 版面向原型/离线批处理;实时降噪请用 `cpp/`(2.85ms/帧), - 编译方法见 `cpp/README.md`。 +## 复现 -**Q: 想自己重新编译模型?** -A: 进入 `model_convert/`,按 README 准备 Pulsar2 7.0 Docker 后运行 - `bash compile_pulsar2.sh`。 +完整导出与编译流程(含双芯 Pulsar2 配置与校准数据)见 +GitHub:https://github.com/ml-inory/rnnoise.axera diff --git a/config.json b/config.json index 9e26dfeeb6e641a33dae4961196235bdb965b21b..0967ef424bce6791893e9a57bb952f80fd536e93 100644 --- a/config.json +++ b/config.json @@ -1 +1 @@ -{} \ No newline at end of file +{} diff --git a/cpp/ax620e/CMakeLists.txt b/cpp/ax620e/CMakeLists.txt new file mode 100644 index 0000000000000000000000000000000000000000..4c4ab983adb2e1939c8474d29fe46a53cbe93077 --- /dev/null +++ b/cpp/ax620e/CMakeLists.txt @@ -0,0 +1,37 @@ +cmake_minimum_required(VERSION 3.15) +project(rnnoise_ax620e_sdk LANGUAGES CXX C) + +set(CMAKE_CXX_STANDARD 14) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +include_directories(include src/rnnoise) + +# 原版 rnnoise 信号处理(denoise/pitch/FFT/LPC/表格)。 +# 注意:不编译 rnn.c(其 compute_rnn 由 src/rnnoise_ax.cpp 用 AX Engine 替换)。 +add_library(rnnoise_c STATIC + src/rnnoise/denoise.c + src/rnnoise/pitch.c + src/rnnoise/kiss_fft.c + src/rnnoise/celt_lpc.c + src/rnnoise/nnet.c + src/rnnoise/nnet_default.c + src/rnnoise/parse_lpcnet_weights.c + src/rnnoise/rnnoise_data.c + src/rnnoise/rnnoise_tables.c +) + +add_library(rnnoise_ax620e_sdk STATIC + src/model_runner.cpp + src/rnnoise_ax.cpp +) +target_include_directories(rnnoise_ax620e_sdk PUBLIC include) +target_link_libraries(rnnoise_ax620e_sdk PRIVATE rnnoise_c) + +add_executable(model_example examples/main.cpp) +target_link_libraries(model_example PRIVATE rnnoise_ax620e_sdk) + +if(AX_RUNTIME_ROOT) + target_include_directories(rnnoise_ax620e_sdk PUBLIC ${AX_RUNTIME_ROOT}/include) + target_link_directories(rnnoise_ax620e_sdk PUBLIC ${AX_RUNTIME_ROOT}/lib) + target_link_libraries(rnnoise_ax620e_sdk PUBLIC ax_engine ax_sys pthread dl atomic) +endif() diff --git a/cpp/ax620e/README.md b/cpp/ax620e/README.md new file mode 100644 index 0000000000000000000000000000000000000000..cef77a27e269ca3ae26f1135fe73eafd06bfdc5c --- /dev/null +++ b/cpp/ax620e/README.md @@ -0,0 +1,29 @@ +# rnnoise-ax620e C++ SDK + +48kHz 单声道实时降噪,RNNoise 原版 C 信号处理 + AX Engine(NPU3)网络推理。 + +## 编译(AX620E 板端/交叉环境) + +```bash +export AX_RUNTIME_ROOT=/path/to/axruntime # 含 include/ax_engine_api.h 与 lib/libax_engine.so +mkdir -p build && cd build +cmake .. -DAX_RUNTIME_ROOT=$AX_RUNTIME_ROOT +make -j$(nproc) +``` + +## 运行 + +```bash +./build/model_example model.axmodel in.pcm out.pcm +``` + +输入为 48kHz f32le PCM(16-bit 等价域 ±32768,不做归一化); +输出为同格式去噪 PCM。每帧 480 采样(10ms),vad 打印均值。 + +## 结构 + +- `include/rnnoise_ax.hpp`:`RNNoiseAX` 类(进程内单实例) +- `src/rnnoise_ax.cpp`:以 AX Engine 替换原版 `compute_rnn`(6 输入/7 输出逐帧状态化) +- `src/model_runner.cpp`:AX Engine 会话封装(按张量名映射,含缓存同步) +- `src/rnnoise/`:原版 rnnoise C 信号处理源码(denoise/pitch/FFT/LPC/表格,ISC 许可); + 网络权重已内嵌 AXMODEL,`rnnoise_data.c` 为最小 stub(空权重表 + 零初始化) diff --git a/cpp/ax620e/examples/main.cpp b/cpp/ax620e/examples/main.cpp new file mode 100644 index 0000000000000000000000000000000000000000..635827bf3b058dfd1727da51c3942434fedec925 --- /dev/null +++ b/cpp/ax620e/examples/main.cpp @@ -0,0 +1,58 @@ +// RNNoise AX620E 示例:处理 48k f32 PCM(16-bit 等价域),输出去噪 PCM。 +// 用法: ./model_example model.axmodel in.pcm out.pcm +#include "rnnoise_ax.hpp" + +#include +#include +#include +#include + +int main(int argc, char** argv) { + if (argc != 4) { + std::fprintf(stderr, "用法: %s model.axmodel in.pcm out.pcm\n", argv[0]); + return 1; + } + FILE* in = std::fopen(argv[2], "rb"); + if (!in) { + std::fprintf(stderr, "无法打开输入 %s\n", argv[2]); + return 1; + } + std::fseek(in, 0, SEEK_END); + long bytes = std::ftell(in); + std::fseek(in, 0, SEEK_SET); + std::vector pcm(bytes / sizeof(float)); + if (!pcm.empty()) { + std::fread(pcm.data(), sizeof(float), pcm.size(), in); + } + std::fclose(in); + + const int frame = RNNoiseAX::FrameSize(); + const int frames = static_cast(pcm.size() / frame); + if (frames == 0) { + std::fprintf(stderr, "输入过短\n"); + return 1; + } + + RNNoiseAX denoiser(argv[1]); + std::vector out(pcm.size()); + double vad_sum = 0.0; + auto t0 = std::chrono::steady_clock::now(); + for (int i = 0; i < frames; ++i) { + vad_sum += denoiser.ProcessFrame( + &out[i * frame], &pcm[i * frame]); + } + auto t1 = std::chrono::steady_clock::now(); + double secs = std::chrono::duration(t1 - t0).count(); + + FILE* of = std::fopen(argv[3], "wb"); + if (!of) { + std::fprintf(stderr, "无法写入输出 %s\n", argv[3]); + return 1; + } + std::fwrite(out.data(), sizeof(float), out.size(), of); + std::fclose(of); + + std::printf("frames=%d vad_mean=%.4f per_frame_ms=%.3f out=%s\n", + frames, vad_sum / frames, secs / frames * 1000.0, argv[3]); + return 0; +} diff --git a/cpp/include/model_runner.hpp b/cpp/ax620e/include/model_runner.hpp similarity index 100% rename from cpp/include/model_runner.hpp rename to cpp/ax620e/include/model_runner.hpp diff --git a/cpp/ax620e/include/rnnoise_ax.hpp b/cpp/ax620e/include/rnnoise_ax.hpp new file mode 100644 index 0000000000000000000000000000000000000000..b6741045ea273f34227084c58d21b8543931cfb0 --- /dev/null +++ b/cpp/ax620e/include/rnnoise_ax.hpp @@ -0,0 +1,26 @@ +#pragma once + +#include +#include + +// RNNoise AX620E 实时降噪器:原版 rnnoise C 信号处理 + AX Engine 网络推理。 +class RNNoiseAX { +public: + explicit RNNoiseAX(const std::string& model_path); + ~RNNoiseAX(); + + RNNoiseAX(const RNNoiseAX&) = delete; + RNNoiseAX& operator=(const RNNoiseAX&) = delete; + + static int FrameSize() { return 480; } + + void Reset(); + + // 处理一帧 48k PCM(16-bit 等价 float,±32768 域)。 + // in/out 各至少 FrameSize() 个 float;返回 vad(0~1)。 + float ProcessFrame(float* out, const float* in); + +private: + struct Impl; + std::unique_ptr impl_; +}; diff --git a/cpp/src/model_runner.cpp b/cpp/ax620e/src/model_runner.cpp similarity index 100% rename from cpp/src/model_runner.cpp rename to cpp/ax620e/src/model_runner.cpp diff --git a/cpp/src/rnnoise/_kiss_fft_guts.h b/cpp/ax620e/src/rnnoise/_kiss_fft_guts.h similarity index 100% rename from cpp/src/rnnoise/_kiss_fft_guts.h rename to cpp/ax620e/src/rnnoise/_kiss_fft_guts.h diff --git a/cpp/src/rnnoise/arch.h b/cpp/ax620e/src/rnnoise/arch.h similarity index 100% rename from cpp/src/rnnoise/arch.h rename to cpp/ax620e/src/rnnoise/arch.h diff --git a/cpp/src/rnnoise/celt_lpc.c b/cpp/ax620e/src/rnnoise/celt_lpc.c similarity index 100% rename from cpp/src/rnnoise/celt_lpc.c rename to cpp/ax620e/src/rnnoise/celt_lpc.c diff --git a/cpp/src/rnnoise/celt_lpc.h b/cpp/ax620e/src/rnnoise/celt_lpc.h similarity index 100% rename from cpp/src/rnnoise/celt_lpc.h rename to cpp/ax620e/src/rnnoise/celt_lpc.h diff --git a/cpp/src/rnnoise/common.h b/cpp/ax620e/src/rnnoise/common.h similarity index 100% rename from cpp/src/rnnoise/common.h rename to cpp/ax620e/src/rnnoise/common.h diff --git a/cpp/src/rnnoise/compile.sh b/cpp/ax620e/src/rnnoise/compile.sh similarity index 100% rename from cpp/src/rnnoise/compile.sh rename to cpp/ax620e/src/rnnoise/compile.sh diff --git a/cpp/src/rnnoise/cpu_support.h b/cpp/ax620e/src/rnnoise/cpu_support.h similarity index 100% rename from cpp/src/rnnoise/cpu_support.h rename to cpp/ax620e/src/rnnoise/cpu_support.h diff --git a/cpp/src/rnnoise/denoise.c b/cpp/ax620e/src/rnnoise/denoise.c similarity index 100% rename from cpp/src/rnnoise/denoise.c rename to cpp/ax620e/src/rnnoise/denoise.c diff --git a/cpp/src/rnnoise/denoise.h b/cpp/ax620e/src/rnnoise/denoise.h similarity index 100% rename from cpp/src/rnnoise/denoise.h rename to cpp/ax620e/src/rnnoise/denoise.h diff --git a/cpp/src/rnnoise/dump_features.c b/cpp/ax620e/src/rnnoise/dump_features.c similarity index 100% rename from cpp/src/rnnoise/dump_features.c rename to cpp/ax620e/src/rnnoise/dump_features.c diff --git a/cpp/src/rnnoise/dump_rnnoise_tables.c b/cpp/ax620e/src/rnnoise/dump_rnnoise_tables.c similarity index 100% rename from cpp/src/rnnoise/dump_rnnoise_tables.c rename to cpp/ax620e/src/rnnoise/dump_rnnoise_tables.c diff --git a/cpp/src/rnnoise/kiss_fft.c b/cpp/ax620e/src/rnnoise/kiss_fft.c similarity index 100% rename from cpp/src/rnnoise/kiss_fft.c rename to cpp/ax620e/src/rnnoise/kiss_fft.c diff --git a/cpp/src/rnnoise/kiss_fft.h b/cpp/ax620e/src/rnnoise/kiss_fft.h similarity index 100% rename from cpp/src/rnnoise/kiss_fft.h rename to cpp/ax620e/src/rnnoise/kiss_fft.h diff --git a/cpp/src/rnnoise/nnet.c b/cpp/ax620e/src/rnnoise/nnet.c similarity index 100% rename from cpp/src/rnnoise/nnet.c rename to cpp/ax620e/src/rnnoise/nnet.c diff --git a/cpp/src/rnnoise/nnet.h b/cpp/ax620e/src/rnnoise/nnet.h similarity index 100% rename from cpp/src/rnnoise/nnet.h rename to cpp/ax620e/src/rnnoise/nnet.h diff --git a/cpp/src/rnnoise/nnet_arch.h b/cpp/ax620e/src/rnnoise/nnet_arch.h similarity index 100% rename from cpp/src/rnnoise/nnet_arch.h rename to cpp/ax620e/src/rnnoise/nnet_arch.h diff --git a/cpp/src/rnnoise/nnet_default.c b/cpp/ax620e/src/rnnoise/nnet_default.c similarity index 100% rename from cpp/src/rnnoise/nnet_default.c rename to cpp/ax620e/src/rnnoise/nnet_default.c diff --git a/cpp/src/rnnoise/opus_types.h b/cpp/ax620e/src/rnnoise/opus_types.h similarity index 100% rename from cpp/src/rnnoise/opus_types.h rename to cpp/ax620e/src/rnnoise/opus_types.h diff --git a/cpp/src/rnnoise/parse_lpcnet_weights.c b/cpp/ax620e/src/rnnoise/parse_lpcnet_weights.c similarity index 100% rename from cpp/src/rnnoise/parse_lpcnet_weights.c rename to cpp/ax620e/src/rnnoise/parse_lpcnet_weights.c diff --git a/cpp/src/rnnoise/pitch.c b/cpp/ax620e/src/rnnoise/pitch.c similarity index 100% rename from cpp/src/rnnoise/pitch.c rename to cpp/ax620e/src/rnnoise/pitch.c diff --git a/cpp/src/rnnoise/pitch.h b/cpp/ax620e/src/rnnoise/pitch.h similarity index 100% rename from cpp/src/rnnoise/pitch.h rename to cpp/ax620e/src/rnnoise/pitch.h diff --git a/cpp/src/rnnoise/rnn.c b/cpp/ax620e/src/rnnoise/rnn.c similarity index 100% rename from cpp/src/rnnoise/rnn.c rename to cpp/ax620e/src/rnnoise/rnn.c diff --git a/cpp/src/rnnoise/rnn.h b/cpp/ax620e/src/rnnoise/rnn.h similarity index 100% rename from cpp/src/rnnoise/rnn.h rename to cpp/ax620e/src/rnnoise/rnn.h diff --git a/cpp/src/rnnoise/rnn_train.py b/cpp/ax620e/src/rnnoise/rnn_train.py similarity index 100% rename from cpp/src/rnnoise/rnn_train.py rename to cpp/ax620e/src/rnnoise/rnn_train.py diff --git a/cpp/src/rnnoise/rnnoise.h b/cpp/ax620e/src/rnnoise/rnnoise.h similarity index 100% rename from cpp/src/rnnoise/rnnoise.h rename to cpp/ax620e/src/rnnoise/rnnoise.h diff --git a/cpp/src/rnnoise/rnnoise_data.c b/cpp/ax620e/src/rnnoise/rnnoise_data.c similarity index 100% rename from cpp/src/rnnoise/rnnoise_data.c rename to cpp/ax620e/src/rnnoise/rnnoise_data.c diff --git a/cpp/src/rnnoise/rnnoise_data.h b/cpp/ax620e/src/rnnoise/rnnoise_data.h similarity index 100% rename from cpp/src/rnnoise/rnnoise_data.h rename to cpp/ax620e/src/rnnoise/rnnoise_data.h diff --git a/cpp/src/rnnoise/rnnoise_tables.c b/cpp/ax620e/src/rnnoise/rnnoise_tables.c similarity index 100% rename from cpp/src/rnnoise/rnnoise_tables.c rename to cpp/ax620e/src/rnnoise/rnnoise_tables.c diff --git a/cpp/src/rnnoise/vec.h b/cpp/ax620e/src/rnnoise/vec.h similarity index 100% rename from cpp/src/rnnoise/vec.h rename to cpp/ax620e/src/rnnoise/vec.h diff --git a/cpp/src/rnnoise/vec_avx.h b/cpp/ax620e/src/rnnoise/vec_avx.h similarity index 100% rename from cpp/src/rnnoise/vec_avx.h rename to cpp/ax620e/src/rnnoise/vec_avx.h diff --git a/cpp/src/rnnoise/vec_neon.h b/cpp/ax620e/src/rnnoise/vec_neon.h similarity index 100% rename from cpp/src/rnnoise/vec_neon.h rename to cpp/ax620e/src/rnnoise/vec_neon.h diff --git a/cpp/src/rnnoise/write_weights.c b/cpp/ax620e/src/rnnoise/write_weights.c similarity index 100% rename from cpp/src/rnnoise/write_weights.c rename to cpp/ax620e/src/rnnoise/write_weights.c diff --git a/cpp/src/rnnoise/x86/dnn_x86.h b/cpp/ax620e/src/rnnoise/x86/dnn_x86.h similarity index 100% rename from cpp/src/rnnoise/x86/dnn_x86.h rename to cpp/ax620e/src/rnnoise/x86/dnn_x86.h diff --git a/cpp/src/rnnoise/x86/nnet_avx2.c b/cpp/ax620e/src/rnnoise/x86/nnet_avx2.c similarity index 100% rename from cpp/src/rnnoise/x86/nnet_avx2.c rename to cpp/ax620e/src/rnnoise/x86/nnet_avx2.c diff --git a/cpp/src/rnnoise/x86/nnet_sse4_1.c b/cpp/ax620e/src/rnnoise/x86/nnet_sse4_1.c similarity index 100% rename from cpp/src/rnnoise/x86/nnet_sse4_1.c rename to cpp/ax620e/src/rnnoise/x86/nnet_sse4_1.c diff --git a/cpp/src/rnnoise/x86/x86_arch_macros.h b/cpp/ax620e/src/rnnoise/x86/x86_arch_macros.h similarity index 100% rename from cpp/src/rnnoise/x86/x86_arch_macros.h rename to cpp/ax620e/src/rnnoise/x86/x86_arch_macros.h diff --git a/cpp/src/rnnoise/x86/x86_dnn_map.c b/cpp/ax620e/src/rnnoise/x86/x86_dnn_map.c similarity index 100% rename from cpp/src/rnnoise/x86/x86_dnn_map.c rename to cpp/ax620e/src/rnnoise/x86/x86_dnn_map.c diff --git a/cpp/src/rnnoise/x86/x86cpu.c b/cpp/ax620e/src/rnnoise/x86/x86cpu.c similarity index 100% rename from cpp/src/rnnoise/x86/x86cpu.c rename to cpp/ax620e/src/rnnoise/x86/x86cpu.c diff --git a/cpp/src/rnnoise/x86/x86cpu.h b/cpp/ax620e/src/rnnoise/x86/x86cpu.h similarity index 100% rename from cpp/src/rnnoise/x86/x86cpu.h rename to cpp/ax620e/src/rnnoise/x86/x86cpu.h diff --git a/cpp/src/rnnoise_ax.cpp b/cpp/ax620e/src/rnnoise_ax.cpp similarity index 100% rename from cpp/src/rnnoise_ax.cpp rename to cpp/ax620e/src/rnnoise_ax.cpp diff --git a/cpp/CMakeLists.txt b/cpp/ax650/CMakeLists.txt similarity index 100% rename from cpp/CMakeLists.txt rename to cpp/ax650/CMakeLists.txt diff --git a/cpp/README.md b/cpp/ax650/README.md similarity index 100% rename from cpp/README.md rename to cpp/ax650/README.md diff --git a/cpp/examples/main.cpp b/cpp/ax650/examples/main.cpp similarity index 100% rename from cpp/examples/main.cpp rename to cpp/ax650/examples/main.cpp diff --git a/cpp/ax650/include/model_runner.hpp b/cpp/ax650/include/model_runner.hpp new file mode 100644 index 0000000000000000000000000000000000000000..cdf196ccb462156f322f6c51022b47edb422290a --- /dev/null +++ b/cpp/ax650/include/model_runner.hpp @@ -0,0 +1,24 @@ +#pragma once + +#include +#include + +// AX Engine 会话封装:加载 AXMODEL,按张量名映射输入输出。 +class ModelRunner { +public: + explicit ModelRunner(const std::string& model_path, + const std::string& model_name = "rnnoise"); + ~ModelRunner(); + + ModelRunner(const ModelRunner&) = delete; + ModelRunner& operator=(const ModelRunner&) = delete; + + // 6 输入(features/conv1_mem/conv2_mem/gru1_s/gru2_s/gru3_s) + // -> 7 输出(gains/vad/conv1_mem_new/conv2_mem_new/gru1_s_new/gru2_s_new/gru3_s_new) + std::vector> Run( + const std::vector>& inputs); + +private: + struct Impl; + Impl* impl_; +}; diff --git a/cpp/include/rnnoise_ax.hpp b/cpp/ax650/include/rnnoise_ax.hpp similarity index 100% rename from cpp/include/rnnoise_ax.hpp rename to cpp/ax650/include/rnnoise_ax.hpp diff --git a/cpp/ax650/src/model_runner.cpp b/cpp/ax650/src/model_runner.cpp new file mode 100644 index 0000000000000000000000000000000000000000..c540ef2d9a87506d4d305fd731024b90494000bf --- /dev/null +++ b/cpp/ax650/src/model_runner.cpp @@ -0,0 +1,202 @@ +#include "model_runner.hpp" + +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace { + +// 老版本 ax_sys_api.h 未声明这两个缓存同步接口,这里补充声明(板端 libax_sys 已导出)。 +extern "C" AX_S32 AX_SYS_MflushCache(AX_U64 phy_addr, AX_VOID* vir_addr, AX_U32 size); +extern "C" AX_S32 AX_SYS_MinvalidateCache(AX_U64 phy_addr, AX_VOID* vir_addr, AX_U32 size); + +std::vector read_binary(const std::string& path) { + std::ifstream file(path, std::ios::binary); + if (!file) { + throw std::runtime_error("failed to open " + path); + } + return std::vector( + std::istreambuf_iterator(file), + std::istreambuf_iterator()); +} + +void check_ax(int ret, const char* message) { + if (ret != 0) { + throw std::runtime_error(message); + } +} + +const char* kInputNames[6] = { + "features", "conv1_mem", "conv2_mem", + "gru1_s", "gru2_s", "gru3_s", +}; +const char* kOutputNames[7] = { + "gains", "vad", + "conv1_mem_new", "conv2_mem_new", + "gru1_s_new", "gru2_s_new", "gru3_s_new", +}; + +// 兼容不同 AX SDK 版本:新版 AX_ENGINE_RunSyncV2(handle, context, io), +// 旧版 AX_ENGINE_Run(context, io)。 +namespace axrun { +template +auto run(H h, C c, IO* io, int) + -> decltype(AX_ENGINE_RunSyncV2(h, c, io)) { + return AX_ENGINE_RunSyncV2(h, c, io); +} +template +auto run(H /*h*/, C c, IO* io, long) + -> decltype(AX_ENGINE_Run(c, io)) { + return AX_ENGINE_Run(c, io); +} +} // namespace axrun + +} // namespace + +struct ModelRunner::Impl { + AX_ENGINE_HANDLE handle = nullptr; + AX_ENGINE_CONTEXT_T context = nullptr; + AX_ENGINE_IO_INFO_T* info = nullptr; + AX_ENGINE_IO_T io {}; + std::vector buffers; + std::vector input_idx; + std::vector output_idx; + std::vector model; + + explicit Impl(const std::string& model_path, const std::string& model_name) + : model(read_binary(model_path)) { + check_ax(AX_SYS_Init(), "AX_SYS_Init failed"); + + AX_ENGINE_NPU_ATTR_T npu_attr; + std::memset(&npu_attr, 0, sizeof(npu_attr)); + npu_attr.eHardMode = static_cast(0); + check_ax(AX_ENGINE_Init(&npu_attr), "AX_ENGINE_Init failed"); + + AX_ENGINE_HANDLE_EXTRA_T extra; + std::memset(&extra, 0, sizeof(extra)); + extra.pName = const_cast( + reinterpret_cast(model_name.c_str())); + check_ax( + AX_ENGINE_CreateHandleV2( + &handle, model.data(), + static_cast(model.size()), &extra), + "AX_ENGINE_CreateHandleV2 failed"); + check_ax( + AX_ENGINE_CreateContextV2(handle, &context), + "AX_ENGINE_CreateContextV2 failed"); + check_ax(AX_ENGINE_GetIOInfo(handle, &info), "AX_ENGINE_GetIOInfo failed"); + if (!info || info->nInputSize < 6 || info->nOutputSize < 7) { + throw std::runtime_error("model IO mismatch (expect 6 in / 7 out)"); + } + + // 按张量名建立索引映射 + input_idx.resize(6, -1); + output_idx.resize(7, -1); + std::unordered_map in_map, out_map; + for (AX_U32 i = 0; i < info->nInputSize; ++i) { + const char* nm = info->pInputs[i].pName; + in_map[nm ? nm : ""] = static_cast(i); + } + for (AX_U32 i = 0; i < info->nOutputSize; ++i) { + const char* nm = info->pOutputs[i].pName; + out_map[nm ? nm : ""] = static_cast(i); + } + for (int i = 0; i < 6; ++i) { + auto it = in_map.find(kInputNames[i]); + if (it == in_map.end()) { + throw std::runtime_error(std::string("missing input ") + kInputNames[i]); + } + input_idx[i] = it->second; + } + for (int i = 0; i < 7; ++i) { + auto it = out_map.find(kOutputNames[i]); + if (it == out_map.end()) { + throw std::runtime_error(std::string("missing output ") + kOutputNames[i]); + } + output_idx[i] = it->second; + } + + buffers.resize(info->nInputSize + info->nOutputSize); + io.pInputs = buffers.data(); + io.nInputSize = info->nInputSize; + io.pOutputs = buffers.data() + info->nInputSize; + io.nOutputSize = info->nOutputSize; + for (AX_U32 i = 0; i < info->nInputSize; ++i) { + std::memset(&buffers[i], 0, sizeof(buffers[i])); + buffers[i].nSize = info->pInputs[i].nSize; + check_ax( + AX_SYS_MemAllocCached( + &buffers[i].phyAddr, &buffers[i].pVirAddr, + buffers[i].nSize, 128, + reinterpret_cast("model_input")), + "AX_SYS_MemAllocCached(input) failed"); + } + for (AX_U32 i = 0; i < info->nOutputSize; ++i) { + AX_ENGINE_IO_BUFFER_T& buf = buffers[info->nInputSize + i]; + std::memset(&buf, 0, sizeof(buf)); + buf.nSize = info->pOutputs[i].nSize; + check_ax( + AX_SYS_MemAllocCached( + &buf.phyAddr, &buf.pVirAddr, buf.nSize, 128, + reinterpret_cast("model_output")), + "AX_SYS_MemAllocCached(output) failed"); + } + } + + ~Impl() { + for (auto& item : buffers) { + if (item.phyAddr) { + AX_SYS_MemFree(item.phyAddr, item.pVirAddr); + } + } + if (handle) { + AX_ENGINE_DestroyHandle(handle); + } + AX_ENGINE_Deinit(); + AX_SYS_Deinit(); + } +}; + +ModelRunner::ModelRunner(const std::string& model_path, + const std::string& model_name) + : impl_(new Impl(model_path, model_name)) {} + +ModelRunner::~ModelRunner() { + delete impl_; +} + +std::vector> ModelRunner::Run( + const std::vector>& inputs) { + if (inputs.size() != 6) { + throw std::runtime_error("rnnoise expects 6 inputs"); + } + for (int i = 0; i < 6; ++i) { + const size_t bytes = inputs[i].size() * sizeof(float); + AX_ENGINE_IO_BUFFER_T& buf = impl_->buffers[impl_->input_idx[i]]; + if (bytes > buf.nSize) { + throw std::runtime_error("input larger than model tensor"); + } + std::memcpy(buf.pVirAddr, inputs[i].data(), bytes); + AX_SYS_MflushCache(buf.phyAddr, buf.pVirAddr, + static_cast(bytes)); + } + check_ax(axrun::run(impl_->handle, impl_->context, &impl_->io, 0), + "AX_ENGINE_Run failed"); + + std::vector> outputs(7); + for (int i = 0; i < 7; ++i) { + const AX_ENGINE_IO_BUFFER_T& buf = + impl_->buffers[impl_->info->nInputSize + impl_->output_idx[i]]; + AX_SYS_MinvalidateCache(buf.phyAddr, buf.pVirAddr, buf.nSize); + const size_t count = buf.nSize / sizeof(float); + const auto* src = static_cast(buf.pVirAddr); + outputs[i].assign(src, src + count); + } + return outputs; +} diff --git a/cpp/ax650/src/rnnoise/_kiss_fft_guts.h b/cpp/ax650/src/rnnoise/_kiss_fft_guts.h new file mode 100644 index 0000000000000000000000000000000000000000..17392b3e90977b74049049b7fa01a9f8cb0fc093 --- /dev/null +++ b/cpp/ax650/src/rnnoise/_kiss_fft_guts.h @@ -0,0 +1,182 @@ +/*Copyright (c) 2003-2004, Mark Borgerding + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE + LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR + CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF + SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS + INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN + CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) + ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE + POSSIBILITY OF SUCH DAMAGE.*/ + +#ifndef KISS_FFT_GUTS_H +#define KISS_FFT_GUTS_H + +#define MIN(a,b) ((a)<(b) ? (a):(b)) +#define MAX(a,b) ((a)>(b) ? (a):(b)) + +/* kiss_fft.h + defines kiss_fft_scalar as either short or a float type + and defines + typedef struct { kiss_fft_scalar r; kiss_fft_scalar i; }kiss_fft_cpx; */ +#include "kiss_fft.h" + +/* + Explanation of macros dealing with complex math: + + C_MUL(m,a,b) : m = a*b + C_FIXDIV( c , div ) : if a fixed point impl., c /= div. noop otherwise + C_SUB( res, a,b) : res = a - b + C_SUBFROM( res , a) : res -= a + C_ADDTO( res , a) : res += a + * */ +#ifdef FIXED_POINT +#include "arch.h" + + +#define SAMP_MAX 2147483647 +#define TWID_MAX 32767 +#define TRIG_UPSCALE 1 + +#define SAMP_MIN -SAMP_MAX + + +# define S_MUL(a,b) MULT16_32_Q15(b, a) + +# define C_MUL(m,a,b) \ + do{ (m).r = SUB32_ovflw(S_MUL((a).r,(b).r) , S_MUL((a).i,(b).i)); \ + (m).i = ADD32_ovflw(S_MUL((a).r,(b).i) , S_MUL((a).i,(b).r)); }while(0) + +# define C_MULC(m,a,b) \ + do{ (m).r = ADD32_ovflw(S_MUL((a).r,(b).r) , S_MUL((a).i,(b).i)); \ + (m).i = SUB32_ovflw(S_MUL((a).i,(b).r) , S_MUL((a).r,(b).i)); }while(0) + +# define C_MULBYSCALAR( c, s ) \ + do{ (c).r = S_MUL( (c).r , s ) ;\ + (c).i = S_MUL( (c).i , s ) ; }while(0) + +# define DIVSCALAR(x,k) \ + (x) = S_MUL( x, (TWID_MAX-((k)>>1))/(k)+1 ) + +# define C_FIXDIV(c,div) \ + do { DIVSCALAR( (c).r , div); \ + DIVSCALAR( (c).i , div); }while (0) + +#define C_ADD( res, a,b)\ + do {(res).r=ADD32_ovflw((a).r,(b).r); (res).i=ADD32_ovflw((a).i,(b).i); \ + }while(0) +#define C_SUB( res, a,b)\ + do {(res).r=SUB32_ovflw((a).r,(b).r); (res).i=SUB32_ovflw((a).i,(b).i); \ + }while(0) +#define C_ADDTO( res , a)\ + do {(res).r = ADD32_ovflw((res).r, (a).r); (res).i = ADD32_ovflw((res).i,(a).i);\ + }while(0) + +#define C_SUBFROM( res , a)\ + do {(res).r = ADD32_ovflw((res).r,(a).r); (res).i = SUB32_ovflw((res).i,(a).i); \ + }while(0) + +#if defined(OPUS_ARM_INLINE_ASM) +#include "arm/kiss_fft_armv4.h" +#endif + +#if defined(OPUS_ARM_INLINE_EDSP) +#include "arm/kiss_fft_armv5e.h" +#endif +#if defined(MIPSr1_ASM) +#include "mips/kiss_fft_mipsr1.h" +#endif + +#else /* not FIXED_POINT*/ + +# define S_MUL(a,b) ( (a)*(b) ) +#define C_MUL(m,a,b) \ + do{ (m).r = (a).r*(b).r - (a).i*(b).i;\ + (m).i = (a).r*(b).i + (a).i*(b).r; }while(0) +#define C_MULC(m,a,b) \ + do{ (m).r = (a).r*(b).r + (a).i*(b).i;\ + (m).i = (a).i*(b).r - (a).r*(b).i; }while(0) + +#define C_MUL4(m,a,b) C_MUL(m,a,b) + +# define C_FIXDIV(c,div) /* NOOP */ +# define C_MULBYSCALAR( c, s ) \ + do{ (c).r *= (s);\ + (c).i *= (s); }while(0) +#endif + +#ifndef CHECK_OVERFLOW_OP +# define CHECK_OVERFLOW_OP(a,op,b) /* noop */ +#endif + +#ifndef C_ADD +#define C_ADD( res, a,b)\ + do { \ + CHECK_OVERFLOW_OP((a).r,+,(b).r)\ + CHECK_OVERFLOW_OP((a).i,+,(b).i)\ + (res).r=(a).r+(b).r; (res).i=(a).i+(b).i; \ + }while(0) +#define C_SUB( res, a,b)\ + do { \ + CHECK_OVERFLOW_OP((a).r,-,(b).r)\ + CHECK_OVERFLOW_OP((a).i,-,(b).i)\ + (res).r=(a).r-(b).r; (res).i=(a).i-(b).i; \ + }while(0) +#define C_ADDTO( res , a)\ + do { \ + CHECK_OVERFLOW_OP((res).r,+,(a).r)\ + CHECK_OVERFLOW_OP((res).i,+,(a).i)\ + (res).r += (a).r; (res).i += (a).i;\ + }while(0) + +#define C_SUBFROM( res , a)\ + do {\ + CHECK_OVERFLOW_OP((res).r,-,(a).r)\ + CHECK_OVERFLOW_OP((res).i,-,(a).i)\ + (res).r -= (a).r; (res).i -= (a).i; \ + }while(0) +#endif /* C_ADD defined */ + +#ifdef FIXED_POINT +/*# define KISS_FFT_COS(phase) TRIG_UPSCALE*floor(MIN(32767,MAX(-32767,.5+32768 * cos (phase)))) +# define KISS_FFT_SIN(phase) TRIG_UPSCALE*floor(MIN(32767,MAX(-32767,.5+32768 * sin (phase))))*/ +# define KISS_FFT_COS(phase) floor(.5+TWID_MAX*cos (phase)) +# define KISS_FFT_SIN(phase) floor(.5+TWID_MAX*sin (phase)) +# define HALF_OF(x) ((x)>>1) +#elif defined(USE_SIMD) +# define KISS_FFT_COS(phase) _mm_set1_ps( cos(phase) ) +# define KISS_FFT_SIN(phase) _mm_set1_ps( sin(phase) ) +# define HALF_OF(x) ((x)*_mm_set1_ps(.5f)) +#else +# define KISS_FFT_COS(phase) (kiss_fft_scalar) cos(phase) +# define KISS_FFT_SIN(phase) (kiss_fft_scalar) sin(phase) +# define HALF_OF(x) ((x)*.5f) +#endif + +#define kf_cexp(x,phase) \ + do{ \ + (x)->r = KISS_FFT_COS(phase);\ + (x)->i = KISS_FFT_SIN(phase);\ + }while(0) + +#define kf_cexp2(x,phase) \ + do{ \ + (x)->r = TRIG_UPSCALE*celt_cos_norm((phase));\ + (x)->i = TRIG_UPSCALE*celt_cos_norm((phase)-32768);\ +}while(0) + +#endif /* KISS_FFT_GUTS_H */ diff --git a/cpp/ax650/src/rnnoise/arch.h b/cpp/ax650/src/rnnoise/arch.h new file mode 100644 index 0000000000000000000000000000000000000000..52de62334b71a7d07fac3daa4ae6c104d63d0718 --- /dev/null +++ b/cpp/ax650/src/rnnoise/arch.h @@ -0,0 +1,261 @@ +/* Copyright (c) 2003-2008 Jean-Marc Valin + Copyright (c) 2007-2008 CSIRO + Copyright (c) 2007-2009 Xiph.Org Foundation + Written by Jean-Marc Valin */ +/** + @file arch.h + @brief Various architecture definitions for CELT +*/ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef ARCH_H +#define ARCH_H + +#include "opus_types.h" +#include "common.h" + +# if !defined(__GNUC_PREREQ) +# if defined(__GNUC__)&&defined(__GNUC_MINOR__) +# define __GNUC_PREREQ(_maj,_min) \ + ((__GNUC__<<16)+__GNUC_MINOR__>=((_maj)<<16)+(_min)) +# else +# define __GNUC_PREREQ(_maj,_min) 0 +# endif +# endif + +#define CELT_SIG_SCALE 32768.f + +#define celt_fatal(str) _celt_fatal(str, __FILE__, __LINE__); +#ifdef ENABLE_ASSERTIONS +#include +#include +#ifdef __GNUC__ +__attribute__((noreturn)) +#endif +static OPUS_INLINE void _celt_fatal(const char *str, const char *file, int line) +{ + fprintf (stderr, "Fatal (internal) error in %s, line %d: %s\n", file, line, str); + abort(); +} +#define celt_assert(cond) {if (!(cond)) {celt_fatal("assertion failed: " #cond);}} +#define celt_assert2(cond, message) {if (!(cond)) {celt_fatal("assertion failed: " #cond "\n" message);}} +#else +#define celt_assert(cond) +#define celt_assert2(cond, message) +#endif + +#define IMUL32(a,b) ((a)*(b)) + +#define MIN16(a,b) ((a) < (b) ? (a) : (b)) /**< Minimum 16-bit value. */ +#define MAX16(a,b) ((a) > (b) ? (a) : (b)) /**< Maximum 16-bit value. */ +#define MIN32(a,b) ((a) < (b) ? (a) : (b)) /**< Minimum 32-bit value. */ +#define MAX32(a,b) ((a) > (b) ? (a) : (b)) /**< Maximum 32-bit value. */ +#define IMIN(a,b) ((a) < (b) ? (a) : (b)) /**< Minimum int value. */ +#define IMAX(a,b) ((a) > (b) ? (a) : (b)) /**< Maximum int value. */ +#define UADD32(a,b) ((a)+(b)) +#define USUB32(a,b) ((a)-(b)) + +/* Set this if opus_int64 is a native type of the CPU. */ +/* Assume that all LP64 architectures have fast 64-bit types; also x86_64 + (which can be ILP32 for x32) and Win64 (which is LLP64). */ +#if defined(__x86_64__) || defined(__LP64__) || defined(_WIN64) +#define OPUS_FAST_INT64 1 +#else +#define OPUS_FAST_INT64 0 +#endif + +#define PRINT_MIPS(file) + +#ifdef FIXED_POINT + +typedef opus_int16 opus_val16; +typedef opus_int32 opus_val32; +typedef opus_int64 opus_val64; + +typedef opus_val32 celt_sig; +typedef opus_val16 celt_norm; +typedef opus_val32 celt_ener; + +#define Q15ONE 32767 + +#define SIG_SHIFT 12 +/* Safe saturation value for 32-bit signals. Should be less than + 2^31*(1-0.85) to avoid blowing up on DC at deemphasis.*/ +#define SIG_SAT (300000000) + +#define NORM_SCALING 16384 + +#define DB_SHIFT 10 + +#define EPSILON 1 +#define VERY_SMALL 0 +#define VERY_LARGE16 ((opus_val16)32767) +#define Q15_ONE ((opus_val16)32767) + +#define SCALEIN(a) (a) +#define SCALEOUT(a) (a) + +#define ABS16(x) ((x) < 0 ? (-(x)) : (x)) +#define ABS32(x) ((x) < 0 ? (-(x)) : (x)) + +static OPUS_INLINE opus_int16 SAT16(opus_int32 x) { + return x > 32767 ? 32767 : x < -32768 ? -32768 : (opus_int16)x; +} + +#ifdef FIXED_DEBUG +#include "fixed_debug.h" +#else + +#include "fixed_generic.h" + +#ifdef OPUS_ARM_PRESUME_AARCH64_NEON_INTR +#include "arm/fixed_arm64.h" +#elif OPUS_ARM_INLINE_EDSP +#include "arm/fixed_armv5e.h" +#elif defined (OPUS_ARM_INLINE_ASM) +#include "arm/fixed_armv4.h" +#elif defined (BFIN_ASM) +#include "fixed_bfin.h" +#elif defined (TI_C5X_ASM) +#include "fixed_c5x.h" +#elif defined (TI_C6X_ASM) +#include "fixed_c6x.h" +#endif + +#endif + +#else /* FIXED_POINT */ + +typedef float opus_val16; +typedef float opus_val32; +typedef float opus_val64; + +typedef float celt_sig; +typedef float celt_norm; +typedef float celt_ener; + +#ifdef FLOAT_APPROX +/* This code should reliably detect NaN/inf even when -ffast-math is used. + Assumes IEEE 754 format. */ +static OPUS_INLINE int celt_isnan(float x) +{ + union {float f; opus_uint32 i;} in; + in.f = x; + return ((in.i>>23)&0xFF)==0xFF && (in.i&0x007FFFFF)!=0; +} +#else +#ifdef __FAST_MATH__ +#error Cannot build libopus with -ffast-math unless FLOAT_APPROX is defined. This could result in crashes on extreme (e.g. NaN) input +#endif +#define celt_isnan(x) ((x)!=(x)) +#endif + +#define Q15ONE 1.0f + +#define NORM_SCALING 1.f + +#define EPSILON 1e-15f +#define VERY_SMALL 1e-30f +#define VERY_LARGE16 1e15f +#define Q15_ONE ((opus_val16)1.f) + +/* This appears to be the same speed as C99's fabsf() but it's more portable. */ +#define ABS16(x) ((float)fabs(x)) +#define ABS32(x) ((float)fabs(x)) + +#define QCONST16(x,bits) (x) +#define QCONST32(x,bits) (x) + +#define NEG16(x) (-(x)) +#define NEG32(x) (-(x)) +#define NEG32_ovflw(x) (-(x)) +#define EXTRACT16(x) (x) +#define EXTEND32(x) (x) +#define SHR16(a,shift) (a) +#define SHL16(a,shift) (a) +#define SHR32(a,shift) (a) +#define SHL32(a,shift) (a) +#define PSHR32(a,shift) (a) +#define VSHR32(a,shift) (a) + +#define PSHR(a,shift) (a) +#define SHR(a,shift) (a) +#define SHL(a,shift) (a) +#define SATURATE(x,a) (x) +#define SATURATE16(x) (x) + +#define ROUND16(a,shift) (a) +#define SROUND16(a,shift) (a) +#define HALF16(x) (.5f*(x)) +#define HALF32(x) (.5f*(x)) + +#define ADD16(a,b) ((a)+(b)) +#define SUB16(a,b) ((a)-(b)) +#define ADD32(a,b) ((a)+(b)) +#define SUB32(a,b) ((a)-(b)) +#define ADD32_ovflw(a,b) ((a)+(b)) +#define SUB32_ovflw(a,b) ((a)-(b)) +#define MULT16_16_16(a,b) ((a)*(b)) +#define MULT16_16(a,b) ((opus_val32)(a)*(opus_val32)(b)) +#define MAC16_16(c,a,b) ((c)+(opus_val32)(a)*(opus_val32)(b)) + +#define MULT16_32_Q15(a,b) ((a)*(b)) +#define MULT16_32_Q16(a,b) ((a)*(b)) + +#define MULT32_32_Q31(a,b) ((a)*(b)) + +#define MAC16_32_Q15(c,a,b) ((c)+(a)*(b)) +#define MAC16_32_Q16(c,a,b) ((c)+(a)*(b)) + +#define MULT16_16_Q11_32(a,b) ((a)*(b)) +#define MULT16_16_Q11(a,b) ((a)*(b)) +#define MULT16_16_Q13(a,b) ((a)*(b)) +#define MULT16_16_Q14(a,b) ((a)*(b)) +#define MULT16_16_Q15(a,b) ((a)*(b)) +#define MULT16_16_P15(a,b) ((a)*(b)) +#define MULT16_16_P13(a,b) ((a)*(b)) +#define MULT16_16_P14(a,b) ((a)*(b)) +#define MULT16_32_P16(a,b) ((a)*(b)) + +#define DIV32_16(a,b) (((opus_val32)(a))/(opus_val16)(b)) +#define DIV32(a,b) (((opus_val32)(a))/(opus_val32)(b)) + +#define SCALEIN(a) ((a)*CELT_SIG_SCALE) +#define SCALEOUT(a) ((a)*(1/CELT_SIG_SCALE)) + +#define SIG2WORD16(x) (x) + +#endif /* !FIXED_POINT */ + +#ifndef GLOBAL_STACK_SIZE +#ifdef FIXED_POINT +#define GLOBAL_STACK_SIZE 120000 +#else +#define GLOBAL_STACK_SIZE 120000 +#endif +#endif + +#endif /* ARCH_H */ diff --git a/cpp/ax650/src/rnnoise/celt_lpc.c b/cpp/ax650/src/rnnoise/celt_lpc.c new file mode 100644 index 0000000000000000000000000000000000000000..750f9a8ae1a0fc803df2553926697ad9bf831faf --- /dev/null +++ b/cpp/ax650/src/rnnoise/celt_lpc.c @@ -0,0 +1,174 @@ +/* Copyright (c) 2009-2010 Xiph.Org Foundation + Written by Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include "celt_lpc.h" +#include "arch.h" +#include "common.h" +#include "pitch.h" +#include "denoise.h" + +void rnn_lpc( + opus_val16 *_lpc, /* out: [0...p-1] LPC coefficients */ +const opus_val32 *ac, /* in: [0...p] autocorrelation values */ +int p +) +{ + int i, j; + opus_val32 r; + opus_val32 error = ac[0]; +#ifdef FIXED_POINT + opus_val32 lpc[LPC_ORDER]; +#else + float *lpc = _lpc; +#endif + + RNN_CLEAR(lpc, p); + if (ac[0] != 0) + { + for (i = 0; i < p; i++) { + /* Sum up this iteration's reflection coefficient */ + opus_val32 rr = 0; + for (j = 0; j < i; j++) + rr += MULT32_32_Q31(lpc[j],ac[i - j]); + rr += SHR32(ac[i + 1],3); + r = -SHL32(rr,3)/error; + /* Update LPC coefficients and total error */ + lpc[i] = SHR32(r,3); + for (j = 0; j < (i+1)>>1; j++) + { + opus_val32 tmp1, tmp2; + tmp1 = lpc[j]; + tmp2 = lpc[i-1-j]; + lpc[j] = tmp1 + MULT32_32_Q31(r,tmp2); + lpc[i-1-j] = tmp2 + MULT32_32_Q31(r,tmp1); + } + + error = error - MULT32_32_Q31(MULT32_32_Q31(r,r),error); + /* Bail out once we get 30 dB gain */ +#ifdef FIXED_POINT + if (error0); + celt_assert(n<=PITCH_BUF_SIZE/2) + celt_assert(overlap>=0); + if (overlap == 0) + { + xptr = x; + } else { + for (i=0;i0) + { + for(i=0;i= 536870912) + { + int shift2=1; + if (ac[0] >= 1073741824) + shift2++; + for (i=0;i<=lag;i++) + ac[i] = SHR32(ac[i], shift2); + shift += shift2; + } +#endif + + return shift; +} diff --git a/cpp/ax650/src/rnnoise/celt_lpc.h b/cpp/ax650/src/rnnoise/celt_lpc.h new file mode 100644 index 0000000000000000000000000000000000000000..5abd61a0f84533478a27ea08017497361bbd529b --- /dev/null +++ b/cpp/ax650/src/rnnoise/celt_lpc.h @@ -0,0 +1,45 @@ +/* Copyright (c) 2009-2010 Xiph.Org Foundation + Written by Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef PLC_H +#define PLC_H + +#include "arch.h" +#include "common.h" + +#if defined(OPUS_X86_MAY_HAVE_SSE4_1) +#include "x86/celt_lpc_sse.h" +#endif + +#define LPC_ORDER 24 + +void rnn_lpc(opus_val16 *_lpc, const opus_val32 *ac, int p); + +int rnn_autocorr(const opus_val16 *x, opus_val32 *ac, + const opus_val16 *window, int overlap, int lag, int n); + +#endif /* PLC_H */ diff --git a/cpp/ax650/src/rnnoise/common.h b/cpp/ax650/src/rnnoise/common.h new file mode 100644 index 0000000000000000000000000000000000000000..f9095ca5e85ed8686f7af5c3e63dd963a726f52b --- /dev/null +++ b/cpp/ax650/src/rnnoise/common.h @@ -0,0 +1,56 @@ + + +#ifndef COMMON_H +#define COMMON_H + +#include "stdlib.h" +#include "string.h" + +#define RNN_INLINE inline +#define OPUS_INLINE inline + + +/** RNNoise wrapper for malloc(). To do your own dynamic allocation, all you need t +o do is replace this function and rnnoise_free */ +#ifndef OVERRIDE_RNNOISE_ALLOC +static RNN_INLINE void *rnnoise_alloc (size_t size) +{ + return malloc(size); +} +#endif + +/** RNNoise wrapper for free(). To do your own dynamic allocation, all you need to do is replace this function and rnnoise_alloc */ +#ifndef OVERRIDE_RNNOISE_FREE +static RNN_INLINE void rnnoise_free (void *ptr) +{ + free(ptr); +} +#endif + +/** Copy n elements from src to dst. The 0* term provides compile-time type checking */ +#ifndef OVERRIDE_RNN_COPY +#define RNN_COPY(dst, src, n) (memcpy((dst), (src), (n)*sizeof(*(dst)) + 0*((dst)-(src)) )) +#endif + +/** Copy n elements from src to dst, allowing overlapping regions. The 0* term + provides compile-time type checking */ +#ifndef OVERRIDE_RNN_MOVE +#define RNN_MOVE(dst, src, n) (memmove((dst), (src), (n)*sizeof(*(dst)) + 0*((dst)-(src)) )) +#endif + +/** Set n elements of dst to zero */ +#ifndef OVERRIDE_RNN_CLEAR +#define RNN_CLEAR(dst, n) (memset((dst), 0, (n)*sizeof(*(dst)))) +#endif + +# if !defined(OPUS_GNUC_PREREQ) +# if defined(__GNUC__)&&defined(__GNUC_MINOR__) +# define OPUS_GNUC_PREREQ(_maj,_min) \ + ((__GNUC__<<16)+__GNUC_MINOR__>=((_maj)<<16)+(_min)) +# else +# define OPUS_GNUC_PREREQ(_maj,_min) 0 +# endif +# endif + + +#endif diff --git a/cpp/ax650/src/rnnoise/compile.sh b/cpp/ax650/src/rnnoise/compile.sh new file mode 100644 index 0000000000000000000000000000000000000000..4b2ea5387e90144450fe13954c5d6ee21aab614f --- /dev/null +++ b/cpp/ax650/src/rnnoise/compile.sh @@ -0,0 +1,3 @@ +#!/bin/sh + +gcc -DTRAINING=1 -Wall -W -O3 -g -I../include denoise.c kiss_fft.c pitch.c celt_lpc.c rnn.c rnn_data.c -o denoise_training -lm diff --git a/cpp/ax650/src/rnnoise/cpu_support.h b/cpp/ax650/src/rnnoise/cpu_support.h new file mode 100644 index 0000000000000000000000000000000000000000..9de21f3a4b963b64c0d0fcccf4c0e397e01cde31 --- /dev/null +++ b/cpp/ax650/src/rnnoise/cpu_support.h @@ -0,0 +1,53 @@ +/* Copyright (c) 2010 Xiph.Org Foundation + * Copyright (c) 2013 Parrot */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef CPU_SUPPORT_H +#define CPU_SUPPORT_H + +#include "opus_types.h" +#include "common.h" + +#ifdef RNN_ENABLE_X86_RTCD + +#include "x86/x86cpu.h" +/* We currently support 5 x86 variants: + * arch[0] -> sse2 + * arch[1] -> sse4.1 + * arch[2] -> avx2 + */ +#define OPUS_ARCHMASK 3 +int rnn_select_arch(void); + +#else +#define OPUS_ARCHMASK 0 + +static OPUS_INLINE int rnn_select_arch(void) +{ + return 0; +} +#endif +#endif diff --git a/cpp/ax650/src/rnnoise/denoise.c b/cpp/ax650/src/rnnoise/denoise.c new file mode 100644 index 0000000000000000000000000000000000000000..b6fc3d4a90e13184e5b94b892796a8288d8260db --- /dev/null +++ b/cpp/ax650/src/rnnoise/denoise.c @@ -0,0 +1,505 @@ +/* Copyright (c) 2024 Jean-Marc Valin + * Copyright (c) 2018 Gregor Richards + * Copyright (c) 2017 Mozilla */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#include +#include +#include "kiss_fft.h" +#include "common.h" +#include "denoise.h" +#include +#include "rnnoise.h" +#include "pitch.h" +#include "arch.h" +#include "rnn.h" +#include "cpu_support.h" + +#define SQUARE(x) ((x)*(x)) + + +#ifndef TRAINING +#define TRAINING 0 +#endif + + +/* ERB bandwidths going in reverse from 20 kHz and then replacing the 700 and 800 + with just 750 because having 32 bands is convenient for the DNN. + B(1)=400; + for k=2:35 + B(k) = B(k-1) - max(2, round(24.7*(4.37*B(k-1)/20+1)/50)); + end + printf("%d, ", B(end:-1:1)); + printf("\n") +*/ +const int eband20ms[NB_BANDS+2] = { +/*0 100 200 300 400 500 600 750 900 1.1 1.2 1.4 1.6 1.8 2.1 2.4 2.7 3.0 3.4 3.9 4.4 4.9 5.5 6.2 7.0 7.9 8.8 9.9 11.2 12.6 14.1 15.9 17.8 20.0*/ + 0, 2, 4, 6, 8, 10, 12, 15, 18, 21, 24, 28, 32, 36, 41, 47, 53, 60, 68, 77, 87, 98, 110, 124, 140, 157, 176, 198, 223, 251, 282, 317, 356, 400}; + + +struct DenoiseState { + RNNoise model; +#if !TRAINING + int arch; +#endif + float analysis_mem[FRAME_SIZE]; + int memid; + float synthesis_mem[FRAME_SIZE]; + float pitch_buf[PITCH_BUF_SIZE]; + float pitch_enh_buf[PITCH_BUF_SIZE]; + float last_gain; + int last_period; + float mem_hp_x[2]; + float lastg[NB_BANDS]; + RNNState rnn; + kiss_fft_cpx delayed_X[FREQ_SIZE]; + kiss_fft_cpx delayed_P[FREQ_SIZE]; + float delayed_Ex[NB_BANDS], delayed_Ep[NB_BANDS]; + float delayed_Exp[NB_BANDS]; + +}; + +static void compute_band_energy(float *bandE, const kiss_fft_cpx *X) { + int i; + float sum[NB_BANDS+2] = {0}; + for (i=0;iblob = NULL; + model->const_blob = ptr; + model->blob_len = len; + return model; +} + +RNNModel *rnnoise_model_from_filename(const char *filename) { + RNNModel *model; + FILE *f = fopen(filename, "rb"); + model = rnnoise_model_from_file(f); + model->file = f; + return model; +} + +RNNModel *rnnoise_model_from_file(FILE *f) { + RNNModel *model; + model = malloc(sizeof(*model)); + model->file = NULL; + + fseek(f, 0, SEEK_END); + model->blob_len = ftell(f); + fseek(f, 0, SEEK_SET); + + model->const_blob = NULL; + model->blob = malloc(model->blob_len); + if (fread(model->blob, model->blob_len, 1, f) != 1) + { + rnnoise_model_free(model); + return NULL; + } + return model; +} + +void rnnoise_model_free(RNNModel *model) { + if (model->file != NULL) fclose(model->file); + if (model->blob != NULL) free(model->blob); + free(model); +} + +int rnnoise_get_size(void) { + return sizeof(DenoiseState); +} + +int rnnoise_get_frame_size(void) { + return FRAME_SIZE; +} + +int rnnoise_init(DenoiseState *st, RNNModel *model) { + memset(st, 0, sizeof(*st)); +#if !TRAINING + if (model != NULL) { + WeightArray *list; + int ret = 1; + parse_weights(&list, model->blob ? model->blob : model->const_blob, model->blob_len); + if (list != NULL) { + ret = init_rnnoise(&st->model, list); + opus_free(list); + } + if (ret != 0) return -1; + } +#ifndef USE_WEIGHTS_FILE + else { + int ret = init_rnnoise(&st->model, rnnoise_arrays); + if (ret != 0) return -1; + } +#endif + st->arch = rnn_select_arch(); +#else + (void)model; +#endif + return 0; +} + +DenoiseState *rnnoise_create(RNNModel *model) { + int ret; + DenoiseState *st; + st = malloc(rnnoise_get_size()); + ret = rnnoise_init(st, model); + if (ret != 0) { + free(st); + return NULL; + } + return st; +} + +void rnnoise_destroy(DenoiseState *st) { + free(st); +} + +#if TRAINING +extern int lowpass; +extern int band_lp; +#endif + +void rnn_frame_analysis(DenoiseState *st, kiss_fft_cpx *X, float *Ex, const float *in) { + int i; + float x[WINDOW_SIZE]; + RNN_COPY(x, st->analysis_mem, FRAME_SIZE); + for (i=0;ianalysis_mem, in, FRAME_SIZE); + apply_window(x); + forward_transform(X, x); +#if TRAINING + for (i=lowpass;i>1]; + int pitch_index; + float gain; + float *(pre[1]); + float follow, logMax; + rnn_frame_analysis(st, X, Ex, in); + RNN_MOVE(st->pitch_buf, &st->pitch_buf[FRAME_SIZE], PITCH_BUF_SIZE-FRAME_SIZE); + RNN_COPY(&st->pitch_buf[PITCH_BUF_SIZE-FRAME_SIZE], in, FRAME_SIZE); + pre[0] = &st->pitch_buf[0]; + rnn_pitch_downsample(pre, pitch_buf, PITCH_BUF_SIZE, 1); + rnn_pitch_search(pitch_buf+(PITCH_MAX_PERIOD>>1), pitch_buf, PITCH_FRAME_SIZE, + PITCH_MAX_PERIOD-3*PITCH_MIN_PERIOD, &pitch_index); + pitch_index = PITCH_MAX_PERIOD-pitch_index; + + gain = rnn_remove_doubling(pitch_buf, PITCH_MAX_PERIOD, PITCH_MIN_PERIOD, + PITCH_FRAME_SIZE, &pitch_index, st->last_period, st->last_gain); + st->last_period = pitch_index; + st->last_gain = gain; + for (i=0;ipitch_buf[PITCH_BUF_SIZE-WINDOW_SIZE-pitch_index+i]; + apply_window(p); + forward_transform(P, p); + compute_band_energy(Ep, P); + compute_band_corr(Exp, X, P); + for (i=0;isynthesis_mem[i]; + RNN_COPY(st->synthesis_mem, &x[FRAME_SIZE], FRAME_SIZE); +} + +void rnn_biquad(float *y, float mem[2], const float *x, const float *b, const float *a, int N) { + int i; + for (i=0;ig[i]) r[i] = 1; + else r[i] = Exp[i]*(1-g[i])/(.001 + g[i]*(1-Exp[i])); + r[i] = MIN16(1, MAX16(0, r[i])); +#else + if (Exp[i]>g[i]) r[i] = 1; + else r[i] = SQUARE(Exp[i])*(1-SQUARE(g[i]))/(.001 + SQUARE(g[i])*(1-SQUARE(Exp[i]))); + r[i] = sqrt(MIN16(1, MAX16(0, r[i]))); +#endif + r[i] *= sqrt(Ex[i]/(1e-8+Ep[i])); + } + interp_band_gain(rf, r); + for (i=0;imem_hp_x, in, b_hp, a_hp, FRAME_SIZE); + silence = rnn_compute_frame_features(st, X, P, Ex, Ep, Exp, features, x); + + if (!silence) { +#if !TRAINING + compute_rnn(&st->model, &st->rnn, g, &vad_prob, features, st->arch); +#endif + rnn_pitch_filter(st->delayed_X, st->delayed_P, st->delayed_Ex, st->delayed_Ep, st->delayed_Exp, g); + for (i=0;ilastg[i]); + /* Compensate for energy change across frame when computing the threshold gain. + Avoids leaking noise when energy increases (e.g. transient noise). */ + st->lastg[i] = MIN16(1.f, g[i]*(st->delayed_Ex[i]+1e-3)/(Ex[i]+1e-3)); + } + interp_band_gain(gf, g); +#if 1 + for (i=0;idelayed_X[i].r *= gf[i]; + st->delayed_X[i].i *= gf[i]; + } +#endif + } + frame_synthesis(st, out, st->delayed_X); + + RNN_COPY(st->delayed_X, X, FREQ_SIZE); + RNN_COPY(st->delayed_P, P, FREQ_SIZE); + RNN_COPY(st->delayed_Ex, Ex, NB_BANDS); + RNN_COPY(st->delayed_Ep, Ep, NB_BANDS); + RNN_COPY(st->delayed_Exp, Exp, NB_BANDS); + return vad_prob; +} + diff --git a/cpp/ax650/src/rnnoise/denoise.h b/cpp/ax650/src/rnnoise/denoise.h new file mode 100644 index 0000000000000000000000000000000000000000..ccfecd9a7e2d6e0b337a93c29a9d1a8acb2b171a --- /dev/null +++ b/cpp/ax650/src/rnnoise/denoise.h @@ -0,0 +1,56 @@ +/* Copyright (c) 2017 Mozilla */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#include "rnnoise.h" +#include "kiss_fft.h" +#include "nnet.h" + +#define FRAME_SIZE 480 +#define WINDOW_SIZE (2*FRAME_SIZE) +#define FREQ_SIZE (FRAME_SIZE + 1) +#define NB_BANDS 32 +#define NB_FEATURES (2*NB_BANDS+1) + + +#define PITCH_MIN_PERIOD 60 +#define PITCH_MAX_PERIOD 768 +#define PITCH_FRAME_SIZE 960 +#define PITCH_BUF_SIZE (PITCH_MAX_PERIOD+PITCH_FRAME_SIZE) + +extern const WeightArray rnnoise_arrays[]; + +extern const int eband20ms[]; + + +void rnn_biquad(float *y, float mem[2], const float *x, const float *b, const float *a, int N); + +void rnn_pitch_filter(kiss_fft_cpx *X, const kiss_fft_cpx *P, const float *Ex, const float *Ep, + const float *Exp, const float *g); + +void rnn_frame_analysis(DenoiseState *st, kiss_fft_cpx *X, float *Ex, const float *in); + +int rnn_compute_frame_features(DenoiseState *st, kiss_fft_cpx *X, kiss_fft_cpx *P, + float *Ex, float *Ep, float *Exp, float *features, const float *in); diff --git a/cpp/ax650/src/rnnoise/dump_features.c b/cpp/ax650/src/rnnoise/dump_features.c new file mode 100644 index 0000000000000000000000000000000000000000..dc2d8c3914d8a6939be90c7289249f2337656e02 --- /dev/null +++ b/cpp/ax650/src/rnnoise/dump_features.c @@ -0,0 +1,499 @@ +/* Copyright (c) 2024 Jean-Marc Valin + * Copyright (c) 2017 Mozilla */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + + +#include +#include +#include +#include +#include +#include "rnnoise.h" +#include "common.h" +#include "denoise.h" +#include "arch.h" +#include "kiss_fft.h" +#include "src/_kiss_fft_guts.h" + +int lowpass = FREQ_SIZE; +int band_lp = NB_BANDS; + +#define SEQUENCE_LENGTH 2000 +#define SEQUENCE_SAMPLES (SEQUENCE_LENGTH*FRAME_SIZE) + +#define RIR_FFT_SIZE 65536 +#define RIR_MAX_DURATION (RIR_FFT_SIZE/2) +#define FILENAME_MAX_SIZE 1000 + +struct rir_list { + int nb_rirs; + int block_size; + kiss_fft_state *fft; + kiss_fft_cpx **rir; + kiss_fft_cpx **early; +}; + +kiss_fft_cpx *load_rir(const char *rir_file, kiss_fft_state *fft, int early) { + kiss_fft_cpx *x, *X; + float rir[RIR_MAX_DURATION]; + int len; + int i; + FILE *f; + f = fopen(rir_file, "rb"); + if (f==NULL) { + fprintf(stderr, "cannot open %s: %s\n", rir_file, strerror(errno)); + exit(1); + } + x = (kiss_fft_cpx*)calloc(fft->nfft, sizeof(*x)); + X = (kiss_fft_cpx*)calloc(fft->nfft, sizeof(*X)); + len = fread(rir, sizeof(*rir), RIR_MAX_DURATION, f); + if (early) { + for (i=0;i<240;i++) { + rir[480+i] *= (1 - i/240.f); + } + RNN_CLEAR(&rir[240+480], RIR_MAX_DURATION-240-480); + } + for (i=0;inb_rirs = 0; + allocated = 2; + rirs->fft = rnn_fft_alloc_twiddles(RIR_FFT_SIZE, NULL, NULL, NULL, 0); + rirs->rir = malloc(allocated*sizeof(rirs->rir[0])); + rirs->early = malloc(allocated*sizeof(rirs->early[0])); + while (fgets(rir_filename, FILENAME_MAX_SIZE, f) != NULL) { + /* Chop trailing newline. */ + rir_filename[strcspn(rir_filename, "\n")] = 0; + if (rirs->nb_rirs+1 > allocated) { + allocated *= 2; + rirs->rir = realloc(rirs->rir, allocated*sizeof(rirs->rir[0])); + rirs->early = realloc(rirs->early, allocated*sizeof(rirs->early[0])); + } + rirs->rir[rirs->nb_rirs] = load_rir(rir_filename, rirs->fft, 0); + rirs->early[rirs->nb_rirs] = load_rir(rir_filename, rirs->fft, 1); + rirs->nb_rirs++; + } + fclose(f); +} + +void rir_filter_sequence(const struct rir_list *rirs, float *audio, int rir_id, int early) { + int i; + kiss_fft_cpx x[RIR_FFT_SIZE] = {{0,0}}; + kiss_fft_cpx y[RIR_FFT_SIZE] = {{0,0}}; + kiss_fft_cpx X[RIR_FFT_SIZE] = {{0,0}}; + const kiss_fft_cpx *Y; + if (early) Y = rirs->early[rir_id]; + else Y = rirs->rir[rir_id]; + i=0; + while (ifft, x, X); + for (j=0;jfft, X, y); + for (j=0;j0) { + float r, theta; + r = rand()/(double)RAND_MAX; + r = .7*r*r; + theta = rand()/(double)RAND_MAX; + theta = M_PI*theta*theta; + a[0] = -2*r*cos(theta); + a[1] = r*r; + } else { + float r0,r1; + r0 = 1.4*uni_rand(); + r1 = 1.4*uni_rand(); + a[0] = -r0-r1; + a[1] = r0*r1; + } +} + +static void rand_resp(float *a, float *b) { + rand_filt(a); + rand_filt(b); +} + +short speech16[SEQUENCE_LENGTH*FRAME_SIZE]; +short noise16[SEQUENCE_LENGTH*FRAME_SIZE]; +short fgnoise16[SEQUENCE_LENGTH*FRAME_SIZE]; +float x[SEQUENCE_LENGTH*FRAME_SIZE]; +float n[SEQUENCE_LENGTH*FRAME_SIZE]; +float fn[SEQUENCE_LENGTH*FRAME_SIZE]; +float xn[SEQUENCE_LENGTH*FRAME_SIZE]; + +#define P00 0.99f +#define P01 0.01f +#define P10 0.01f +#define P11 0.99f +#define LOGIT_SCALE 0.5f + +static void viterbi_vad(const float *E, int *vad) { + int i; + float Enoise, Esig; + int back[SEQUENCE_LENGTH][2]; + float curr; + Enoise = Esig = 1e-30; + for (i=0;i (1-curr)*P01) { + back[i][1] = 1; + prior = curr*P11; + } else { + back[i][1] = 0; + prior = (1-curr)*P01; + } + pspeech = prior*p0; + + if ((1-curr)*P00 > curr*P10) { + back[i][0] = 0; + prior = (1-curr)*P00; + } else { + back[i][0] = 1; + prior = curr*P10; + } + pnoise = prior*(1-p0); + curr = pspeech / (pspeech + pnoise); + /*printf("%f ", curr);*/ + } + vad[SEQUENCE_LENGTH-1] = curr > .5; + for (i=SEQUENCE_LENGTH-2;i>=0;i--) { + if (vad[i+1]) { + vad[i] = back[i+1][1]; + } else { + vad[i] = back[i+1][0]; + } + } + for (i=0;i=1;i--) { + if (vad[i-1]) vad[i] = 1; + } +} + +static void clear_vad(float *x, int *vad) { + int i; + int active = vad[0]; + for (i=0;i=1 && vad[i]==0 && vad[i-1]==0) { + int j; + for (j=0;j6) { + if (strcmp(argv[1], "-rir_list")==0) { + rir_filename = argv[2]; + argv+=2; + argc-=2; + } + } + if (argc!=6) { + fprintf(stderr, "usage: %s [-rir_list list] \n", argv0); + return 1; + } + f1 = fopen(argv[1], "rb"); + f2 = fopen(argv[2], "rb"); + f3 = fopen(argv[3], "rb"); + fout = fopen(argv[4], "wb"); + + fseek(f1, 0, SEEK_END); + speech_length = ftell(f1); + fseek(f1, 0, SEEK_SET); + + fseek(f2, 0, SEEK_END); + noise_length = ftell(f2); + fseek(f2, 0, SEEK_SET); + + fseek(f3, 0, SEEK_END); + fgnoise_length = ftell(f3); + fseek(f3, 0, SEEK_SET); + + maxCount = atoi(argv[5]); + if (rir_filename) load_rir_list(rir_filename, &rirs); + for (count=0;count speech_length-(long)sizeof(speech16)) speech_pos = speech_length-sizeof(speech16); + if (noise_pos > noise_length-(long)sizeof(noise16)) noise_pos = noise_length-sizeof(noise16); + if (fgnoise_pos > fgnoise_length-(long)sizeof(fgnoise16)) fgnoise_pos = fgnoise_length-sizeof(fgnoise16); + speech_pos -= speech_pos&1; + noise_pos -= noise_pos&1; + fgnoise_pos -= fgnoise_pos&1; + fseek(f1, speech_pos, SEEK_SET); + fseek(f2, noise_pos, SEEK_SET); + fseek(f3, fgnoise_pos, SEEK_SET); + fread(speech16, sizeof(speech16), 1, f1); + fread(noise16, sizeof(noise16), 1, f2); + fread(fgnoise16, sizeof(fgnoise16), 1, f3); + if (rand()%4) start_pos = 0; + else start_pos = -(int)(1000*log(rand()/(float)RAND_MAX)); + start_pos = IMIN(start_pos, SEQUENCE_LENGTH*FRAME_SIZE); + + speech_gain = pow(10., (-45+randf(45.f)+randf(10.f))/20.); + noise_gain = pow(10., (-30+randf(40.f)+randf(15.f))/20.); + fgnoise_gain = pow(10., (-30+randf(40.f)+randf(15.f))/20.); + if (rand()%8==0) noise_gain = 0; + if (rand()%8!=0) fgnoise_gain = 0; + if (rand()%12==0) { + noise_gain *= 0.03; + fgnoise_gain *= 0.03; + } + noise_gain *= speech_gain; + fgnoise_gain *= speech_gain; + rand_resp(a_noise, b_noise); + rand_resp(a_fgnoise, b_fgnoise); + rand_resp(a_sig, b_sig); + lowpass = FREQ_SIZE * 3000./24000. * pow(50., rand()/(double)RAND_MAX); + for (i=0;i lowpass) { + band_lp = i; + break; + } + } + + for (frame=0;frame 1) g[i] = 1; + if (silence || i > band_lp) g[i] = -1; + if (Ey[i] < 5e-2 && Ex[i] < 5e-2) g[i] = -1; + if (vad_target==0 && noise_gain==0 && fgnoise_gain==0) g[i] = -1; + } +#if 0 + { + short tmp[FRAME_SIZE]; + for (j=0;j +#include +#include "denoise.h" +#include "kiss_fft.h" + +#define OVERLAP_SIZE FRAME_SIZE + +int main(void) { + int i; + FILE *file; + kiss_fft_state *kfft; + float half_window[OVERLAP_SIZE]; + float dct_table[NB_BANDS*NB_BANDS]; + + file=fopen("rnnoise_tables.c", "wb"); + fprintf(file, "/* The contents of this file was automatically generated by dump_rnnoise_tables.c*/\n\n"); + fprintf(file, "#ifdef HAVE_CONFIG_H\n"); + fprintf(file, "#include \"config.h\"\n"); + fprintf(file, "#endif\n"); + + fprintf(file, "#include \"kiss_fft.h\"\n\n"); + + kfft = rnn_fft_alloc_twiddles(WINDOW_SIZE, NULL, NULL, NULL, 0); + + fprintf(file, "static const arch_fft_state arch_fft = {0, NULL};\n\n"); + + fprintf (file, "static const opus_int32 fft_bitrev[%d] = {\n", kfft->nfft); + for (i=0;infft;i++) + fprintf (file, "%d,%c", kfft->bitrev[i],(i+16)%15==0?'\n':' '); + fprintf (file, "};\n\n"); + + fprintf (file, "static const kiss_twiddle_cpx fft_twiddles[%d] = {\n", kfft->nfft); + for (i=0;infft;i++) + fprintf (file, "{%#0.9gf, %#0.9gf},%c", kfft->twiddles[i].r, kfft->twiddles[i].i,(i+3)%2==0?'\n':' '); + fprintf (file, "};\n\n"); + + + fprintf(file, "const kiss_fft_state rnn_kfft = {\n"); + fprintf(file, "%d, /* nfft */\n", kfft->nfft); + fprintf(file, "%#0.8gf, /* scale */\n", kfft->scale); + fprintf(file, "%d, /* shift */\n", kfft->shift); + fprintf(file, "{"); + for (i=0;i<2*MAXFACTORS;i++) { + fprintf(file, "%d, ", kfft->factors[i]); + } + fprintf(file, "}, /* factors */\n"); + fprintf(file, "fft_bitrev, /* bitrev*/\n"); + fprintf(file, "fft_twiddles, /* twiddles*/\n"); + fprintf(file, "(arch_fft_state *)&arch_fft, /* arch_fft*/\n"); + + fprintf(file, "};\n\n"); + + for (i=0;itwiddles; + /* m is guaranteed to be a multiple of 4. */ + for (j=0;jtwiddles[fstride*m]; +#endif + for (i=0;itwiddles; + /* For non-custom modes, m is guaranteed to be a multiple of 4. */ + k=m; + do { + + C_MUL(scratch[1],Fout[m] , *tw1); + C_MUL(scratch[2],Fout[m2] , *tw2); + + C_ADD(scratch[3],scratch[1],scratch[2]); + C_SUB(scratch[0],scratch[1],scratch[2]); + tw1 += fstride; + tw2 += fstride*2; + + Fout[m].r = SUB32_ovflw(Fout->r, HALF_OF(scratch[3].r)); + Fout[m].i = SUB32_ovflw(Fout->i, HALF_OF(scratch[3].i)); + + C_MULBYSCALAR( scratch[0] , epi3.i ); + + C_ADDTO(*Fout,scratch[3]); + + Fout[m2].r = ADD32_ovflw(Fout[m].r, scratch[0].i); + Fout[m2].i = SUB32_ovflw(Fout[m].i, scratch[0].r); + + Fout[m].r = SUB32_ovflw(Fout[m].r, scratch[0].i); + Fout[m].i = ADD32_ovflw(Fout[m].i, scratch[0].r); + + ++Fout; + } while(--k); + } +} + + +#ifndef OVERRIDE_kf_bfly5 +static void kf_bfly5( + kiss_fft_cpx * Fout, + const size_t fstride, + const kiss_fft_state *st, + int m, + int N, + int mm + ) +{ + kiss_fft_cpx *Fout0,*Fout1,*Fout2,*Fout3,*Fout4; + int i, u; + kiss_fft_cpx scratch[13]; + const kiss_twiddle_cpx *tw; + kiss_twiddle_cpx ya,yb; + kiss_fft_cpx * Fout_beg = Fout; + +#ifdef FIXED_POINT + ya.r = 10126; + ya.i = -31164; + yb.r = -26510; + yb.i = -19261; +#else + ya = st->twiddles[fstride*m]; + yb = st->twiddles[fstride*2*m]; +#endif + tw=st->twiddles; + + for (i=0;ir = ADD32_ovflw(Fout0->r, ADD32_ovflw(scratch[7].r, scratch[8].r)); + Fout0->i = ADD32_ovflw(Fout0->i, ADD32_ovflw(scratch[7].i, scratch[8].i)); + + scratch[5].r = ADD32_ovflw(scratch[0].r, ADD32_ovflw(S_MUL(scratch[7].r,ya.r), S_MUL(scratch[8].r,yb.r))); + scratch[5].i = ADD32_ovflw(scratch[0].i, ADD32_ovflw(S_MUL(scratch[7].i,ya.r), S_MUL(scratch[8].i,yb.r))); + + scratch[6].r = ADD32_ovflw(S_MUL(scratch[10].i,ya.i), S_MUL(scratch[9].i,yb.i)); + scratch[6].i = NEG32_ovflw(ADD32_ovflw(S_MUL(scratch[10].r,ya.i), S_MUL(scratch[9].r,yb.i))); + + C_SUB(*Fout1,scratch[5],scratch[6]); + C_ADD(*Fout4,scratch[5],scratch[6]); + + scratch[11].r = ADD32_ovflw(scratch[0].r, ADD32_ovflw(S_MUL(scratch[7].r,yb.r), S_MUL(scratch[8].r,ya.r))); + scratch[11].i = ADD32_ovflw(scratch[0].i, ADD32_ovflw(S_MUL(scratch[7].i,yb.r), S_MUL(scratch[8].i,ya.r))); + scratch[12].r = SUB32_ovflw(S_MUL(scratch[9].i,ya.i), S_MUL(scratch[10].i,yb.i)); + scratch[12].i = SUB32_ovflw(S_MUL(scratch[10].r,yb.i), S_MUL(scratch[9].r,ya.i)); + + C_ADD(*Fout2,scratch[11],scratch[12]); + C_SUB(*Fout3,scratch[11],scratch[12]); + + ++Fout0;++Fout1;++Fout2;++Fout3;++Fout4; + } + } +} +#endif /* OVERRIDE_kf_bfly5 */ + + +#endif + + +#ifdef CUSTOM_MODES + +static +void compute_bitrev_table( + int Fout, + opus_int32 *f, + const size_t fstride, + int in_stride, + opus_int16 * factors, + const kiss_fft_state *st + ) +{ + const int p=*factors++; /* the radix */ + const int m=*factors++; /* stage's fft length/p */ + + /*printf ("fft %d %d %d %d %d %d\n", p*m, m, p, s2, fstride*in_stride, N);*/ + if (m==1) + { + int j; + for (j=0;j32000 || (opus_int32)p*(opus_int32)p > n) + p = n; /* no more factors, skip to end */ + } + n /= p; +#ifdef RADIX_TWO_ONLY + if (p!=2 && p != 4) +#else + if (p>5) +#endif + { + return 0; + } + facbuf[2*stages] = p; + if (p==2 && stages > 1) + { + facbuf[2*stages] = 4; + facbuf[2] = 2; + } + stages++; + } while (n > 1); + n = nbak; + /* Reverse the order to get the radix 4 at the end, so we can use the + fast degenerate case. It turns out that reversing the order also + improves the noise behaviour. */ + for (i=0;i= memneeded) + st = (kiss_fft_state*)mem; + *lenmem = memneeded; + } + if (st) { + opus_int32 *bitrev; + kiss_twiddle_cpx *twiddles; + + st->nfft=nfft; +#ifdef FIXED_POINT + st->scale_shift = celt_ilog2(st->nfft); + if (st->nfft == 1<scale_shift) + st->scale = Q15ONE; + else + st->scale = (1073741824+st->nfft/2)/st->nfft>>(15-st->scale_shift); +#else + st->scale = 1.f/nfft; +#endif + if (base != NULL) + { + st->twiddles = base->twiddles; + st->shift = 0; + while (st->shift < 32 && nfft<shift != base->nfft) + st->shift++; + if (st->shift>=32) + goto fail; + } else { + st->twiddles = twiddles = (kiss_twiddle_cpx*)KISS_FFT_MALLOC(sizeof(kiss_twiddle_cpx)*nfft); + compute_twiddles(twiddles, nfft); + st->shift = -1; + } + if (!kf_factor(nfft,st->factors)) + { + goto fail; + } + + /* bitrev */ + st->bitrev = bitrev = (opus_int32*)KISS_FFT_MALLOC(sizeof(opus_int32)*nfft); + if (st->bitrev==NULL) + goto fail; + compute_bitrev_table(0, bitrev, 1,1, st->factors,st); + + /* Initialize architecture specific fft parameters */ + if (rnn_fft_alloc_arch(st, arch)) + goto fail; + } + return st; +fail: + rnn_fft_free(st, arch); + return NULL; +} + +kiss_fft_state *rnn_fft_alloc(int nfft,void * mem,size_t * lenmem, int arch) +{ + return rnn_fft_alloc_twiddles(nfft, mem, lenmem, NULL, arch); +} + +void rnn_fft_free_arch_c(kiss_fft_state *st) { + (void)st; +} + +void rnn_fft_free(const kiss_fft_state *cfg, int arch) +{ + if (cfg) + { + rnn_fft_free_arch((kiss_fft_state *)cfg, arch); + opus_free((opus_int32*)cfg->bitrev); + if (cfg->shift < 0) + opus_free((kiss_twiddle_cpx*)cfg->twiddles); + opus_free((kiss_fft_state*)cfg); + } +} + +#endif /* CUSTOM_MODES */ + +void rnn_fft_impl(const kiss_fft_state *st,kiss_fft_cpx *fout) +{ + int m2, m; + int p; + int L; + int fstride[MAXFACTORS]; + int i; + int shift; + + /* st->shift can be -1 */ + shift = st->shift>0 ? st->shift : 0; + + fstride[0] = 1; + L=0; + do { + p = st->factors[2*L]; + m = st->factors[2*L+1]; + fstride[L+1] = fstride[L]*p; + L++; + } while(m!=1); + m = st->factors[2*L-1]; + for (i=L-1;i>=0;i--) + { + if (i!=0) + m2 = st->factors[2*i-1]; + else + m2 = 1; + switch (st->factors[2*i]) + { + case 2: + kf_bfly2(fout, m, fstride[i]); + break; + case 4: + kf_bfly4(fout,fstride[i]<scale_shift-1; +#endif + scale = st->scale; + + celt_assert2 (fin != fout, "In-place FFT not supported"); + /* Bit-reverse the input */ + for (i=0;infft;i++) + { + kiss_fft_cpx x = fin[i]; + fout[st->bitrev[i]].r = SHR32(MULT16_32_Q16(scale, x.r), scale_shift); + fout[st->bitrev[i]].i = SHR32(MULT16_32_Q16(scale, x.i), scale_shift); + } + rnn_fft_impl(st, fout); +} + + +void rnn_ifft_c(const kiss_fft_state *st,const kiss_fft_cpx *fin,kiss_fft_cpx *fout) +{ + int i; + celt_assert2 (fin != fout, "In-place FFT not supported"); + /* Bit-reverse the input */ + for (i=0;infft;i++) + fout[st->bitrev[i]] = fin[i]; + for (i=0;infft;i++) + fout[i].i = -fout[i].i; + rnn_fft_impl(st, fout); + for (i=0;infft;i++) + fout[i].i = -fout[i].i; +} diff --git a/cpp/ax650/src/rnnoise/kiss_fft.h b/cpp/ax650/src/rnnoise/kiss_fft.h new file mode 100644 index 0000000000000000000000000000000000000000..52387f21f9caaeb171559a23f20d0a6e57641c7c --- /dev/null +++ b/cpp/ax650/src/rnnoise/kiss_fft.h @@ -0,0 +1,203 @@ +/*Copyright (c) 2003-2004, Mark Borgerding + Lots of modifications by Jean-Marc Valin + Copyright (c) 2005-2007, Xiph.Org Foundation + Copyright (c) 2008, Xiph.Org Foundation, CSIRO + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE + LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR + CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF + SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS + INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN + CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) + ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE + POSSIBILITY OF SUCH DAMAGE.*/ + +#ifndef KISS_FFT_H +#define KISS_FFT_H + +#include +#include +#include "arch.h" + +#include +#define opus_alloc(x) malloc(x) +#define opus_free(x) free(x) + +#ifdef __cplusplus +extern "C" { +#endif + +#ifdef USE_SIMD +# include +# define kiss_fft_scalar __m128 +#define KISS_FFT_MALLOC(nbytes) memalign(16,nbytes) +#else +#define KISS_FFT_MALLOC opus_alloc +#endif + +#ifdef FIXED_POINT +#include "arch.h" + +# define kiss_fft_scalar opus_int32 +# define kiss_twiddle_scalar opus_int16 + + +#else +# ifndef kiss_fft_scalar +/* default is float */ +# define kiss_fft_scalar float +# define kiss_twiddle_scalar float +# define KF_SUFFIX _celt_single +# endif +#endif + +typedef struct { + kiss_fft_scalar r; + kiss_fft_scalar i; +}kiss_fft_cpx; + +typedef struct { + kiss_twiddle_scalar r; + kiss_twiddle_scalar i; +}kiss_twiddle_cpx; + +#define MAXFACTORS 8 +/* e.g. an fft of length 128 has 4 factors + as far as kissfft is concerned + 4*4*4*2 + */ + +typedef struct arch_fft_state{ + int is_supported; + void *priv; +} arch_fft_state; + +typedef struct kiss_fft_state{ + int nfft; + opus_val16 scale; +#ifdef FIXED_POINT + int scale_shift; +#endif + int shift; + opus_int16 factors[2*MAXFACTORS]; + const opus_int32 *bitrev; + const kiss_twiddle_cpx *twiddles; + arch_fft_state *arch_fft; +} kiss_fft_state; + +#if defined(HAVE_ARM_NE10) +#include "arm/fft_arm.h" +#endif + +/*typedef struct kiss_fft_state* kiss_fft_cfg;*/ + +/** + * opus_fft_alloc + * + * Initialize a FFT (or IFFT) algorithm's cfg/state buffer. + * + * typical usage: kiss_fft_cfg mycfg=opus_fft_alloc(1024,0,NULL,NULL); + * + * The return value from fft_alloc is a cfg buffer used internally + * by the fft routine or NULL. + * + * If lenmem is NULL, then opus_fft_alloc will allocate a cfg buffer using malloc. + * The returned value should be free()d when done to avoid memory leaks. + * + * The state can be placed in a user supplied buffer 'mem': + * If lenmem is not NULL and mem is not NULL and *lenmem is large enough, + * then the function places the cfg in mem and the size used in *lenmem + * and returns mem. + * + * If lenmem is not NULL and ( mem is NULL or *lenmem is not large enough), + * then the function returns NULL and places the minimum cfg + * buffer size in *lenmem. + * */ + +kiss_fft_state *rnn_fft_alloc_twiddles(int nfft,void * mem,size_t * lenmem, const kiss_fft_state *base, int arch); + +kiss_fft_state *rnn_fft_alloc(int nfft,void * mem,size_t * lenmem, int arch); + +/** + * opus_fft(cfg,in_out_buf) + * + * Perform an FFT on a complex input buffer. + * for a forward FFT, + * fin should be f[0] , f[1] , ... ,f[nfft-1] + * fout will be F[0] , F[1] , ... ,F[nfft-1] + * Note that each element is complex and can be accessed like + f[k].r and f[k].i + * */ +void rnn_fft_c(const kiss_fft_state *cfg,const kiss_fft_cpx *fin,kiss_fft_cpx *fout); +void rnn_ifft_c(const kiss_fft_state *cfg,const kiss_fft_cpx *fin,kiss_fft_cpx *fout); + +void rnn_fft_impl(const kiss_fft_state *st,kiss_fft_cpx *fout); +void rnn_ifft_impl(const kiss_fft_state *st,kiss_fft_cpx *fout); + +void rnn_fft_free(const kiss_fft_state *cfg, int arch); + + +void rnn_fft_free_arch_c(kiss_fft_state *st); +int rnn_fft_alloc_arch_c(kiss_fft_state *st); + +#if !defined(OVERRIDE_OPUS_FFT) +/* Is run-time CPU detection enabled on this platform? */ +#if defined(OPUS_HAVE_RTCD) && (defined(HAVE_ARM_NE10)) + +extern int (*const OPUS_FFT_ALLOC_ARCH_IMPL[OPUS_ARCHMASK+1])( + kiss_fft_state *st); + +#define opus_fft_alloc_arch(_st, arch) \ + ((*OPUS_FFT_ALLOC_ARCH_IMPL[(arch)&OPUS_ARCHMASK])(_st)) + +extern void (*const OPUS_FFT_FREE_ARCH_IMPL[OPUS_ARCHMASK+1])( + kiss_fft_state *st); +#define opus_fft_free_arch(_st, arch) \ + ((*OPUS_FFT_FREE_ARCH_IMPL[(arch)&OPUS_ARCHMASK])(_st)) + +extern void (*const OPUS_FFT[OPUS_ARCHMASK+1])(const kiss_fft_state *cfg, + const kiss_fft_cpx *fin, kiss_fft_cpx *fout); +#define opus_fft(_cfg, _fin, _fout, arch) \ + ((*OPUS_FFT[(arch)&OPUS_ARCHMASK])(_cfg, _fin, _fout)) + +extern void (*const OPUS_IFFT[OPUS_ARCHMASK+1])(const kiss_fft_state *cfg, + const kiss_fft_cpx *fin, kiss_fft_cpx *fout); +#define opus_ifft(_cfg, _fin, _fout, arch) \ + ((*OPUS_IFFT[(arch)&OPUS_ARCHMASK])(_cfg, _fin, _fout)) + +#else /* else for if defined(OPUS_HAVE_RTCD) && (defined(HAVE_ARM_NE10)) */ + +#define rnn_fft_alloc_arch(_st, arch) \ + ((void)(arch), rnn_fft_alloc_arch_c(_st)) + +#define rnn_fft_free_arch(_st, arch) \ + ((void)(arch), rnn_fft_free_arch_c(_st)) + +#define rnn_fft(_cfg, _fin, _fout, arch) \ + ((void)(arch), rnn_fft_c(_cfg, _fin, _fout)) + +#define rnn_ifft(_cfg, _fin, _fout, arch) \ + ((void)(arch), rnn_ifft_c(_cfg, _fin, _fout)) + +#endif /* end if defined(OPUS_HAVE_RTCD) && (defined(HAVE_ARM_NE10)) */ +#endif /* end if !defined(OVERRIDE_OPUS_FFT) */ + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/cpp/ax650/src/rnnoise/nnet.c b/cpp/ax650/src/rnnoise/nnet.c new file mode 100644 index 0000000000000000000000000000000000000000..3cafea6051a858ba316a32478b59255863fb7222 --- /dev/null +++ b/cpp/ax650/src/rnnoise/nnet.c @@ -0,0 +1,123 @@ +/* Copyright (c) 2018 Mozilla + 2008-2011 Octasic Inc. + 2012-2017 Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#include +#include "opus_types.h" +#include "arch.h" +#include "nnet.h" +#include "common.h" +#include "vec.h" + +#ifdef ENABLE_OSCE +#include "osce.h" +#endif + +#ifdef NO_OPTIMIZATIONS +#if defined(_MSC_VER) +#pragma message ("Compiling without any vectorization. This code will be very slow") +#else +#warning Compiling without any vectorization. This code will be very slow +#endif +#endif + + +#define SOFTMAX_HACK + + +void compute_generic_dense(const LinearLayer *layer, float *output, const float *input, int activation, int arch) +{ + compute_linear(layer, output, input, arch); + compute_activation(output, output, layer->nb_outputs, activation, arch); +} + +#define MAX_RNN_NEURONS_ALL 1024 + +void compute_generic_gru(const LinearLayer *input_weights, const LinearLayer *recurrent_weights, float *state, const float *in, int arch) +{ + int i; + int N; + float zrh[3*MAX_RNN_NEURONS_ALL]; + float recur[3*MAX_RNN_NEURONS_ALL]; + float *z; + float *r; + float *h; + celt_assert(3*recurrent_weights->nb_inputs == recurrent_weights->nb_outputs); + celt_assert(input_weights->nb_outputs == recurrent_weights->nb_outputs); + N = recurrent_weights->nb_inputs; + z = zrh; + r = &zrh[N]; + h = &zrh[2*N]; + celt_assert(recurrent_weights->nb_outputs <= 3*MAX_RNN_NEURONS_ALL); + celt_assert(in != state); + compute_linear(input_weights, zrh, in, arch); + compute_linear(recurrent_weights, recur, state, arch); + for (i=0;i<2*N;i++) + zrh[i] += recur[i]; + compute_activation(zrh, zrh, 2*N, ACTIVATION_SIGMOID, arch); + for (i=0;inb_inputs == layer->nb_outputs); + compute_linear(layer, act2, input, arch); + compute_activation(act2, act2, layer->nb_outputs, ACTIVATION_SIGMOID, arch); + if (input == output) { + /* Give a vectorization hint to the compiler for the in-place case. */ + for (i=0;inb_outputs;i++) output[i] = output[i]*act2[i]; + } else { + for (i=0;inb_outputs;i++) output[i] = input[i]*act2[i]; + } +} + +#define MAX_CONV_INPUTS_ALL 1024 + +void compute_generic_conv1d(const LinearLayer *layer, float *output, float *mem, const float *input, int input_size, int activation, int arch) +{ + float tmp[MAX_CONV_INPUTS_ALL]; + celt_assert(input != output); + celt_assert(layer->nb_inputs <= MAX_CONV_INPUTS_ALL); + if (layer->nb_inputs!=input_size) RNN_COPY(tmp, mem, layer->nb_inputs-input_size); + RNN_COPY(&tmp[layer->nb_inputs-input_size], input, input_size); + compute_linear(layer, output, tmp, arch); + compute_activation(output, output, layer->nb_outputs, activation, arch); + if (layer->nb_inputs!=input_size) RNN_COPY(mem, &tmp[input_size], layer->nb_inputs-input_size); +} diff --git a/cpp/ax650/src/rnnoise/nnet.h b/cpp/ax650/src/rnnoise/nnet.h new file mode 100644 index 0000000000000000000000000000000000000000..6b8023bdb03b9dfa49728a3dbe76c1f32bc8fc6d --- /dev/null +++ b/cpp/ax650/src/rnnoise/nnet.h @@ -0,0 +1,169 @@ +/* Copyright (c) 2018 Mozilla + Copyright (c) 2017 Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef NNET_H_ +#define NNET_H_ + +#include +#include "opus_types.h" + +#define ACTIVATION_LINEAR 0 +#define ACTIVATION_SIGMOID 1 +#define ACTIVATION_TANH 2 +#define ACTIVATION_RELU 3 +#define ACTIVATION_SOFTMAX 4 +#define ACTIVATION_SWISH 5 + +#define WEIGHT_BLOB_VERSION 0 +#define WEIGHT_BLOCK_SIZE 64 +typedef struct { + const char *name; + int type; + int size; + const void *data; +} WeightArray; + +#define WEIGHT_TYPE_float 0 +#define WEIGHT_TYPE_int 1 +#define WEIGHT_TYPE_qweight 2 +#define WEIGHT_TYPE_int8 3 + +typedef struct { + char head[4]; + int version; + int type; + int size; + int block_size; + char name[44]; +} WeightHead; + +/* Generic sparse affine transformation. */ +typedef struct { + const float *bias; + const float *subias; + const opus_int8 *weights; + const float *float_weights; + const int *weights_idx; + const float *diag; + const float *scale; + int nb_inputs; + int nb_outputs; +} LinearLayer; + +/* Generic sparse affine transformation. */ +typedef struct { + const float *bias; + const float *float_weights; + int in_channels; + int out_channels; + int ktime; + int kheight; +} Conv2dLayer; + + +/* Changes some symbol names to add the rnn_ prefix so we don't get conflicts with Opus. */ +#define linear_init rnn_linear_init +#define conv2d_init rnn_conv2d_init +#define compute_generic_dense rnn_compute_generic_dense +#define compute_generic_gru rnn_compute_generic_gru +#define compute_generic_conv1d rnn_compute_generic_conv1d +#define compute_glu rnn_compute_glu + +#define parse_weights rnn_parse_weights + +#define compute_linear_c rnn_compute_linear_c +#define compute_activation_c rnn_compute_activation_c +#define compute_conv2d_c rnn_compute_conv2d_c +#define compute_linear_sse4_1 rnn_compute_linear_sse4_1 +#define compute_activation_sse4_1 rnn_compute_activation_sse4_1 +#define compute_conv2d_sse4_1 rnn_compute_conv2d_sse4_1 +#define compute_linear_avx2 rnn_compute_linear_avx2 +#define compute_activation_avx2 rnn_compute_activation_avx2 +#define compute_conv2d_avx2 rnn_compute_conv2d_avx2 + + +void compute_generic_dense(const LinearLayer *layer, float *output, const float *input, int activation, int arch); +void compute_generic_gru(const LinearLayer *input_weights, const LinearLayer *recurrent_weights, float *state, const float *in, int arch); +void compute_generic_conv1d(const LinearLayer *layer, float *output, float *mem, const float *input, int input_size, int activation, int arch); +void compute_glu(const LinearLayer *layer, float *output, const float *input, int arch); + + +int parse_weights(WeightArray **list, const void *data, int len); + + + +int linear_init(LinearLayer *layer, const WeightArray *arrays, + const char *bias, + const char *subias, + const char *weights, + const char *float_weights, + const char *weights_idx, + const char *diag, + const char *scale, + int nb_inputs, + int nb_outputs); + +int conv2d_init(Conv2dLayer *layer, const WeightArray *arrays, + const char *bias, + const char *float_weights, + int in_channels, + int out_channels, + int ktime, + int kheight); + + +void compute_linear_c(const LinearLayer *linear, float *out, const float *in); +void compute_activation_c(float *output, const float *input, int N, int activation); +void compute_conv2d_c(const Conv2dLayer *conv, float *out, float *mem, const float *in, int height, int hstride, int activation); + +#ifdef RNN_ENABLE_X86_RTCD +#include "x86/dnn_x86.h" +#endif + +#ifndef OVERRIDE_COMPUTE_LINEAR +#define compute_linear(linear, out, in, arch) ((void)(arch),compute_linear_c(linear, out, in)) +#endif + +#ifndef OVERRIDE_COMPUTE_ACTIVATION +#define compute_activation(output, input, N, activation, arch) ((void)(arch),compute_activation_c(output, input, N, activation)) +#endif + +#ifndef OVERRIDE_COMPUTE_CONV2D +#define compute_conv2d(conv, out, mem, in, height, hstride, activation, arch) ((void)(arch),compute_conv2d_c(conv, out, mem, in, height, hstride, activation)) +#endif + +#if defined(__x86_64__) && !defined(RNN_ENABLE_X86_RTCD) && !defined(__AVX2__) +#if defined(_MSC_VER) +#pragma message ("Only SSE and SSE2 are available. On newer machines, enable SSSE3/AVX/AVX2 to get better performance") +#else +#warning "Only SSE and SSE2 are available. On newer machines, enable SSSE3/AVX/AVX2 using -march= to get better performance" +#endif +#endif + + + +#endif /* NNET_H_ */ diff --git a/cpp/ax650/src/rnnoise/nnet_arch.h b/cpp/ax650/src/rnnoise/nnet_arch.h new file mode 100644 index 0000000000000000000000000000000000000000..dfd5eaadeaf6b8ee00257de58d0c114e32d52aa3 --- /dev/null +++ b/cpp/ax650/src/rnnoise/nnet_arch.h @@ -0,0 +1,257 @@ +/* Copyright (c) 2018-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef NNET_ARCH_H +#define NNET_ARCH_H + +#include "nnet.h" +#include "arch.h" +#include "common.h" +#include "vec.h" + +#define CAT_SUFFIX2(a,b) a ## b +#define CAT_SUFFIX(a,b) CAT_SUFFIX2(a, b) + +#define RTCD_SUF(name) CAT_SUFFIX(name, RTCD_ARCH) + +# if !defined(OPUS_GNUC_PREREQ) +# if defined(__GNUC__)&&defined(__GNUC_MINOR__) +# define OPUS_GNUC_PREREQ(_maj,_min) \ + ((__GNUC__<<16)+__GNUC_MINOR__>=((_maj)<<16)+(_min)) +# else +# define OPUS_GNUC_PREREQ(_maj,_min) 0 +# endif +# endif + + +/* Force vectorization on for DNN code because some of the loops rely on + compiler vectorization rather than explicitly using intrinsics. */ +#if OPUS_GNUC_PREREQ(5,1) +#define GCC_POP_OPTIONS +#pragma GCC push_options +#pragma GCC optimize("tree-vectorize") +#endif + + +#define MAX_ACTIVATIONS (4096) + +static OPUS_INLINE void vec_swish(float *y, const float *x, int N) +{ + int i; + float tmp[MAX_ACTIVATIONS]; + celt_assert(N <= MAX_ACTIVATIONS); + vec_sigmoid(tmp, x, N); + for (i=0;ibias; + M = linear->nb_inputs; + N = linear->nb_outputs; + if (linear->float_weights != NULL) { + if (linear->weights_idx != NULL) sparse_sgemv8x4(out, linear->float_weights, linear->weights_idx, N, in); + else sgemv(out, linear->float_weights, N, M, N, in); + } else if (linear->weights != NULL) { + if (linear->weights_idx != NULL) sparse_cgemv8x4(out, linear->weights, linear->weights_idx, linear->scale, N, M, in); + else cgemv8x4(out, linear->weights, linear->scale, N, M, in); + /* Only use SU biases on for integer matrices on SU archs. */ +#ifdef USE_SU_BIAS + bias = linear->subias; +#endif + } + else RNN_CLEAR(out, N); + if (bias != NULL) { + for (i=0;idiag) { + /* Diag is only used for GRU recurrent weights. */ + celt_assert(3*M == N); + for (i=0;idiag[i]*in[i]; + out[i+M] += linear->diag[i+M]*in[i]; + out[i+2*M] += linear->diag[i+2*M]*in[i]; + } + } +} + +/* Computes non-padded convolution for input [ ksize1 x in_channels x (len2+ksize2) ], + kernel [ out_channels x in_channels x ksize1 x ksize2 ], + storing the output as [ out_channels x len2 ]. + We assume that the output dimension along the ksize1 axis is 1, + i.e. processing one frame at a time. */ +static void conv2d_float(float *out, const float *weights, int in_channels, int out_channels, int ktime, int kheight, const float *in, int height, int hstride) +{ + int i; + int in_stride; + in_stride = height+kheight-1; + for (i=0;iin_channels*(height+conv->kheight-1); + celt_assert(conv->ktime*time_stride <= MAX_CONV2D_INPUTS); + RNN_COPY(in_buf, mem, (conv->ktime-1)*time_stride); + RNN_COPY(&in_buf[(conv->ktime-1)*time_stride], in, time_stride); + RNN_COPY(mem, &in_buf[time_stride], (conv->ktime-1)*time_stride); + bias = conv->bias; + if (conv->kheight == 3 && conv->ktime == 3) + conv2d_3x3_float(out, conv->float_weights, conv->in_channels, conv->out_channels, in_buf, height, hstride); + else + conv2d_float(out, conv->float_weights, conv->in_channels, conv->out_channels, conv->ktime, conv->kheight, in_buf, height, hstride); + if (bias != NULL) { + for (i=0;iout_channels;i++) { + int j; + for (j=0;jout_channels;i++) { + RTCD_SUF(compute_activation_)(&out[i*hstride], &out[i*hstride], height, activation); + } +} + +#ifdef GCC_POP_OPTIONS +#pragma GCC pop_options +#endif + +#endif diff --git a/cpp/ax650/src/rnnoise/nnet_default.c b/cpp/ax650/src/rnnoise/nnet_default.c new file mode 100644 index 0000000000000000000000000000000000000000..4316f0fba3c5dca1085f46c5c97304ecbfb2e993 --- /dev/null +++ b/cpp/ax650/src/rnnoise/nnet_default.c @@ -0,0 +1,35 @@ +/* Copyright (c) 2018-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + + +#define RTCD_ARCH c + +#include "nnet_arch.h" diff --git a/cpp/ax650/src/rnnoise/opus_types.h b/cpp/ax650/src/rnnoise/opus_types.h new file mode 100644 index 0000000000000000000000000000000000000000..71808266655a1502105ea4c68f15e8ff0be932e4 --- /dev/null +++ b/cpp/ax650/src/rnnoise/opus_types.h @@ -0,0 +1,159 @@ +/* (C) COPYRIGHT 1994-2002 Xiph.Org Foundation */ +/* Modified by Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ +/* opus_types.h based on ogg_types.h from libogg */ + +/** + @file opus_types.h + @brief Opus reference implementation types +*/ +#ifndef OPUS_TYPES_H +#define OPUS_TYPES_H + +/* Use the real stdint.h if it's there (taken from Paul Hsieh's pstdint.h) */ +#if (defined(__STDC__) && __STDC__ && defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L) || (defined(__GNUC__) && (defined(_STDINT_H) || defined(_STDINT_H_)) || defined (HAVE_STDINT_H)) +#include + + typedef int16_t opus_int16; + typedef uint16_t opus_uint16; + typedef int32_t opus_int32; + typedef uint32_t opus_uint32; +#elif defined(_WIN32) + +# if defined(__CYGWIN__) +# include <_G_config.h> + typedef _G_int32_t opus_int32; + typedef _G_uint32_t opus_uint32; + typedef _G_int16 opus_int16; + typedef _G_uint16 opus_uint16; +# elif defined(__MINGW32__) + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef int opus_int32; + typedef unsigned int opus_uint32; +# elif defined(__MWERKS__) + typedef int opus_int32; + typedef unsigned int opus_uint32; + typedef short opus_int16; + typedef unsigned short opus_uint16; +# else + /* MSVC/Borland */ + typedef __int32 opus_int32; + typedef unsigned __int32 opus_uint32; + typedef __int16 opus_int16; + typedef unsigned __int16 opus_uint16; +# endif + +#elif defined(__MACOS__) + +# include + typedef SInt16 opus_int16; + typedef UInt16 opus_uint16; + typedef SInt32 opus_int32; + typedef UInt32 opus_uint32; + +#elif (defined(__APPLE__) && defined(__MACH__)) /* MacOS X Framework build */ + +# include + typedef int16_t opus_int16; + typedef u_int16_t opus_uint16; + typedef int32_t opus_int32; + typedef u_int32_t opus_uint32; + +#elif defined(__BEOS__) + + /* Be */ +# include + typedef int16 opus_int16; + typedef u_int16 opus_uint16; + typedef int32_t opus_int32; + typedef u_int32_t opus_uint32; + +#elif defined (__EMX__) + + /* OS/2 GCC */ + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef int opus_int32; + typedef unsigned int opus_uint32; + +#elif defined (DJGPP) + + /* DJGPP */ + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef int opus_int32; + typedef unsigned int opus_uint32; + +#elif defined(R5900) + + /* PS2 EE */ + typedef int opus_int32; + typedef unsigned opus_uint32; + typedef short opus_int16; + typedef unsigned short opus_uint16; + +#elif defined(__SYMBIAN32__) + + /* Symbian GCC */ + typedef signed short opus_int16; + typedef unsigned short opus_uint16; + typedef signed int opus_int32; + typedef unsigned int opus_uint32; + +#elif defined(CONFIG_TI_C54X) || defined (CONFIG_TI_C55X) + + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef long opus_int32; + typedef unsigned long opus_uint32; + +#elif defined(CONFIG_TI_C6X) + + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef int opus_int32; + typedef unsigned int opus_uint32; + +#else + + /* Give up, take a reasonable guess */ + typedef short opus_int16; + typedef unsigned short opus_uint16; + typedef int opus_int32; + typedef unsigned int opus_uint32; + +#endif + +#define opus_int int /* used for counters etc; at least 16 bits */ +#define opus_int64 long long +#define opus_int8 signed char + +#define opus_uint unsigned int /* used for counters etc; at least 16 bits */ +#define opus_uint64 unsigned long long +#define opus_uint8 unsigned char + +#endif /* OPUS_TYPES_H */ diff --git a/cpp/ax650/src/rnnoise/parse_lpcnet_weights.c b/cpp/ax650/src/rnnoise/parse_lpcnet_weights.c new file mode 100644 index 0000000000000000000000000000000000000000..bdd50d7ea160a695741addf9126ad69fb646524b --- /dev/null +++ b/cpp/ax650/src/rnnoise/parse_lpcnet_weights.c @@ -0,0 +1,237 @@ +/* Copyright (c) 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#include +#include "nnet.h" + +#define SPARSE_BLOCK_SIZE 32 + +static int parse_record(const void **data, int *len, WeightArray *array) { + WeightHead *h = (WeightHead *)*data; + if (*len < WEIGHT_BLOCK_SIZE) return -1; + if (h->block_size < h->size) return -1; + if (h->block_size > *len-WEIGHT_BLOCK_SIZE) return -1; + if (h->name[sizeof(h->name)-1] != 0) return -1; + if (h->size < 0) return -1; + array->name = h->name; + array->type = h->type; + array->size = h->size; + array->data = (void*)((unsigned char*)(*data)+WEIGHT_BLOCK_SIZE); + + *data = (void*)((unsigned char*)*data + h->block_size+WEIGHT_BLOCK_SIZE); + *len -= h->block_size+WEIGHT_BLOCK_SIZE; + return array->size; +} + +int parse_weights(WeightArray **list, const void *data, int len) +{ + int nb_arrays=0; + int capacity=20; + *list = calloc(capacity*sizeof(WeightArray), 1); + while (len > 0) { + int ret; + WeightArray array = {NULL, 0, 0, 0}; + ret = parse_record(&data, &len, &array); + if (ret > 0) { + if (nb_arrays+1 >= capacity) { + /* Make sure there's room for the ending NULL element too. */ + capacity = capacity*3/2; + *list = realloc(*list, capacity*sizeof(WeightArray)); + } + (*list)[nb_arrays++] = array; + } else { + free(*list); + *list = NULL; + return -1; + } + } + (*list)[nb_arrays].name=NULL; + return nb_arrays; +} + +static const void *find_array_entry(const WeightArray *arrays, const char *name) { + while (arrays->name && strcmp(arrays->name, name) != 0) arrays++; + return arrays; +} + +static const void *find_array_check(const WeightArray *arrays, const char *name, int size) { + const WeightArray *a = find_array_entry(arrays, name); + if (a->name && a->size == size) return a->data; + else return NULL; +} + +static const void *opt_array_check(const WeightArray *arrays, const char *name, int size, int *error) { + const WeightArray *a = find_array_entry(arrays, name); + *error = (a->name != NULL && a->size != size); + if (a->name && a->size == size) return a->data; + else return NULL; +} + +static const void *find_idx_check(const WeightArray *arrays, const char *name, int nb_in, int nb_out, int *total_blocks) { + int remain; + const int *idx; + const WeightArray *a = find_array_entry(arrays, name); + *total_blocks = 0; + if (a == NULL) return NULL; + idx = a->data; + remain = a->size/sizeof(int); + while (remain > 0) { + int nb_blocks; + int i; + nb_blocks = *idx++; + if (remain < nb_blocks+1) return NULL; + for (i=0;i= nb_in || (pos&0x3)) return NULL; + } + nb_out -= 8; + remain -= nb_blocks+1; + *total_blocks += nb_blocks; + } + if (nb_out != 0) return NULL; + return a->data; +} + +int linear_init(LinearLayer *layer, const WeightArray *arrays, + const char *bias, + const char *subias, + const char *weights, + const char *float_weights, + const char *weights_idx, + const char *diag, + const char *scale, + int nb_inputs, + int nb_outputs) +{ + int err; + layer->bias = NULL; + layer->subias = NULL; + layer->weights = NULL; + layer->float_weights = NULL; + layer->weights_idx = NULL; + layer->diag = NULL; + layer->scale = NULL; + if (bias != NULL) { + if ((layer->bias = find_array_check(arrays, bias, nb_outputs*sizeof(layer->bias[0]))) == NULL) return 1; + } + if (subias != NULL) { + if ((layer->subias = find_array_check(arrays, subias, nb_outputs*sizeof(layer->subias[0]))) == NULL) return 1; + } + if (weights_idx != NULL) { + int total_blocks; + if ((layer->weights_idx = find_idx_check(arrays, weights_idx, nb_inputs, nb_outputs, &total_blocks)) == NULL) return 1; + if (weights != NULL) { + if ((layer->weights = find_array_check(arrays, weights, SPARSE_BLOCK_SIZE*total_blocks*sizeof(layer->weights[0]))) == NULL) return 1; + } + if (float_weights != NULL) { + layer->float_weights = opt_array_check(arrays, float_weights, SPARSE_BLOCK_SIZE*total_blocks*sizeof(layer->float_weights[0]), &err); + if (err) return 1; + } + } else { + if (weights != NULL) { + if ((layer->weights = find_array_check(arrays, weights, nb_inputs*nb_outputs*sizeof(layer->weights[0]))) == NULL) return 1; + } + if (float_weights != NULL) { + layer->float_weights = opt_array_check(arrays, float_weights, nb_inputs*nb_outputs*sizeof(layer->float_weights[0]), &err); + if (err) return 1; + } + } + if (diag != NULL) { + if ((layer->diag = find_array_check(arrays, diag, nb_outputs*sizeof(layer->diag[0]))) == NULL) return 1; + } + if (weights != NULL) { + if ((layer->scale = find_array_check(arrays, scale, nb_outputs*sizeof(layer->scale[0]))) == NULL) return 1; + } + layer->nb_inputs = nb_inputs; + layer->nb_outputs = nb_outputs; + return 0; +} + +int conv2d_init(Conv2dLayer *layer, const WeightArray *arrays, + const char *bias, + const char *float_weights, + int in_channels, + int out_channels, + int ktime, + int kheight) +{ + int err; + layer->bias = NULL; + layer->float_weights = NULL; + if (bias != NULL) { + if ((layer->bias = find_array_check(arrays, bias, out_channels*sizeof(layer->bias[0]))) == NULL) return 1; + } + if (float_weights != NULL) { + layer->float_weights = opt_array_check(arrays, float_weights, in_channels*out_channels*ktime*kheight*sizeof(layer->float_weights[0]), &err); + if (err) return 1; + } + layer->in_channels = in_channels; + layer->out_channels = out_channels; + layer->ktime = ktime; + layer->kheight = kheight; + return 0; +} + + + +#if 0 +#include +#include +#include +#include +#include + +int main() +{ + int fd; + void *data; + int len; + int nb_arrays; + int i; + WeightArray *list; + struct stat st; + const char *filename = "weights_blob.bin"; + stat(filename, &st); + len = st.st_size; + fd = open(filename, O_RDONLY); + data = mmap(NULL, len, PROT_READ, MAP_SHARED, fd, 0); + printf("size is %d\n", len); + nb_arrays = parse_weights(&list, data, len); + for (i=0;i0) + { + opus_val16 num; + opus_val32 xcorr16; + xcorr16 = EXTRACT16(VSHR32(xcorr[i], xshift)); +#ifndef FIXED_POINT + /* Considering the range of xcorr16, this should avoid both underflows + and overflows (inf) when squaring xcorr16 */ + xcorr16 *= 1e-12f; +#endif + num = MULT16_16_Q15(xcorr16,xcorr16); + if (MULT16_32_Q15(num,best_den[1]) > MULT16_32_Q15(best_num[1],Syy)) + { + if (MULT16_32_Q15(num,best_den[0]) > MULT16_32_Q15(best_num[0],Syy)) + { + best_num[1] = best_num[0]; + best_den[1] = best_den[0]; + best_pitch[1] = best_pitch[0]; + best_num[0] = num; + best_den[0] = Syy; + best_pitch[0] = i; + } else { + best_num[1] = num; + best_den[1] = Syy; + best_pitch[1] = i; + } + } + } + Syy += SHR32(MULT16_16(y[i+len],y[i+len]),yshift) - SHR32(MULT16_16(y[i],y[i]),yshift); + Syy = MAX32(1, Syy); + } +} + +static void celt_fir5(const opus_val16 *x, + const opus_val16 *num, + opus_val16 *y, + int N, + opus_val16 *mem) +{ + int i; + opus_val16 num0, num1, num2, num3, num4; + opus_val32 mem0, mem1, mem2, mem3, mem4; + num0=num[0]; + num1=num[1]; + num2=num[2]; + num3=num[3]; + num4=num[4]; + mem0=mem[0]; + mem1=mem[1]; + mem2=mem[2]; + mem3=mem[3]; + mem4=mem[4]; + for (i=0;i>1;i++) + x_lp[i] = SHR32(HALF32(HALF32(x[0][(2*i-1)]+x[0][(2*i+1)])+x[0][2*i]), shift); + x_lp[0] = SHR32(HALF32(HALF32(x[0][1])+x[0][0]), shift); + if (C==2) + { + for (i=1;i>1;i++) + x_lp[i] += SHR32(HALF32(HALF32(x[1][(2*i-1)]+x[1][(2*i+1)])+x[1][2*i]), shift); + x_lp[0] += SHR32(HALF32(HALF32(x[1][1])+x[1][0]), shift); + } + + rnn_autocorr(x_lp, ac, NULL, 0, + 4, len>>1); + + /* Noise floor -40 dB */ +#ifdef FIXED_POINT + ac[0] += SHR32(ac[0],13); +#else + ac[0] *= 1.0001f; +#endif + /* Lag windowing */ + for (i=1;i<=4;i++) + { + /*ac[i] *= exp(-.5*(2*M_PI*.002*i)*(2*M_PI*.002*i));*/ +#ifdef FIXED_POINT + ac[i] -= MULT16_32_Q15(2*i*i, ac[i]); +#else + ac[i] -= ac[i]*(.008f*i)*(.008f*i); +#endif + } + + rnn_lpc(lpc, ac, 4); + for (i=0;i<4;i++) + { + tmp = MULT16_16_Q15(QCONST16(.9f,15), tmp); + lpc[i] = MULT16_16_Q15(lpc[i], tmp); + } + /* Add a zero */ + lpc2[0] = lpc[0] + QCONST16(.8f,SIG_SHIFT); + lpc2[1] = lpc[1] + MULT16_16_Q15(c1,lpc[0]); + lpc2[2] = lpc[2] + MULT16_16_Q15(c1,lpc[1]); + lpc2[3] = lpc[3] + MULT16_16_Q15(c1,lpc[2]); + lpc2[4] = MULT16_16_Q15(c1,lpc[3]); + celt_fir5(x_lp, lpc2, x_lp, len>>1, mem); +} + +void rnn_pitch_xcorr(const opus_val16 *_x, const opus_val16 *_y, + opus_val32 *xcorr, int len, int max_pitch) +{ + +#if 0 /* This is a simple version of the pitch correlation that should work + well on DSPs like Blackfin and TI C5x/C6x */ + int i, j; +#ifdef FIXED_POINT + opus_val32 maxcorr=1; +#endif + for (i=0;i0); + celt_assert((((unsigned char *)_x-(unsigned char *)NULL)&3)==0); + for (i=0;i>2]; + opus_val16 y_lp4[(PITCH_FRAME_SIZE+PITCH_MAX_PERIOD)>>2]; + opus_val32 xcorr[PITCH_MAX_PERIOD>>1]; + + celt_assert(len <= PITCH_FRAME_SIZE); + celt_assert(max_pitch <= PITCH_MAX_PERIOD); + celt_assert(len>0); + celt_assert(max_pitch>0); + lag = len+max_pitch; + + + /* Downsample by 2 again */ + for (j=0;j>2;j++) + x_lp4[j] = x_lp[2*j]; + for (j=0;j>2;j++) + y_lp4[j] = y[2*j]; + +#ifdef FIXED_POINT + xmax = celt_maxabs16(x_lp4, len>>2); + ymax = celt_maxabs16(y_lp4, lag>>2); + shift = celt_ilog2(MAX32(1, MAX32(xmax, ymax)))-11; + if (shift>0) + { + for (j=0;j>2;j++) + x_lp4[j] = SHR16(x_lp4[j], shift); + for (j=0;j>2;j++) + y_lp4[j] = SHR16(y_lp4[j], shift); + /* Use double the shift for a MAC */ + shift *= 2; + } else { + shift = 0; + } +#endif + + /* Coarse search with 4x decimation */ + +#ifdef FIXED_POINT + maxcorr = +#endif + rnn_pitch_xcorr(x_lp4, y_lp4, xcorr, len>>2, max_pitch>>2); + + find_best_pitch(xcorr, y_lp4, len>>2, max_pitch>>2, best_pitch +#ifdef FIXED_POINT + , 0, maxcorr +#endif + ); + + /* Finer search with 2x decimation */ +#ifdef FIXED_POINT + maxcorr=1; +#endif + for (i=0;i>1;i++) + { + opus_val32 sum; + xcorr[i] = 0; + if (abs(i-2*best_pitch[0])>2 && abs(i-2*best_pitch[1])>2) + continue; +#ifdef FIXED_POINT + sum = 0; + for (j=0;j>1;j++) + sum += SHR32(MULT16_16(x_lp[j],y[i+j]), shift); +#else + sum = celt_inner_prod(x_lp, y+i, len>>1); +#endif + xcorr[i] = MAX32(-1, sum); +#ifdef FIXED_POINT + maxcorr = MAX32(maxcorr, sum); +#endif + } + find_best_pitch(xcorr, y, len>>1, max_pitch>>1, best_pitch +#ifdef FIXED_POINT + , shift+1, maxcorr +#endif + ); + + /* Refine by pseudo-interpolation */ + if (best_pitch[0]>0 && best_pitch[0]<(max_pitch>>1)-1) + { + opus_val32 a, b, c; + a = xcorr[best_pitch[0]-1]; + b = xcorr[best_pitch[0]]; + c = xcorr[best_pitch[0]+1]; + if ((c-a) > MULT16_32_Q15(QCONST16(.7f,15),b-a)) + offset = 1; + else if ((a-c) > MULT16_32_Q15(QCONST16(.7f,15),b-c)) + offset = -1; + else + offset = 0; + } else { + offset = 0; + } + *pitch = 2*best_pitch[0]-offset; +} + +#ifdef FIXED_POINT +static opus_val16 compute_pitch_gain(opus_val32 xy, opus_val32 xx, opus_val32 yy) +{ + opus_val32 x2y2; + int sx, sy, shift; + opus_val32 g; + opus_val16 den; + if (xy == 0 || xx == 0 || yy == 0) + return 0; + sx = celt_ilog2(xx)-14; + sy = celt_ilog2(yy)-14; + shift = sx + sy; + x2y2 = SHR32(MULT16_16(VSHR32(xx, sx), VSHR32(yy, sy)), 14); + if (shift & 1) { + if (x2y2 < 32768) + { + x2y2 <<= 1; + shift--; + } else { + x2y2 >>= 1; + shift++; + } + } + den = celt_rsqrt_norm(x2y2); + g = MULT16_32_Q15(den, xy); + g = VSHR32(g, (shift>>1)-1); + return EXTRACT16(MIN32(g, Q15ONE)); +} +#else +static opus_val16 compute_pitch_gain(opus_val32 xy, opus_val32 xx, opus_val32 yy) +{ + return xy/sqrt(1+xx*yy); +} +#endif + +static const int second_check[16] = {0, 0, 3, 2, 3, 2, 5, 2, 3, 2, 3, 2, 5, 2, 3, 2}; +opus_val16 rnn_remove_doubling(opus_val16 *x, int maxperiod, int minperiod, + int N, int *T0_, int prev_period, opus_val16 prev_gain) +{ + int k, i, T, T0; + opus_val16 g, g0; + opus_val16 pg; + opus_val32 xy,xx,yy,xy2; + opus_val32 xcorr[3]; + opus_val32 best_xy, best_yy; + int offset; + int minperiod0; + opus_val32 yy_lookup[PITCH_MAX_PERIOD+1]; + + celt_assert(maxperiod <= PITCH_MAX_PERIOD); + + minperiod0 = minperiod; + maxperiod /= 2; + minperiod /= 2; + *T0_ /= 2; + prev_period /= 2; + N /= 2; + x += maxperiod; + if (*T0_>=maxperiod) + *T0_=maxperiod-1; + + T = T0 = *T0_; + dual_inner_prod(x, x, x-T0, N, &xx, &xy); + yy_lookup[0] = xx; + yy=xx; + for (i=1;i<=maxperiod;i++) + { + yy = yy+MULT16_16(x[-i],x[-i])-MULT16_16(x[N-i],x[N-i]); + yy_lookup[i] = MAX32(0, yy); + } + yy = yy_lookup[T0]; + best_xy = xy; + best_yy = yy; + g = g0 = compute_pitch_gain(xy, xx, yy); + /* Look for any pitch at T/k */ + for (k=2;k<=15;k++) + { + int T1, T1b; + opus_val16 g1; + opus_val16 cont=0; + opus_val16 thresh; + T1 = (2*T0+k)/(2*k); + if (T1 < minperiod) + break; + /* Look for another strong correlation at T1b */ + if (k==2) + { + if (T1+T0>maxperiod) + T1b = T0; + else + T1b = T0+T1; + } else + { + T1b = (2*second_check[k]*T0+k)/(2*k); + } + dual_inner_prod(x, &x[-T1], &x[-T1b], N, &xy, &xy2); + xy = HALF32(xy + xy2); + yy = HALF32(yy_lookup[T1] + yy_lookup[T1b]); + g1 = compute_pitch_gain(xy, xx, yy); + if (abs(T1-prev_period)<=1) + cont = prev_gain; + else if (abs(T1-prev_period)<=2 && 5*k*k < T0) + cont = HALF16(prev_gain); + else + cont = 0; + thresh = MAX16(QCONST16(.3f,15), MULT16_16_Q15(QCONST16(.7f,15),g0)-cont); + /* Bias against very high pitch (very short period) to avoid false-positives + due to short-term correlation */ + if (T1<3*minperiod) + thresh = MAX16(QCONST16(.4f,15), MULT16_16_Q15(QCONST16(.85f,15),g0)-cont); + else if (T1<2*minperiod) + thresh = MAX16(QCONST16(.5f,15), MULT16_16_Q15(QCONST16(.9f,15),g0)-cont); + if (g1 > thresh) + { + best_xy = xy; + best_yy = yy; + T = T1; + g = g1; + } + } + best_xy = MAX32(0, best_xy); + if (best_yy <= best_xy) + pg = Q15ONE; + else + pg = best_xy/(best_yy+1); + + for (k=0;k<3;k++) + xcorr[k] = celt_inner_prod(x, x-(T+k-1), N); + if ((xcorr[2]-xcorr[0]) > MULT16_32_Q15(QCONST16(.7f,15),xcorr[1]-xcorr[0])) + offset = 1; + else if ((xcorr[0]-xcorr[2]) > MULT16_32_Q15(QCONST16(.7f,15),xcorr[1]-xcorr[2])) + offset = -1; + else + offset = 0; + if (pg > g) + pg = g; + *T0_ = 2*T+offset; + + if (*T0_=3); + y_3=0; /* gcc doesn't realize that y_3 can't be used uninitialized */ + y_0=*y++; + y_1=*y++; + y_2=*y++; + for (j=0;j +#include "opus_types.h" +#include "common.h" +#include "arch.h" +#include "rnn.h" +#include "rnnoise_data.h" +#include + + +#define INPUT_SIZE 42 + + +void compute_rnn(const RNNoise *model, RNNState *rnn, float *gains, float *vad, const float *input, int arch) { + float tmp[MAX_NEURONS]; + float cat[CONV2_OUT_SIZE + GRU1_OUT_SIZE + GRU2_OUT_SIZE + GRU3_OUT_SIZE]; + /*for (int i=0;iconv1, tmp, rnn->conv1_state, input, CONV1_IN_SIZE, ACTIVATION_TANH, arch); + compute_generic_conv1d(&model->conv2, cat, rnn->conv2_state, tmp, CONV2_IN_SIZE, ACTIVATION_TANH, arch); + compute_generic_gru(&model->gru1_input, &model->gru1_recurrent, rnn->gru1_state, cat, arch); + compute_generic_gru(&model->gru2_input, &model->gru2_recurrent, rnn->gru2_state, rnn->gru1_state, arch); + compute_generic_gru(&model->gru3_input, &model->gru3_recurrent, rnn->gru3_state, rnn->gru2_state, arch); + RNN_COPY(&cat[CONV2_OUT_SIZE], rnn->gru1_state, GRU1_OUT_SIZE); + RNN_COPY(&cat[CONV2_OUT_SIZE+GRU1_OUT_SIZE], rnn->gru2_state, GRU2_OUT_SIZE); + RNN_COPY(&cat[CONV2_OUT_SIZE+GRU1_OUT_SIZE+GRU2_OUT_SIZE], rnn->gru3_state, GRU3_OUT_SIZE); + compute_generic_dense(&model->dense_out, gains, cat, ACTIVATION_SIGMOID, arch); + compute_generic_dense(&model->vad_dense, vad, cat, ACTIVATION_SIGMOID, arch); + /*for (int i=0;i<22;i++) printf("%f ", gains[i]);printf("\n");*/ + /*printf("%f\n", *vad);*/ +} diff --git a/cpp/ax650/src/rnnoise/rnn.h b/cpp/ax650/src/rnnoise/rnn.h new file mode 100644 index 0000000000000000000000000000000000000000..60a8871e7a89161e57c037cf89e2ecb6035269e3 --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnn.h @@ -0,0 +1,49 @@ +/* Copyright (c) 2017 Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef RNN_H_ +#define RNN_H_ + +#include "rnnoise.h" +#include "rnnoise_data.h" + +#include "opus_types.h" + +#define WEIGHTS_SCALE (1.f/256) + +#define MAX_NEURONS 1024 + + +typedef struct { + float conv1_state[CONV1_STATE_SIZE]; + float conv2_state[CONV2_STATE_SIZE]; + float gru1_state[GRU1_STATE_SIZE]; + float gru2_state[GRU2_STATE_SIZE]; + float gru3_state[GRU3_STATE_SIZE]; +} RNNState; +void compute_rnn(const RNNoise *model, RNNState *rnn, float *gains, float *vad, const float *input, int arch); + +#endif /* RNN_H_ */ diff --git a/cpp/ax650/src/rnnoise/rnn_train.py b/cpp/ax650/src/rnnoise/rnn_train.py new file mode 100644 index 0000000000000000000000000000000000000000..b561963d9c42aa8c58135daf72a4b3622ba9ce29 --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnn_train.py @@ -0,0 +1,66 @@ +#!/usr/bin/python + +from __future__ import print_function + +from keras.models import Sequential +from keras.models import Model +from keras.layers import Input +from keras.layers import Dense +from keras.layers import LSTM +from keras.layers import GRU +from keras.layers import SimpleRNN +from keras.layers import Dropout +from keras import losses +import h5py + +from keras import backend as K +import numpy as np + +print('Build model...') +main_input = Input(shape=(None, 22), name='main_input') +#x = Dense(44, activation='relu')(main_input) +#x = GRU(44, dropout=0.0, recurrent_dropout=0.0, activation='tanh', recurrent_activation='sigmoid', return_sequences=True)(x) +x=main_input +x = GRU(128, activation='tanh', recurrent_activation='sigmoid', return_sequences=True)(x) +#x = GRU(128, return_sequences=True)(x) +#x = GRU(22, activation='relu', return_sequences=True)(x) +x = Dense(22, activation='sigmoid')(x) +#x = Dense(22, activation='softplus')(x) +model = Model(inputs=main_input, outputs=x) + +batch_size = 32 + +print('Loading data...') +with h5py.File('denoise_data.h5', 'r') as hf: + all_data = hf['denoise_data'][:] +print('done.') + +window_size = 500 + +nb_sequences = len(all_data)//window_size +print(nb_sequences, ' sequences') +x_train = all_data[:nb_sequences*window_size, :-22] +x_train = np.reshape(x_train, (nb_sequences, window_size, 22)) + +y_train = np.copy(all_data[:nb_sequences*window_size, -22:]) +y_train = np.reshape(y_train, (nb_sequences, window_size, 22)) + +#y_train = -20*np.log10(np.add(y_train, .03)); + +all_data = 0; +x_train = x_train.astype('float32') +y_train = y_train.astype('float32') + +print(len(x_train), 'train sequences. x shape =', x_train.shape, 'y shape = ', y_train.shape) + +# try using different optimizers and different optimizer configs +model.compile(loss='mean_squared_error', + optimizer='adam', + metrics=['binary_accuracy']) + +print('Train...') +model.fit(x_train, y_train, + batch_size=batch_size, + epochs=200, + validation_data=(x_train, y_train)) +model.save("newweights.hdf5") diff --git a/cpp/ax650/src/rnnoise/rnnoise.h b/cpp/ax650/src/rnnoise/rnnoise.h new file mode 100644 index 0000000000000000000000000000000000000000..81d8f62341ab39e51da0bd7332b905d79f36b782 --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnnoise.h @@ -0,0 +1,131 @@ +/* Copyright (c) 2018 Gregor Richards + * Copyright (c) 2017 Mozilla */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef RNNOISE_H +#define RNNOISE_H 1 + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +#ifndef RNNOISE_EXPORT +# if defined(WIN32) +# if defined(RNNOISE_BUILD) && defined(DLL_EXPORT) +# define RNNOISE_EXPORT __declspec(dllexport) +# else +# define RNNOISE_EXPORT +# endif +# elif defined(__GNUC__) && defined(RNNOISE_BUILD) +# define RNNOISE_EXPORT __attribute__ ((visibility ("default"))) +# else +# define RNNOISE_EXPORT +# endif +#endif + +typedef struct DenoiseState DenoiseState; +typedef struct RNNModel RNNModel; + +/** + * Return the size of DenoiseState + */ +RNNOISE_EXPORT int rnnoise_get_size(void); + +/** + * Return the number of samples processed by rnnoise_process_frame at a time + */ +RNNOISE_EXPORT int rnnoise_get_frame_size(void); + +/** + * Initializes a pre-allocated DenoiseState + * + * If model is NULL the default model is used. + * + * See: rnnoise_create() and rnnoise_model_from_file() + */ +RNNOISE_EXPORT int rnnoise_init(DenoiseState *st, RNNModel *model); + +/** + * Allocate and initialize a DenoiseState + * + * If model is NULL the default model is used. + * + * The returned pointer MUST be freed with rnnoise_destroy(). + */ +RNNOISE_EXPORT DenoiseState *rnnoise_create(RNNModel *model); + +/** + * Free a DenoiseState produced by rnnoise_create. + * + * The optional custom model must be freed by rnnoise_model_free() after. + */ +RNNOISE_EXPORT void rnnoise_destroy(DenoiseState *st); + +/** + * Denoise a frame of samples + * + * in and out must be at least rnnoise_get_frame_size() large. + */ +RNNOISE_EXPORT float rnnoise_process_frame(DenoiseState *st, float *out, const float *in); + +/** + * Load a model from a memory buffer + * + * It must be deallocated with rnnoise_model_free() and the buffer must remain + * valid until after the returned object is destroyed. + */ +RNNOISE_EXPORT RNNModel *rnnoise_model_from_buffer(const void *ptr, int len); + + +/** + * Load a model from a file + * + * It must be deallocated with rnnoise_model_free() and the file must not be + * closed until the returned object is destroyed. + */ +RNNOISE_EXPORT RNNModel *rnnoise_model_from_file(FILE *f); + +/** + * Load a model from a file name + * + * It must be deallocated with rnnoise_model_free() + */ +RNNOISE_EXPORT RNNModel *rnnoise_model_from_filename(const char *filename); + +/** + * Free a custom model + * + * It must be called after all the DenoiseStates referring to it are freed. + */ +RNNOISE_EXPORT void rnnoise_model_free(RNNModel *model); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/cpp/ax650/src/rnnoise/rnnoise_data.c b/cpp/ax650/src/rnnoise/rnnoise_data.c new file mode 100644 index 0000000000000000000000000000000000000000..d45db0f95308829e5d31840c64687e35c76344b4 --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnnoise_data.c @@ -0,0 +1,18 @@ +/* RNNoise 网络权重占位(SDK 版)。 + * + * 本交付包的网络权重已内嵌到 AXMODEL 中,compute_rnn 由 AX Engine 推理替换, + * 不再需要原版 rnnoise_data.c 的 75MB 权重数组。此处仅保留空权重表与 + * init_rnnoise 空实现(memset 清零),以保持原版 denoise.c 的初始化链路。 + */ +#include + +#include "rnnoise_data.h" + +const WeightArray rnnoise_arrays[] = {{NULL, 0, 0, NULL}}; +const WeightArray n[] = {{NULL, 0, 0, NULL}}; + +int init_rnnoise(RNNoise *model, const WeightArray *arrays) { + (void)arrays; + memset(model, 0, sizeof(*model)); + return 0; +} diff --git a/cpp/ax650/src/rnnoise/rnnoise_data.h b/cpp/ax650/src/rnnoise/rnnoise_data.h new file mode 100644 index 0000000000000000000000000000000000000000..7955a7c95b1531a5e2a5a19fdce3d4354896695b --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnnoise_data.h @@ -0,0 +1,55 @@ + +#ifndef RNNOISE_DATA_H +#define RNNOISE_DATA_H + +#include "nnet.h" + + +#define CONV1_OUT_SIZE 128 + +#define CONV1_IN_SIZE 65 + +#define CONV1_STATE_SIZE (65 * (2)) + +#define CONV1_DELAY 1 + +#define CONV2_OUT_SIZE 384 + +#define CONV2_IN_SIZE 128 + +#define CONV2_STATE_SIZE (128 * (2)) + +#define CONV2_DELAY 1 + +#define GRU1_OUT_SIZE 384 + +#define GRU1_STATE_SIZE 384 + +#define GRU2_OUT_SIZE 384 + +#define GRU2_STATE_SIZE 384 + +#define GRU3_OUT_SIZE 384 + +#define GRU3_STATE_SIZE 384 + +#define DENSE_OUT_OUT_SIZE 32 + +#define VAD_DENSE_OUT_SIZE 1 + +typedef struct { + LinearLayer conv1; + LinearLayer conv2; + LinearLayer gru1_input; + LinearLayer gru1_recurrent; + LinearLayer gru2_input; + LinearLayer gru2_recurrent; + LinearLayer gru3_input; + LinearLayer gru3_recurrent; + LinearLayer dense_out; + LinearLayer vad_dense; +} RNNoise; + +int init_rnnoise(RNNoise *model, const WeightArray *arrays); + +#endif /* RNNOISE_DATA_H */ diff --git a/cpp/ax650/src/rnnoise/rnnoise_tables.c b/cpp/ax650/src/rnnoise/rnnoise_tables.c new file mode 100644 index 0000000000000000000000000000000000000000..adc8a28f451274bbd2156c13a3dfa883ac5d0f82 --- /dev/null +++ b/cpp/ax650/src/rnnoise/rnnoise_tables.c @@ -0,0 +1,874 @@ +/* The contents of this file was automatically generated by dump_rnnoise_tables.c*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif +#include "kiss_fft.h" + +static const arch_fft_state arch_fft = {0, NULL}; + +static const opus_int32 fft_bitrev[960] = { +0, 192, 384, 576, 768, 64, 256, 448, 640, 832, 128, 320, 512, 704, 896, +16, 208, 400, 592, 784, 80, 272, 464, 656, 848, 144, 336, 528, 720, 912, +32, 224, 416, 608, 800, 96, 288, 480, 672, 864, 160, 352, 544, 736, 928, +48, 240, 432, 624, 816, 112, 304, 496, 688, 880, 176, 368, 560, 752, 944, +4, 196, 388, 580, 772, 68, 260, 452, 644, 836, 132, 324, 516, 708, 900, +20, 212, 404, 596, 788, 84, 276, 468, 660, 852, 148, 340, 532, 724, 916, +36, 228, 420, 612, 804, 100, 292, 484, 676, 868, 164, 356, 548, 740, 932, +52, 244, 436, 628, 820, 116, 308, 500, 692, 884, 180, 372, 564, 756, 948, +8, 200, 392, 584, 776, 72, 264, 456, 648, 840, 136, 328, 520, 712, 904, +24, 216, 408, 600, 792, 88, 280, 472, 664, 856, 152, 344, 536, 728, 920, +40, 232, 424, 616, 808, 104, 296, 488, 680, 872, 168, 360, 552, 744, 936, +56, 248, 440, 632, 824, 120, 312, 504, 696, 888, 184, 376, 568, 760, 952, +12, 204, 396, 588, 780, 76, 268, 460, 652, 844, 140, 332, 524, 716, 908, +28, 220, 412, 604, 796, 92, 284, 476, 668, 860, 156, 348, 540, 732, 924, +44, 236, 428, 620, 812, 108, 300, 492, 684, 876, 172, 364, 556, 748, 940, +60, 252, 444, 636, 828, 124, 316, 508, 700, 892, 188, 380, 572, 764, 956, +1, 193, 385, 577, 769, 65, 257, 449, 641, 833, 129, 321, 513, 705, 897, +17, 209, 401, 593, 785, 81, 273, 465, 657, 849, 145, 337, 529, 721, 913, +33, 225, 417, 609, 801, 97, 289, 481, 673, 865, 161, 353, 545, 737, 929, +49, 241, 433, 625, 817, 113, 305, 497, 689, 881, 177, 369, 561, 753, 945, +5, 197, 389, 581, 773, 69, 261, 453, 645, 837, 133, 325, 517, 709, 901, +21, 213, 405, 597, 789, 85, 277, 469, 661, 853, 149, 341, 533, 725, 917, +37, 229, 421, 613, 805, 101, 293, 485, 677, 869, 165, 357, 549, 741, 933, +53, 245, 437, 629, 821, 117, 309, 501, 693, 885, 181, 373, 565, 757, 949, +9, 201, 393, 585, 777, 73, 265, 457, 649, 841, 137, 329, 521, 713, 905, +25, 217, 409, 601, 793, 89, 281, 473, 665, 857, 153, 345, 537, 729, 921, +41, 233, 425, 617, 809, 105, 297, 489, 681, 873, 169, 361, 553, 745, 937, +57, 249, 441, 633, 825, 121, 313, 505, 697, 889, 185, 377, 569, 761, 953, +13, 205, 397, 589, 781, 77, 269, 461, 653, 845, 141, 333, 525, 717, 909, +29, 221, 413, 605, 797, 93, 285, 477, 669, 861, 157, 349, 541, 733, 925, +45, 237, 429, 621, 813, 109, 301, 493, 685, 877, 173, 365, 557, 749, 941, +61, 253, 445, 637, 829, 125, 317, 509, 701, 893, 189, 381, 573, 765, 957, +2, 194, 386, 578, 770, 66, 258, 450, 642, 834, 130, 322, 514, 706, 898, +18, 210, 402, 594, 786, 82, 274, 466, 658, 850, 146, 338, 530, 722, 914, +34, 226, 418, 610, 802, 98, 290, 482, 674, 866, 162, 354, 546, 738, 930, +50, 242, 434, 626, 818, 114, 306, 498, 690, 882, 178, 370, 562, 754, 946, +6, 198, 390, 582, 774, 70, 262, 454, 646, 838, 134, 326, 518, 710, 902, +22, 214, 406, 598, 790, 86, 278, 470, 662, 854, 150, 342, 534, 726, 918, +38, 230, 422, 614, 806, 102, 294, 486, 678, 870, 166, 358, 550, 742, 934, +54, 246, 438, 630, 822, 118, 310, 502, 694, 886, 182, 374, 566, 758, 950, +10, 202, 394, 586, 778, 74, 266, 458, 650, 842, 138, 330, 522, 714, 906, +26, 218, 410, 602, 794, 90, 282, 474, 666, 858, 154, 346, 538, 730, 922, +42, 234, 426, 618, 810, 106, 298, 490, 682, 874, 170, 362, 554, 746, 938, +58, 250, 442, 634, 826, 122, 314, 506, 698, 890, 186, 378, 570, 762, 954, +14, 206, 398, 590, 782, 78, 270, 462, 654, 846, 142, 334, 526, 718, 910, +30, 222, 414, 606, 798, 94, 286, 478, 670, 862, 158, 350, 542, 734, 926, +46, 238, 430, 622, 814, 110, 302, 494, 686, 878, 174, 366, 558, 750, 942, +62, 254, 446, 638, 830, 126, 318, 510, 702, 894, 190, 382, 574, 766, 958, +3, 195, 387, 579, 771, 67, 259, 451, 643, 835, 131, 323, 515, 707, 899, +19, 211, 403, 595, 787, 83, 275, 467, 659, 851, 147, 339, 531, 723, 915, +35, 227, 419, 611, 803, 99, 291, 483, 675, 867, 163, 355, 547, 739, 931, +51, 243, 435, 627, 819, 115, 307, 499, 691, 883, 179, 371, 563, 755, 947, +7, 199, 391, 583, 775, 71, 263, 455, 647, 839, 135, 327, 519, 711, 903, +23, 215, 407, 599, 791, 87, 279, 471, 663, 855, 151, 343, 535, 727, 919, +39, 231, 423, 615, 807, 103, 295, 487, 679, 871, 167, 359, 551, 743, 935, +55, 247, 439, 631, 823, 119, 311, 503, 695, 887, 183, 375, 567, 759, 951, +11, 203, 395, 587, 779, 75, 267, 459, 651, 843, 139, 331, 523, 715, 907, +27, 219, 411, 603, 795, 91, 283, 475, 667, 859, 155, 347, 539, 731, 923, +43, 235, 427, 619, 811, 107, 299, 491, 683, 875, 171, 363, 555, 747, 939, +59, 251, 443, 635, 827, 123, 315, 507, 699, 891, 187, 379, 571, 763, 955, +15, 207, 399, 591, 783, 79, 271, 463, 655, 847, 143, 335, 527, 719, 911, +31, 223, 415, 607, 799, 95, 287, 479, 671, 863, 159, 351, 543, 735, 927, +47, 239, 431, 623, 815, 111, 303, 495, 687, 879, 175, 367, 559, 751, 943, +63, 255, 447, 639, 831, 127, 319, 511, 703, 895, 191, 383, 575, 767, 959, +}; + +static const kiss_twiddle_cpx fft_twiddles[960] = { +{1.00000000f, -0.00000000f}, {0.999978602f, -0.00654493785f}, +{0.999914348f, -0.0130895954f}, {0.999807239f, -0.0196336918f}, +{0.999657333f, -0.0261769481f}, {0.999464571f, -0.0327190831f}, +{0.999229014f, -0.0392598175f}, {0.998950660f, -0.0457988679f}, +{0.998629510f, -0.0523359552f}, {0.998265624f, -0.0588708036f}, +{0.997858942f, -0.0654031262f}, {0.997409463f, -0.0719326511f}, +{0.996917307f, -0.0784590989f}, {0.996382475f, -0.0849821791f}, +{0.995804906f, -0.0915016159f}, {0.995184720f, -0.0980171412f}, +{0.994521916f, -0.104528464f}, {0.993816435f, -0.111035310f}, +{0.993068457f, -0.117537394f}, {0.992277920f, -0.124034449f}, +{0.991444886f, -0.130526185f}, {0.990569353f, -0.137012348f}, +{0.989651382f, -0.143492624f}, {0.988691032f, -0.149966761f}, +{0.987688363f, -0.156434461f}, {0.986643314f, -0.162895471f}, +{0.985556066f, -0.169349506f}, {0.984426558f, -0.175796285f}, +{0.983254910f, -0.182235524f}, {0.982041121f, -0.188666970f}, +{0.980785251f, -0.195090324f}, {0.979487419f, -0.201505318f}, +{0.978147626f, -0.207911685f}, {0.976765871f, -0.214309156f}, +{0.975342333f, -0.220697433f}, {0.973876953f, -0.227076262f}, +{0.972369909f, -0.233445361f}, {0.970821202f, -0.239804462f}, +{0.969230890f, -0.246153295f}, {0.967599094f, -0.252491564f}, +{0.965925813f, -0.258819044f}, {0.964211166f, -0.265135437f}, +{0.962455213f, -0.271440446f}, {0.960658073f, -0.277733833f}, +{0.958819747f, -0.284015357f}, {0.956940353f, -0.290284663f}, +{0.955019951f, -0.296541572f}, {0.953058660f, -0.302785784f}, +{0.951056540f, -0.309017003f}, {0.949013650f, -0.315234989f}, +{0.946930110f, -0.321439475f}, {0.944806039f, -0.327630192f}, +{0.942641497f, -0.333806872f}, {0.940436542f, -0.339969248f}, +{0.938191354f, -0.346117049f}, {0.935905933f, -0.352250040f}, +{0.933580399f, -0.358367950f}, {0.931214929f, -0.364470512f}, +{0.928809524f, -0.370557427f}, {0.926364362f, -0.376628488f}, +{0.923879504f, -0.382683426f}, {0.921355128f, -0.388721973f}, +{0.918791234f, -0.394743860f}, {0.916187942f, -0.400748819f}, +{0.913545430f, -0.406736642f}, {0.910863817f, -0.412707031f}, +{0.908143163f, -0.418659747f}, {0.905383646f, -0.424594522f}, +{0.902585268f, -0.430511087f}, {0.899748266f, -0.436409235f}, +{0.896872759f, -0.442288697f}, {0.893958807f, -0.448149204f}, +{0.891006529f, -0.453990489f}, {0.888016105f, -0.459812373f}, +{0.884987652f, -0.465614527f}, {0.881921291f, -0.471396744f}, +{0.878817141f, -0.477158755f}, {0.875675321f, -0.482900351f}, +{0.872496009f, -0.488621235f}, {0.869279325f, -0.494321197f}, +{0.866025388f, -0.500000000f}, {0.862734377f, -0.505657375f}, +{0.859406412f, -0.511293113f}, {0.856041610f, -0.516906917f}, +{0.852640152f, -0.522498548f}, {0.849202156f, -0.528067827f}, +{0.845727801f, -0.533614516f}, {0.842217207f, -0.539138317f}, +{0.838670552f, -0.544639051f}, {0.835087955f, -0.550116420f}, +{0.831469595f, -0.555570245f}, {0.827815652f, -0.561000228f}, +{0.824126184f, -0.566406250f}, {0.820401430f, -0.571787953f}, +{0.816641569f, -0.577145219f}, {0.812846661f, -0.582477689f}, +{0.809017003f, -0.587785244f}, {0.805152655f, -0.593067646f}, +{0.801253796f, -0.598324597f}, {0.797320664f, -0.603555918f}, +{0.793353319f, -0.608761430f}, {0.789352059f, -0.613940835f}, +{0.785316944f, -0.619093955f}, {0.781248152f, -0.624220550f}, +{0.777145982f, -0.629320383f}, {0.773010433f, -0.634393275f}, +{0.768841803f, -0.639438987f}, {0.764640272f, -0.644457340f}, +{0.760405958f, -0.649448037f}, {0.756139100f, -0.654410958f}, +{0.751839817f, -0.659345806f}, {0.747508347f, -0.664252460f}, +{0.743144810f, -0.669130623f}, {0.738749504f, -0.673980117f}, +{0.734322488f, -0.678800762f}, {0.729864061f, -0.683592319f}, +{0.725374401f, -0.688354552f}, {0.720853567f, -0.693087339f}, +{0.716301918f, -0.697790444f}, {0.711719632f, -0.702463686f}, +{0.707106769f, -0.707106769f}, {0.702463686f, -0.711719632f}, +{0.697790444f, -0.716301918f}, {0.693087339f, -0.720853567f}, +{0.688354552f, -0.725374401f}, {0.683592319f, -0.729864061f}, +{0.678800762f, -0.734322488f}, {0.673980117f, -0.738749504f}, +{0.669130623f, -0.743144810f}, {0.664252460f, -0.747508347f}, +{0.659345806f, -0.751839817f}, {0.654410958f, -0.756139100f}, +{0.649448037f, -0.760405958f}, {0.644457340f, -0.764640272f}, +{0.639438987f, -0.768841803f}, {0.634393275f, -0.773010433f}, +{0.629320383f, -0.777145982f}, {0.624220550f, -0.781248152f}, +{0.619093955f, -0.785316944f}, {0.613940835f, -0.789352059f}, +{0.608761430f, -0.793353319f}, {0.603555918f, -0.797320664f}, +{0.598324597f, -0.801253796f}, {0.593067646f, -0.805152655f}, +{0.587785244f, -0.809017003f}, {0.582477689f, -0.812846661f}, +{0.577145219f, -0.816641569f}, {0.571787953f, -0.820401430f}, +{0.566406250f, -0.824126184f}, {0.561000228f, -0.827815652f}, +{0.555570245f, -0.831469595f}, {0.550116420f, -0.835087955f}, +{0.544639051f, -0.838670552f}, {0.539138317f, -0.842217207f}, +{0.533614516f, -0.845727801f}, {0.528067827f, -0.849202156f}, +{0.522498548f, -0.852640152f}, {0.516906917f, -0.856041610f}, +{0.511293113f, -0.859406412f}, {0.505657375f, -0.862734377f}, +{0.500000000f, -0.866025388f}, {0.494321197f, -0.869279325f}, +{0.488621235f, -0.872496009f}, {0.482900351f, -0.875675321f}, +{0.477158755f, -0.878817141f}, {0.471396744f, -0.881921291f}, +{0.465614527f, -0.884987652f}, {0.459812373f, -0.888016105f}, +{0.453990489f, -0.891006529f}, {0.448149204f, -0.893958807f}, +{0.442288697f, -0.896872759f}, {0.436409235f, -0.899748266f}, +{0.430511087f, -0.902585268f}, {0.424594522f, -0.905383646f}, +{0.418659747f, -0.908143163f}, {0.412707031f, -0.910863817f}, +{0.406736642f, -0.913545430f}, {0.400748819f, -0.916187942f}, +{0.394743860f, -0.918791234f}, {0.388721973f, -0.921355128f}, +{0.382683426f, -0.923879504f}, {0.376628488f, -0.926364362f}, +{0.370557427f, -0.928809524f}, {0.364470512f, -0.931214929f}, +{0.358367950f, -0.933580399f}, {0.352250040f, -0.935905933f}, +{0.346117049f, -0.938191354f}, {0.339969248f, -0.940436542f}, +{0.333806872f, -0.942641497f}, {0.327630192f, -0.944806039f}, +{0.321439475f, -0.946930110f}, {0.315234989f, -0.949013650f}, +{0.309017003f, -0.951056540f}, {0.302785784f, -0.953058660f}, +{0.296541572f, -0.955019951f}, {0.290284663f, -0.956940353f}, +{0.284015357f, -0.958819747f}, {0.277733833f, -0.960658073f}, +{0.271440446f, -0.962455213f}, {0.265135437f, -0.964211166f}, +{0.258819044f, -0.965925813f}, {0.252491564f, -0.967599094f}, +{0.246153295f, -0.969230890f}, {0.239804462f, -0.970821202f}, +{0.233445361f, -0.972369909f}, {0.227076262f, -0.973876953f}, +{0.220697433f, -0.975342333f}, {0.214309156f, -0.976765871f}, +{0.207911685f, -0.978147626f}, {0.201505318f, -0.979487419f}, +{0.195090324f, -0.980785251f}, {0.188666970f, -0.982041121f}, +{0.182235524f, -0.983254910f}, {0.175796285f, -0.984426558f}, +{0.169349506f, -0.985556066f}, {0.162895471f, -0.986643314f}, +{0.156434461f, -0.987688363f}, {0.149966761f, -0.988691032f}, +{0.143492624f, -0.989651382f}, {0.137012348f, -0.990569353f}, +{0.130526185f, -0.991444886f}, {0.124034449f, -0.992277920f}, +{0.117537394f, -0.993068457f}, {0.111035310f, -0.993816435f}, +{0.104528464f, -0.994521916f}, {0.0980171412f, -0.995184720f}, +{0.0915016159f, -0.995804906f}, {0.0849821791f, -0.996382475f}, +{0.0784590989f, -0.996917307f}, {0.0719326511f, -0.997409463f}, +{0.0654031262f, -0.997858942f}, {0.0588708036f, -0.998265624f}, +{0.0523359552f, -0.998629510f}, {0.0457988679f, -0.998950660f}, +{0.0392598175f, -0.999229014f}, {0.0327190831f, -0.999464571f}, +{0.0261769481f, -0.999657333f}, {0.0196336918f, -0.999807239f}, +{0.0130895954f, -0.999914348f}, {0.00654493785f, -0.999978602f}, +{6.12323426e-17f, -1.00000000f}, {-0.00654493785f, -0.999978602f}, +{-0.0130895954f, -0.999914348f}, {-0.0196336918f, -0.999807239f}, +{-0.0261769481f, -0.999657333f}, {-0.0327190831f, -0.999464571f}, +{-0.0392598175f, -0.999229014f}, {-0.0457988679f, -0.998950660f}, +{-0.0523359552f, -0.998629510f}, {-0.0588708036f, -0.998265624f}, +{-0.0654031262f, -0.997858942f}, {-0.0719326511f, -0.997409463f}, +{-0.0784590989f, -0.996917307f}, {-0.0849821791f, -0.996382475f}, +{-0.0915016159f, -0.995804906f}, {-0.0980171412f, -0.995184720f}, +{-0.104528464f, -0.994521916f}, {-0.111035310f, -0.993816435f}, +{-0.117537394f, -0.993068457f}, {-0.124034449f, -0.992277920f}, +{-0.130526185f, -0.991444886f}, {-0.137012348f, -0.990569353f}, +{-0.143492624f, -0.989651382f}, {-0.149966761f, -0.988691032f}, +{-0.156434461f, -0.987688363f}, {-0.162895471f, -0.986643314f}, +{-0.169349506f, -0.985556066f}, {-0.175796285f, -0.984426558f}, +{-0.182235524f, -0.983254910f}, {-0.188666970f, -0.982041121f}, +{-0.195090324f, -0.980785251f}, {-0.201505318f, -0.979487419f}, +{-0.207911685f, -0.978147626f}, {-0.214309156f, -0.976765871f}, +{-0.220697433f, -0.975342333f}, {-0.227076262f, -0.973876953f}, +{-0.233445361f, -0.972369909f}, {-0.239804462f, -0.970821202f}, +{-0.246153295f, -0.969230890f}, {-0.252491564f, -0.967599094f}, +{-0.258819044f, -0.965925813f}, {-0.265135437f, -0.964211166f}, +{-0.271440446f, -0.962455213f}, {-0.277733833f, -0.960658073f}, +{-0.284015357f, -0.958819747f}, {-0.290284663f, -0.956940353f}, +{-0.296541572f, -0.955019951f}, {-0.302785784f, -0.953058660f}, +{-0.309017003f, -0.951056540f}, {-0.315234989f, -0.949013650f}, +{-0.321439475f, -0.946930110f}, {-0.327630192f, -0.944806039f}, +{-0.333806872f, -0.942641497f}, {-0.339969248f, -0.940436542f}, +{-0.346117049f, -0.938191354f}, {-0.352250040f, -0.935905933f}, +{-0.358367950f, -0.933580399f}, {-0.364470512f, -0.931214929f}, +{-0.370557427f, -0.928809524f}, {-0.376628488f, -0.926364362f}, +{-0.382683426f, -0.923879504f}, {-0.388721973f, -0.921355128f}, +{-0.394743860f, -0.918791234f}, {-0.400748819f, -0.916187942f}, +{-0.406736642f, -0.913545430f}, {-0.412707031f, -0.910863817f}, +{-0.418659747f, -0.908143163f}, {-0.424594522f, -0.905383646f}, +{-0.430511087f, -0.902585268f}, {-0.436409235f, -0.899748266f}, +{-0.442288697f, -0.896872759f}, {-0.448149204f, -0.893958807f}, +{-0.453990489f, -0.891006529f}, {-0.459812373f, -0.888016105f}, +{-0.465614527f, -0.884987652f}, {-0.471396744f, -0.881921291f}, +{-0.477158755f, -0.878817141f}, {-0.482900351f, -0.875675321f}, +{-0.488621235f, -0.872496009f}, {-0.494321197f, -0.869279325f}, +{-0.500000000f, -0.866025388f}, {-0.505657375f, -0.862734377f}, +{-0.511293113f, -0.859406412f}, {-0.516906917f, -0.856041610f}, +{-0.522498548f, -0.852640152f}, {-0.528067827f, -0.849202156f}, +{-0.533614516f, -0.845727801f}, {-0.539138317f, -0.842217207f}, +{-0.544639051f, -0.838670552f}, {-0.550116420f, -0.835087955f}, +{-0.555570245f, -0.831469595f}, {-0.561000228f, -0.827815652f}, +{-0.566406250f, -0.824126184f}, {-0.571787953f, -0.820401430f}, +{-0.577145219f, -0.816641569f}, {-0.582477689f, -0.812846661f}, +{-0.587785244f, -0.809017003f}, {-0.593067646f, -0.805152655f}, +{-0.598324597f, -0.801253796f}, {-0.603555918f, -0.797320664f}, +{-0.608761430f, -0.793353319f}, {-0.613940835f, -0.789352059f}, +{-0.619093955f, -0.785316944f}, {-0.624220550f, -0.781248152f}, +{-0.629320383f, -0.777145982f}, {-0.634393275f, -0.773010433f}, +{-0.639438987f, -0.768841803f}, {-0.644457340f, -0.764640272f}, +{-0.649448037f, -0.760405958f}, {-0.654410958f, -0.756139100f}, +{-0.659345806f, -0.751839817f}, {-0.664252460f, -0.747508347f}, +{-0.669130623f, -0.743144810f}, {-0.673980117f, -0.738749504f}, +{-0.678800762f, -0.734322488f}, {-0.683592319f, -0.729864061f}, +{-0.688354552f, -0.725374401f}, {-0.693087339f, -0.720853567f}, +{-0.697790444f, -0.716301918f}, {-0.702463686f, -0.711719632f}, +{-0.707106769f, -0.707106769f}, {-0.711719632f, -0.702463686f}, +{-0.716301918f, -0.697790444f}, {-0.720853567f, -0.693087339f}, +{-0.725374401f, -0.688354552f}, {-0.729864061f, -0.683592319f}, +{-0.734322488f, -0.678800762f}, {-0.738749504f, -0.673980117f}, +{-0.743144810f, -0.669130623f}, {-0.747508347f, -0.664252460f}, +{-0.751839817f, -0.659345806f}, {-0.756139100f, -0.654410958f}, +{-0.760405958f, -0.649448037f}, {-0.764640272f, -0.644457340f}, +{-0.768841803f, -0.639438987f}, {-0.773010433f, -0.634393275f}, +{-0.777145982f, -0.629320383f}, {-0.781248152f, -0.624220550f}, +{-0.785316944f, -0.619093955f}, {-0.789352059f, -0.613940835f}, +{-0.793353319f, -0.608761430f}, {-0.797320664f, -0.603555918f}, +{-0.801253796f, -0.598324597f}, {-0.805152655f, -0.593067646f}, +{-0.809017003f, -0.587785244f}, {-0.812846661f, -0.582477689f}, +{-0.816641569f, -0.577145219f}, {-0.820401430f, -0.571787953f}, +{-0.824126184f, -0.566406250f}, {-0.827815652f, -0.561000228f}, +{-0.831469595f, -0.555570245f}, {-0.835087955f, -0.550116420f}, +{-0.838670552f, -0.544639051f}, {-0.842217207f, -0.539138317f}, +{-0.845727801f, -0.533614516f}, {-0.849202156f, -0.528067827f}, +{-0.852640152f, -0.522498548f}, {-0.856041610f, -0.516906917f}, +{-0.859406412f, -0.511293113f}, {-0.862734377f, -0.505657375f}, +{-0.866025388f, -0.500000000f}, {-0.869279325f, -0.494321197f}, +{-0.872496009f, -0.488621235f}, {-0.875675321f, -0.482900351f}, +{-0.878817141f, -0.477158755f}, {-0.881921291f, -0.471396744f}, +{-0.884987652f, -0.465614527f}, {-0.888016105f, -0.459812373f}, +{-0.891006529f, -0.453990489f}, {-0.893958807f, -0.448149204f}, +{-0.896872759f, -0.442288697f}, {-0.899748266f, -0.436409235f}, +{-0.902585268f, -0.430511087f}, {-0.905383646f, -0.424594522f}, +{-0.908143163f, -0.418659747f}, {-0.910863817f, -0.412707031f}, +{-0.913545430f, -0.406736642f}, {-0.916187942f, -0.400748819f}, +{-0.918791234f, -0.394743860f}, {-0.921355128f, -0.388721973f}, +{-0.923879504f, -0.382683426f}, {-0.926364362f, -0.376628488f}, +{-0.928809524f, -0.370557427f}, {-0.931214929f, -0.364470512f}, +{-0.933580399f, -0.358367950f}, {-0.935905933f, -0.352250040f}, +{-0.938191354f, -0.346117049f}, {-0.940436542f, -0.339969248f}, +{-0.942641497f, -0.333806872f}, {-0.944806039f, -0.327630192f}, +{-0.946930110f, -0.321439475f}, {-0.949013650f, -0.315234989f}, +{-0.951056540f, -0.309017003f}, {-0.953058660f, -0.302785784f}, +{-0.955019951f, -0.296541572f}, {-0.956940353f, -0.290284663f}, +{-0.958819747f, -0.284015357f}, {-0.960658073f, -0.277733833f}, +{-0.962455213f, -0.271440446f}, {-0.964211166f, -0.265135437f}, +{-0.965925813f, -0.258819044f}, {-0.967599094f, -0.252491564f}, +{-0.969230890f, -0.246153295f}, {-0.970821202f, -0.239804462f}, +{-0.972369909f, -0.233445361f}, {-0.973876953f, -0.227076262f}, +{-0.975342333f, -0.220697433f}, {-0.976765871f, -0.214309156f}, +{-0.978147626f, -0.207911685f}, {-0.979487419f, -0.201505318f}, +{-0.980785251f, -0.195090324f}, {-0.982041121f, -0.188666970f}, +{-0.983254910f, -0.182235524f}, {-0.984426558f, -0.175796285f}, +{-0.985556066f, -0.169349506f}, {-0.986643314f, -0.162895471f}, +{-0.987688363f, -0.156434461f}, {-0.988691032f, -0.149966761f}, +{-0.989651382f, -0.143492624f}, {-0.990569353f, -0.137012348f}, +{-0.991444886f, -0.130526185f}, {-0.992277920f, -0.124034449f}, +{-0.993068457f, -0.117537394f}, {-0.993816435f, -0.111035310f}, +{-0.994521916f, -0.104528464f}, {-0.995184720f, -0.0980171412f}, +{-0.995804906f, -0.0915016159f}, {-0.996382475f, -0.0849821791f}, +{-0.996917307f, -0.0784590989f}, {-0.997409463f, -0.0719326511f}, +{-0.997858942f, -0.0654031262f}, {-0.998265624f, -0.0588708036f}, +{-0.998629510f, -0.0523359552f}, {-0.998950660f, -0.0457988679f}, +{-0.999229014f, -0.0392598175f}, {-0.999464571f, -0.0327190831f}, +{-0.999657333f, -0.0261769481f}, {-0.999807239f, -0.0196336918f}, +{-0.999914348f, -0.0130895954f}, {-0.999978602f, -0.00654493785f}, +{-1.00000000f, -1.22464685e-16f}, {-0.999978602f, 0.00654493785f}, +{-0.999914348f, 0.0130895954f}, {-0.999807239f, 0.0196336918f}, +{-0.999657333f, 0.0261769481f}, {-0.999464571f, 0.0327190831f}, +{-0.999229014f, 0.0392598175f}, {-0.998950660f, 0.0457988679f}, +{-0.998629510f, 0.0523359552f}, {-0.998265624f, 0.0588708036f}, +{-0.997858942f, 0.0654031262f}, {-0.997409463f, 0.0719326511f}, +{-0.996917307f, 0.0784590989f}, {-0.996382475f, 0.0849821791f}, +{-0.995804906f, 0.0915016159f}, {-0.995184720f, 0.0980171412f}, +{-0.994521916f, 0.104528464f}, {-0.993816435f, 0.111035310f}, +{-0.993068457f, 0.117537394f}, {-0.992277920f, 0.124034449f}, +{-0.991444886f, 0.130526185f}, {-0.990569353f, 0.137012348f}, +{-0.989651382f, 0.143492624f}, {-0.988691032f, 0.149966761f}, +{-0.987688363f, 0.156434461f}, {-0.986643314f, 0.162895471f}, +{-0.985556066f, 0.169349506f}, {-0.984426558f, 0.175796285f}, +{-0.983254910f, 0.182235524f}, {-0.982041121f, 0.188666970f}, +{-0.980785251f, 0.195090324f}, {-0.979487419f, 0.201505318f}, +{-0.978147626f, 0.207911685f}, {-0.976765871f, 0.214309156f}, +{-0.975342333f, 0.220697433f}, {-0.973876953f, 0.227076262f}, +{-0.972369909f, 0.233445361f}, {-0.970821202f, 0.239804462f}, +{-0.969230890f, 0.246153295f}, {-0.967599094f, 0.252491564f}, +{-0.965925813f, 0.258819044f}, {-0.964211166f, 0.265135437f}, +{-0.962455213f, 0.271440446f}, {-0.960658073f, 0.277733833f}, +{-0.958819747f, 0.284015357f}, {-0.956940353f, 0.290284663f}, +{-0.955019951f, 0.296541572f}, {-0.953058660f, 0.302785784f}, +{-0.951056540f, 0.309017003f}, {-0.949013650f, 0.315234989f}, +{-0.946930110f, 0.321439475f}, {-0.944806039f, 0.327630192f}, +{-0.942641497f, 0.333806872f}, {-0.940436542f, 0.339969248f}, +{-0.938191354f, 0.346117049f}, {-0.935905933f, 0.352250040f}, +{-0.933580399f, 0.358367950f}, {-0.931214929f, 0.364470512f}, +{-0.928809524f, 0.370557427f}, {-0.926364362f, 0.376628488f}, +{-0.923879504f, 0.382683426f}, {-0.921355128f, 0.388721973f}, +{-0.918791234f, 0.394743860f}, {-0.916187942f, 0.400748819f}, +{-0.913545430f, 0.406736642f}, {-0.910863817f, 0.412707031f}, +{-0.908143163f, 0.418659747f}, {-0.905383646f, 0.424594522f}, +{-0.902585268f, 0.430511087f}, {-0.899748266f, 0.436409235f}, +{-0.896872759f, 0.442288697f}, {-0.893958807f, 0.448149204f}, +{-0.891006529f, 0.453990489f}, {-0.888016105f, 0.459812373f}, +{-0.884987652f, 0.465614527f}, {-0.881921291f, 0.471396744f}, +{-0.878817141f, 0.477158755f}, {-0.875675321f, 0.482900351f}, +{-0.872496009f, 0.488621235f}, {-0.869279325f, 0.494321197f}, +{-0.866025388f, 0.500000000f}, {-0.862734377f, 0.505657375f}, +{-0.859406412f, 0.511293113f}, {-0.856041610f, 0.516906917f}, +{-0.852640152f, 0.522498548f}, {-0.849202156f, 0.528067827f}, +{-0.845727801f, 0.533614516f}, {-0.842217207f, 0.539138317f}, +{-0.838670552f, 0.544639051f}, {-0.835087955f, 0.550116420f}, +{-0.831469595f, 0.555570245f}, {-0.827815652f, 0.561000228f}, +{-0.824126184f, 0.566406250f}, {-0.820401430f, 0.571787953f}, +{-0.816641569f, 0.577145219f}, {-0.812846661f, 0.582477689f}, +{-0.809017003f, 0.587785244f}, {-0.805152655f, 0.593067646f}, +{-0.801253796f, 0.598324597f}, {-0.797320664f, 0.603555918f}, +{-0.793353319f, 0.608761430f}, {-0.789352059f, 0.613940835f}, +{-0.785316944f, 0.619093955f}, {-0.781248152f, 0.624220550f}, +{-0.777145982f, 0.629320383f}, {-0.773010433f, 0.634393275f}, +{-0.768841803f, 0.639438987f}, {-0.764640272f, 0.644457340f}, +{-0.760405958f, 0.649448037f}, {-0.756139100f, 0.654410958f}, +{-0.751839817f, 0.659345806f}, {-0.747508347f, 0.664252460f}, +{-0.743144810f, 0.669130623f}, {-0.738749504f, 0.673980117f}, +{-0.734322488f, 0.678800762f}, {-0.729864061f, 0.683592319f}, +{-0.725374401f, 0.688354552f}, {-0.720853567f, 0.693087339f}, +{-0.716301918f, 0.697790444f}, {-0.711719632f, 0.702463686f}, +{-0.707106769f, 0.707106769f}, {-0.702463686f, 0.711719632f}, +{-0.697790444f, 0.716301918f}, {-0.693087339f, 0.720853567f}, +{-0.688354552f, 0.725374401f}, {-0.683592319f, 0.729864061f}, +{-0.678800762f, 0.734322488f}, {-0.673980117f, 0.738749504f}, +{-0.669130623f, 0.743144810f}, {-0.664252460f, 0.747508347f}, +{-0.659345806f, 0.751839817f}, {-0.654410958f, 0.756139100f}, +{-0.649448037f, 0.760405958f}, {-0.644457340f, 0.764640272f}, +{-0.639438987f, 0.768841803f}, {-0.634393275f, 0.773010433f}, +{-0.629320383f, 0.777145982f}, {-0.624220550f, 0.781248152f}, +{-0.619093955f, 0.785316944f}, {-0.613940835f, 0.789352059f}, +{-0.608761430f, 0.793353319f}, {-0.603555918f, 0.797320664f}, +{-0.598324597f, 0.801253796f}, {-0.593067646f, 0.805152655f}, +{-0.587785244f, 0.809017003f}, {-0.582477689f, 0.812846661f}, +{-0.577145219f, 0.816641569f}, {-0.571787953f, 0.820401430f}, +{-0.566406250f, 0.824126184f}, {-0.561000228f, 0.827815652f}, +{-0.555570245f, 0.831469595f}, {-0.550116420f, 0.835087955f}, +{-0.544639051f, 0.838670552f}, {-0.539138317f, 0.842217207f}, +{-0.533614516f, 0.845727801f}, {-0.528067827f, 0.849202156f}, +{-0.522498548f, 0.852640152f}, {-0.516906917f, 0.856041610f}, +{-0.511293113f, 0.859406412f}, {-0.505657375f, 0.862734377f}, +{-0.500000000f, 0.866025388f}, {-0.494321197f, 0.869279325f}, +{-0.488621235f, 0.872496009f}, {-0.482900351f, 0.875675321f}, +{-0.477158755f, 0.878817141f}, {-0.471396744f, 0.881921291f}, +{-0.465614527f, 0.884987652f}, {-0.459812373f, 0.888016105f}, +{-0.453990489f, 0.891006529f}, {-0.448149204f, 0.893958807f}, +{-0.442288697f, 0.896872759f}, {-0.436409235f, 0.899748266f}, +{-0.430511087f, 0.902585268f}, {-0.424594522f, 0.905383646f}, +{-0.418659747f, 0.908143163f}, {-0.412707031f, 0.910863817f}, +{-0.406736642f, 0.913545430f}, {-0.400748819f, 0.916187942f}, +{-0.394743860f, 0.918791234f}, {-0.388721973f, 0.921355128f}, +{-0.382683426f, 0.923879504f}, {-0.376628488f, 0.926364362f}, +{-0.370557427f, 0.928809524f}, {-0.364470512f, 0.931214929f}, +{-0.358367950f, 0.933580399f}, {-0.352250040f, 0.935905933f}, +{-0.346117049f, 0.938191354f}, {-0.339969248f, 0.940436542f}, +{-0.333806872f, 0.942641497f}, {-0.327630192f, 0.944806039f}, +{-0.321439475f, 0.946930110f}, {-0.315234989f, 0.949013650f}, +{-0.309017003f, 0.951056540f}, {-0.302785784f, 0.953058660f}, +{-0.296541572f, 0.955019951f}, {-0.290284663f, 0.956940353f}, +{-0.284015357f, 0.958819747f}, {-0.277733833f, 0.960658073f}, +{-0.271440446f, 0.962455213f}, {-0.265135437f, 0.964211166f}, +{-0.258819044f, 0.965925813f}, {-0.252491564f, 0.967599094f}, +{-0.246153295f, 0.969230890f}, {-0.239804462f, 0.970821202f}, +{-0.233445361f, 0.972369909f}, {-0.227076262f, 0.973876953f}, +{-0.220697433f, 0.975342333f}, {-0.214309156f, 0.976765871f}, +{-0.207911685f, 0.978147626f}, {-0.201505318f, 0.979487419f}, +{-0.195090324f, 0.980785251f}, {-0.188666970f, 0.982041121f}, +{-0.182235524f, 0.983254910f}, {-0.175796285f, 0.984426558f}, +{-0.169349506f, 0.985556066f}, {-0.162895471f, 0.986643314f}, +{-0.156434461f, 0.987688363f}, {-0.149966761f, 0.988691032f}, +{-0.143492624f, 0.989651382f}, {-0.137012348f, 0.990569353f}, +{-0.130526185f, 0.991444886f}, {-0.124034449f, 0.992277920f}, +{-0.117537394f, 0.993068457f}, {-0.111035310f, 0.993816435f}, +{-0.104528464f, 0.994521916f}, {-0.0980171412f, 0.995184720f}, +{-0.0915016159f, 0.995804906f}, {-0.0849821791f, 0.996382475f}, +{-0.0784590989f, 0.996917307f}, {-0.0719326511f, 0.997409463f}, +{-0.0654031262f, 0.997858942f}, {-0.0588708036f, 0.998265624f}, +{-0.0523359552f, 0.998629510f}, {-0.0457988679f, 0.998950660f}, +{-0.0392598175f, 0.999229014f}, {-0.0327190831f, 0.999464571f}, +{-0.0261769481f, 0.999657333f}, {-0.0196336918f, 0.999807239f}, +{-0.0130895954f, 0.999914348f}, {-0.00654493785f, 0.999978602f}, +{-1.83697015e-16f, 1.00000000f}, {0.00654493785f, 0.999978602f}, +{0.0130895954f, 0.999914348f}, {0.0196336918f, 0.999807239f}, +{0.0261769481f, 0.999657333f}, {0.0327190831f, 0.999464571f}, +{0.0392598175f, 0.999229014f}, {0.0457988679f, 0.998950660f}, +{0.0523359552f, 0.998629510f}, {0.0588708036f, 0.998265624f}, +{0.0654031262f, 0.997858942f}, {0.0719326511f, 0.997409463f}, +{0.0784590989f, 0.996917307f}, {0.0849821791f, 0.996382475f}, +{0.0915016159f, 0.995804906f}, {0.0980171412f, 0.995184720f}, +{0.104528464f, 0.994521916f}, {0.111035310f, 0.993816435f}, +{0.117537394f, 0.993068457f}, {0.124034449f, 0.992277920f}, +{0.130526185f, 0.991444886f}, {0.137012348f, 0.990569353f}, +{0.143492624f, 0.989651382f}, {0.149966761f, 0.988691032f}, +{0.156434461f, 0.987688363f}, {0.162895471f, 0.986643314f}, +{0.169349506f, 0.985556066f}, {0.175796285f, 0.984426558f}, +{0.182235524f, 0.983254910f}, {0.188666970f, 0.982041121f}, +{0.195090324f, 0.980785251f}, {0.201505318f, 0.979487419f}, +{0.207911685f, 0.978147626f}, {0.214309156f, 0.976765871f}, +{0.220697433f, 0.975342333f}, {0.227076262f, 0.973876953f}, +{0.233445361f, 0.972369909f}, {0.239804462f, 0.970821202f}, +{0.246153295f, 0.969230890f}, {0.252491564f, 0.967599094f}, +{0.258819044f, 0.965925813f}, {0.265135437f, 0.964211166f}, +{0.271440446f, 0.962455213f}, {0.277733833f, 0.960658073f}, +{0.284015357f, 0.958819747f}, {0.290284663f, 0.956940353f}, +{0.296541572f, 0.955019951f}, {0.302785784f, 0.953058660f}, +{0.309017003f, 0.951056540f}, {0.315234989f, 0.949013650f}, +{0.321439475f, 0.946930110f}, {0.327630192f, 0.944806039f}, +{0.333806872f, 0.942641497f}, {0.339969248f, 0.940436542f}, +{0.346117049f, 0.938191354f}, {0.352250040f, 0.935905933f}, +{0.358367950f, 0.933580399f}, {0.364470512f, 0.931214929f}, +{0.370557427f, 0.928809524f}, {0.376628488f, 0.926364362f}, +{0.382683426f, 0.923879504f}, {0.388721973f, 0.921355128f}, +{0.394743860f, 0.918791234f}, {0.400748819f, 0.916187942f}, +{0.406736642f, 0.913545430f}, {0.412707031f, 0.910863817f}, +{0.418659747f, 0.908143163f}, {0.424594522f, 0.905383646f}, +{0.430511087f, 0.902585268f}, {0.436409235f, 0.899748266f}, +{0.442288697f, 0.896872759f}, {0.448149204f, 0.893958807f}, +{0.453990489f, 0.891006529f}, {0.459812373f, 0.888016105f}, +{0.465614527f, 0.884987652f}, {0.471396744f, 0.881921291f}, +{0.477158755f, 0.878817141f}, {0.482900351f, 0.875675321f}, +{0.488621235f, 0.872496009f}, {0.494321197f, 0.869279325f}, +{0.500000000f, 0.866025388f}, {0.505657375f, 0.862734377f}, +{0.511293113f, 0.859406412f}, {0.516906917f, 0.856041610f}, +{0.522498548f, 0.852640152f}, {0.528067827f, 0.849202156f}, +{0.533614516f, 0.845727801f}, {0.539138317f, 0.842217207f}, +{0.544639051f, 0.838670552f}, {0.550116420f, 0.835087955f}, +{0.555570245f, 0.831469595f}, {0.561000228f, 0.827815652f}, +{0.566406250f, 0.824126184f}, {0.571787953f, 0.820401430f}, +{0.577145219f, 0.816641569f}, {0.582477689f, 0.812846661f}, +{0.587785244f, 0.809017003f}, {0.593067646f, 0.805152655f}, +{0.598324597f, 0.801253796f}, {0.603555918f, 0.797320664f}, +{0.608761430f, 0.793353319f}, {0.613940835f, 0.789352059f}, +{0.619093955f, 0.785316944f}, {0.624220550f, 0.781248152f}, +{0.629320383f, 0.777145982f}, {0.634393275f, 0.773010433f}, +{0.639438987f, 0.768841803f}, {0.644457340f, 0.764640272f}, +{0.649448037f, 0.760405958f}, {0.654410958f, 0.756139100f}, +{0.659345806f, 0.751839817f}, {0.664252460f, 0.747508347f}, +{0.669130623f, 0.743144810f}, {0.673980117f, 0.738749504f}, +{0.678800762f, 0.734322488f}, {0.683592319f, 0.729864061f}, +{0.688354552f, 0.725374401f}, {0.693087339f, 0.720853567f}, +{0.697790444f, 0.716301918f}, {0.702463686f, 0.711719632f}, +{0.707106769f, 0.707106769f}, {0.711719632f, 0.702463686f}, +{0.716301918f, 0.697790444f}, {0.720853567f, 0.693087339f}, +{0.725374401f, 0.688354552f}, {0.729864061f, 0.683592319f}, +{0.734322488f, 0.678800762f}, {0.738749504f, 0.673980117f}, +{0.743144810f, 0.669130623f}, {0.747508347f, 0.664252460f}, +{0.751839817f, 0.659345806f}, {0.756139100f, 0.654410958f}, +{0.760405958f, 0.649448037f}, {0.764640272f, 0.644457340f}, +{0.768841803f, 0.639438987f}, {0.773010433f, 0.634393275f}, +{0.777145982f, 0.629320383f}, {0.781248152f, 0.624220550f}, +{0.785316944f, 0.619093955f}, {0.789352059f, 0.613940835f}, +{0.793353319f, 0.608761430f}, {0.797320664f, 0.603555918f}, +{0.801253796f, 0.598324597f}, {0.805152655f, 0.593067646f}, +{0.809017003f, 0.587785244f}, {0.812846661f, 0.582477689f}, +{0.816641569f, 0.577145219f}, {0.820401430f, 0.571787953f}, +{0.824126184f, 0.566406250f}, {0.827815652f, 0.561000228f}, +{0.831469595f, 0.555570245f}, {0.835087955f, 0.550116420f}, +{0.838670552f, 0.544639051f}, {0.842217207f, 0.539138317f}, +{0.845727801f, 0.533614516f}, {0.849202156f, 0.528067827f}, +{0.852640152f, 0.522498548f}, {0.856041610f, 0.516906917f}, +{0.859406412f, 0.511293113f}, {0.862734377f, 0.505657375f}, +{0.866025388f, 0.500000000f}, {0.869279325f, 0.494321197f}, +{0.872496009f, 0.488621235f}, {0.875675321f, 0.482900351f}, +{0.878817141f, 0.477158755f}, {0.881921291f, 0.471396744f}, +{0.884987652f, 0.465614527f}, {0.888016105f, 0.459812373f}, +{0.891006529f, 0.453990489f}, {0.893958807f, 0.448149204f}, +{0.896872759f, 0.442288697f}, {0.899748266f, 0.436409235f}, +{0.902585268f, 0.430511087f}, {0.905383646f, 0.424594522f}, +{0.908143163f, 0.418659747f}, {0.910863817f, 0.412707031f}, +{0.913545430f, 0.406736642f}, {0.916187942f, 0.400748819f}, +{0.918791234f, 0.394743860f}, {0.921355128f, 0.388721973f}, +{0.923879504f, 0.382683426f}, {0.926364362f, 0.376628488f}, +{0.928809524f, 0.370557427f}, {0.931214929f, 0.364470512f}, +{0.933580399f, 0.358367950f}, {0.935905933f, 0.352250040f}, +{0.938191354f, 0.346117049f}, {0.940436542f, 0.339969248f}, +{0.942641497f, 0.333806872f}, {0.944806039f, 0.327630192f}, +{0.946930110f, 0.321439475f}, {0.949013650f, 0.315234989f}, +{0.951056540f, 0.309017003f}, {0.953058660f, 0.302785784f}, +{0.955019951f, 0.296541572f}, {0.956940353f, 0.290284663f}, +{0.958819747f, 0.284015357f}, {0.960658073f, 0.277733833f}, +{0.962455213f, 0.271440446f}, {0.964211166f, 0.265135437f}, +{0.965925813f, 0.258819044f}, {0.967599094f, 0.252491564f}, +{0.969230890f, 0.246153295f}, {0.970821202f, 0.239804462f}, +{0.972369909f, 0.233445361f}, {0.973876953f, 0.227076262f}, +{0.975342333f, 0.220697433f}, {0.976765871f, 0.214309156f}, +{0.978147626f, 0.207911685f}, {0.979487419f, 0.201505318f}, +{0.980785251f, 0.195090324f}, {0.982041121f, 0.188666970f}, +{0.983254910f, 0.182235524f}, {0.984426558f, 0.175796285f}, +{0.985556066f, 0.169349506f}, {0.986643314f, 0.162895471f}, +{0.987688363f, 0.156434461f}, {0.988691032f, 0.149966761f}, +{0.989651382f, 0.143492624f}, {0.990569353f, 0.137012348f}, +{0.991444886f, 0.130526185f}, {0.992277920f, 0.124034449f}, +{0.993068457f, 0.117537394f}, {0.993816435f, 0.111035310f}, +{0.994521916f, 0.104528464f}, {0.995184720f, 0.0980171412f}, +{0.995804906f, 0.0915016159f}, {0.996382475f, 0.0849821791f}, +{0.996917307f, 0.0784590989f}, {0.997409463f, 0.0719326511f}, +{0.997858942f, 0.0654031262f}, {0.998265624f, 0.0588708036f}, +{0.998629510f, 0.0523359552f}, {0.998950660f, 0.0457988679f}, +{0.999229014f, 0.0392598175f}, {0.999464571f, 0.0327190831f}, +{0.999657333f, 0.0261769481f}, {0.999807239f, 0.0196336918f}, +{0.999914348f, 0.0130895954f}, {0.999978602f, 0.00654493785f}, +}; + +const kiss_fft_state rnn_kfft = { +960, /* nfft */ +0.0010416667f, /* scale */ +-1, /* shift */ +{5, 192, 3, 64, 4, 16, 4, 4, 4, 1, 0, 0, 0, 0, 0, 0, }, /* factors */ +fft_bitrev, /* bitrev*/ +fft_twiddles, /* twiddles*/ +(arch_fft_state *)&arch_fft, /* arch_fft*/ +}; + +const float rnn_half_window[] = { +4.20549168e-06f, 3.78491532e-05f, 0.000105135041f, 0.000206060256f, 0.000340620492f, +0.000508809986f, 0.000710621476f, 0.000946046319f, 0.00121507444f, 0.00151769421f, +0.00185389258f, 0.00222365512f, 0.00262696599f, 0.00306380726f, 0.00353416055f, +0.00403800514f, 0.00457531959f, 0.00514607970f, 0.00575026125f, 0.00638783723f, +0.00705878017f, 0.00776306028f, 0.00850064680f, 0.00927150715f, 0.0100756064f, +0.0109129101f, 0.0117833801f, 0.0126869772f, 0.0136236614f, 0.0145933898f, +0.0155961197f, 0.0166318044f, 0.0177003983f, 0.0188018531f, 0.0199361145f, +0.0211031344f, 0.0223028567f, 0.0235352255f, 0.0248001851f, 0.0260976739f, +0.0274276342f, 0.0287899990f, 0.0301847085f, 0.0316116922f, 0.0330708846f, +0.0345622115f, 0.0360856056f, 0.0376409888f, 0.0392282903f, 0.0408474281f, +0.0424983241f, 0.0441808924f, 0.0458950549f, 0.0476407260f, 0.0494178124f, +0.0512262285f, 0.0530658774f, 0.0549366735f, 0.0568385124f, 0.0587713011f, +0.0607349351f, 0.0627293140f, 0.0647543296f, 0.0668098852f, 0.0688958541f, +0.0710121393f, 0.0731586292f, 0.0753351897f, 0.0775417164f, 0.0797780901f, +0.0820441842f, 0.0843398646f, 0.0866650119f, 0.0890194997f, 0.0914031938f, +0.0938159525f, 0.0962576419f, 0.0987281203f, 0.101227246f, 0.103754878f, +0.106310867f, 0.108895063f, 0.111507311f, 0.114147455f, 0.116815343f, +0.119510807f, 0.122233689f, 0.124983832f, 0.127761051f, 0.130565181f, +0.133396059f, 0.136253506f, 0.139137328f, 0.142047361f, 0.144983411f, +0.147945285f, 0.150932819f, 0.153945804f, 0.156984031f, 0.160047337f, +0.163135484f, 0.166248307f, 0.169385567f, 0.172547072f, 0.175732598f, +0.178941950f, 0.182174906f, 0.185431242f, 0.188710734f, 0.192013159f, +0.195338294f, 0.198685899f, 0.202055752f, 0.205447599f, 0.208861232f, +0.212296382f, 0.215752810f, 0.219230279f, 0.222728521f, 0.226247311f, +0.229786381f, 0.233345464f, 0.236924306f, 0.240522653f, 0.244140238f, +0.247776777f, 0.251432031f, 0.255105674f, 0.258797467f, 0.262507141f, +0.266234398f, 0.269978970f, 0.273740560f, 0.277518868f, 0.281313598f, +0.285124481f, 0.288951218f, 0.292793512f, 0.296651065f, 0.300523549f, +0.304410696f, 0.308312178f, 0.312227666f, 0.316156894f, 0.320099503f, +0.324055225f, 0.328023702f, 0.332004637f, 0.335997701f, 0.340002567f, +0.344018906f, 0.348046392f, 0.352084726f, 0.356133521f, 0.360192508f, +0.364261299f, 0.368339598f, 0.372427016f, 0.376523286f, 0.380627990f, +0.384740859f, 0.388861477f, 0.392989576f, 0.397124738f, 0.401266664f, +0.405414969f, 0.409569323f, 0.413729399f, 0.417894781f, 0.422065198f, +0.426240236f, 0.430419534f, 0.434602767f, 0.438789606f, 0.442979604f, +0.447172493f, 0.451367885f, 0.455565393f, 0.459764689f, 0.463965416f, +0.468167186f, 0.472369671f, 0.476572484f, 0.480775267f, 0.484977663f, +0.489179343f, 0.493379891f, 0.497579008f, 0.501776278f, 0.505971372f, +0.510163903f, 0.514353573f, 0.518539906f, 0.522722721f, 0.526901484f, +0.531075954f, 0.535245717f, 0.539410412f, 0.543569744f, 0.547723293f, +0.551870763f, 0.556011736f, 0.560145974f, 0.564273000f, 0.568392515f, +0.572504222f, 0.576607704f, 0.580702662f, 0.584788740f, 0.588865638f, +0.592932940f, 0.596990347f, 0.601037502f, 0.605074167f, 0.609099925f, +0.613114417f, 0.617117405f, 0.621108532f, 0.625087440f, 0.629053831f, +0.633007407f, 0.636947870f, 0.640874863f, 0.644788086f, 0.648687243f, +0.652572036f, 0.656442165f, 0.660297334f, 0.664137185f, 0.667961538f, +0.671769977f, 0.675562322f, 0.679338276f, 0.683097482f, 0.686839759f, +0.690564752f, 0.694272280f, 0.697961986f, 0.701633692f, 0.705287039f, +0.708921850f, 0.712537885f, 0.716134787f, 0.719712436f, 0.723270535f, +0.726808906f, 0.730327189f, 0.733825266f, 0.737302899f, 0.740759790f, +0.744195819f, 0.747610688f, 0.751004279f, 0.754376352f, 0.757726669f, +0.761055112f, 0.764361382f, 0.767645359f, 0.770906866f, 0.774145722f, +0.777361751f, 0.780554771f, 0.783724606f, 0.786871076f, 0.789994121f, +0.793093503f, 0.796169102f, 0.799220800f, 0.802248418f, 0.805251837f, +0.808230937f, 0.811185598f, 0.814115703f, 0.817021132f, 0.819901764f, +0.822757542f, 0.825588286f, 0.828393936f, 0.831174433f, 0.833929658f, +0.836659551f, 0.839363992f, 0.842042983f, 0.844696403f, 0.847324252f, +0.849926353f, 0.852502763f, 0.855053425f, 0.857578218f, 0.860077202f, +0.862550259f, 0.864997447f, 0.867418647f, 0.869813919f, 0.872183204f, +0.874526560f, 0.876843870f, 0.879135191f, 0.881400526f, 0.883639932f, +0.885853291f, 0.888040781f, 0.890202343f, 0.892337978f, 0.894447744f, +0.896531701f, 0.898589849f, 0.900622249f, 0.902628958f, 0.904610038f, +0.906565487f, 0.908495426f, 0.910399914f, 0.912279010f, 0.914132774f, +0.915961266f, 0.917764664f, 0.919542909f, 0.921296239f, 0.923024654f, +0.924728215f, 0.926407158f, 0.928061485f, 0.929691315f, 0.931296766f, +0.932878017f, 0.934435070f, 0.935968161f, 0.937477291f, 0.938962698f, +0.940424502f, 0.941862822f, 0.943277776f, 0.944669485f, 0.946038187f, +0.947383940f, 0.948706925f, 0.950007319f, 0.951285243f, 0.952540874f, +0.953774393f, 0.954985917f, 0.956175685f, 0.957343817f, 0.958490491f, +0.959615886f, 0.960720181f, 0.961803555f, 0.962866247f, 0.963908315f, +0.964930058f, 0.965931594f, 0.966913164f, 0.967874944f, 0.968817174f, +0.969739914f, 0.970643520f, 0.971528113f, 0.972393870f, 0.973241091f, +0.974069893f, 0.974880517f, 0.975673139f, 0.976447999f, 0.977205336f, +0.977945268f, 0.978668094f, 0.979374051f, 0.980063200f, 0.980735898f, +0.981392324f, 0.982032716f, 0.982657254f, 0.983266115f, 0.983859658f, +0.984437943f, 0.985001266f, 0.985549867f, 0.986083925f, 0.986603677f, +0.987109363f, 0.987601161f, 0.988079309f, 0.988544047f, 0.988995552f, +0.989434063f, 0.989859879f, 0.990273118f, 0.990674019f, 0.991062820f, +0.991439700f, 0.991804957f, 0.992158771f, 0.992501318f, 0.992832899f, +0.993153632f, 0.993463814f, 0.993763626f, 0.994053245f, 0.994332969f, +0.994602919f, 0.994863331f, 0.995114446f, 0.995356441f, 0.995589554f, +0.995813966f, 0.996029854f, 0.996237516f, 0.996437073f, 0.996628702f, +0.996812642f, 0.996989131f, 0.997158289f, 0.997320294f, 0.997475445f, +0.997623861f, 0.997765720f, 0.997901261f, 0.998030603f, 0.998153925f, +0.998271465f, 0.998383403f, 0.998489857f, 0.998591006f, 0.998687088f, +0.998778164f, 0.998864532f, 0.998946249f, 0.999023557f, 0.999096513f, +0.999165416f, 0.999230266f, 0.999291301f, 0.999348700f, 0.999402523f, +0.999453008f, 0.999500215f, 0.999544322f, 0.999585509f, 0.999623775f, +0.999659419f, 0.999692440f, 0.999723017f, 0.999751270f, 0.999777317f, +0.999801278f, 0.999823213f, 0.999843359f, 0.999861658f, 0.999878347f, +0.999893486f, 0.999907196f, 0.999919534f, 0.999930561f, 0.999940455f, +0.999949217f, 0.999957025f, 0.999963880f, 0.999969840f, 0.999975085f, +0.999979615f, 0.999983490f, 0.999986768f, 0.999989510f, 0.999991834f, +0.999993742f, 0.999995291f, 0.999996543f, 0.999997556f, 0.999998271f, +0.999998868f, 0.999999285f, 0.999999523f, 0.999999762f, 0.999999881f, +0.999999940f, 1.00000000f, 1.00000000f, 1.00000000f, 1.00000000f, +}; + +const float rnn_dct_table[] = { +0.707106769f, 0.998795450f, 0.995184720f, 0.989176512f, 0.980785251f, +0.970031261f, 0.956940353f, 0.941544056f, 0.923879504f, 0.903989315f, +0.881921291f, 0.857728601f, 0.831469595f, 0.803207517f, 0.773010433f, +0.740951121f, 0.707106769f, 0.671558976f, 0.634393275f, 0.595699310f, +0.555570245f, 0.514102757f, 0.471396744f, 0.427555084f, 0.382683426f, +0.336889863f, 0.290284663f, 0.242980182f, 0.195090324f, 0.146730468f, +0.0980171412f, 0.0490676761f, 0.707106769f, 0.989176512f, 0.956940353f, +0.903989315f, 0.831469595f, 0.740951121f, 0.634393275f, 0.514102757f, +0.382683426f, 0.242980182f, 0.0980171412f, -0.0490676761f, -0.195090324f, +-0.336889863f, -0.471396744f, -0.595699310f, -0.707106769f, -0.803207517f, +-0.881921291f, -0.941544056f, -0.980785251f, -0.998795450f, -0.995184720f, +-0.970031261f, -0.923879504f, -0.857728601f, -0.773010433f, -0.671558976f, +-0.555570245f, -0.427555084f, -0.290284663f, -0.146730468f, 0.707106769f, +0.970031261f, 0.881921291f, 0.740951121f, 0.555570245f, 0.336889863f, +0.0980171412f, -0.146730468f, -0.382683426f, -0.595699310f, -0.773010433f, +-0.903989315f, -0.980785251f, -0.998795450f, -0.956940353f, -0.857728601f, +-0.707106769f, -0.514102757f, -0.290284663f, -0.0490676761f, 0.195090324f, +0.427555084f, 0.634393275f, 0.803207517f, 0.923879504f, 0.989176512f, +0.995184720f, 0.941544056f, 0.831469595f, 0.671558976f, 0.471396744f, +0.242980182f, 0.707106769f, 0.941544056f, 0.773010433f, 0.514102757f, +0.195090324f, -0.146730468f, -0.471396744f, -0.740951121f, -0.923879504f, +-0.998795450f, -0.956940353f, -0.803207517f, -0.555570245f, -0.242980182f, +0.0980171412f, 0.427555084f, 0.707106769f, 0.903989315f, 0.995184720f, +0.970031261f, 0.831469595f, 0.595699310f, 0.290284663f, -0.0490676761f, +-0.382683426f, -0.671558976f, -0.881921291f, -0.989176512f, -0.980785251f, +-0.857728601f, -0.634393275f, -0.336889863f, 0.707106769f, 0.903989315f, +0.634393275f, 0.242980182f, -0.195090324f, -0.595699310f, -0.881921291f, +-0.998795450f, -0.923879504f, -0.671558976f, -0.290284663f, 0.146730468f, +0.555570245f, 0.857728601f, 0.995184720f, 0.941544056f, 0.707106769f, +0.336889863f, -0.0980171412f, -0.514102757f, -0.831469595f, -0.989176512f, +-0.956940353f, -0.740951121f, -0.382683426f, 0.0490676761f, 0.471396744f, +0.803207517f, 0.980785251f, 0.970031261f, 0.773010433f, 0.427555084f, +0.707106769f, 0.857728601f, 0.471396744f, -0.0490676761f, -0.555570245f, +-0.903989315f, -0.995184720f, -0.803207517f, -0.382683426f, 0.146730468f, +0.634393275f, 0.941544056f, 0.980785251f, 0.740951121f, 0.290284663f, +-0.242980182f, -0.707106769f, -0.970031261f, -0.956940353f, -0.671558976f, +-0.195090324f, 0.336889863f, 0.773010433f, 0.989176512f, 0.923879504f, +0.595699310f, 0.0980171412f, -0.427555084f, -0.831469595f, -0.998795450f, +-0.881921291f, -0.514102757f, 0.707106769f, 0.803207517f, 0.290284663f, +-0.336889863f, -0.831469595f, -0.998795450f, -0.773010433f, -0.242980182f, +0.382683426f, 0.857728601f, 0.995184720f, 0.740951121f, 0.195090324f, +-0.427555084f, -0.881921291f, -0.989176512f, -0.707106769f, -0.146730468f, +0.471396744f, 0.903989315f, 0.980785251f, 0.671558976f, 0.0980171412f, +-0.514102757f, -0.923879504f, -0.970031261f, -0.634393275f, -0.0490676761f, +0.555570245f, 0.941544056f, 0.956940353f, 0.595699310f, 0.707106769f, +0.740951121f, 0.0980171412f, -0.595699310f, -0.980785251f, -0.857728601f, +-0.290284663f, 0.427555084f, 0.923879504f, 0.941544056f, 0.471396744f, +-0.242980182f, -0.831469595f, -0.989176512f, -0.634393275f, 0.0490676761f, +0.707106769f, 0.998795450f, 0.773010433f, 0.146730468f, -0.555570245f, +-0.970031261f, -0.881921291f, -0.336889863f, 0.382683426f, 0.903989315f, +0.956940353f, 0.514102757f, -0.195090324f, -0.803207517f, -0.995184720f, +-0.671558976f, 0.707106769f, 0.671558976f, -0.0980171412f, -0.803207517f, +-0.980785251f, -0.514102757f, 0.290284663f, 0.903989315f, 0.923879504f, +0.336889863f, -0.471396744f, -0.970031261f, -0.831469595f, -0.146730468f, +0.634393275f, 0.998795450f, 0.707106769f, -0.0490676761f, -0.773010433f, +-0.989176512f, -0.555570245f, 0.242980182f, 0.881921291f, 0.941544056f, +0.382683426f, -0.427555084f, -0.956940353f, -0.857728601f, -0.195090324f, +0.595699310f, 0.995184720f, 0.740951121f, 0.707106769f, 0.595699310f, +-0.290284663f, -0.941544056f, -0.831469595f, -0.0490676761f, 0.773010433f, +0.970031261f, 0.382683426f, -0.514102757f, -0.995184720f, -0.671558976f, +0.195090324f, 0.903989315f, 0.881921291f, 0.146730468f, -0.707106769f, +-0.989176512f, -0.471396744f, 0.427555084f, 0.980785251f, 0.740951121f, +-0.0980171412f, -0.857728601f, -0.923879504f, -0.242980182f, 0.634393275f, +0.998795450f, 0.555570245f, -0.336889863f, -0.956940353f, -0.803207517f, +0.707106769f, 0.514102757f, -0.471396744f, -0.998795450f, -0.555570245f, +0.427555084f, 0.995184720f, 0.595699310f, -0.382683426f, -0.989176512f, +-0.634393275f, 0.336889863f, 0.980785251f, 0.671558976f, -0.290284663f, +-0.970031261f, -0.707106769f, 0.242980182f, 0.956940353f, 0.740951121f, +-0.195090324f, -0.941544056f, -0.773010433f, 0.146730468f, 0.923879504f, +0.803207517f, -0.0980171412f, -0.903989315f, -0.831469595f, 0.0490676761f, +0.881921291f, 0.857728601f, 0.707106769f, 0.427555084f, -0.634393275f, +-0.970031261f, -0.195090324f, 0.803207517f, 0.881921291f, -0.0490676761f, +-0.923879504f, -0.740951121f, 0.290284663f, 0.989176512f, 0.555570245f, +-0.514102757f, -0.995184720f, -0.336889863f, 0.707106769f, 0.941544056f, +0.0980171412f, -0.857728601f, -0.831469595f, 0.146730468f, 0.956940353f, +0.671558976f, -0.382683426f, -0.998795450f, -0.471396744f, 0.595699310f, +0.980785251f, 0.242980182f, -0.773010433f, -0.903989315f, 0.707106769f, +0.336889863f, -0.773010433f, -0.857728601f, 0.195090324f, 0.989176512f, +0.471396744f, -0.671558976f, -0.923879504f, 0.0490676761f, 0.956940353f, +0.595699310f, -0.555570245f, -0.970031261f, -0.0980171412f, 0.903989315f, +0.707106769f, -0.427555084f, -0.995184720f, -0.242980182f, 0.831469595f, +0.803207517f, -0.290284663f, -0.998795450f, -0.382683426f, 0.740951121f, +0.881921291f, -0.146730468f, -0.980785251f, -0.514102757f, 0.634393275f, +0.941544056f, 0.707106769f, 0.242980182f, -0.881921291f, -0.671558976f, +0.555570245f, 0.941544056f, -0.0980171412f, -0.989176512f, -0.382683426f, +0.803207517f, 0.773010433f, -0.427555084f, -0.980785251f, -0.0490676761f, +0.956940353f, 0.514102757f, -0.707106769f, -0.857728601f, 0.290284663f, +0.998795450f, 0.195090324f, -0.903989315f, -0.634393275f, 0.595699310f, +0.923879504f, -0.146730468f, -0.995184720f, -0.336889863f, 0.831469595f, +0.740951121f, -0.471396744f, -0.970031261f, 0.707106769f, 0.146730468f, +-0.956940353f, -0.427555084f, 0.831469595f, 0.671558976f, -0.634393275f, +-0.857728601f, 0.382683426f, 0.970031261f, -0.0980171412f, -0.998795450f, +-0.195090324f, 0.941544056f, 0.471396744f, -0.803207517f, -0.707106769f, +0.595699310f, 0.881921291f, -0.336889863f, -0.980785251f, 0.0490676761f, +0.995184720f, 0.242980182f, -0.923879504f, -0.514102757f, 0.773010433f, +0.740951121f, -0.555570245f, -0.903989315f, 0.290284663f, 0.989176512f, +0.707106769f, 0.0490676761f, -0.995184720f, -0.146730468f, 0.980785251f, +0.242980182f, -0.956940353f, -0.336889863f, 0.923879504f, 0.427555084f, +-0.881921291f, -0.514102757f, 0.831469595f, 0.595699310f, -0.773010433f, +-0.671558976f, 0.707106769f, 0.740951121f, -0.634393275f, -0.803207517f, +0.555570245f, 0.857728601f, -0.471396744f, -0.903989315f, 0.382683426f, +0.941544056f, -0.290284663f, -0.970031261f, 0.195090324f, 0.989176512f, +-0.0980171412f, -0.998795450f, 0.707106769f, -0.0490676761f, -0.995184720f, +0.146730468f, 0.980785251f, -0.242980182f, -0.956940353f, 0.336889863f, +0.923879504f, -0.427555084f, -0.881921291f, 0.514102757f, 0.831469595f, +-0.595699310f, -0.773010433f, 0.671558976f, 0.707106769f, -0.740951121f, +-0.634393275f, 0.803207517f, 0.555570245f, -0.857728601f, -0.471396744f, +0.903989315f, 0.382683426f, -0.941544056f, -0.290284663f, 0.970031261f, +0.195090324f, -0.989176512f, -0.0980171412f, 0.998795450f, 0.707106769f, +-0.146730468f, -0.956940353f, 0.427555084f, 0.831469595f, -0.671558976f, +-0.634393275f, 0.857728601f, 0.382683426f, -0.970031261f, -0.0980171412f, +0.998795450f, -0.195090324f, -0.941544056f, 0.471396744f, 0.803207517f, +-0.707106769f, -0.595699310f, 0.881921291f, 0.336889863f, -0.980785251f, +-0.0490676761f, 0.995184720f, -0.242980182f, -0.923879504f, 0.514102757f, +0.773010433f, -0.740951121f, -0.555570245f, 0.903989315f, 0.290284663f, +-0.989176512f, 0.707106769f, -0.242980182f, -0.881921291f, 0.671558976f, +0.555570245f, -0.941544056f, -0.0980171412f, 0.989176512f, -0.382683426f, +-0.803207517f, 0.773010433f, 0.427555084f, -0.980785251f, 0.0490676761f, +0.956940353f, -0.514102757f, -0.707106769f, 0.857728601f, 0.290284663f, +-0.998795450f, 0.195090324f, 0.903989315f, -0.634393275f, -0.595699310f, +0.923879504f, 0.146730468f, -0.995184720f, 0.336889863f, 0.831469595f, +-0.740951121f, -0.471396744f, 0.970031261f, 0.707106769f, -0.336889863f, +-0.773010433f, 0.857728601f, 0.195090324f, -0.989176512f, 0.471396744f, +0.671558976f, -0.923879504f, -0.0490676761f, 0.956940353f, -0.595699310f, +-0.555570245f, 0.970031261f, -0.0980171412f, -0.903989315f, 0.707106769f, +0.427555084f, -0.995184720f, 0.242980182f, 0.831469595f, -0.803207517f, +-0.290284663f, 0.998795450f, -0.382683426f, -0.740951121f, 0.881921291f, +0.146730468f, -0.980785251f, 0.514102757f, 0.634393275f, -0.941544056f, +0.707106769f, -0.427555084f, -0.634393275f, 0.970031261f, -0.195090324f, +-0.803207517f, 0.881921291f, 0.0490676761f, -0.923879504f, 0.740951121f, +0.290284663f, -0.989176512f, 0.555570245f, 0.514102757f, -0.995184720f, +0.336889863f, 0.707106769f, -0.941544056f, 0.0980171412f, 0.857728601f, +-0.831469595f, -0.146730468f, 0.956940353f, -0.671558976f, -0.382683426f, +0.998795450f, -0.471396744f, -0.595699310f, 0.980785251f, -0.242980182f, +-0.773010433f, 0.903989315f, 0.707106769f, -0.514102757f, -0.471396744f, +0.998795450f, -0.555570245f, -0.427555084f, 0.995184720f, -0.595699310f, +-0.382683426f, 0.989176512f, -0.634393275f, -0.336889863f, 0.980785251f, +-0.671558976f, -0.290284663f, 0.970031261f, -0.707106769f, -0.242980182f, +0.956940353f, -0.740951121f, -0.195090324f, 0.941544056f, -0.773010433f, +-0.146730468f, 0.923879504f, -0.803207517f, -0.0980171412f, 0.903989315f, +-0.831469595f, -0.0490676761f, 0.881921291f, -0.857728601f, 0.707106769f, +-0.595699310f, -0.290284663f, 0.941544056f, -0.831469595f, 0.0490676761f, +0.773010433f, -0.970031261f, 0.382683426f, 0.514102757f, -0.995184720f, +0.671558976f, 0.195090324f, -0.903989315f, 0.881921291f, -0.146730468f, +-0.707106769f, 0.989176512f, -0.471396744f, -0.427555084f, 0.980785251f, +-0.740951121f, -0.0980171412f, 0.857728601f, -0.923879504f, 0.242980182f, +0.634393275f, -0.998795450f, 0.555570245f, 0.336889863f, -0.956940353f, +0.803207517f, 0.707106769f, -0.671558976f, -0.0980171412f, 0.803207517f, +-0.980785251f, 0.514102757f, 0.290284663f, -0.903989315f, 0.923879504f, +-0.336889863f, -0.471396744f, 0.970031261f, -0.831469595f, 0.146730468f, +0.634393275f, -0.998795450f, 0.707106769f, 0.0490676761f, -0.773010433f, +0.989176512f, -0.555570245f, -0.242980182f, 0.881921291f, -0.941544056f, +0.382683426f, 0.427555084f, -0.956940353f, 0.857728601f, -0.195090324f, +-0.595699310f, 0.995184720f, -0.740951121f, 0.707106769f, -0.740951121f, +0.0980171412f, 0.595699310f, -0.980785251f, 0.857728601f, -0.290284663f, +-0.427555084f, 0.923879504f, -0.941544056f, 0.471396744f, 0.242980182f, +-0.831469595f, 0.989176512f, -0.634393275f, -0.0490676761f, 0.707106769f, +-0.998795450f, 0.773010433f, -0.146730468f, -0.555570245f, 0.970031261f, +-0.881921291f, 0.336889863f, 0.382683426f, -0.903989315f, 0.956940353f, +-0.514102757f, -0.195090324f, 0.803207517f, -0.995184720f, 0.671558976f, +0.707106769f, -0.803207517f, 0.290284663f, 0.336889863f, -0.831469595f, +0.998795450f, -0.773010433f, 0.242980182f, 0.382683426f, -0.857728601f, +0.995184720f, -0.740951121f, 0.195090324f, 0.427555084f, -0.881921291f, +0.989176512f, -0.707106769f, 0.146730468f, 0.471396744f, -0.903989315f, +0.980785251f, -0.671558976f, 0.0980171412f, 0.514102757f, -0.923879504f, +0.970031261f, -0.634393275f, 0.0490676761f, 0.555570245f, -0.941544056f, +0.956940353f, -0.595699310f, 0.707106769f, -0.857728601f, 0.471396744f, +0.0490676761f, -0.555570245f, 0.903989315f, -0.995184720f, 0.803207517f, +-0.382683426f, -0.146730468f, 0.634393275f, -0.941544056f, 0.980785251f, +-0.740951121f, 0.290284663f, 0.242980182f, -0.707106769f, 0.970031261f, +-0.956940353f, 0.671558976f, -0.195090324f, -0.336889863f, 0.773010433f, +-0.989176512f, 0.923879504f, -0.595699310f, 0.0980171412f, 0.427555084f, +-0.831469595f, 0.998795450f, -0.881921291f, 0.514102757f, 0.707106769f, +-0.903989315f, 0.634393275f, -0.242980182f, -0.195090324f, 0.595699310f, +-0.881921291f, 0.998795450f, -0.923879504f, 0.671558976f, -0.290284663f, +-0.146730468f, 0.555570245f, -0.857728601f, 0.995184720f, -0.941544056f, +0.707106769f, -0.336889863f, -0.0980171412f, 0.514102757f, -0.831469595f, +0.989176512f, -0.956940353f, 0.740951121f, -0.382683426f, -0.0490676761f, +0.471396744f, -0.803207517f, 0.980785251f, -0.970031261f, 0.773010433f, +-0.427555084f, 0.707106769f, -0.941544056f, 0.773010433f, -0.514102757f, +0.195090324f, 0.146730468f, -0.471396744f, 0.740951121f, -0.923879504f, +0.998795450f, -0.956940353f, 0.803207517f, -0.555570245f, 0.242980182f, +0.0980171412f, -0.427555084f, 0.707106769f, -0.903989315f, 0.995184720f, +-0.970031261f, 0.831469595f, -0.595699310f, 0.290284663f, 0.0490676761f, +-0.382683426f, 0.671558976f, -0.881921291f, 0.989176512f, -0.980785251f, +0.857728601f, -0.634393275f, 0.336889863f, 0.707106769f, -0.970031261f, +0.881921291f, -0.740951121f, 0.555570245f, -0.336889863f, 0.0980171412f, +0.146730468f, -0.382683426f, 0.595699310f, -0.773010433f, 0.903989315f, +-0.980785251f, 0.998795450f, -0.956940353f, 0.857728601f, -0.707106769f, +0.514102757f, -0.290284663f, 0.0490676761f, 0.195090324f, -0.427555084f, +0.634393275f, -0.803207517f, 0.923879504f, -0.989176512f, 0.995184720f, +-0.941544056f, 0.831469595f, -0.671558976f, 0.471396744f, -0.242980182f, +0.707106769f, -0.989176512f, 0.956940353f, -0.903989315f, 0.831469595f, +-0.740951121f, 0.634393275f, -0.514102757f, 0.382683426f, -0.242980182f, +0.0980171412f, 0.0490676761f, -0.195090324f, 0.336889863f, -0.471396744f, +0.595699310f, -0.707106769f, 0.803207517f, -0.881921291f, 0.941544056f, +-0.980785251f, 0.998795450f, -0.995184720f, 0.970031261f, -0.923879504f, +0.857728601f, -0.773010433f, 0.671558976f, -0.555570245f, 0.427555084f, +-0.290284663f, 0.146730468f, 0.707106769f, -0.998795450f, 0.995184720f, +-0.989176512f, 0.980785251f, -0.970031261f, 0.956940353f, -0.941544056f, +0.923879504f, -0.903989315f, 0.881921291f, -0.857728601f, 0.831469595f, +-0.803207517f, 0.773010433f, -0.740951121f, 0.707106769f, -0.671558976f, +0.634393275f, -0.595699310f, 0.555570245f, -0.514102757f, 0.471396744f, +-0.427555084f, 0.382683426f, -0.336889863f, 0.290284663f, -0.242980182f, +0.195090324f, -0.146730468f, 0.0980171412f, -0.0490676761f, }; diff --git a/cpp/ax650/src/rnnoise/vec.h b/cpp/ax650/src/rnnoise/vec.h new file mode 100644 index 0000000000000000000000000000000000000000..71b7afbb8b0289165155eff2c5956f23b8c605ed --- /dev/null +++ b/cpp/ax650/src/rnnoise/vec.h @@ -0,0 +1,388 @@ +/* Copyright (c) 2018 Mozilla + 2008-2011 Octasic Inc. + 2012-2017 Jean-Marc Valin */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef VEC_H +#define VEC_H + +#include "opus_types.h" +#include "common.h" +#include +#include "arch.h" +#include "x86/x86_arch_macros.h" + + +#if defined(__AVX__) || defined(__SSE2__) +#include "vec_avx.h" +#elif (defined(__ARM_NEON__) || defined(__ARM_NEON)) && !defined(DISABLE_NEON) +#include "vec_neon.h" +#else + +#define MAX_INPUTS (2048) + +#define NO_OPTIMIZATIONS + +static inline void sgemv16x1(float *out, const float *weights, int rows, int cols, int col_stride, const float *x) +{ + int i, j; + RNN_CLEAR(out, rows); + for (i=0;i +#include +#include "x86/x86cpu.h" + +#define MAX_INPUTS (2048) + +#define USE_SU_BIAS + +#ifndef __SSE_4_1__ +static inline __m128 mm_floor_ps(__m128 x) { + __m128 half = _mm_set1_ps(0.5); + return _mm_cvtepi32_ps(_mm_cvtps_epi32(_mm_sub_ps(x, half))); +} +#undef _mm_floor_ps +#define _mm_floor_ps(x) mm_floor_ps(x) +#endif + + +/* If we don't have AVX available, emulate what we need with SSE up to 4.1. */ +#ifndef __AVX__ + +typedef struct { + __m128 lo; + __m128 hi; +} mm256_emu; +#define __m256 mm256_emu + +static inline mm256_emu mm256_loadu_ps(const float *src) { + mm256_emu ret; + ret.lo = _mm_loadu_ps(&src[0]); + ret.hi = _mm_loadu_ps(&src[4]); + return ret; +} +#define _mm256_loadu_ps(src) mm256_loadu_ps(src) + + +static inline void mm256_storeu_ps(float *dst, mm256_emu src) { + _mm_storeu_ps(dst, src.lo); + _mm_storeu_ps(&dst[4], src.hi); +} +#define _mm256_storeu_ps(dst, src) mm256_storeu_ps(dst, src) + + +static inline mm256_emu mm256_setzero_ps(void) { + mm256_emu ret; + ret.lo = _mm_setzero_ps(); + ret.hi = ret.lo; + return ret; +} +#define _mm256_setzero_ps mm256_setzero_ps + +static inline mm256_emu mm256_broadcast_ss(const float *x) { + mm256_emu ret; + ret.lo = _mm_set1_ps(*x); + ret.hi = ret.lo; + return ret; +} +#define _mm256_broadcast_ss(x) mm256_broadcast_ss(x) + +static inline mm256_emu mm256_set1_ps(float x) { + mm256_emu ret; + ret.lo = _mm_set1_ps(x); + ret.hi = ret.lo; + return ret; +} +#define _mm256_set1_ps(x) mm256_set1_ps(x) + + + +static inline mm256_emu mm256_mul_ps(mm256_emu a, mm256_emu b) { + mm256_emu ret; + ret.lo = _mm_mul_ps(a.lo, b.lo); + ret.hi = _mm_mul_ps(a.hi, b.hi); + return ret; +} +#define _mm256_mul_ps(a,b) mm256_mul_ps(a,b) + +static inline mm256_emu mm256_add_ps(mm256_emu a, mm256_emu b) { + mm256_emu ret; + ret.lo = _mm_add_ps(a.lo, b.lo); + ret.hi = _mm_add_ps(a.hi, b.hi); + return ret; +} +#define _mm256_add_ps(a,b) mm256_add_ps(a,b) + + +static inline mm256_emu mm256_max_ps(mm256_emu a, mm256_emu b) { + mm256_emu ret; + ret.lo = _mm_max_ps(a.lo, b.lo); + ret.hi = _mm_max_ps(a.hi, b.hi); + return ret; +} +#define _mm256_max_ps(a,b) mm256_max_ps(a,b) + +static inline mm256_emu mm256_min_ps(mm256_emu a, mm256_emu b) { + mm256_emu ret; + ret.lo = _mm_min_ps(a.lo, b.lo); + ret.hi = _mm_min_ps(a.hi, b.hi); + return ret; +} +#define _mm256_min_ps(a,b) mm256_min_ps(a,b) + +static inline mm256_emu mm256_rcp_ps(mm256_emu a) { + mm256_emu ret; + ret.lo = _mm_rcp_ps(a.lo); + ret.hi = _mm_rcp_ps(a.hi); + return ret; +} +#define _mm256_rcp_ps(a) mm256_rcp_ps(a) + + +static inline __m128 mm256_extractf128_ps(mm256_emu x, int i) { + return (i==0) ? x.lo : x.hi; +} +#undef _mm256_extractf128_ps +#define _mm256_extractf128_ps(x,i) mm256_extractf128_ps(x,i) + +static inline mm256_emu mm256_insertf128_ps(mm256_emu dst, __m128 src, int i) { + if (i==0) dst.lo = src; + else dst.hi = src; + return dst; +} +#undef _mm256_insertf128_ps +#define _mm256_insertf128_ps(dst,src,i) mm256_insertf128_ps(dst,src,i) + +#endif /* __AVX__ */ + + + +/* If we don't have AVX2 available, emulate what we need with SSE up to 4.1. */ +#ifndef __AVX2__ + +typedef struct { + __m128i lo; + __m128i hi; +} mm256i_emu; +typedef __m256i real_m256i; +#define __m256i mm256i_emu + +static inline mm256i_emu mm256_setzero_si256(void) { + mm256i_emu ret; + ret.lo = _mm_setzero_si128(); + ret.hi = ret.lo; + return ret; +} +#define _mm256_setzero_si256 mm256_setzero_si256 + + +static inline mm256i_emu mm256_loadu_si256(const mm256i_emu *src) { + mm256i_emu ret; + ret.lo = _mm_loadu_si128((const __m128i*)src); + ret.hi = _mm_loadu_si128(&((const __m128i*)src)[1]); + return ret; +} +#define _mm256_loadu_si256(src) mm256_loadu_si256(src) + + +static inline void mm256_storeu_si256(mm256i_emu *dst, mm256i_emu src) { + _mm_storeu_si128((__m128i*)dst, src.lo); + _mm_storeu_si128(&((__m128i*)dst)[1], src.hi); +} +#define _mm256_storeu_si256(dst, src) mm256_storeu_si256(dst, src) + + +static inline mm256i_emu mm256_broadcastd_epi32(__m128i x) { + mm256i_emu ret; + ret.hi = ret.lo = _mm_shuffle_epi32(x, 0); + return ret; +} +#define _mm256_broadcastd_epi32(x) mm256_broadcastd_epi32(x) + + +static inline mm256i_emu mm256_set1_epi32(int x) { + mm256i_emu ret; + ret.lo = _mm_set1_epi32(x); + ret.hi = ret.lo; + return ret; +} +#define _mm256_set1_epi32(x) mm256_set1_epi32(x) + +static inline mm256i_emu mm256_set1_epi16(int x) { + mm256i_emu ret; + ret.lo = _mm_set1_epi16(x); + ret.hi = ret.lo; + return ret; +} +#define _mm256_set1_epi16(x) mm256_set1_epi16(x) + + +static inline mm256i_emu mm256_add_epi32(mm256i_emu a, mm256i_emu b) { + mm256i_emu ret; + ret.lo = _mm_add_epi32(a.lo, b.lo); + ret.hi = _mm_add_epi32(a.hi, b.hi); + return ret; +} +#define _mm256_add_epi32(a,b) mm256_add_epi32(a,b) + +static inline mm256i_emu mm256_madd_epi16(mm256i_emu a, mm256i_emu b) { + mm256i_emu ret; + ret.lo = _mm_madd_epi16(a.lo, b.lo); + ret.hi = _mm_madd_epi16(a.hi, b.hi); + return ret; +} +#define _mm256_madd_epi16(a,b) mm256_madd_epi16(a,b) + +static inline mm256i_emu mm256_maddubs_epi16(mm256i_emu a, mm256i_emu b) { + mm256i_emu ret; + ret.lo = _mm_maddubs_epi16(a.lo, b.lo); + ret.hi = _mm_maddubs_epi16(a.hi, b.hi); + return ret; +} +#define _mm256_maddubs_epi16(a,b) mm256_maddubs_epi16(a,b) + + + +/* Emulating the conversion functions is tricky because they use __m256i but are defined in AVX. + So we need to make a special when only AVX is available. */ +#ifdef __AVX__ + +typedef union { + mm256i_emu fake; + real_m256i real; +} mm256_union; + +static inline __m256 mm256_cvtepi32_ps(mm256i_emu a) { + mm256_union src; + src.fake = a; + return _mm256_cvtepi32_ps(src.real); +} +#define _mm256_cvtepi32_ps(a) mm256_cvtepi32_ps(a) + +static inline mm256i_emu mm256_cvtps_epi32(__m256 a) { + mm256_union ret; + ret.real = _mm256_cvtps_epi32(a); + return ret.fake; +} +#define _mm256_cvtps_epi32(a) mm256_cvtps_epi32(a) + + +#else + +static inline mm256_emu mm256_cvtepi32_ps(mm256i_emu a) { + mm256_emu ret; + ret.lo = _mm_cvtepi32_ps(a.lo); + ret.hi = _mm_cvtepi32_ps(a.hi); + return ret; +} +#define _mm256_cvtepi32_ps(a) mm256_cvtepi32_ps(a) + +static inline mm256i_emu mm256_cvtps_epi32(mm256_emu a) { + mm256i_emu ret; + ret.lo = _mm_cvtps_epi32(a.lo); + ret.hi = _mm_cvtps_epi32(a.hi); + return ret; +} +#define _mm256_cvtps_epi32(a) mm256_cvtps_epi32(a) + +#endif /* __AVX__ */ + + +#endif /* __AVX2__ */ + +/* In case we don't have FMA, make it a mul and an add. */ +#if !(defined(__FMA__) && defined(__AVX__)) +#define _mm256_fmadd_ps(a,b,c) _mm256_add_ps(_mm256_mul_ps(a, b), c) +#define _mm_fmadd_ps(a,b,c) _mm_add_ps(_mm_mul_ps(a, b), c) +#endif + +#ifdef __AVX2__ +static inline __m256 exp8_approx(__m256 X) +{ + const __m256 K0 = _mm256_set1_ps(0.99992522f); + const __m256 K1 = _mm256_set1_ps(0.69583354f); + const __m256 K2 = _mm256_set1_ps(0.22606716f); + const __m256 K3 = _mm256_set1_ps(0.078024523f); + const __m256 log2_E = _mm256_set1_ps(1.44269504f); + const __m256 max_in = _mm256_set1_ps(50.f); + const __m256 min_in = _mm256_set1_ps(-50.f); + __m256 XF, Y; + __m256i I; + X = _mm256_mul_ps(X, log2_E); + X = _mm256_max_ps(min_in, _mm256_min_ps(max_in, X)); + XF = _mm256_floor_ps(X); + I = _mm256_cvtps_epi32(XF); + X = _mm256_sub_ps(X, XF); + Y = _mm256_fmadd_ps(_mm256_fmadd_ps(_mm256_fmadd_ps(K3, X, K2), X, K1), X, K0); + I = _mm256_slli_epi32(I, 23); + Y = _mm256_castsi256_ps(_mm256_add_epi32(I, _mm256_castps_si256(Y))); + return Y; +} + +static inline void vector_ps_to_epi8(unsigned char *x, const float *_x, int len) { + int i; + __m256 const127 = _mm256_set1_ps(127.f); + for (i=0;i +#include "opus_types.h" +#include "common.h" + +#if defined(__arm__) && !defined(__aarch64__) && (__ARM_ARCH < 8 || !defined(__clang__)) +/* Emulate vcvtnq_s32_f32() for ARMv7 Neon. */ +static OPUS_INLINE int32x4_t vcvtnq_s32_f32(float32x4_t x) { + return vrshrq_n_s32(vcvtq_n_s32_f32(x, 8), 8); +} + +static OPUS_INLINE int16x8_t vpaddq_s16(int16x8_t a, int16x8_t b) { + return vcombine_s16(vpadd_s16(vget_low_s16(a), vget_high_s16(a)), vpadd_s16(vget_low_s16(b), vget_high_s16(b))); +} + +static OPUS_INLINE int16x8_t vmull_high_s8(int8x16_t a, int8x16_t b) { + return vmull_s8(vget_high_s8(a), vget_high_s8(b)); +} +#endif + +#ifdef __ARM_FEATURE_FMA +/* If we can, force the compiler to use an FMA instruction rather than break + vmlaq_f32() into fmul/fadd. */ +#define vmlaq_f32(a,b,c) vfmaq_f32(a,b,c) +#endif + +#ifndef LPCNET_TEST +static inline float32x4_t exp4_approx(float32x4_t x) { + int32x4_t i; + float32x4_t xf; + + x = vmaxq_f32(vminq_f32(x, vdupq_n_f32(88.f)), vdupq_n_f32(-88.f)); + + /* express exp(x) as exp2(x/log(2)), add 127 for the exponent later */ + x = vmlaq_f32(vdupq_n_f32(127.f), x, vdupq_n_f32(1.44269504f)); + + /* split into integer and fractional parts */ + i = vcvtq_s32_f32(x); + xf = vcvtq_f32_s32(i); + x = vsubq_f32(x, xf); + + float32x4_t K0 = vdupq_n_f32(0.99992522f); + float32x4_t K1 = vdupq_n_f32(0.69583354f); + float32x4_t K2 = vdupq_n_f32(0.22606716f); + float32x4_t K3 = vdupq_n_f32(0.078024523f); + float32x4_t Y = vmlaq_f32(K0, x, vmlaq_f32(K1, x, vmlaq_f32(K2, K3, x))); + + /* compute 2^i */ + float32x4_t exponent = vreinterpretq_f32_s32(vshlq_n_s32(i, 23)); + + Y = vmulq_f32(Y, exponent); + return Y; +} + +static inline float32x4_t tanh4_approx(float32x4_t X) +{ + const float32x4_t N0 = vdupq_n_f32(952.52801514f); + const float32x4_t N1 = vdupq_n_f32(96.39235687f); + const float32x4_t N2 = vdupq_n_f32(0.60863042f); + const float32x4_t D0 = vdupq_n_f32(952.72399902f); + const float32x4_t D1 = vdupq_n_f32(413.36801147f); + const float32x4_t D2 = vdupq_n_f32(11.88600922f); + const float32x4_t max_out = vdupq_n_f32(1.f); + const float32x4_t min_out = vdupq_n_f32(-1.f); + float32x4_t X2, num, den; + X2 = vmulq_f32(X, X); + num = vmlaq_f32(N0, X2, vmlaq_f32(N1, N2, X2)); + den = vmlaq_f32(D0, X2, vmlaq_f32(D1, D2, X2)); + num = vmulq_f32(num, X); + den = vrecpeq_f32(den); + num = vmulq_f32(num, den); + return vmaxq_f32(min_out, vminq_f32(max_out, num)); +} + +static inline float32x4_t sigmoid4_approx(float32x4_t X) +{ + const float32x4_t N0 = vdupq_n_f32(238.13200378f); + const float32x4_t N1 = vdupq_n_f32(6.02452230f); + const float32x4_t N2 = vdupq_n_f32(0.00950985f); + const float32x4_t D0 = vdupq_n_f32(952.72399902f); + const float32x4_t D1 = vdupq_n_f32(103.34200287f); + const float32x4_t D2 = vdupq_n_f32(0.74287558f); + const float32x4_t half = vdupq_n_f32(0.5f); + const float32x4_t max_out = vdupq_n_f32(1.f); + const float32x4_t min_out = vdupq_n_f32(0.f); + float32x4_t X2, num, den; + X2 = vmulq_f32(X, X); + num = vmlaq_f32(N0, X2, vmlaq_f32(N1, N2, X2)); + den = vmlaq_f32(D0, X2, vmlaq_f32(D1, D2, X2)); + num = vmulq_f32(num, X); + den = vrecpeq_f32(den); + num = vmlaq_f32(half, num, den); + return vmaxq_f32(min_out, vminq_f32(max_out, num)); +} + +static inline float lpcnet_exp(float x) +{ + float out[4]; + float32x4_t X, Y; + X = vdupq_n_f32(x); + Y = exp4_approx(X); + vst1q_f32(out, Y); + return out[0]; +} + +static inline float tanh_approx(float x) +{ + float out[4]; + float32x4_t X, Y; + X = vdupq_n_f32(x); + Y = tanh4_approx(X); + vst1q_f32(out, Y); + return out[0]; +} + +static inline float sigmoid_approx(float x) +{ + float out[4]; + float32x4_t X, Y; + X = vdupq_n_f32(x); + Y = sigmoid4_approx(X); + vst1q_f32(out, Y); + return out[0]; +} + +static inline void softmax(float *y, const float *x, int N) +{ + int i; + for (i=0;i +#include +#include +#include "nnet.h" +#include "arch.h" +#include "nnet.h" + +/* This is a bit of a hack because we need to build nnet_data.c and plc_data.c without USE_WEIGHTS_FILE, + but USE_WEIGHTS_FILE is defined in config.h. */ +#undef HAVE_CONFIG_H +#ifdef USE_WEIGHTS_FILE +#undef USE_WEIGHTS_FILE +#endif +#include "rnnoise_data.c" + +void write_weights(const WeightArray *list, FILE *fout) +{ + int i=0; + unsigned char zeros[WEIGHT_BLOCK_SIZE] = {0}; + while (list[i].name != NULL) { + WeightHead h; + if (strlen(list[i].name) >= sizeof(h.name) - 1) { + printf("[write_weights] warning: name %s too long\n", list[i].name); + } + memcpy(h.head, "DNNw", 4); + h.version = WEIGHT_BLOB_VERSION; + h.type = list[i].type; + h.size = list[i].size; + h.block_size = (h.size+WEIGHT_BLOCK_SIZE-1)/WEIGHT_BLOCK_SIZE*WEIGHT_BLOCK_SIZE; + RNN_CLEAR(h.name, sizeof(h.name)); + strncpy(h.name, list[i].name, sizeof(h.name)); + h.name[sizeof(h.name)-1] = 0; + celt_assert(sizeof(h) == WEIGHT_BLOCK_SIZE); + fwrite(&h, 1, WEIGHT_BLOCK_SIZE, fout); + fwrite(list[i].data, 1, h.size, fout); + fwrite(zeros, 1, h.block_size-h.size, fout); + i++; + } +} + +int main(void) +{ + FILE *fout = fopen("weights_blob.bin", "w"); + write_weights(rnnoise_arrays, fout); + fclose(fout); + return 0; +} diff --git a/cpp/ax650/src/rnnoise/x86/dnn_x86.h b/cpp/ax650/src/rnnoise/x86/dnn_x86.h new file mode 100644 index 0000000000000000000000000000000000000000..bf1069867494e2b50662c4653606c91b6bd083d8 --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/dnn_x86.h @@ -0,0 +1,85 @@ +/* Copyright (c) 2011-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef DNN_X86_H +#define DNN_X86_H + +#include "cpu_support.h" +#include "opus_types.h" + +void compute_linear_sse4_1(const LinearLayer *linear, float *out, const float *in); +void compute_activation_sse4_1(float *output, const float *input, int N, int activation); +void compute_conv2d_sse4_1(const Conv2dLayer *conv, float *out, float *mem, const float *in, int height, int hstride, int activation); + +void compute_linear_avx2(const LinearLayer *linear, float *out, const float *in); +void compute_activation_avx2(float *output, const float *input, int N, int activation); +void compute_conv2d_avx2(const Conv2dLayer *conv, float *out, float *mem, const float *in, int height, int hstride, int activation); + + + +#ifdef RNN_ENABLE_X86_RTCD + +extern void (*const RNN_COMPUTE_LINEAR_IMPL[OPUS_ARCHMASK + 1])( + const LinearLayer *linear, + float *out, + const float *in + ); +#define OVERRIDE_COMPUTE_LINEAR +#define compute_linear(linear, out, in, arch) \ + ((*RNN_COMPUTE_LINEAR_IMPL[(arch) & OPUS_ARCHMASK])(linear, out, in)) + + +extern void (*const RNN_COMPUTE_ACTIVATION_IMPL[OPUS_ARCHMASK + 1])( + float *output, + const float *input, + int N, + int activation + ); +#define OVERRIDE_COMPUTE_ACTIVATION +#define compute_activation(output, input, N, activation, arch) \ + ((*RNN_COMPUTE_ACTIVATION_IMPL[(arch) & OPUS_ARCHMASK])(output, input, N, activation)) + + +extern void (*const RNN_COMPUTE_CONV2D_IMPL[OPUS_ARCHMASK + 1])( + const Conv2dLayer *conv, + float *out, + float *mem, + const float *in, + int height, + int hstride, + int activation + ); +#define OVERRIDE_COMPUTE_CONV2D +#define compute_conv2d(conv, out, mem, in, height, hstride, activation, arch) \ + ((*RNN_COMPUTE_CONV2D_IMPL[(arch) & OPUS_ARCHMASK])(conv, out, mem, in, height, hstride, activation)) + + +#endif + + + +#endif /* DNN_X86_H */ diff --git a/cpp/ax650/src/rnnoise/x86/nnet_avx2.c b/cpp/ax650/src/rnnoise/x86/nnet_avx2.c new file mode 100644 index 0000000000000000000000000000000000000000..41037fcc89e2304ee0aa5b46a7814f7f4568f47d --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/nnet_avx2.c @@ -0,0 +1,40 @@ +/* Copyright (c) 2018-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include "x86/x86_arch_macros.h" + +#ifndef __AVX2__ +#error nnet_avx2.c is being compiled without AVX2 enabled +#endif + +#define RTCD_ARCH avx2 + +#include "nnet_arch.h" diff --git a/cpp/ax650/src/rnnoise/x86/nnet_sse4_1.c b/cpp/ax650/src/rnnoise/x86/nnet_sse4_1.c new file mode 100644 index 0000000000000000000000000000000000000000..224926e5a9754a6d28ab98d4fde2a3abea43410d --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/nnet_sse4_1.c @@ -0,0 +1,40 @@ +/* Copyright (c) 2018-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include "x86/x86_arch_macros.h" + +#ifndef __SSE4_1__ +#error nnet_sse4_1.c is being compiled without SSE4.1 enabled +#endif + +#define RTCD_ARCH sse4_1 + +#include "nnet_arch.h" diff --git a/cpp/ax650/src/rnnoise/x86/x86_arch_macros.h b/cpp/ax650/src/rnnoise/x86/x86_arch_macros.h new file mode 100644 index 0000000000000000000000000000000000000000..975b443e9311fe015182b555a581d35b1cb0a936 --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/x86_arch_macros.h @@ -0,0 +1,47 @@ +/* Copyright (c) 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef _MSC_VER + +# ifdef OPUS_X86_MAY_HAVE_SSE +# ifndef __SSE__ +# define __SSE__ +# endif +# endif + +# ifdef OPUS_X86_MAY_HAVE_SSE2 +# ifndef __SSE2__ +# define __SSE2__ +# endif +# endif + +# ifdef OPUS_X86_MAY_HAVE_SSE4_1 +# ifndef __SSE4_1__ +# define __SSE4_1__ +# endif +# endif + +#endif diff --git a/cpp/ax650/src/rnnoise/x86/x86_dnn_map.c b/cpp/ax650/src/rnnoise/x86/x86_dnn_map.c new file mode 100644 index 0000000000000000000000000000000000000000..a16fa147f61867973cc9f3014ec9c637ed661b05 --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/x86_dnn_map.c @@ -0,0 +1,74 @@ +/* Copyright (c) 2018-2019 Mozilla + 2023 Amazon */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include "x86/x86cpu.h" +#include "nnet.h" + +#ifdef RNN_ENABLE_X86_RTCD + + +void (*const RNN_COMPUTE_LINEAR_IMPL[OPUS_ARCHMASK + 1])( + const LinearLayer *linear, + float *out, + const float *in +) = { + compute_linear_c, /* non-sse */ + MAY_HAVE_SSE4_1(compute_linear), /* sse4.1 */ + MAY_HAVE_AVX2(compute_linear) /* avx */ +}; + +void (*const RNN_COMPUTE_ACTIVATION_IMPL[OPUS_ARCHMASK + 1])( + float *output, + const float *input, + int N, + int activation +) = { + compute_activation_c, /* non-sse */ + MAY_HAVE_SSE4_1(compute_activation), /* sse4.1 */ + MAY_HAVE_AVX2(compute_activation) /* avx */ +}; + +void (*const RNN_COMPUTE_CONV2D_IMPL[OPUS_ARCHMASK + 1])( + const Conv2dLayer *conv, + float *out, + float *mem, + const float *in, + int height, + int hstride, + int activation +) = { + compute_conv2d_c, /* non-sse */ + MAY_HAVE_SSE4_1(compute_conv2d), /* sse4.1 */ + MAY_HAVE_AVX2(compute_conv2d) /* avx */ +}; + + +#endif diff --git a/cpp/ax650/src/rnnoise/x86/x86cpu.c b/cpp/ax650/src/rnnoise/x86/x86cpu.c new file mode 100644 index 0000000000000000000000000000000000000000..370c909be6dc91488566172577ee5efb6ce7d05e --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/x86cpu.c @@ -0,0 +1,166 @@ +/* Copyright (c) 2014, Cisco Systems, INC + Written by XiangMingZhu WeiZhou MinPeng YanWang + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include "cpu_support.h" +#include "pitch.h" +#include "x86cpu.h" + +#ifdef RNN_ENABLE_X86_RTCD + +#if defined(_MSC_VER) + +#include +static _inline void cpuid(unsigned int CPUInfo[4], unsigned int InfoType) +{ + __cpuid((int*)CPUInfo, InfoType); +} + +#else + +#if defined(CPU_INFO_BY_C) +#include +#endif + +static void cpuid(unsigned int CPUInfo[4], unsigned int InfoType) +{ +#if defined(CPU_INFO_BY_ASM) +#if defined(__i386__) && defined(__PIC__) +/* %ebx is PIC register in 32-bit, so mustn't clobber it. */ + __asm__ __volatile__ ( + "xchg %%ebx, %1\n" + "cpuid\n" + "xchg %%ebx, %1\n": + "=a" (CPUInfo[0]), + "=r" (CPUInfo[1]), + "=c" (CPUInfo[2]), + "=d" (CPUInfo[3]) : + /* We clear ECX to avoid a valgrind false-positive prior to v3.17.0. */ + "0" (InfoType), "2" (0) + ); +#else + __asm__ __volatile__ ( + "cpuid": + "=a" (CPUInfo[0]), + "=b" (CPUInfo[1]), + "=c" (CPUInfo[2]), + "=d" (CPUInfo[3]) : + /* We clear ECX to avoid a valgrind false-positive prior to v3.17.0. */ + "0" (InfoType), "2" (0) + ); +#endif +#elif defined(CPU_INFO_BY_C) + /* We use __get_cpuid_count to clear ECX to avoid a valgrind false-positive + prior to v3.17.0.*/ + if (!__get_cpuid_count(InfoType, 0, &(CPUInfo[0]), &(CPUInfo[1]), &(CPUInfo[2]), &(CPUInfo[3]))) { + /* Our function cannot fail, but __get_cpuid{_count} can. + Returning all zeroes will effectively disable all SIMD, which is + what we want on CPUs that don't support CPUID. */ + CPUInfo[3] = CPUInfo[2] = CPUInfo[1] = CPUInfo[0] = 0; + } +#else +# error "Configured to use x86 RTCD, but no CPU detection method available. " \ + "Reconfigure with --disable-rtcd (or send patches)." +#endif +} + +#endif + +typedef struct CPU_Feature{ + /* SIMD: 128-bit */ + int HW_SSE; + int HW_SSE2; + int HW_SSE41; + /* SIMD: 256-bit */ + int HW_AVX2; +} CPU_Feature; + +static void rnn_cpu_feature_check(CPU_Feature *cpu_feature) +{ + unsigned int info[4]; + unsigned int nIds = 0; + + cpuid(info, 0); + nIds = info[0]; + + if (nIds >= 1){ + cpuid(info, 1); + cpu_feature->HW_SSE = (info[3] & (1 << 25)) != 0; + cpu_feature->HW_SSE2 = (info[3] & (1 << 26)) != 0; + cpu_feature->HW_SSE41 = (info[2] & (1 << 19)) != 0; + cpu_feature->HW_AVX2 = (info[2] & (1 << 28)) != 0 && (info[2] & (1 << 12)) != 0; + if (cpu_feature->HW_AVX2 && nIds >= 7) { + cpuid(info, 7); + cpu_feature->HW_AVX2 = cpu_feature->HW_AVX2 && (info[1] & (1 << 5)) != 0; + } else { + cpu_feature->HW_AVX2 = 0; + } + } + else { + cpu_feature->HW_SSE = 0; + cpu_feature->HW_SSE2 = 0; + cpu_feature->HW_SSE41 = 0; + cpu_feature->HW_AVX2 = 0; + } +} + +static int rnn_select_arch_impl(void) +{ + CPU_Feature cpu_feature; + int arch; + + rnn_cpu_feature_check(&cpu_feature); + + arch = 0; + if (!cpu_feature.HW_SSE41) + { + return arch; + } + arch++; + + if (!cpu_feature.HW_AVX2) + { + return arch; + } + arch++; + + return arch; +} + +int rnn_select_arch(void) { + int arch = rnn_select_arch_impl(); +#ifdef FUZZING + /* Randomly downgrade the architecture. */ + arch = rand()%(arch+1); +#endif + return arch; +} + +#endif diff --git a/cpp/ax650/src/rnnoise/x86/x86cpu.h b/cpp/ax650/src/rnnoise/x86/x86cpu.h new file mode 100644 index 0000000000000000000000000000000000000000..e214abab547a8f6550163dd15bac5e5131a981db --- /dev/null +++ b/cpp/ax650/src/rnnoise/x86/x86cpu.h @@ -0,0 +1,88 @@ +/* Copyright (c) 2014, Cisco Systems, INC + Written by XiangMingZhu WeiZhou MinPeng YanWang + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER + OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#if !defined(X86CPU_H) +# define X86CPU_H + +# define MAY_HAVE_SSE4_1(name) name ## _sse4_1 + +# define MAY_HAVE_AVX2(name) name ## _avx2 + +# ifdef RNN_ENABLE_X86_RTCD +int opus_select_arch(void); +# endif + +# if defined(__SSE2__) +# include "common.h" + +/*MOVD should not impose any alignment restrictions, but the C standard does, + and UBSan will report errors if we actually make unaligned accesses. + Use this to work around those restrictions (which should hopefully all get + optimized to a single MOVD instruction). + GCC implemented _mm_loadu_si32() since GCC 11; HOWEVER, there is a bug! + https://gcc.gnu.org/bugzilla/show_bug.cgi?id=99754 + LLVM implemented _mm_loadu_si32() since Clang 8.0, however the + __clang_major__ version number macro is unreliable, as vendors + (specifically, Apple) will use different numbering schemes than upstream. + Clang's advice is "use feature detection", but they do not provide feature + detection support for specific SIMD functions. + We follow the approach from the SIMDe project and instead detect unrelated + features that should be available in the version we want (see + ).*/ +# if defined(__clang__) +# if __has_warning("-Wextra-semi-stmt") || \ + __has_builtin(__builtin_rotateleft32) +# define OPUS_CLANG_8 (1) +# endif +# endif +# if !defined(_MSC_VER) && !OPUS_GNUC_PREREQ(11,3) && !defined(OPUS_CLANG_8) +# include +# include + +# ifdef _mm_loadu_si32 +# undef _mm_loadu_si32 +# endif +# define _mm_loadu_si32 WORKAROUND_mm_loadu_si32 +static inline __m128i WORKAROUND_mm_loadu_si32(void const* mem_addr) { + int val; + memcpy(&val, mem_addr, sizeof(val)); + return _mm_cvtsi32_si128(val); +} +# elif defined(_MSC_VER) + /* MSVC needs this for _mm_loadu_si32 */ +# include +# endif + +# define OP_CVTEPI8_EPI32_M32(x) \ + (_mm_cvtepi8_epi32(_mm_loadu_si32(x))) + +# define OP_CVTEPI16_EPI32_M64(x) \ + (_mm_cvtepi16_epi32(_mm_loadl_epi64((__m128i *)(void*)(x)))) + +# endif + +#endif diff --git a/cpp/ax650/src/rnnoise_ax.cpp b/cpp/ax650/src/rnnoise_ax.cpp new file mode 100644 index 0000000000000000000000000000000000000000..c757bd558ebdb549da63c3ef7c323dadc214ac55 --- /dev/null +++ b/cpp/ax650/src/rnnoise_ax.cpp @@ -0,0 +1,115 @@ +// RNNoise AX SDK:用 AX Engine 替换原版 compute_rnn(网络推理), +// 其余信号处理(biquad/FFT/pitch/合成)沿用原版 C 实现。 +#include "rnnoise_ax.hpp" + +#include "model_runner.hpp" + +#include +#include +#include +#include + +extern "C" { +#include "denoise.h" +#include "rnn.h" +} + +namespace { + +// 单实例模型 runner(compute_rnn 无状态上下文可用,采用全局注册方式; +// 同一时刻只允许一个 RNNoiseAX 实例执行推理)。 +std::mutex g_runner_mu; +ModelRunner* g_ax_runner = nullptr; + +extern "C" void rnnoise_ax_set_runner(ModelRunner* runner) { + std::lock_guard lk(g_runner_mu); + g_ax_runner = runner; +} + +extern "C" void compute_rnn(const RNNoise* model, RNNState* rnn, + float* gains, float* vad, + const float* input, int arch) { + (void)model; + (void)arch; + ModelRunner* runner = nullptr; + { + std::lock_guard lk(g_runner_mu); + runner = g_ax_runner; + } + if (runner == nullptr) { + throw std::runtime_error("compute_rnn: AX runner 未初始化"); + } + + std::vector> feeds = { + std::vector(input, input + NB_FEATURES), + std::vector(rnn->conv1_state, + rnn->conv1_state + CONV1_STATE_SIZE), + std::vector(rnn->conv2_state, + rnn->conv2_state + CONV2_STATE_SIZE), + std::vector(rnn->gru1_state, + rnn->gru1_state + GRU1_OUT_SIZE), + std::vector(rnn->gru2_state, + rnn->gru2_state + GRU2_OUT_SIZE), + std::vector(rnn->gru3_state, + rnn->gru3_state + GRU3_OUT_SIZE), + }; + std::vector> outs = runner->Run(feeds); + + std::memcpy(gains, outs[0].data(), NB_BANDS * sizeof(float)); + *vad = outs[1][0]; + std::memcpy(rnn->conv1_state, outs[2].data(), + CONV1_STATE_SIZE * sizeof(float)); + std::memcpy(rnn->conv2_state, outs[3].data(), + CONV2_STATE_SIZE * sizeof(float)); + std::memcpy(rnn->gru1_state, outs[4].data(), + GRU1_OUT_SIZE * sizeof(float)); + std::memcpy(rnn->gru2_state, outs[5].data(), + GRU2_OUT_SIZE * sizeof(float)); + std::memcpy(rnn->gru3_state, outs[6].data(), + GRU3_OUT_SIZE * sizeof(float)); +} + +} // namespace + +struct RNNoiseAX::Impl { + std::unique_ptr runner; + DenoiseState* state = nullptr; + + explicit Impl(const std::string& model_path) + : runner(new ModelRunner(model_path)), + state(rnnoise_create(nullptr)) { + if (state == nullptr) { + throw std::runtime_error("rnnoise_create 失败"); + } + rnnoise_ax_set_runner(runner.get()); + } + + ~Impl() { + rnnoise_ax_set_runner(nullptr); + if (state) { + rnnoise_destroy(state); + } + } +}; + +RNNoiseAX::RNNoiseAX(const std::string& model_path) + : impl_(new Impl(model_path)) {} + +RNNoiseAX::~RNNoiseAX() = default; + +void RNNoiseAX::Reset() { + if (!impl_) { + return; + } + rnnoise_ax_set_runner(nullptr); + rnnoise_destroy(impl_->state); + impl_->state = rnnoise_create(nullptr); + if (impl_->state == nullptr) { + throw std::runtime_error("rnnoise_create 失败"); + } + rnnoise_ax_set_runner(impl_->runner.get()); +} + +float RNNoiseAX::ProcessFrame(float* out, const float* in) { + return rnnoise_process_frame(impl_->state, out, in); +} diff --git a/python/demo.py b/python/demo.py index 9f3de03b86a4db22bc5572d2cb93442ee6fa2b84..d3902404c8d4c926e7fec98f86120838d7e450ce 100644 --- a/python/demo.py +++ b/python/demo.py @@ -1,41 +1,71 @@ -"""RNNoise AX650 降噪一键演示(复制即用)。 +"""RNNoise 48k 实时降噪演示(AX650 / AX620E 双芯)。 -板端:bash setup.sh && bash run.sh(自带 sample_speech.pcm 演示样本)。 -非 AX 主机:提示在板端运行后正常退出。 +用法: + python3 demo.py --chip ax650 # 默认处理 python/sample_speech.pcm + python3 demo.py --chip ax620e --input in.pcm + +输入格式:48kHz f32le PCM(16-bit 等价域,±32768,不做归一化);也可传 16-bit WAV。 """ +import argparse import sys +import wave from pathlib import Path +import numpy as np + ROOT = Path(__file__).resolve().parents[1] +CHIP_MODELS = { + "ax650": ROOT / "rnnoise_ax650" / "model.axmodel", + "ax620e": ROOT / "rnnoise_ax620e" / "model.axmodel", +} -try: - import axengine # noqa: F401 - AX_AVAILABLE = True -except Exception: - AX_AVAILABLE = False + +def load_pcm(path: Path) -> np.ndarray: + if path.suffix.lower() == ".wav": + with wave.open(str(path), "rb") as w: + assert w.getframerate() == 48000, "仅支持 48kHz WAV" + assert w.getsampwidth() == 2, "仅支持 16-bit PCM WAV" + raw = w.readframes(w.getnframes()) + return np.frombuffer(raw, dtype=" None: + parser = argparse.ArgumentParser(description="RNNoise 48k 实时降噪(AX650/AX620E)") + parser.add_argument("--chip", choices=["ax650", "ax620e"], default="ax650", + help="目标芯片,决定使用哪个 axmodel") + parser.add_argument("--input", default=str(ROOT / "python" / "sample_speech.pcm"), + help="48k f32 PCM 或 16-bit WAV") + parser.add_argument("--output-dir", default="output") + args = parser.parse_args() + + try: + import axengine # noqa: F401 + AX_AVAILABLE = True + except Exception: + AX_AVAILABLE = False if not AX_AVAILABLE: print("当前主机没有 AX 芯片(pyaxengine 不可用),无法运行 NPU 推理。") - print("本交付包为 NPU 专用版,请在 AX650/AX630 板端执行:") - print(" bash setup.sh && bash run.sh") + print("请在对应 AX 板端执行:python3 python/demo.py --chip ax650|ax620e") return - import numpy as np - from rnnoise_ax650_sdk import RNNoiseDenoiser, dsp + sys.path.insert(0, str(ROOT / "python")) + from rnnoise_sdk import RNNoiseDenoiser, dsp - model = ROOT / "models" / "model.axmodel" - sample = ROOT / "python" / "sample_speech.pcm" + model = CHIP_MODELS[args.chip] + if not model.is_file(): + print(f"模型不存在: {model}(请确认仓库完整)") + sys.exit(1) + pcm = load_pcm(Path(args.input)) + print(f"chip: {args.chip} | model: {model.name} | " + f"input: {pcm.size / 48000:.2f}s ({pcm.size // dsp.FRAME_SIZE} 帧)") denoiser = RNNoiseDenoiser(str(model)) - pcm = np.fromfile(sample, dtype=np.float32) out, vads = denoiser.process(pcm) - out_dir = ROOT / "output" + out_dir = Path(args.output_dir) out_dir.mkdir(exist_ok=True) out.astype(np.float32).tofile(out_dir / "out.pcm") np.save(out_dir / "vad.npy", vads) - print(f"模型加载成功:{model}") - print(f"帧数: {vads.size}(每帧 10ms)") + print(f"backend: {denoiser.backend}") print(f"语音存在比例: {float((vads > 0.5).mean()):.2f}") print(f"输出已保存: {out_dir / 'out.pcm'}") diff --git a/python/rnnoise_ax650_sdk/README.md b/python/rnnoise_sdk/README.md similarity index 84% rename from python/rnnoise_ax650_sdk/README.md rename to python/rnnoise_sdk/README.md index 4a639a526e1215c41d6f72553b404bdc74db26e6..5174d8f1cc1050ef33fa189db9126c6fb177d556 100644 --- a/python/rnnoise_ax650_sdk/README.md +++ b/python/rnnoise_sdk/README.md @@ -1,6 +1,6 @@ -# rnnoise-ax650 Python SDK +# rnnoise-ax Python SDK -48kHz 单声道实时降噪(RNNoise,AX650 NPU3 编译)。 +48kHz 单声道实时降噪(RNNoise,AX620E NPU3 编译)。 - 输入:480 采样/帧 float32(16-bit PCM 等价域 ±32768,不做归一化) - 输出:去噪帧(480) + vad @@ -10,7 +10,7 @@ 频谱合成),对照 `c_ref` 逐帧验证:特征 cosine≥0.995、去噪输出 cosine≥0.995 ```python -from rnnoise_ax650_sdk import RNNoiseDenoiser +from rnnoise_sdk import RNNoiseDenoiser import numpy as np denoiser = RNNoiseDenoiser("model.axmodel") diff --git a/python/rnnoise_ax650_sdk/__init__.py b/python/rnnoise_sdk/__init__.py similarity index 100% rename from python/rnnoise_ax650_sdk/__init__.py rename to python/rnnoise_sdk/__init__.py diff --git a/python/rnnoise_ax650_sdk/dsp.py b/python/rnnoise_sdk/dsp.py similarity index 100% rename from python/rnnoise_ax650_sdk/dsp.py rename to python/rnnoise_sdk/dsp.py diff --git a/python/rnnoise_ax650_sdk/example.py b/python/rnnoise_sdk/example.py similarity index 97% rename from python/rnnoise_ax650_sdk/example.py rename to python/rnnoise_sdk/example.py index c455d3ade37a2b6b2a37414798b417b69ab52c41..0318ef1b6ed5fbce59a992a8e77b90f1fc20799b 100644 --- a/python/rnnoise_ax650_sdk/example.py +++ b/python/rnnoise_sdk/example.py @@ -15,7 +15,7 @@ import numpy as np sys.path.insert(0, str(Path(__file__).resolve().parents[1])) -from rnnoise_ax650_sdk import RNNoiseDenoiser, dsp # noqa: E402 +from rnnoise_sdk import RNNoiseDenoiser, dsp # noqa: E402 def load_pcm(path: Path) -> np.ndarray: diff --git a/python/rnnoise_ax650_sdk/inference.py b/python/rnnoise_sdk/inference.py similarity index 82% rename from python/rnnoise_ax650_sdk/inference.py rename to python/rnnoise_sdk/inference.py index 9ff2fd0a36f610830cea8963b772af0d4e4228b4..c1f3f477c8f61263c93f2ba59edd26997a7e78e8 100644 --- a/python/rnnoise_ax650_sdk/inference.py +++ b/python/rnnoise_sdk/inference.py @@ -1,4 +1,4 @@ -"""RNNoise AX650 推理会话(NPU 专用发布版:无 onnxruntime/torch 回退)。""" +"""RNNoise AX620E 推理会话(开发版:默认 AX Engine,本机可用 onnxruntime CPU 回退验证)。""" import numpy as np from . import dsp @@ -16,19 +16,19 @@ _STATE_INPUTS = ["conv1_mem", "conv2_mem", "gru1_s", "gru2_s", "gru3_s"] class RNNoiseDenoiser: - """48k 单声道实时降噪器(AX 芯片端到端,无 CPU 回退)。""" + """48k 单声道实时降噪器(默认 AX 芯片;本机验证时回退 onnxruntime CPU)。""" def __init__(self, model_path, providers=None): + self.backend = "axengine" try: import axengine as axe - except ImportError as exc: - raise RuntimeError( - "SDK 为 NPU 专用发布版,仅支持在 AX 芯片上运行;请先安装 " - "requirements.txt 并在板端执行(无 onnxruntime/torch 回退)" - ) from exc - self.session = axe.InferenceSession( - model_path, providers=providers or [DEFAULT_PROVIDER]) - self.backend = "axengine" + self.session = axe.InferenceSession( + model_path, providers=providers or [DEFAULT_PROVIDER]) + except Exception: + import onnxruntime as ort + self.session = ort.InferenceSession( + model_path, providers=["CPUExecutionProvider"]) + self.backend = "onnxruntime" self.input_names = [i.name for i in self.session.get_inputs()] self.output_names = [o.name for o in self.session.get_outputs()] self.reset() diff --git a/python/rnnoise_sdk/inference_npu.py b/python/rnnoise_sdk/inference_npu.py new file mode 100644 index 0000000000000000000000000000000000000000..0b148bad2e51f552547b5d6f890636c7cbcd5d7d --- /dev/null +++ b/python/rnnoise_sdk/inference_npu.py @@ -0,0 +1,77 @@ +"""RNNoise AX620E 推理会话(NPU 专用发布版:无 onnxruntime/torch 回退)。""" +import numpy as np + +from . import dsp + +DEFAULT_PROVIDER = "AxEngineExecutionProvider" + +INPUT_NAMES = ["features", "conv1_mem", "conv2_mem", + "gru1_s", "gru2_s", "gru3_s"] +INPUT_SHAPES = {"features": (1, 65), "conv1_mem": (1, 130), + "conv2_mem": (1, 256), "gru1_s": (1, 384), + "gru2_s": (1, 384), "gru3_s": (1, 384)} +OUTPUT_NAMES = ["gains", "vad", "conv1_mem_new", "conv2_mem_new", + "gru1_s_new", "gru2_s_new", "gru3_s_new"] +_STATE_INPUTS = ["conv1_mem", "conv2_mem", "gru1_s", "gru2_s", "gru3_s"] + + +class RNNoiseDenoiser: + """48k 单声道实时降噪器(AX 芯片端到端,无 CPU 回退)。""" + + def __init__(self, model_path, providers=None): + try: + import axengine as axe + except ImportError as exc: + raise RuntimeError( + "SDK 为 NPU 专用发布版,仅支持在 AX 芯片上运行;请先安装 " + "requirements.txt 并在板端执行(无 onnxruntime/torch 回退)" + ) from exc + self.session = axe.InferenceSession( + model_path, providers=providers or [DEFAULT_PROVIDER]) + self.backend = "axengine" + self.input_names = [i.name for i in self.session.get_inputs()] + self.output_names = [o.name for o in self.session.get_outputs()] + self.reset() + + def reset(self): + self.st = dsp.RNNoiseState() + self._states = {k: np.zeros(INPUT_SHAPES[k], dtype=np.float32) + for k in _STATE_INPUTS} + + def process_frame(self, frame): + frame = np.asarray(frame, dtype=np.float32).reshape(-1) + if frame.size != dsp.FRAME_SIZE: + raise ValueError(f"帧长必须为 {dsp.FRAME_SIZE},实际 {frame.size}") + ana = dsp.analyze_frame(self.st, frame) + if ana["silence"]: + out, _ = dsp.synthesize_frame(self.st, ana, None, 0.0) + return out, 0.0 + feeds = { + "features": np.ascontiguousarray( + ana["features"][None, :].astype(np.float32)), + } + feeds.update({k: np.ascontiguousarray(v) + for k, v in self._states.items()}) + outs = self.session.run(None, feeds) + out_idx = {n: i for i, n in enumerate(self.output_names)} + gains = outs[out_idx["gains"]] + vad = outs[out_idx["vad"]] + for k in _STATE_INPUTS: + self._states[k] = np.asarray( + outs[out_idx[k + "_new"]], dtype=np.float32) + out, vad = dsp.synthesize_frame(self.st, ana, gains, vad) + return out, float(vad) + + def process(self, pcm): + pcm = np.asarray(pcm, dtype=np.float32) + if pcm.ndim == 1: + n = pcm.size // dsp.FRAME_SIZE + frames = pcm[:n * dsp.FRAME_SIZE].reshape(n, dsp.FRAME_SIZE) + else: + frames = pcm.reshape(-1, dsp.FRAME_SIZE) + outs, vads = [], [] + for fr in frames: + o, v = self.process_frame(fr) + outs.append(o) + vads.append(v) + return np.concatenate(outs), np.asarray(vads, dtype=np.float32) diff --git a/python/rnnoise_ax650_sdk/postprocess.py b/python/rnnoise_sdk/postprocess.py similarity index 100% rename from python/rnnoise_ax650_sdk/postprocess.py rename to python/rnnoise_sdk/postprocess.py diff --git a/python/rnnoise_ax650_sdk/preprocess.py b/python/rnnoise_sdk/preprocess.py similarity index 100% rename from python/rnnoise_ax650_sdk/preprocess.py rename to python/rnnoise_sdk/preprocess.py diff --git a/python/rnnoise_ax650_sdk/requirements.txt b/python/rnnoise_sdk/requirements.txt similarity index 100% rename from python/rnnoise_ax650_sdk/requirements.txt rename to python/rnnoise_sdk/requirements.txt diff --git a/reports/ax620e/compile_report.md b/reports/ax620e/compile_report.md new file mode 100644 index 0000000000000000000000000000000000000000..4242efdc57c057a74cb854829a4585bbc757bd0f --- /dev/null +++ b/reports/ax620e/compile_report.md @@ -0,0 +1,7 @@ +# Compile Report + +- image: docker-registry.aitsw.axera-tech.com/pulsar2:20260810-temp-0d4427ff +- target: AX620E +- input: features:1x65,conv1_mem:1x130,conv2_mem:1x256,gru1_s:1x384,gru2_s:1x384,gru3_s:1x384 +- src_dtype: FP32 +- size: 2925.5 KB diff --git a/reports/ax620e/export_report.md b/reports/ax620e/export_report.md new file mode 100644 index 0000000000000000000000000000000000000000..ef33c4d5a92f7393a17fb2a947d18015e20081d4 --- /dev/null +++ b/reports/ax620e/export_report.md @@ -0,0 +1,8 @@ +# Export Report (AX620E 复用) + +- ONNX: export/model.onnx(opset 13, 静态 shape, 6 输入 / 7 输出),复用 rnnoise-ax650 已验证产物 +- 权重来源: 官方 rnnoise_data.c(float 数组 + 稀疏重建 + diag),tanh/sigmoid 复刻 C 端有理逼近 +- 原对分验证(198 帧真实语音特征序列): gains cosine 1.000000, vad cosine 1.000000(ax650 任务完成) +- 本次复核: onnxruntime 加载通过,全静态 shape,用 calib_data 首帧推理输出 shape 正确且有限 +- 校准数据: calib_data/.tar.gz,来自真实语音(speech/speech-echo/speech-reverb + 合成噪声 6dB),每输入 40 帧特征+状态轨迹(real 业务数据) +- 状态语义: 逐帧推理,conv1/conv2 mem 各保留 2 帧,GRU 状态 384x3 diff --git a/reports/ax620e/simulate_report.md b/reports/ax620e/simulate_report.md new file mode 100644 index 0000000000000000000000000000000000000000..06b4721849fdb9747d2d920744d1ff36c81ad5db --- /dev/null +++ b/reports/ax620e/simulate_report.md @@ -0,0 +1,24 @@ +# Simulate Report + +Method: pulsar2 run(无 AX620E 板回退仿真) +target: AX620E NPU2 +frames: 30~129(预热 30 帧后状态化推理 100 帧) + +- gains_cosine: 0.998734 +- gains_mae: 0.017887 +- vad_cosine: 0.999971 +- vad_mae: 0.002027 +- conv1_mem_new_cosine: 0.999442 +- conv1_mem_new_mae: 0.001393 +- conv2_mem_new_cosine: 0.997951 +- conv2_mem_new_mae: 0.019488 +- gru1_s_new_cosine: 0.996999 +- gru1_s_new_mae: 0.022802 +- gru2_s_new_cosine: 0.996491 +- gru2_s_new_mae: 0.026434 +- gru3_s_new_cosine: 0.996274 +- gru3_s_new_mae: 0.031290 +- gains_max_abs_diff: 0.285138 +- vad_max_abs_diff: 0.053815 + +结论:gains cosine 0.9987 ≥ 0.99 ✅(参考 AX650 板端 gains 0.9991,U16 链路一致) \ No newline at end of file diff --git a/reports/compile_report.md b/reports/ax650/compile_report.md similarity index 100% rename from reports/compile_report.md rename to reports/ax650/compile_report.md diff --git a/reports/export_report.md b/reports/ax650/export_report.md similarity index 100% rename from reports/export_report.md rename to reports/ax650/export_report.md diff --git a/reports/runonboard_report.md b/reports/ax650/runonboard_report.md similarity index 100% rename from reports/runonboard_report.md rename to reports/ax650/runonboard_report.md diff --git a/reports/simulate_report.md b/reports/ax650/simulate_report.md similarity index 100% rename from reports/simulate_report.md rename to reports/ax650/simulate_report.md diff --git a/rnnoise_ax620e/model.axmodel b/rnnoise_ax620e/model.axmodel new file mode 100644 index 0000000000000000000000000000000000000000..e5bd75069e536f4b202ae0f949889421e95e5156 --- /dev/null +++ b/rnnoise_ax620e/model.axmodel @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:581c5243b7a6e8fbedc20e5b17b040e1783767d8194fa0409866cdc0c21c3ec9 +size 2995746 diff --git a/rnnoise_ax620e/model_meta.json b/rnnoise_ax620e/model_meta.json new file mode 100644 index 0000000000000000000000000000000000000000..7b984ee91ca3196c8d609ea22b63f872064848fb --- /dev/null +++ b/rnnoise_ax620e/model_meta.json @@ -0,0 +1,140 @@ +{ + "model_name": "rnnoise-ax620e", + "framework": "pytorch->onnx", + "task": "noise_suppression", + "route": "general", + "opset": 13, + "inputs": [ + { + "name": "features", + "shape": [ + 1, + 65 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "conv1_mem", + "shape": [ + 1, + 130 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "conv2_mem", + "shape": [ + 1, + 256 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru1_s", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru2_s", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru3_s", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + } + ], + "outputs": [ + { + "name": "gains", + "shape": [ + 1, + 32 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "vad", + "shape": [ + 1, + 1 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "conv1_mem_new", + "shape": [ + 1, + 130 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "conv2_mem_new", + "shape": [ + 1, + 256 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru1_s_new", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru2_s_new", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + }, + { + "name": "gru3_s_new", + "shape": [ + 1, + 384 + ], + "dtype": "float32", + "layout": "NC" + } + ], + "stateful": true, + "frame_size": 480, + "sample_rate": 48000, + "preprocess": "48k PCM 帧(480) -> biquad HP -> FFT(960) -> 32 波段能量/DCT + pitch 特征 -> 65 维特征", + "input_domain": "float 等价 16-bit PCM(±32768 量级,不做归一化,与官方 demo 一致)", + "postprocess": "32 gains + vad -> gain 平滑/限幅 -> pitch filter -> 频谱合成 -> 480 样本", + "sdk_interface": { + "entry": "rnnoise_process_frame", + "args": "float32 帧(480) + 状态(内部维护)", + "returns": "float32 去噪帧 + vad" + }, + "license": "isc" +} \ No newline at end of file diff --git a/models/model.axmodel b/rnnoise_ax650/model.axmodel similarity index 100% rename from models/model.axmodel rename to rnnoise_ax650/model.axmodel diff --git a/models/model_meta.json b/rnnoise_ax650/model_meta.json similarity index 100% rename from models/model_meta.json rename to rnnoise_ax650/model_meta.json diff --git a/run.sh b/run.sh deleted file mode 100644 index b62a8039512f680738553984514c402959c57671..0000000000000000000000000000000000000000 --- a/run.sh +++ /dev/null @@ -1,7 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -echo "=== 运行推理 ===" -if command -v python3 >/dev/null 2>&1; then PY=python3; else PY=python; fi -"$PY" python/demo.py -# ./cpp/build/model_example models/model.axmodel diff --git a/setup.sh b/setup.sh deleted file mode 100644 index ba0819e8826b6dafd7c791e82941fa00ab724728..0000000000000000000000000000000000000000 --- a/setup.sh +++ /dev/null @@ -1,21 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -echo "=== 安装依赖 ===" -PIP_INDEX_URL="${PIP_INDEX_URL:-https://mirrors.aliyun.com/pypi/simple/}" -if command -v python3 >/dev/null 2>&1; then PY=python3; else PY=python; fi -if "$PY" -m pip --version >/dev/null 2>&1; then - "$PY" -m pip install -i "$PIP_INDEX_URL" -r python/requirements.txt -elif command -v pip >/dev/null 2>&1; then - pip install -i "$PIP_INDEX_URL" -r python/requirements.txt -else - echo "⚠ 未找到 pip:本机仅用于查看/自测,NPU 推理请在 AX 板端执行(板端自带 pip)。" -fi - -echo "C++ SDK: 请先安装 AX650 BSP SDK,然后:" -# export AX_RUNTIME_ROOT=/path/to/axruntime -# mkdir -p cpp/build && cd cpp/build -# cmake .. -DCMAKE_TOOLCHAIN_FILE=${AX_RUNTIME_ROOT}/toolchain.cmake -# make -j$(nproc) - -echo "✅ 环境准备完成"