diff --git a/include/eddy/core/model_configs.hpp b/include/eddy/core/model_configs.hpp index 09eedd7..d98dd01 100644 --- a/include/eddy/core/model_configs.hpp +++ b/include/eddy/core/model_configs.hpp @@ -98,6 +98,18 @@ namespace model_configs { .repo_subdir = "en/int8" }; + // NVIDIA parakeet-unified-en-0.6b: stateless FastConformer-RNNT (English). + // Streams via a fixed [left|chunk|right] attention window (static shapes -> + // NPU-compatible), served by the eddy::nemotron backend's window-streaming + // path (no caches/prompt; auto-detected from metadata). Points at the + // streaming IR; an offline (dynamic, CPU/GPU) variant also exists at fp16/. + inline const ModelConfig PARAKEET_UNIFIED = { + .repo_id = "FluidInference/parakeet-unified-en-0.6b-ov", + .required_files = NEMOTRON_FILES, + .cache_subdir = "parakeet-unified", + .repo_subdir = "streaming/fp16" + }; + // Model name lookup map inline const std::map MODEL_MAP = { {"parakeet-v2", PARAKEET_V2}, @@ -105,7 +117,8 @@ namespace model_configs { {"nemotron-streaming", NEMOTRON_STREAMING}, {"nemotron-streaming-int8", NEMOTRON_STREAMING_INT8}, {"nemotron-speech-streaming", NEMOTRON_SPEECH}, - {"nemotron-speech-streaming-int8", NEMOTRON_SPEECH_INT8} + {"nemotron-speech-streaming-int8", NEMOTRON_SPEECH_INT8}, + {"parakeet-unified", PARAKEET_UNIFIED} }; // Default model diff --git a/include/eddy/models/nemotron/nemotron_featurizer.hpp b/include/eddy/models/nemotron/nemotron_featurizer.hpp index 0a4333e..0c49e2e 100644 --- a/include/eddy/models/nemotron/nemotron_featurizer.hpp +++ b/include/eddy/models/nemotron/nemotron_featurizer.hpp @@ -23,7 +23,13 @@ class MelFeaturizer { // Nemotron defaults: 16 kHz, 128 mels, 25 ms Hann window (400 samples), // 10 ms hop (160), 512-pt FFT, preemphasis 0.97, log guard 6e-8 (the value // stored in the exported IR; NeMo's 2^-24 rounds to this at fp16). - explicit MelFeaturizer(int sample_rate = 16000, int n_mels = 128); + // + // normalize_per_feature: apply NeMo's "per_feature" normalization — subtract + // each mel bin's mean and divide by its std (ddof=1, +1e-5) over the valid + // frames. Nemotron models export raw log-mel (false); parakeet-unified + // exports normalized features (true). + explicit MelFeaturizer(int sample_rate = 16000, int n_mels = 128, + bool normalize_per_feature = false); // Compute log-mel for `n` samples of 16 kHz mono float PCM. `valid_samples` // is the number of non-padding samples (frames whose index >= valid_samples/ @@ -45,6 +51,7 @@ class MelFeaturizer { int n_freq_; // n_fft_/2 + 1 float preemph_; float log_guard_; + bool normalize_per_feature_; std::vector window_; // [n_fft_]: Hann(win_length_) centred, 0 elsewhere std::vector mel_fb_; // [n_mels_ * n_freq_], row-major (slaney) diff --git a/src/models/nemotron/nemotron_featurizer.cpp b/src/models/nemotron/nemotron_featurizer.cpp index d894968..fe66c39 100644 --- a/src/models/nemotron/nemotron_featurizer.cpp +++ b/src/models/nemotron/nemotron_featurizer.cpp @@ -39,7 +39,7 @@ inline double mel_to_hz(double mel) { } // namespace -MelFeaturizer::MelFeaturizer(int sample_rate, int n_mels) +MelFeaturizer::MelFeaturizer(int sample_rate, int n_mels, bool normalize_per_feature) : sample_rate_(sample_rate), n_mels_(n_mels), n_fft_(512), @@ -47,7 +47,8 @@ MelFeaturizer::MelFeaturizer(int sample_rate, int n_mels) win_length_(static_cast(sample_rate * 0.025 + 0.5)), // 25 ms -> 400 n_freq_(512 / 2 + 1), preemph_(0.97f), - log_guard_(6e-8f) { + log_guard_(6e-8f), + normalize_per_feature_(normalize_per_feature) { // The featurizer is calibrated for NeMo's 16 kHz Nemotron config (25 ms / // 10 ms framing -> win 400 / hop 160, 512-pt FFT). A 512 FFT only fits the // window if win_length <= n_fft; guard so an unexpected sample rate fails @@ -193,6 +194,28 @@ void MelFeaturizer::compute(const float* audio, std::size_t n, int valid_samples out_mel[static_cast(mbin) * frames + f] = std::log(acc + log_guard_); } } + + // NeMo "per_feature" normalization: per mel bin, subtract the mean and divide + // by the std (sample std, ddof=1, +1e-5) over the valid frames; invalid + // (length-masked) frames stay zero. Matches the exported preprocessor. + if (normalize_per_feature_ && valid_frames >= 2) { + const double inv_n = 1.0 / static_cast(valid_frames); + const double inv_nm1 = 1.0 / static_cast(valid_frames - 1); + for (int mbin = 0; mbin < n_mels_; ++mbin) { + float* row = &out_mel[static_cast(mbin) * frames]; + double mean = 0.0; + for (std::size_t t = 0; t < valid_frames; ++t) mean += row[t]; + mean *= inv_n; + double var = 0.0; + for (std::size_t t = 0; t < valid_frames; ++t) { + const double d = row[t] - mean; + var += d * d; + } + const double sd = std::sqrt(var * inv_nm1) + 1e-5; + for (std::size_t t = 0; t < valid_frames; ++t) + row[t] = static_cast((row[t] - mean) / sd); + } + } } } // namespace eddy::nemotron diff --git a/src/models/nemotron/nemotron_openvino.cpp b/src/models/nemotron/nemotron_openvino.cpp index 4b87c3b..43b81ea 100644 --- a/src/models/nemotron/nemotron_openvino.cpp +++ b/src/models/nemotron/nemotron_openvino.cpp @@ -145,6 +145,21 @@ struct OpenVINONemotron::Impl { // backend serves both variants. bool has_prompt = false; + // Whether the encoder carries cross-chunk caches (Nemotron cache-aware). False + // for the parakeet-unified fixed-window streaming encoder (stateless, sliding + // [left|chunk|right] window). Auto-detected from the encoder's input ports. + bool has_cache = false; + + // parakeet-unified fixed-window streaming geometry (metadata "streaming":true). + bool window_streaming = false; + int subsampling = 8; + int left_enc = 0, chunk_enc = 0, right_enc = 0; // encoder frames per window + int window_mel_frames = 0; // mel frames the encoder expects + + // NeMo "per_feature" mel normalization (parakeet-unified). Nemotron exports + // raw log-mel and leaves this false. + bool feature_normalize = false; + ov::element::Type token_et = ov::element::i32; std::once_flag compile_once; @@ -225,16 +240,39 @@ void OpenVINONemotron::ensure_compiled() const { for (const auto& d : arr) s.push_back(d.get()); return s; }; - // These two keys have no sensible default (they size the encoder caches), - // so require them explicitly with a path-aware message rather than letting - // nlohmann's bare "key not found" propagate from a truncated metadata.json. - if (!m.contains("cache_channel_shape") || !m.contains("cache_time_shape")) { - throw std::runtime_error( - "Nemotron metadata missing required cache shape keys " - "(cache_channel_shape / cache_time_shape): " + impl_->paths.metadata_json); + // Cache shapes: present for cache-aware encoders (Nemotron), absent for the + // stateless parakeet-unified streaming encoder. Parse if present; the + // requirement (when the encoder actually has cache inputs) is enforced + // after compile, once has_cache is known. + if (m.contains("cache_channel_shape")) impl_->cache_channel_shape = to_shape(m.at("cache_channel_shape")); + if (m.contains("cache_time_shape")) impl_->cache_time_shape = to_shape(m.at("cache_time_shape")); + + // parakeet-unified fixed-window streaming geometry. + impl_->window_streaming = m.value("streaming", false) && m.contains("context_encoder_frames"); + impl_->subsampling = m.value("subsampling_factor", 8); + impl_->feature_normalize = m.value("feature_normalize", false); + if (impl_->window_streaming) { + const auto& ctx = m.at("context_encoder_frames"); + impl_->left_enc = ctx.value("left", 0); + impl_->chunk_enc = ctx.value("chunk", 0); + impl_->right_enc = ctx.value("right", 0); + impl_->window_mel_frames = m.value("window_mel_frames", 0); + // chunk_enc is the loop stride in run_window_streaming(); a missing/zero + // value would never advance the sliding window (infinite loop), and a + // zero window_mel_frames would build a zero-sized mel tensor. The + // window-streaming path also skips the cache-aware geometry guard below, + // so there is no secondary net — fail loudly here. + if (impl_->chunk_enc <= 0) { + throw std::runtime_error( + "parakeet-unified metadata missing/invalid context_encoder_frames.chunk: " + + impl_->paths.metadata_json); + } + if (impl_->window_mel_frames <= 0) { + throw std::runtime_error( + "parakeet-unified metadata missing/invalid window_mel_frames: " + + impl_->paths.metadata_json); + } } - impl_->cache_channel_shape = to_shape(m.at("cache_channel_shape")); - impl_->cache_time_shape = to_shape(m.at("cache_time_shape")); if (m.contains("prompt_dictionary")) { for (auto& [k, v] : m["prompt_dictionary"].items()) { @@ -252,13 +290,14 @@ void OpenVINONemotron::ensure_compiled() const { // The IR preprocessor is dynamic-shaped (NPU-incompatible) and adds an // OV inference per chunk; the native featurizer reproduces it exactly // (validated to fp16-storage precision). paths.preprocessor is unused. - impl_->featurizer = std::make_unique(impl_->sample_rate, impl_->mel_features); + impl_->featurizer = std::make_unique(impl_->sample_rate, impl_->mel_features, + impl_->feature_normalize); // Guard: the featurizer's framing must match the model's expected geometry. // A full chunk (chunk_mel_frames * sample_rate/100 samples) must yield // chunk_mel_frames + 1 mel frames; otherwise metadata (sample_rate / // chunk_mel_frames) disagrees with the hardcoded 10 ms hop and the mel would // silently misalign with the encoder. - { + if (!impl_->window_streaming) { const size_t cs = impl_->chunk_samples(); std::vector probe(cs, 0.0f); std::vector mel_probe; @@ -274,6 +313,24 @@ void OpenVINONemotron::ensure_compiled() const { std::to_string(impl_->sample_rate) + "). The model's featurizer " "config differs from the hardcoded 25 ms/10 ms framing."); } + } else { + // Window-streaming counterpart: a full [left|chunk|right] window must yield + // exactly window_mel_frames mel frames, or the static-shape NPU encoder + // would silently receive a truncated / over-copied mel (run_window_streaming + // clamps to win_mel without diagnosing the mismatch). + const size_t win_s = static_cast(impl_->left_enc + impl_->chunk_enc + impl_->right_enc) * + static_cast(impl_->subsampling) * + (static_cast(impl_->sample_rate) / 100); + std::vector probe(win_s, 0.0f); + std::vector mel_probe; + size_t probe_frames = 0; + impl_->featurizer->compute(probe.data(), win_s, static_cast(win_s), mel_probe, probe_frames); + if (probe_frames != static_cast(impl_->window_mel_frames)) { + throw std::runtime_error( + "parakeet-unified featurizer geometry mismatch: produced " + + std::to_string(probe_frames) + " mel frames for a window, expected " + + "window_mel_frames=" + std::to_string(impl_->window_mel_frames) + "."); + } } // --- Compile models on the chosen device --- @@ -291,8 +348,17 @@ void OpenVINONemotron::ensure_compiled() const { // multilingual encoder has a "prompt_id" input; the English speech-streaming // encoder does not. Drives whether transcribe() feeds a prompt_id tensor. impl_->has_prompt = false; + impl_->has_cache = false; for (const auto& in : impl_->encoder.inputs()) { - if (in.get_names().count("prompt_id")) { impl_->has_prompt = true; break; } + if (in.get_names().count("cache_channel")) impl_->has_cache = true; + if (in.get_names().count("prompt_id")) { impl_->has_prompt = true; } + } + // A cache-aware encoder must have its cache shapes in metadata. + if (impl_->has_cache && + (impl_->cache_channel_shape.empty() || impl_->cache_time_shape.empty())) { + throw std::runtime_error( + "Nemotron encoder has cache inputs but metadata is missing " + "cache_channel_shape / cache_time_shape: " + impl_->paths.metadata_json); } impl_->token_et = impl_->decoder.input("token").get_element_type(); @@ -317,6 +383,124 @@ void OpenVINONemotron::ensure_compiled() const { }); } +// parakeet-unified fixed-window streaming decode: the encoder is stateless and +// runs over a sliding [left|chunk|right] window of audio; only the chunk's +// encoder frames are decoded, with LSTM state carried across windows. Plain +// greedy RNNT (no caches, no prompt). Reuses the C++ featurizer and the same +// decoder/joint as the cache-aware path. +static TranscriptionResult run_window_streaming( + OpenVINONemotron::Impl& I, const std::vector& pcm, + std::chrono::steady_clock::time_point t_start) { + const size_t bins = static_cast(I.mel_features); + const long efs = static_cast(I.subsampling) * (I.sample_rate / 100); // encoder-frame samples + const long chunk_s = static_cast(I.chunk_enc) * efs; + const long left_s = static_cast(I.left_enc) * efs; + const long win_s = static_cast(I.left_enc + I.chunk_enc + I.right_enc) * efs; + const size_t win_mel = static_cast(I.window_mel_frames); + + const ov::Shape lstm_shape{static_cast(I.decoder_layers), 1, + static_cast(I.decoder_hidden)}; + ov::Tensor h(ov::element::f32, lstm_shape), c(ov::element::f32, lstm_shape); + std::memset(h.data(), 0, h.get_byte_size()); + std::memset(c.data(), 0, c.get_byte_size()); + int last_token = I.blank_idx; + std::vector all_tokens; + + ov::Tensor token(I.token_et, ov::Shape{1, 1}); + const ov::Tensor token_length = make_i32(1); + const ov::Tensor mel_length = make_i32(static_cast(win_mel)); + ov::Tensor mel_in(ov::element::f32, ov::Shape{1, bins, win_mel}); + std::vector window(static_cast(win_s)); + std::vector mel_scratch; + + const long N = static_cast(pcm.size()); + for (long pos = 0; pos < N; pos += chunk_s) { + // Audio window [pos-left_s, pos+chunk_s+right_s), zero-padded at the edges. + const long start = pos - left_s; + for (long i = 0; i < win_s; ++i) { + const long src = start + i; + window[static_cast(i)] = (src >= 0 && src < N) ? pcm[static_cast(src)] : 0.0f; + } + // Real (non-padded) sample count within the window: the boundary windows are + // zero-padded at the left (first chunk) or right (last chunk) edge, and those + // pad samples must be excluded from per-feature normalization so the stats + // match NeMo's preprocessor (which normalizes over valid frames only). + const long real_lo = std::max(0L, -start); + const long real_hi = std::min(win_s, N - start); + const int valid_n = static_cast(std::max(0L, real_hi - real_lo)); + size_t t_mel = 0; + I.featurizer->compute(window.data(), window.size(), valid_n, mel_scratch, t_mel); + float* md = mel_in.data(); + std::memset(md, 0, mel_in.get_byte_size()); + const size_t copyT = std::min(t_mel, win_mel); + for (size_t b = 0; b < bins; ++b) + for (size_t t = 0; t < copyT; ++t) md[b * win_mel + t] = mel_scratch[b * t_mel + t]; + + I.encoder_req.set_tensor("mel", mel_in); + I.encoder_req.set_tensor("mel_length", mel_length); + I.encoder_req.infer(); + const ov::Tensor encoded = I.encoder_req.get_tensor("encoded"); // [1, D, T_enc] + const ov::Shape es = encoded.get_shape(); + const size_t enc_d = es[1], t_enc = es[2]; + const float* ed = encoded.data(); + + // Decode only the chunk's encoder frames [left_enc : left_enc+chunk_enc]. + const size_t keep_lo = std::min(static_cast(I.left_enc), t_enc); + const size_t keep_hi = std::min(static_cast(I.left_enc + I.chunk_enc), t_enc); + ov::Tensor enc_step(ov::element::f32, ov::Shape{1, enc_d, 1}); + for (size_t t = keep_lo; t < keep_hi; ++t) { + float* sp = enc_step.data(); + for (size_t ch = 0; ch < enc_d; ++ch) sp[ch] = ed[ch * t_enc + t]; + for (size_t sym = 0; sym < I.config.max_symbols_per_frame; ++sym) { + if (I.token_et == ov::element::i64) token.data()[0] = last_token; + else token.data()[0] = last_token; + I.decoder_req.set_tensor("token", token); + I.decoder_req.set_tensor("token_length", token_length); + I.decoder_req.set_tensor("h_in", h); + I.decoder_req.set_tensor("c_in", c); + I.decoder_req.infer(); + const ov::Tensor dec_out = I.decoder_req.get_tensor("decoder_out"); + I.joint_req.set_tensor("encoder", enc_step); + I.joint_req.set_tensor("decoder", dec_out); + I.joint_req.infer(); + const ov::Tensor logits = I.joint_req.get_tensor("logits"); + const float* lg = logits.data(); + const size_t vsz = logits.get_size(); + if (vsz == 0) break; // empty joint output (IR shape mismatch) — don't read lg[0] + int best = 0; + float best_score = lg[0]; + for (size_t i = 1; i < vsz; ++i) + if (lg[i] > best_score) { best_score = lg[i]; best = static_cast(i); } + if (best == I.blank_idx) break; + all_tokens.push_back(best); + last_token = best; + const ov::Tensor h_out = I.decoder_req.get_tensor("h_out"); + const ov::Tensor c_out = I.decoder_req.get_tensor("c_out"); + std::memcpy(h.data(), h_out.data(), h.get_byte_size()); + std::memcpy(c.data(), c_out.data(), c.get_byte_size()); + } + } + } + + auto piece = [&](int tok) -> const std::string& { + static const std::string empty; + return (tok >= 0 && tok < static_cast(I.vocab.size())) ? I.vocab[tok] : empty; + }; + TranscriptionResult result; + result.prompt_id_used = 0; + result.token_ids = all_tokens; + std::string body; + for (int tok : all_tokens) { + if (tok == I.blank_idx || tok >= I.vocab_size) continue; + if (I.lang_tag_token_ids.count(tok)) continue; // monolingual: no lang tags + body += piece(tok); + } + result.text = finalize_text(body); + result.latency_ms = + std::chrono::duration(std::chrono::steady_clock::now() - t_start).count(); + return result; +} + TranscriptionResult OpenVINONemotron::transcribe(const std::vector& pcm) { ensure_compiled(); std::lock_guard lock(impl_->infer_guard); @@ -324,6 +508,14 @@ TranscriptionResult OpenVINONemotron::transcribe(const std::vector& pcm) const auto t_start = std::chrono::steady_clock::now(); auto& I = *impl_; + + // parakeet-unified fixed-window streaming: stateless encoder over a sliding + // [left|chunk|right] window (no caches/prompt). Separate path from the + // cache-aware Nemotron loop below. + if (I.window_streaming) { + return run_window_streaming(I, pcm, t_start); + } + const size_t bins = static_cast(I.mel_features); const size_t total = static_cast(I.total_mel_frames); const size_t pre_cache = static_cast(I.pre_encode_cache);