| 122 | } |
| 123 | |
| 124 | void validate_tokenizer_config(const VibeVoiceTokenizerConfig & config, bool require_decoder) { |
| 125 | if (!config.causal) { |
| 126 | throw std::runtime_error("VibeVoice tokenizer loader expects causal tokenizers"); |
| 127 | } |
| 128 | if (config.channels <= 0 || config.vae_dim <= 0 || config.encoder_n_filters <= 0) { |
| 129 | throw std::runtime_error("VibeVoice tokenizer config dimensions must be positive"); |
| 130 | } |
| 131 | if (require_decoder && config.decoder_n_filters <= 0) { |
| 132 | throw std::runtime_error("VibeVoice acoustic tokenizer decoder_n_filters must be positive"); |
| 133 | } |
| 134 | if (config.mixer_layer != "depthwise_conv") { |
| 135 | throw std::runtime_error("VibeVoice tokenizer loader expects depthwise_conv mixer"); |
| 136 | } |
| 137 | if (config.pad_mode != "constant") { |
| 138 | throw std::runtime_error("VibeVoice tokenizer loader expects constant tokenizer padding"); |
| 139 | } |
| 140 | if (config.conv_norm != "none") { |
| 141 | throw std::runtime_error("VibeVoice tokenizer loader expects conv_norm none"); |
| 142 | } |
| 143 | if (config.layernorm != "RMSNorm") { |
| 144 | throw std::runtime_error("VibeVoice tokenizer loader expects RMSNorm blocks"); |
| 145 | } |
| 146 | if (!config.layernorm_elementwise_affine) { |
| 147 | throw std::runtime_error("VibeVoice tokenizer loader expects affine RMSNorm"); |
| 148 | } |
| 149 | if (!config.conv_bias) { |
| 150 | throw std::runtime_error("VibeVoice tokenizer loader expects convolution bias"); |
| 151 | } |
| 152 | if (!(config.layer_scale_init_value > 0.0F)) { |
| 153 | throw std::runtime_error("VibeVoice tokenizer loader expects layer scale tensors"); |
| 154 | } |
| 155 | if (config.encoder_ratios.empty()) { |
| 156 | throw std::runtime_error("VibeVoice tokenizer encoder ratios must not be empty"); |
| 157 | } |
| 158 | } |
| 159 | |
| 160 | modules::Conv1dWeights load_conv1d( |
| 161 | core::BackendWeightStore & store, |
no test coverage detected