| 110 | config.ddpm_batch_mul = json::optional_i64(value, "ddpm_batch_mul", config.ddpm_batch_mul); |
| 111 | config.ddpm_beta_schedule = json::optional_string(value, "ddpm_beta_schedule", config.ddpm_beta_schedule); |
| 112 | config.ddpm_num_inference_steps = |
| 113 | json::optional_i64(value, "ddpm_num_inference_steps", config.ddpm_num_inference_steps); |
| 114 | config.ddpm_num_steps = json::optional_i64(value, "ddpm_num_steps", config.ddpm_num_steps); |
| 115 | config.diffusion_type = json::optional_string(value, "diffusion_type", config.diffusion_type); |
| 116 | config.head_ffn_ratio = json::optional_f32(value, "head_ffn_ratio", config.head_ffn_ratio); |
| 117 | config.head_layers = json::optional_i64(value, "head_layers", config.head_layers); |
| 118 | config.hidden_size = json::require_i64(value, "hidden_size"); |
| 119 | config.latent_size = json::require_i64(value, "latent_size"); |
| 120 | config.prediction_type = json::optional_string(value, "prediction_type", config.prediction_type); |
| 121 | config.rms_norm_eps = json::optional_f32(value, "rms_norm_eps", config.rms_norm_eps); |
| 122 | config.speech_vae_dim = json::optional_i64(value, "speech_vae_dim", config.latent_size); |
| 123 | engine::io::require_positive(config.ddpm_batch_mul, "diffusion ddpm_batch_mul"); |
| 124 | engine::io::require_positive(config.ddpm_num_inference_steps, "diffusion ddpm_num_inference_steps"); |
| 125 | engine::io::require_positive(config.ddpm_num_steps, "diffusion ddpm_num_steps"); |
| 126 | engine::io::require_positive(config.head_layers, "diffusion head_layers"); |
| 127 | engine::io::require_positive(config.hidden_size, "diffusion hidden_size"); |
| 128 | engine::io::require_positive(config.latent_size, "diffusion latent_size"); |
| 129 | engine::io::require_positive(config.speech_vae_dim, "diffusion speech_vae_dim"); |
| 130 | if (config.diffusion_type != "ddpm") { |
| 131 | throw std::runtime_error("VibeVoice config diffusion_type mismatch"); |
| 132 | } |
| 133 | if (config.prediction_type != "v_prediction") { |
| 134 | throw std::runtime_error("VibeVoice config diffusion prediction_type mismatch"); |
| 135 | } |
| 136 | if (config.ddpm_beta_schedule != "cosine") { |
| 137 | throw std::runtime_error("VibeVoice config diffusion beta schedule mismatch"); |
| 138 | } |
| 139 | if (config.speech_vae_dim != config.latent_size) { |
| 140 | throw std::runtime_error("VibeVoice diffusion speech_vae_dim must match latent_size"); |
| 141 | } |
| 142 | return config; |
| 143 | } |
| 144 | |
| 145 | int64_t require_acoustic_vae_dim(const engine::io::json::Value & root) { |
| 146 | // VibeVoice-7B's config.json spells this key "acostic_vae_dim". |
no test coverage detected