| 230 | if (const auto repaint_mode = json_optional_string(value, "repaint_mode")) { |
| 231 | set_option(request.options, "repaint_mode", *repaint_mode); |
| 232 | } |
| 233 | if (const auto repaint_strength = json_optional_float(value, "repaint_strength")) { |
| 234 | set_option(request.options, "repaint_strength", std::to_string(*repaint_strength)); |
| 235 | } |
| 236 | set_option_from_json_field(request.options, value, "seed", "seed"); |
| 237 | set_option_from_json_field(request.options, value, "max_tokens", "max_tokens"); |
| 238 | set_option_from_json_field(request.options, value, "max_steps", "max_steps"); |
| 239 | set_option_from_json_field(request.options, value, "temperature", "temperature"); |
| 240 | set_option_from_json_field(request.options, value, "top_k", "top_k"); |
| 241 | set_option_from_json_field(request.options, value, "top_p", "top_p"); |
| 242 | set_option_from_json_field(request.options, value, "repetition_penalty", "repetition_penalty"); |
| 243 | set_option_from_json_field(request.options, value, "do_sample", "do_sample"); |
| 244 | set_option_from_json_field(request.options, value, "num_beams", "num_beams"); |
| 245 | set_option_from_json_field(request.options, value, "guidance_scale", "guidance_scale"); |
| 246 | set_option_from_json_field(request.options, value, "num_inference_steps", "num_inference_steps"); |
| 247 | set_option_from_json_field(request.options, value, "text_chunk_size", "text_chunk_size"); |
| 248 | set_option_from_json_field(request.options, value, "text_chunk_mode", "text_chunk_mode"); |
| 249 | set_option_from_json_field(request.options, value, "audio_chunk_seconds", "audio_chunk_seconds"); |
| 250 | set_option_from_json_field(request.options, value, "audio_chunk_mode", "audio_chunk_mode"); |
| 251 | set_option_from_json_field(request.options, value, "return_timestamps", "return_timestamps"); |
| 252 | set_option_from_json_field(request.options, value, "use_prosody_code", "use_prosody_code"); |
| 253 | set_option_from_json_field(request.options, value, "predict_target_prosody", "predict_target_prosody"); |
| 254 | set_option_from_json_field(request.options, value, "use_pitch_shift", "use_pitch_shift"); |
| 255 | set_option_from_json_field(request.options, value, "source_shift_steps", "source_shift_steps"); |
| 256 | set_option_from_json_field(request.options, value, "prosody_shift_steps", "prosody_shift_steps"); |
| 257 | set_option_from_json_field(request.options, value, "style_shift_steps", "style_shift_steps"); |
| 258 | set_option_from_json_field(request.options, value, "target_duration_seconds", "target_duration_seconds"); |
| 259 | set_option_from_json_field(request.options, value, "reference_duration_seconds", "reference_duration_seconds"); |
| 260 | if (const auto reference_text = json_optional_string(value, "reference_text")) { |
| 261 | set_option(request.options, "reference_text", *reference_text); |
| 262 | } |
| 263 | if (const auto instruct = json_optional_string(value, "instruct")) { |
| 264 | set_option(request.options, "instruct", *instruct); |
| 265 | } |
| 266 | return request; |
| 267 | } |
| 268 | |
| 269 | engine::runtime::TaskRequest build_request_from_cli(int argc, char ** argv) { |
| 270 | engine::runtime::TaskRequest request; |
| 271 | const auto language = find_arg(argc, argv, "--language").value_or(""); |
| 272 | if (const auto text = find_arg(argc, argv, "--text")) { |
| 273 | request.text_input = engine::runtime::Transcript{*text, language}; |
| 274 | } |
| 275 | // Read outside the branch below: only a live stdin source uses them, but an option the CLI |
| 276 | // never looks up cannot be told apart from a misspelling. |
| 277 | const int input_rate = parse_int_arg(argc, argv, "--input-rate", 16000); |
| 278 | const int input_channels = parse_int_arg(argc, argv, "--input-channels", 1); |
| 279 | if (const auto audio_path = find_arg(argc, argv, "--audio")) { |
| 280 | if (is_stdin_audio_source(*audio_path)) { |
| 281 | // Live PCM arrives chunk by chunk, so only the format contract is known up front. |
| 282 | // The samples stay empty; the streaming driver pulls them from stdin instead. |
| 283 | request.audio_input = engine::runtime::AudioBuffer{input_rate, input_channels, {}}; |
| 284 | } else { |
| 285 | request.audio_input = read_audio_buffer(std::filesystem::path(*audio_path)); |
| 286 | } |
| 287 | } |
| 288 | engine::runtime::VoiceCondition voice; |
| 289 | bool has_voice = false; |
no test coverage detected