| 439 | std::string vad_chunks_to_json(const std::vector<engine::runtime::TimeSpan> & chunks) { |
| 440 | std::ostringstream out; |
| 441 | out << "["; |
| 442 | for (size_t i = 0; i < chunks.size(); ++i) { |
| 443 | if (i != 0) { |
| 444 | out << ","; |
| 445 | } |
| 446 | out << "{\"index\":" << i |
| 447 | << ",\"start_sample\":" << chunks[i].start_sample |
| 448 | << ",\"end_sample\":" << chunks[i].end_sample |
| 449 | << "}"; |
| 450 | } |
| 451 | out << "]"; |
| 452 | return out.str(); |
| 453 | } |
| 454 | |
| 455 | void write_vad_chunks_output( |
| 456 | const engine::runtime::TaskResult & result, |
| 457 | const engine::runtime::AudioBuffer & audio, |
| 458 | const std::filesystem::path & path, |
| 459 | const engine::audio::VadAudioChunkOptions & options) { |
| 460 | if (audio.channels <= 0) { |
| 461 | throw std::runtime_error("VAD chunk planning requires positive audio channels"); |
| 462 | } |
| 463 | if (audio.samples.size() % static_cast<size_t>(audio.channels) != 0) { |
| 464 | throw std::runtime_error("VAD chunk planning requires audio samples divisible by channel count"); |
| 465 | } |
| 466 | const int64_t audio_frames = static_cast<int64_t>(audio.samples.size() / static_cast<size_t>(audio.channels)); |
| 467 | const auto chunks = engine::audio::plan_vad_audio_chunks(result.speech_segments, audio_frames, options); |
| 468 | if (!path.parent_path().empty()) { |
| 469 | std::filesystem::create_directories(path.parent_path()); |
| 470 | } |
| 471 | std::ofstream(path) << vad_chunks_to_json(chunks); |
| 472 | std::cout << "vad_chunks_out=" << path.string() << "\n"; |
| 473 | } |
| 474 | |
| 475 | bool stream_audio_from_stdin(int argc, char ** argv) { |
| 476 | const auto audio = minitts::cli::find_arg(argc, argv, "--audio"); |
| 477 | return audio.has_value() && minitts::cli::is_stdin_audio_source(*audio); |
| 478 | } |
| 479 | |
| 480 | // Only a terminal can take partials as running text; a redirected stream has to keep the |
| 481 | // line-per-update format so pipes and logs stay parseable. |
| 482 | bool stdout_is_terminal() { |
| 483 | #ifdef _WIN32 |
| 484 | return _isatty(_fileno(stdout)) != 0; |
| 485 | #else |
| 486 | return isatty(fileno(stdout)) != 0; |
| 487 | #endif |
| 488 | } |
| 489 | |
| 490 | double duration_ms(std::chrono::steady_clock::duration duration) { |
| 491 | return std::chrono::duration<double, std::milli>(duration).count(); |
| 492 | } |
| 493 | |
| 494 | std::optional<minitts::app::AudioMetricsInfo> audio_metrics_info( |
| 495 | const engine::runtime::TaskRequest & request) { |
| 496 | if (!request.audio_input.has_value()) { |
| 497 | return std::nullopt; |
| 498 | } |
no test coverage detected