Complete long video cotqa processing pipeline using API models. Integrates caption generation, reasoning QA generation, and reformatting.
| 148 | |
| 149 | |
| 150 | class LongVideoPipelineAPI(OperatorABC): |
| 151 | """ |
| 152 | Complete long video cotqa processing pipeline using API models. |
| 153 | Integrates caption generation, reasoning QA generation, and reformatting. |
| 154 | """ |
| 155 | |
| 156 | def __init__( |
| 157 | self, |
| 158 | # VideoInfoFilter parameters |
| 159 | backend: str = "opencv", |
| 160 | ext: bool = False, |
| 161 | |
| 162 | # VideoSceneFilter parameters |
| 163 | frame_skip: int = 0, |
| 164 | start_remove_sec: float = 0.0, |
| 165 | end_remove_sec: float = 0.0, |
| 166 | min_seconds: float = 2.0, |
| 167 | max_seconds: float = 15.0, |
| 168 | use_adaptive_detector: bool = False, |
| 169 | overlap: bool = False, |
| 170 | use_fixed_interval: bool = False, |
| 171 | |
| 172 | # API VLM parameters (for caption generation) |
| 173 | vlm_api_url: str = "https://dashscope.aliyuncs.com/compatible-mode/v1", |
| 174 | vlm_api_key_name: str = "DF_API_KEY", |
| 175 | vlm_model_name: str = "qwen3-vl-8b-instruct", |
| 176 | vlm_max_workers: int = 10, |
| 177 | vlm_timeout: int = 1800, |
| 178 | |
| 179 | # API LLM parameters (for reasoning generation) |
| 180 | llm_api_url: str = "https://dashscope.aliyuncs.com/compatible-mode/v1", |
| 181 | llm_api_key_name: str = "DF_API_KEY", |
| 182 | llm_model_name: str = "qwen-plus", |
| 183 | llm_max_workers: int = 10, |
| 184 | llm_timeout: int = 1800, |
| 185 | |
| 186 | # API LLM parameters (for reasoning reformatting) |
| 187 | reformat_api_url: str = "https://openrouter.ai/api/v1", |
| 188 | reformat_api_key_name: str = "OPENROUTER_API_KEY", |
| 189 | reformat_model_name: str = "openai/gpt-4o", |
| 190 | reformat_max_workers: int = 10, |
| 191 | reformat_timeout: int = 1800, |
| 192 | |
| 193 | # VideoClipGenerator parameters |
| 194 | video_save_dir: str = "./cache/video_clips", |
| 195 | ): |
| 196 | """ |
| 197 | Initialize the long video cotqa pipeline with API models. |
| 198 | |
| 199 | Args: |
| 200 | backend: Video backend for info extraction (opencv, torchvision, av) |
| 201 | ext: Whether to filter non-existent files |
| 202 | frame_skip: Frame skip for scene detection |
| 203 | start_remove_sec: Seconds to remove from start of each scene |
| 204 | end_remove_sec: Seconds to remove from end of each scene |
| 205 | min_seconds: Minimum scene duration |
| 206 | max_seconds: Maximum scene duration |
| 207 | use_adaptive_detector: Whether to use AdaptiveDetector in scene detection |
no outgoing calls
no test coverage detected