Execute the complete long video cotqa pipeline. Args: storage: DataFlow storage object input_video_key: Input video path field name (default: 'video') input_conversation_key: Input conversation field name (default: 'conversation')
(
self,
storage: DataFlowStorage,
input_video_key: str = "video",
input_conversation_key: str = "conversation",
output_key: str = "caption",
)
| 335 | return "long video cotqa pipeline using API models." |
| 336 | |
| 337 | def run( |
| 338 | self, |
| 339 | storage: DataFlowStorage, |
| 340 | input_video_key: str = "video", |
| 341 | input_conversation_key: str = "conversation", |
| 342 | output_key: str = "caption", |
| 343 | ): |
| 344 | """ |
| 345 | Execute the complete long video cotqa pipeline. |
| 346 | |
| 347 | Args: |
| 348 | storage: DataFlow storage object |
| 349 | input_video_key: Input video path field name (default: 'video') |
| 350 | input_conversation_key: Input conversation field name (default: 'conversation') |
| 351 | output_key: Output caption field name (default: 'caption') |
| 352 | |
| 353 | Returns: |
| 354 | str: Output key name |
| 355 | """ |
| 356 | self.logger.info("="*80) |
| 357 | self.logger.info("Running Complete Long Video CoTQA Pipeline (API Version)...") |
| 358 | self.logger.info("="*80) |
| 359 | |
| 360 | # ============================================================ |
| 361 | # STAGE 1: Video Caption Generation |
| 362 | # ============================================================ |
| 363 | self.logger.info("\n" + "="*80) |
| 364 | self.logger.info("STAGE 1: VIDEO CAPTION GENERATION") |
| 365 | self.logger.info("="*80) |
| 366 | |
| 367 | # Step 1: Extract video info |
| 368 | self.logger.info("\n[Step 1/6] Extracting video info...") |
| 369 | self.video_info_filter.run( |
| 370 | storage=storage.step(), |
| 371 | input_video_key=input_video_key, |
| 372 | output_key="video_info", |
| 373 | ) |
| 374 | self.logger.info("✓ Video info extracted") |
| 375 | |
| 376 | # Step 2: Detect video scenes |
| 377 | self.logger.info("\n[Step 2/6] Detecting video scenes...") |
| 378 | self.video_scene_filter.run( |
| 379 | storage=storage.step(), |
| 380 | input_video_key=input_video_key, |
| 381 | video_info_key="video_info", |
| 382 | output_key="video_scene", |
| 383 | ) |
| 384 | self.logger.info("✓ Scene detection complete") |
| 385 | |
| 386 | # Step 3: Generate clip metadata |
| 387 | self.logger.info("\n[Step 3/6] Generating clip metadata...") |
| 388 | self.video_clip_filter.run( |
| 389 | storage=storage.step(), |
| 390 | input_video_key=input_video_key, |
| 391 | video_info_key="video_info", |
| 392 | video_scene_key="video_scene", |
| 393 | output_key="video_clip", |
| 394 | ) |
no test coverage detected