storyboard.py 5.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143
  1. # Copyright (C) 2025 AIDC-AI
  2. #
  3. # Licensed under the Apache License, Version 2.0 (the "License");
  4. # you may not use this file except in compliance with the License.
  5. # You may obtain a copy of the License at
  6. # http://www.apache.org/licenses/LICENSE-2.0
  7. # Unless required by applicable law or agreed to in writing, software
  8. # distributed under the License is distributed on an "AS IS" BASIS,
  9. # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
  10. # See the License for the specific language governing permissions and
  11. # limitations under the License.
  12. """
  13. Storyboard data models for video generation
  14. """
  15. from dataclasses import dataclass, field
  16. from datetime import datetime
  17. from typing import List, Optional, Dict, Any
  18. @dataclass
  19. class StoryboardConfig:
  20. """Storyboard configuration parameters"""
  21. # Required parameters (must come first in dataclass)
  22. media_width: int # Media width (image or video, required)
  23. media_height: int # Media height (image or video, required)
  24. # Task isolation
  25. task_id: Optional[str] = None # Task ID for file isolation (auto-generated if None)
  26. n_storyboard: int = 5 # Number of storyboard frames
  27. min_narration_words: int = 5 # Min narration word count
  28. max_narration_words: int = 20 # Max narration word count
  29. min_image_prompt_words: int = 30 # Min image prompt word count
  30. max_image_prompt_words: int = 60 # Max image prompt word count
  31. # Video parameters (fps only, size is determined by frame template)
  32. video_fps: int = 30 # Frame rate
  33. # Audio parameters
  34. tts_inference_mode: str = "local" # TTS inference mode: "local" or "comfyui"
  35. voice_id: Optional[str] = None # Voice ID (for local: Edge TTS voice ID; for comfyui: workflow-specific)
  36. tts_workflow: Optional[str] = None # TTS workflow filename (for ComfyUI mode, None = use default)
  37. tts_speed: Optional[float] = None # TTS speed multiplier (0.5-2.0, 1.0 = normal)
  38. ref_audio: Optional[str] = None # Reference audio for voice cloning (ComfyUI mode only)
  39. # Media workflow
  40. media_workflow: Optional[str] = None # Media workflow filename (image or video, None = use default)
  41. # Frame template (includes size information in path)
  42. frame_template: str = "1080x1920/default.html" # Template path with size (e.g., "1080x1920/default.html")
  43. template_params: Optional[Dict[str, Any]] = None # Custom template parameters (e.g., {"accent_color": "#ff0000"})
  44. @dataclass
  45. class StoryboardFrame:
  46. """Single storyboard frame"""
  47. index: int # Frame index (0-based)
  48. narration: str # Narration text
  49. image_prompt: str # Image generation prompt (can be None for text-only or video)
  50. # Generated resource paths
  51. audio_path: Optional[str] = None # Audio file path (narration)
  52. media_type: Optional[str] = None # Media type: "image" or "video" (None if no media)
  53. image_path: Optional[str] = None # Original image path (for image type)
  54. video_path: Optional[str] = None # Original video path (for video type, before composition)
  55. composed_image_path: Optional[str] = None # Composed image path (with subtitles, for image type)
  56. video_segment_path: Optional[str] = None # Final video segment path
  57. # Metadata
  58. duration: float = 0.0 # Frame duration (seconds, from audio or video)
  59. created_at: Optional[datetime] = None
  60. def __post_init__(self):
  61. if self.created_at is None:
  62. self.created_at = datetime.now()
  63. @dataclass
  64. class ContentMetadata:
  65. """Content metadata for visual display and narration generation"""
  66. title: str # Content title
  67. author: Optional[str] = None # Author/creator
  68. subtitle: Optional[str] = None # Subtitle
  69. genre: Optional[str] = None # Genre/category
  70. summary: Optional[str] = None # Content summary
  71. publication_year: Optional[str] = None # Publication year
  72. cover_url: Optional[str] = None # Cover/thumbnail image URL
  73. @dataclass
  74. class Storyboard:
  75. """Complete storyboard"""
  76. title: str # Video title
  77. config: StoryboardConfig # Configuration
  78. frames: List[StoryboardFrame] = field(default_factory=list)
  79. # Content metadata (optional)
  80. content_metadata: Optional[ContentMetadata] = None
  81. # Final output
  82. final_video_path: Optional[str] = None
  83. total_duration: float = 0.0
  84. # Metadata
  85. created_at: Optional[datetime] = None
  86. completed_at: Optional[datetime] = None
  87. def __post_init__(self):
  88. if self.created_at is None:
  89. self.created_at = datetime.now()
  90. @property
  91. def is_completed(self) -> bool:
  92. """Check if all frames are processed"""
  93. return all(
  94. frame.video_segment_path is not None
  95. for frame in self.frames
  96. )
  97. @property
  98. def progress(self) -> float:
  99. """Return processing progress (0.0-1.0)"""
  100. if not self.frames:
  101. return 0.0
  102. completed = sum(
  103. 1 for frame in self.frames
  104. if frame.video_segment_path is not None
  105. )
  106. return completed / len(self.frames)
  107. @dataclass
  108. class VideoGenerationResult:
  109. """Video generation result"""
  110. video_path: str # Final video path
  111. storyboard: Storyboard # Complete storyboard
  112. duration: float # Total duration
  113. file_size: int # File size (bytes)
  114. created_at: datetime = field(default_factory=datetime.now)