AswinMathew commited on
Commit
7190fd0
·
verified ·
1 Parent(s): 63e6122

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +4 -0
  2. .gitignore +31 -0
  3. Dockerfile +51 -0
  4. RESEARCH.md +953 -0
  5. TESTED_APIS.md +128 -0
  6. app/__init__.py +0 -0
  7. app/api/__init__.py +0 -0
  8. app/api/auth.py +104 -0
  9. app/api/chapters.py +82 -0
  10. app/api/characters.py +74 -0
  11. app/api/episodes.py +68 -0
  12. app/api/generation.py +241 -0
  13. app/api/novels.py +80 -0
  14. app/api/projects.py +96 -0
  15. app/config.py +38 -0
  16. app/database.py +51 -0
  17. app/main.py +54 -0
  18. app/models/__init__.py +13 -0
  19. app/models/api_usage.py +13 -0
  20. app/models/chapter.py +19 -0
  21. app/models/character.py +27 -0
  22. app/models/critic_result.py +18 -0
  23. app/models/episode.py +25 -0
  24. app/models/generation_job.py +25 -0
  25. app/models/project.py +24 -0
  26. app/models/user.py +18 -0
  27. app/pipeline/__init__.py +0 -0
  28. app/pipeline/orchestrator.py +435 -0
  29. app/pipeline/recap_generator.py +191 -0
  30. app/pipeline/stage_1_ingest.py +39 -0
  31. app/pipeline/stage_2_characters.py +172 -0
  32. app/pipeline/stage_2b_portraits.py +118 -0
  33. app/pipeline/stage_2c_voice_enroll.py +177 -0
  34. app/pipeline/stage_3_scene_parse.py +346 -0
  35. app/pipeline/stage_4_image_gen.py +137 -0
  36. app/pipeline/stage_5_tts.py +242 -0
  37. app/pipeline/stage_6_animation.py +121 -0
  38. app/pipeline/stage_6_video_gen.py +304 -0
  39. app/pipeline/stage_7_assembly.py +278 -0
  40. app/pipeline/stage_8_critic.py +189 -0
  41. app/schemas/__init__.py +19 -0
  42. app/schemas/auth.py +32 -0
  43. app/schemas/chapter.py +20 -0
  44. app/schemas/character.py +38 -0
  45. app/schemas/episode.py +22 -0
  46. app/schemas/generation.py +23 -0
  47. app/schemas/project.py +39 -0
  48. app/services/__init__.py +0 -0
  49. app/services/auth.py +66 -0
  50. app/services/fandom.py +107 -0
.gitattributes CHANGED
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ data/bgm/calm.mp3 filter=lfs diff=lfs merge=lfs -text
37
+ data/bgm/epic.mp3 filter=lfs diff=lfs merge=lfs -text
38
+ data/bgm/sad.mp3 filter=lfs diff=lfs merge=lfs -text
39
+ data/bgm/tense.mp3 filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ dist/
6
+ .venv/
7
+ venv/
8
+
9
+ # Environment
10
+ .env
11
+
12
+ # Workdir (runtime output)
13
+ workdir/
14
+
15
+ # Test outputs (generated images, videos, audio — large binary files)
16
+ test_*_output/
17
+ test_tts_calibration/
18
+ test_*.png
19
+
20
+ # IDE / Claude
21
+ .vscode/
22
+ .idea/
23
+ .claude/
24
+
25
+ # Node
26
+ node_modules/
27
+ .next/
28
+
29
+ # OS
30
+ .DS_Store
31
+ Thumbs.db
Dockerfile ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.12-slim
2
+
3
+ # Install FFmpeg + git
4
+ RUN apt-get update && \
5
+ apt-get install -y --no-install-recommends ffmpeg git && \
6
+ rm -rf /var/lib/apt/lists/*
7
+
8
+ # Create non-root user (HF Spaces requirement)
9
+ RUN useradd -m -u 1000 user
10
+ WORKDIR /app
11
+
12
+ # Install Python dependencies (cache layer)
13
+ RUN pip install --no-cache-dir \
14
+ "fastapi[standard]>=0.115.0" \
15
+ "uvicorn[standard]>=0.32.0" \
16
+ "sqlalchemy[asyncio]>=2.0.0" \
17
+ "aiosqlite>=0.20.0" \
18
+ "pydantic>=2.0.0" \
19
+ "pydantic-settings>=2.0.0" \
20
+ "python-jose[cryptography]>=3.3.0" \
21
+ "passlib[bcrypt]>=1.7.4" \
22
+ "python-multipart>=0.0.9" \
23
+ "httpx>=0.27.0" \
24
+ "edge-tts>=6.1.0" \
25
+ "Pillow>=10.0.0" \
26
+ "pymupdf>=1.24.0" \
27
+ "opencv-python-headless>=4.9.0" \
28
+ "numpy>=1.26.0" \
29
+ "pocket-tts>=0.1.0" \
30
+ "huggingface-hub>=0.20.0" \
31
+ "scipy>=1.12.0" \
32
+ "safetensors>=0.4.0" \
33
+ "pyarrow>=15.0.0" \
34
+ "soundfile>=0.12.0"
35
+
36
+ # Copy application code
37
+ COPY app/ app/
38
+ COPY data/*.json data/
39
+ COPY scripts/ scripts/
40
+ COPY deploy/start.sh .
41
+ RUN chmod +x start.sh
42
+
43
+ # Create directories with correct permissions
44
+ RUN mkdir -p /data/workdir/projects /data/genshin_voices /data/animevox_voices && \
45
+ chown -R user:user /app /data
46
+
47
+ USER user
48
+
49
+ EXPOSE 7860
50
+
51
+ CMD ["./start.sh"]
RESEARCH.md ADDED
@@ -0,0 +1,953 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Web Novel to Anime -- Complete Research Report
2
+
3
+ > Generated: 2026-02-27 | Status: Comprehensive research across 5 domains
4
+
5
+ ---
6
+
7
+ ## TABLE OF CONTENTS
8
+
9
+ 1. [Executive Summary & Verdict](#1-executive-summary--verdict)
10
+ 2. [PLAN A: Free Video Generation APIs](#2-plan-a-free-video-generation-apis)
11
+ 3. [PLAN B: The Hybrid Pipeline](#3-plan-b-the-hybrid-pipeline)
12
+ - 3.1 [Image Generation (Free APIs)](#31-image-generation-free-apis)
13
+ - 3.2 [Character Consistency Solutions](#32-character-consistency-solutions)
14
+ - 3.3 [TTS / Voice Acting](#33-tts--voice-acting)
15
+ - 3.4 [Interpolation & Animation](#34-interpolation--animation)
16
+ - 3.5 [Video Assembly](#35-video-assembly)
17
+ 4. [NLP & Intelligence Layer](#4-nlp--intelligence-layer)
18
+ 5. [System Architecture](#5-system-architecture)
19
+ 6. [Your Hardware Advantage: Intel Arc + OpenVINO](#6-your-hardware-advantage)
20
+ 7. [Implementation Roadmap](#7-implementation-roadmap)
21
+ 8. [Monetization Strategy](#8-monetization-strategy)
22
+ 9. [Final Recommendation](#9-final-recommendation)
23
+
24
+ ---
25
+
26
+ ## 1. EXECUTIVE SUMMARY & VERDICT
27
+
28
+ ### The Honest Truth About Plan A (Pure Video Generation)
29
+
30
+ **Plan A is NOT viable today for free, automated, anime-quality video generation.** Here's why:
31
+
32
+ | Service | Free API? | Anime Quality? | Programmatic? | Verdict |
33
+ |---------|-----------|----------------|---------------|---------|
34
+ | Wan 2.1 (Alibaba) | HuggingFace free inference (slow, queued) | Good anime style | Yes via HF API | **Best free option but too slow/unreliable for production** |
35
+ | CogVideoX (Tsinghua) | HuggingFace free inference | Decent | Yes via HF API | Queue times make it impractical |
36
+ | Kling AI | Free web tier, NO free API | Good | No API on free tier | Unusable for automation |
37
+ | Runway Gen-3 | No free API | Good | Paid API only | Too expensive |
38
+ | Pika Labs | No free API | Decent | No API access | Unusable |
39
+ | Minimax (Hailuo) | Limited free API credits | Good | Yes | Credits run out fast |
40
+ | Luma Dream Machine | No free API | Decent | Paid only | Unusable |
41
+ | Stable Video Diffusion | Open source | Basic (image-to-4sec) | Yes locally | Too short, needs GPU |
42
+ | AnimateDiff | Open source | Good anime | Yes locally | **Needs 8GB+ VRAM** |
43
+ | Google Veo | No public API | Unknown | No | Not available |
44
+
45
+ **The core problem**: Video generation models need massive GPU power (12-24GB VRAM minimum). Free API tiers either:
46
+ - Don't exist (Runway, Pika, Luma)
47
+ - Have no programmatic access (Kling free tier)
48
+ - Are too slow/queued on HuggingFace (Wan, CogVideoX -- minutes of queue time per 4-second clip)
49
+ - Produce only 3-6 second clips (not episodes)
50
+
51
+ **However**, there ARE some usable free options for short video clips via HuggingFace that could supplement Plan B (see section 2 below).
52
+
53
+ ### Why Plan B Is Actually BETTER
54
+
55
+ Plan B (Image + TTS + Interpolation) is not just a fallback -- it's the **superior approach** because:
56
+
57
+ 1. **Character consistency** -- Image gen with IP-Adapter keeps characters looking the same. Video gen cannot do this yet.
58
+ 2. **Controllability** -- You control every frame, every camera angle, every expression. Video gen is a black box.
59
+ 3. **This IS how anime works** -- Real anime uses 8-12 unique drawings per second with camera movements. That's literally Plan B.
60
+ 4. **Free and unlimited** -- Multiple free image gen APIs exist with generous limits.
61
+ 5. **Your Intel Arc GPU helps** -- OpenVINO acceleration makes interpolation and animation feasible on your hardware.
62
+
63
+ ---
64
+
65
+ ## 2. PLAN A: FREE VIDEO GENERATION APIs
66
+
67
+ Despite video gen not being viable as the primary approach, here's every free option found:
68
+
69
+ ### 2.1 Wan 2.1 (Best Free Video Gen Available)
70
+
71
+ - **Source**: Alibaba's open-source video generation model
72
+ - **Access**: HuggingFace Inference API (free, queued) + HuggingFace Spaces
73
+ - **Quality**: Excellent anime-style output, one of the best for anime
74
+ - **Clip Length**: 4-8 seconds
75
+ - **Resolution**: Up to 720p
76
+ - **Limitations**: Queue wait times of 2-15 minutes per clip. Rate limited. Not reliable for batch processing.
77
+ - **Programmatic access**:
78
+ ```python
79
+ from huggingface_hub import InferenceClient
80
+ client = InferenceClient(token="YOUR_HF_TOKEN")
81
+
82
+ # Text-to-video
83
+ video = client.text_to_video(
84
+ "anime girl with silver hair walking through a fantasy forest, detailed anime style",
85
+ model="Wan-AI/Wan2.1-T2V-14B"
86
+ )
87
+ with open("output.mp4", "wb") as f:
88
+ f.write(video)
89
+ ```
90
+ - **Or via Gradio Spaces**:
91
+ ```python
92
+ from gradio_client import Client
93
+ client = Client("Wan-AI/Wan2.1-T2V-14B")
94
+ result = client.predict(prompt="...", api_name="/generate")
95
+ ```
96
+ - **Verdict**: Usable for generating occasional hero clips (opening shots, climactic moments). NOT for generating full episodes.
97
+
98
+ ### 2.2 CogVideoX (Tsinghua)
99
+
100
+ - **Access**: HuggingFace Inference API + Spaces
101
+ - **Quality**: Good, less anime-specific than Wan
102
+ - **Clip Length**: 6 seconds
103
+ - **Programmatic access**: Same pattern as Wan via HF
104
+ - **Verdict**: Backup to Wan 2.1
105
+
106
+ ### 2.3 Stable Video Diffusion (Stability AI)
107
+
108
+ - **Access**: Open source, can run locally
109
+ - **What it does**: Image-to-video (takes a still image, adds subtle motion)
110
+ - **Clip Length**: 3-4 seconds
111
+ - **Hardware**: Needs ~6GB VRAM minimum. **Might work on Intel Arc via OpenVINO** (untested but theoretically possible)
112
+ - **Verdict**: Could complement Plan B by adding subtle motion to generated keyframes. Worth testing on Intel Arc.
113
+
114
+ ### 2.4 Free Cloud GPU Options for Running Video Models
115
+
116
+ | Platform | Free GPU | GPU Type | Time Limit | Video Gen Feasible? |
117
+ |----------|----------|----------|------------|-------------------|
118
+ | Google Colab | Yes | T4 (15GB VRAM) | ~4hrs/day | **Yes** -- can run AnimateDiff, Wan 2.1 small |
119
+ | Kaggle | Yes | T4x2 or P100 | 30hrs/week | **Yes** -- best free option |
120
+ | Lightning.ai | Yes | T4 | 22hrs/month | Marginal |
121
+ | Paperspace | Free tier | M4000 (8GB) | Limited | Tight but possible |
122
+
123
+ **Google Colab + Kaggle Strategy**: Use Kaggle's 30 hours/week to batch-generate video clips. AnimateDiff on a T4 can produce ~4 seconds of anime video per 30-60 seconds of compute. That's about 120-240 4-second clips per week = ~8-16 minutes of video content.
124
+
125
+ ### 2.5 Recommended Use of Video Gen (Hybrid with Plan B)
126
+
127
+ ```
128
+ Plan B generates 95% of the episode (images + camera + TTS + interpolation)
129
+ |
130
+ v
131
+ For KEY MOMENTS (5% of episode):
132
+ - Opening/ending sequence → Wan 2.1 via HuggingFace
133
+ - Dramatic reveals → AnimateDiff via Kaggle/Colab
134
+ - Action climaxes → CogVideoX via HuggingFace
135
+ |
136
+ v
137
+ These premium clips are spliced into the Plan B video
138
+ ```
139
+
140
+ ---
141
+
142
+ ## 3. PLAN B: THE HYBRID PIPELINE
143
+
144
+ ### 3.1 Image Generation (Free APIs)
145
+
146
+ #### TOP TIER: Truly Free with Good Anime Quality
147
+
148
+ **1. Together AI -- FLUX.1 (Best Overall)**
149
+ - **Free**: $5 credits on signup (~500 images). Periodically refreshed for active users.
150
+ - **Quality**: FLUX produces excellent anime art with strong prompt adherence
151
+ - **API**:
152
+ ```python
153
+ from together import Together
154
+ client = Together(api_key="YOUR_KEY")
155
+
156
+ response = client.images.generate(
157
+ model="black-forest-labs/FLUX.1-schnell-Free", # Free model
158
+ prompt="anime girl, silver hair, blue eyes, holding a sword, fantasy forest background, detailed anime illustration",
159
+ width=1024, height=768, n=1
160
+ )
161
+ image_url = response.data[0].url
162
+ ```
163
+ - **Anime-specific**: FLUX.1-schnell is fast and free. Excellent prompt following.
164
+
165
+ **2. HuggingFace Inference API -- Stable Diffusion XL / FLUX**
166
+ - **Free**: Rate-limited (rough estimate: 100-1000 requests/day depending on model load)
167
+ - **Models available**: SDXL, FLUX, SD 1.5 anime finetunes
168
+ - **API**:
169
+ ```python
170
+ from huggingface_hub import InferenceClient
171
+ client = InferenceClient(token="YOUR_HF_TOKEN")
172
+
173
+ image = client.text_to_image(
174
+ "1girl, silver hair, blue eyes, anime style, masterpiece, best quality",
175
+ model="stabilityai/stable-diffusion-xl-base-1.0"
176
+ )
177
+ image.save("output.png")
178
+
179
+ # Anime-specific model:
180
+ image = client.text_to_image(
181
+ "1girl, silver hair, detailed anime illustration",
182
+ model="cagliostrolab/animagine-xl-3.1" # Top anime SDXL model
183
+ )
184
+ ```
185
+ - **Key anime models on HF**: `animagine-xl-3.1`, `CounterfeitXL`, `BluePencilXL`
186
+
187
+ **3. Stability AI Free Tier**
188
+ - **Free**: Limited free credits on signup
189
+ - **Models**: SDXL, SD3, Stable Image Core
190
+ - **API**:
191
+ ```python
192
+ import requests
193
+
194
+ response = requests.post(
195
+ "https://api.stability.ai/v2beta/stable-image/generate/sd3",
196
+ headers={"Authorization": f"Bearer {api_key}"},
197
+ files={"none": ""},
198
+ data={
199
+ "prompt": "anime style illustration, young warrior...",
200
+ "output_format": "png",
201
+ "model": "sd3.5-large",
202
+ }
203
+ )
204
+ ```
205
+
206
+ **4. Pollinations.ai (Completely Free, No API Key)**
207
+ - **Free**: 100% free, no signup, no API key needed
208
+ - **Quality**: Uses FLUX under the hood
209
+ - **API**:
210
+ ```python
211
+ import requests
212
+ import urllib.parse
213
+
214
+ prompt = urllib.parse.quote("anime girl, silver hair, fantasy setting, detailed illustration")
215
+ url = f"https://image.pollinations.ai/prompt/{prompt}?width=1024&height=768&model=flux"
216
+ response = requests.get(url)
217
+ with open("output.png", "wb") as f:
218
+ f.write(response.content)
219
+ ```
220
+ - **Limits**: No hard limits documented, but be respectful with rate
221
+ - **Verdict**: **Best for prototyping and early development. Zero friction.**
222
+
223
+ **5. Google Colab / Kaggle -- Run Your Own**
224
+ - Run SDXL, FLUX, or anime-specialized models on free T4 GPUs
225
+ - Full control over model selection, LoRA, IP-Adapter, ControlNet
226
+ - 30 hrs/week on Kaggle = ~3,000-5,000 images per week
227
+ - This is the **production-grade free option**
228
+
229
+ #### Anime-Specialized Models (All Free/Open Source)
230
+
231
+ | Model | Base | Anime Quality | Best For |
232
+ |-------|------|---------------|----------|
233
+ | Animagine XL 3.1 | SDXL | Excellent | General anime |
234
+ | CounterfeitXL | SDXL | Excellent | Detailed anime |
235
+ | BluePencilXL | SDXL | Very Good | Soft anime style |
236
+ | Anything V5 | SD 1.5 | Good | Classic anime |
237
+ | NovelAI anime model | SD 1.5 | Excellent | Light novel style |
238
+ | Pony Diffusion XL | SDXL | Good | Stylized anime |
239
+
240
+ ### 3.2 Character Consistency Solutions
241
+
242
+ This is **THE hardest problem** in the entire pipeline. Every image gen call produces a different-looking character.
243
+
244
+ #### Solution 1: IP-Adapter (Recommended Primary)
245
+
246
+ IP-Adapter takes a reference image and forces new generations to match the character's appearance.
247
+
248
+ - **How**: Generate one "canonical" image of each character, then use it as reference for ALL subsequent images
249
+ - **Where to run**: Kaggle/Colab with ComfyUI, or via API services that support it
250
+ - **Quality**: ~70-85% consistency (face may drift slightly)
251
+ - **ComfyUI workflow on Kaggle**:
252
+ ```python
253
+ # In a Kaggle notebook:
254
+ # 1. Install ComfyUI
255
+ # 2. Load SDXL + IP-Adapter
256
+ # 3. Provide reference image + new pose/scene prompt
257
+ # 4. Get consistent character in new situation
258
+ ```
259
+
260
+ #### Solution 2: PuLID / InstantID (Best for Face Consistency)
261
+
262
+ - Specifically designed to maintain facial identity across images
263
+ - Takes a single face reference and locks facial features
264
+ - Available as ComfyUI nodes
265
+ - Run on Kaggle/Colab
266
+
267
+ #### Solution 3: Character Sheet + Reference Technique
268
+
269
+ 1. First, generate a character reference sheet (front/side/3-quarter views)
270
+ 2. Use this sheet as IP-Adapter reference for all subsequent generations
271
+ 3. This is how professional anime production works
272
+
273
+ ```python
274
+ CHARACTER_SHEET_PROMPT = """
275
+ character reference sheet, multiple views, front view, side view,
276
+ three-quarter view, same character, {character_description},
277
+ anime style, white background, character turnaround,
278
+ masterpiece, best quality
279
+ """
280
+ ```
281
+
282
+ #### Solution 4: LoRA Training (Most Consistent, More Work)
283
+
284
+ - Train a small LoRA model on 10-20 images of a specific character design
285
+ - Then use this LoRA every time you generate that character
286
+ - Can be done for free on Kaggle (LoRA training takes ~20-30 minutes on T4)
287
+ - Most consistent results but requires upfront work per character
288
+
289
+ #### Solution 5: StoryDiffusion (Emerging Research)
290
+
291
+ - Research project specifically designed for consistent characters across comic/manga panels
292
+ - Maintains character identity without LoRA training
293
+ - Still experimental but very promising for this exact use case
294
+
295
+ #### Recommended Strategy:
296
+
297
+ ```
298
+ 1. Generate character reference sheet (one-time per character)
299
+ 2. Use IP-Adapter + PuLID for all subsequent images
300
+ 3. For protagonist/main characters: train a quick LoRA on Kaggle
301
+ 4. Store all reference images in character database
302
+ 5. Include character visual description in EVERY prompt
303
+ ```
304
+
305
+ ### 3.3 TTS / Voice Acting
306
+
307
+ #### PRIMARY: Edge TTS (Unlimited, Free, Great Quality)
308
+
309
+ ```python
310
+ pip install edge-tts
311
+ ```
312
+
313
+ - **Cost**: Completely free, no API key, no account
314
+ - **Limits**: Effectively unlimited
315
+ - **Emotion support**: SSML with styles: `cheerful`, `sad`, `angry`, `excited`, `terrified`, `whispering`, `shouting`, and more
316
+ - **Voices**: 300+ voices, including Japanese voices for anime feel
317
+ - **Code**:
318
+ ```python
319
+ import edge_tts
320
+ import asyncio
321
+
322
+ async def speak(text, voice="en-US-AriaNeural", style="cheerful", output="out.mp3"):
323
+ if style:
324
+ ssml = f"""
325
+ <speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis'
326
+ xmlns:mstts='https://www.w3.org/2001/mstts' xml:lang='en-US'>
327
+ <voice name='{voice}'>
328
+ <mstts:express-as style='{style}' styledegree='1.5'>
329
+ {text}
330
+ </mstts:express-as>
331
+ </voice>
332
+ </speak>"""
333
+ communicate = edge_tts.Communicate(ssml)
334
+ else:
335
+ communicate = edge_tts.Communicate(text, voice)
336
+ await communicate.save(output)
337
+
338
+ asyncio.run(speak("I won't let you down!", style="excited"))
339
+ ```
340
+
341
+ - **Recommended voice assignments**:
342
+ ```python
343
+ CHARACTER_VOICES = {
344
+ "narrator": {"voice": "en-US-AriaNeural", "style": "narration-professional"},
345
+ "male_hero": {"voice": "en-US-GuyNeural", "style": "chat"},
346
+ "female_lead": {"voice": "en-US-JennyNeural", "style": "cheerful"},
347
+ "villain": {"voice": "en-US-DavisNeural", "style": "angry"},
348
+ "old_mentor": {"voice": "en-US-TonyNeural", "style": "calm"},
349
+ "child": {"voice": "en-US-SaraNeural", "style": "friendly"},
350
+ # Japanese voices for anime feel:
351
+ "jp_female": {"voice": "ja-JP-NanamiNeural", "style": None},
352
+ "jp_male": {"voice": "ja-JP-KeitaNeural", "style": None},
353
+ }
354
+
355
+ EMOTION_TO_STYLE = {
356
+ "happy": "cheerful", "sad": "sad", "angry": "angry",
357
+ "scared": "terrified", "excited": "excited", "calm": "calm",
358
+ "whisper": "whispering", "shout": "shouting", "love": "affectionate",
359
+ "nervous": "embarrassed",
360
+ }
361
+ ```
362
+
363
+ #### SECONDARY: Azure TTS (500K chars/month free, BEST emotion system)
364
+
365
+ - Same voices as Edge TTS but with full API control
366
+ - `styledegree` (0.01 to 2.0) for emotion intensity
367
+ - Role tags: `Girl`, `Boy`, `YoungAdultFemale`, `OlderAdultMale`, `SeniorFemale`
368
+ - Requires Azure account (free tier, credit card on file but charges nothing)
369
+
370
+ #### SPECIAL: Bark by Suno (Local, CPU, Sound Effects)
371
+
372
+ - Open source, runs on CPU (slow: ~30-90s per 10s of audio)
373
+ - **Unique feature**: Inline sound effect tags:
374
+ - `[laughs]`, `[sighs]`, `[gasps]`, `[clears throat]`
375
+ - Great for moments Edge TTS can't handle
376
+ - Use for 5-10% of lines that need sound effects
377
+ - Batch process overnight
378
+
379
+ ```python
380
+ os.environ["SUNO_USE_SMALL_MODELS"] = "True"
381
+ os.environ["SUNO_OFFLOAD_CPU"] = "True"
382
+
383
+ from bark import generate_audio, SAMPLE_RATE
384
+ audio = generate_audio("[sighs] I thought we were friends... [laughs nervously]",
385
+ history_prompt="v2/en_speaker_6")
386
+ ```
387
+
388
+ #### BACKUP: Piper TTS (Local, CPU, Real-Time)
389
+
390
+ - Designed specifically for CPU -- runs in real-time on your hardware
391
+ - No emotion control but perfect for bulk narration
392
+ - Use as offline fallback when internet is unavailable
393
+
394
+ #### Total Free Monthly Capacity:
395
+
396
+ | Service | Free Chars/Month | Emotion | Speed |
397
+ |---------|-----------------|---------|-------|
398
+ | Edge TTS | **Unlimited** | Good SSML | Fast |
399
+ | Azure TTS | **500,000** | **Best** | Fast |
400
+ | Bark (local) | **Unlimited** | **Excellent** ([laughs] etc.) | Slow |
401
+ | Piper (local) | **Unlimited** | None | Real-time |
402
+
403
+ ### 3.4 Interpolation & Animation
404
+
405
+ #### Frame Interpolation: RIFE (Primary Tool)
406
+
407
+ Takes 2 keyframe images and generates smooth intermediate frames.
408
+
409
+ ```
410
+ Frame A → [RIFE generates 7 in-between frames] → Frame B
411
+ Result: 9 frames of smooth animation from just 2 images
412
+ ```
413
+
414
+ - **Best option for your hardware**: `rife-ncnn-vulkan-python`
415
+ - Uses Vulkan backend -- **works natively on Intel Arc**
416
+ - No OpenVINO conversion needed
417
+ - Fast: near real-time at 512x512
418
+
419
+ ```bash
420
+ pip install rife-ncnn-vulkan-python
421
+ ```
422
+ ```python
423
+ from rife_ncnn_vulkan_python import Rife
424
+ rife = Rife(gpuid=0) # Uses Vulkan on Intel Arc
425
+ mid_frame = rife.process(img0, img1) # Returns interpolated frame
426
+ ```
427
+
428
+ - **Alternative**: RIFE via OpenVINO (potentially faster but requires model conversion)
429
+ - **Fallback**: FILM by Google (better for large motion gaps, runs on CPU via TensorFlow)
430
+
431
+ #### Image Animation: SadTalker (For Talking Faces)
432
+
433
+ Takes character image + TTS audio → outputs video of character talking with lip sync.
434
+
435
+ - **CPU mode**: `--cpu` flag, ~30-60s for 10s video at 256x256
436
+ - **Free on HuggingFace Spaces**: Can call remotely via `gradio_client`
437
+
438
+ ```python
439
+ from gradio_client import Client
440
+ client = Client("vinthony/SadTalker")
441
+ result = client.predict("character.png", "speech.wav", "crop", True, True)
442
+ ```
443
+
444
+ #### Image Animation: LivePortrait (For Character Motion)
445
+
446
+ Animates portrait with expressions and head movements.
447
+
448
+ - Has OpenVINO support -- can use Intel Arc GPU
449
+ - ~15-30 FPS with OpenVINO on Intel Arc
450
+
451
+ #### Lip Sync: Rhubarb Lip Sync (CPU, Real-Time)
452
+
453
+ Analyzes audio and outputs timed mouth shape data. Anime only uses ~5-7 mouth shapes.
454
+
455
+ ```python
456
+ # Rhubarb outputs: {"mouthCues": [{"start": 0.00, "end": 0.05, "value": "A"}, ...]}
457
+ # Map to anime mouth shapes: Closed, Small, Medium, Open, Wide
458
+ ```
459
+
460
+ #### Ken Burns Effects: FFmpeg (ZERO AI, Instant)
461
+
462
+ This should be **70-80% of your animation**. Real anime uses this constantly.
463
+
464
+ ```python
465
+ # Slow zoom into image
466
+ ffmpeg -loop 1 -i image.png -vf "zoompan=z='min(zoom+0.001,1.3)':s=1920x1080:fps=24:d=120" \
467
+ -t 5 -c:v libx264 output.mp4
468
+
469
+ # Pan left to right
470
+ ffmpeg -loop 1 -i image.png -vf "zoompan=z=1.2:x='iw/2-(iw/zoom/2)+(-0.3+on/120*0.6)*iw/zoom':..." \
471
+ -t 5 output.mp4
472
+ ```
473
+
474
+ - **Performance**: Instant. FFmpeg is heavily optimized.
475
+ - **Intel QSV**: Use `h264_qsv` encoder for 3-5x faster encoding on Intel Arc
476
+
477
+ #### Anime-Specific Effects (CPU, OpenCV)
478
+
479
+ All of these are trivial on CPU:
480
+
481
+ | Effect | Implementation | Use Case |
482
+ |--------|---------------|----------|
483
+ | Speed lines | OpenCV line drawing overlay | Action scenes |
484
+ | Camera shake | Random frame offset | Impact moments |
485
+ | White flash | Alpha blend with white | Explosions, reveals |
486
+ | Dramatic zoom lines | Radial lines from center | Power-up moments |
487
+ | Vignette | Radial gradient overlay | Moody scenes |
488
+ | Screen shake | Sine wave displacement | Earthquakes, impacts |
489
+
490
+ #### Parallax / 2.5D Effect
491
+
492
+ Use MiDaS depth estimation (has OpenVINO support!) to create depth maps, then move foreground/background at different speeds.
493
+
494
+ ```python
495
+ from openvino.runtime import Core
496
+ core = Core()
497
+ model = core.compile_model("midas_v21_small.xml", "GPU") # Intel Arc
498
+ depth_map = model([image])[0]
499
+ # Separate into layers based on depth → animate at different speeds
500
+ ```
501
+
502
+ ### 3.5 Video Assembly
503
+
504
+ #### FFmpeg Pipeline (Core)
505
+
506
+ ```python
507
+ class AnimeVideoAssembler:
508
+ def images_to_video(self, images, fps=24):
509
+ """Number images → video"""
510
+
511
+ def add_audio(self, video, tts_audio):
512
+ """Combine video + TTS"""
513
+
514
+ def mix_audio(self, tts, bg_music, music_vol=0.15):
515
+ """Mix narration + background music"""
516
+
517
+ def add_subtitles(self, video, srt_file):
518
+ """Burn anime-style subtitles"""
519
+
520
+ def crossfade(self, clip1, clip2, duration=0.5):
521
+ """Scene transitions"""
522
+
523
+ def concat(self, clips):
524
+ """Join all scenes into episode"""
525
+
526
+ def encode_qsv(self, input, output):
527
+ """Intel Arc hardware-accelerated encoding"""
528
+ # ffmpeg -i input -c:v h264_qsv -global_quality 23 output.mp4
529
+ ```
530
+
531
+ #### MoviePy (Python-Friendly Alternative)
532
+
533
+ ```python
534
+ from moviepy.editor import *
535
+
536
+ # Per scene: image + Ken Burns + audio + subtitle
537
+ clip = ImageClip("scene.png", duration=5)
538
+ clip = clip.resize(lambda t: 1 + 0.04*t) # Zoom effect
539
+ clip = clip.set_audio(AudioFileClip("tts.mp3"))
540
+ txt = TextClip("Dialogue here", fontsize=36, color='white', stroke_color='black')
541
+ clip = CompositeVideoClip([clip, txt.set_position(('center','bottom'))])
542
+ ```
543
+
544
+ ---
545
+
546
+ ## 4. NLP & INTELLIGENCE LAYER
547
+
548
+ ### 4.1 Free LLM APIs (for Scene Parsing, Prompt Generation)
549
+
550
+ | Provider | Free Limits | Best Model | Context | Best For |
551
+ |----------|------------|------------|---------|----------|
552
+ | **Google Gemini** | 1500 req/day, 1M TPM | Gemini 2.0 Flash | **1M tokens** | Scene parsing (fit whole chapter in one call) |
553
+ | **Groq** | 14,400 req/day | Llama 3.3 70B | 128K | Batch emotion analysis (fastest) |
554
+ | **Together AI** | $5 free credits | Llama 3.3 70B | 128K | Backup + image gen (FLUX) |
555
+ | **OpenRouter** | 200 req/day | Various free models | Varies | Unified fallback |
556
+ | **HuggingFace** | ~100 req/hour | Specialized NLP models | Varies | Emotion classification |
557
+ | **Cohere** | 1000 calls/month | Command-R | 128K | Structured extraction backup |
558
+ | **SambaNova** | Free during beta | Llama 3.1 405B | 128K | Strongest reasoning |
559
+
560
+ **Strategy**: Gemini as primary (1M context = entire chapter in one call), Groq as fast secondary, others as fallbacks.
561
+
562
+ ### 4.2 Text → Scene Pipeline
563
+
564
+ ```
565
+ Raw Chapter Text
566
+ ↓
567
+ [spaCy + NLTK] -- Fast local NLP (FREE)
568
+ - Sentence tokenization
569
+ - Dialogue extraction (regex)
570
+ - Named Entity Recognition
571
+ - Scene break detection (--- markers, time skips)
572
+ ↓
573
+ [Gemini API] -- Semantic understanding (FREE)
574
+ - Scene segmentation
575
+ - Emotion per line
576
+ - Action extraction
577
+ - Setting descriptions
578
+ ↓
579
+ [Prompt Generator] -- Convert to image prompts
580
+ - Character visual descriptions (from database)
581
+ - Camera angle selection
582
+ - Art style tags
583
+ ↓
584
+ Structured Scene JSON
585
+ ```
586
+
587
+ ### 4.3 Key Libraries
588
+
589
+ ```bash
590
+ pip install spacy nltk google-generativeai groq huggingface-hub
591
+ python -m spacy download en_core_web_lg
592
+ ```
593
+
594
+ ### 4.4 Character Management
595
+
596
+ ```python
597
+ # Each character gets:
598
+ # - Canonical name + aliases
599
+ # - Visual description (50 words, prompt-ready, NO name)
600
+ # - Reference image paths (for IP-Adapter)
601
+ # - Voice assignment (Edge TTS voice + default emotion)
602
+ # - Outfit tracking per chapter
603
+
604
+ # Example:
605
+ {
606
+ "name": "Sarah",
607
+ "aliases": ["the silver-haired girl", "the mage"],
608
+ "visual_prompt": "young woman, long silver hair, blue eyes, slender build, wearing blue mage robes with gold trim, carrying wooden staff",
609
+ "reference_images": ["characters/sarah_ref1.png", "characters/sarah_ref2.png"],
610
+ "voice": {"engine": "edge-tts", "voice": "en-US-JennyNeural", "default_style": "cheerful"},
611
+ }
612
+ ```
613
+
614
+ ### 4.5 Web Novel Scraping
615
+
616
+ - **Primary target**: Royal Road (easy, clean HTML, minimal anti-scraping)
617
+ - **Best existing library**: `lightnovel-crawler` (supports 100+ sites)
618
+ - **Custom scraper**: BeautifulSoup + requests with 2s delay between pages
619
+ - **Legal**: Author opt-in system for commercial use. Public domain novels for demos.
620
+
621
+ ---
622
+
623
+ ## 5. SYSTEM ARCHITECTURE
624
+
625
+ ### 5.1 Full Pipeline Overview
626
+
627
+ ```
628
+ USER INPUT (URL or pasted text)
629
+ │
630
+ ▼
631
+ ┌─────────────────────┐
632
+ │ 1. TEXT INGESTION │ Scraper or paste → raw chapter text
633
+ │ (2 seconds) │ lightnovel-crawler / BeautifulSoup
634
+ └────────┬────────────┘
635
+ │
636
+ ▼
637
+ ┌─────────────────────┐
638
+ │ 2. NLP ANALYSIS │ spaCy + NLTK → entities, dialogue, breaks
639
+ │ (5 seconds) │ Gemini API → scenes, emotions, actions
640
+ └────────┬────────────┘
641
+ │
642
+ ▼
643
+ ┌─────────────────────┐
644
+ │ 3. VISUAL SCRIPT │ Scene → shots → image prompts
645
+ │ (10 seconds) │ Character DB → consistent descriptions
646
+ │ │ Camera angle selection
647
+ └────────┬────────────┘
648
+ │
649
+ ▼
650
+ ┌─────────────────────────────────────────┐
651
+ │ 4. ASSET GENERATION (parallel) │
652
+ │ │
653
+ │ ┌──────────────┐ ┌──────────────────┐ │
654
+ │ │ Image Gen │ │ TTS Generation │ │
655
+ │ │ 15-30 images │ │ All dialogue + │ │
656
+ │ │ via FLUX/SDXL│ │ narration via │ │
657
+ │ │ + IP-Adapter │ │ Edge TTS │ │
658
+ │ │ (2-5 min) │ │ (1-2 min) │ │
659
+ │ └──────┬───────┘ └──────┬───────────┘ │
660
+ └─────────┼─────────────────┼─────────────┘
661
+ │ │
662
+ ▼ ▼
663
+ ┌─────────────────────────────────────────┐
664
+ │ 5. ANIMATION LAYER │
665
+ │ │
666
+ │ • Ken Burns on all scenes (instant) │
667
+ │ • RIFE interpolation between keyframes │
668
+ │ • SadTalker for key dialogue faces │
669
+ │ • Lip sync via Rhubarb │
670
+ │ • Camera shake / effects for action │
671
+ │ • Depth parallax for atmosphere │
672
+ │ (1-3 min) │
673
+ └────────┬────────────────────────────────┘
674
+ │
675
+ ▼
676
+ ┌─────────────────────┐
677
+ │ 6. VIDEO ASSEMBLY │ FFmpeg: concat scenes + add subtitles
678
+ │ (30 seconds) │ Mix TTS + background music
679
+ │ │ Encode with Intel QSV
680
+ └────────┬────────────┘
681
+ │
682
+ ▼
683
+ FINAL MP4 EPISODE
684
+ (3-10 minutes per chapter)
685
+ ```
686
+
687
+ ### 5.2 Tech Stack
688
+
689
+ ```
690
+ Backend: Python 3.11+, FastAPI, Celery + Redis (task queue)
691
+ NLP: spaCy (en_core_web_lg) + NLTK + Gemini API
692
+ Image Gen: FLUX/SDXL via Together AI, HuggingFace, Pollinations, or Kaggle
693
+ Animation: RIFE (rife-ncnn-vulkan), SadTalker, FFmpeg, OpenCV
694
+ TTS: Edge TTS (primary), Bark (effects), Piper (offline fallback)
695
+ Video: FFmpeg + MoviePy, Intel QSV encoding
696
+ Frontend: Next.js + Tailwind CSS + Video.js player
697
+ Storage: Local filesystem → Cloudflare R2 (production)
698
+ Database: SQLite (start) → PostgreSQL (scale)
699
+ ```
700
+
701
+ ### 5.3 Performance Estimates (Your Hardware)
702
+
703
+ | Task | Time | Tool |
704
+ |------|------|------|
705
+ | Scrape 1 chapter | 3s | requests/beautifulsoup |
706
+ | NLP analysis | 5s | spaCy local |
707
+ | LLM scene parsing | 10s | Gemini API |
708
+ | Generate 1 image (API) | 5-15s | FLUX via Together/HF/Pollinations |
709
+ | Generate 1 image (Kaggle) | 10-30s | SDXL + IP-Adapter |
710
+ | RIFE interpolation (1 pair) | 0.5-2s | ncnn-vulkan on Intel Arc |
711
+ | Ken Burns 5s clip | <1s | FFmpeg |
712
+ | TTS for 1 line | 1-3s | Edge TTS |
713
+ | SadTalker 10s video | 30-60s | CPU mode |
714
+ | Depth map | 1-3s | MiDaS + OpenVINO on Intel Arc |
715
+ | Final encoding | 5-10s/min | FFmpeg + QSV |
716
+ | **FULL CHAPTER** | **~5-15 minutes** | **Complete pipeline** |
717
+
718
+ ---
719
+
720
+ ## 6. YOUR HARDWARE ADVANTAGE
721
+
722
+ ### Intel Core Ultra 5 + Intel Arc Graphics
723
+
724
+ Your hardware is actually better than you think for AI:
725
+
726
+ 1. **Intel Arc GPU supports**:
727
+ - OpenVINO (Intel's AI inference toolkit)
728
+ - Vulkan (for ncnn-based models like RIFE)
729
+ - Intel QSV (hardware video encoding -- 3-5x faster)
730
+ - DirectML (Windows AI acceleration)
731
+
732
+ 2. **NPU (Neural Processing Unit)** in Core Ultra:
733
+ - Can run lightweight models (face detection, classification)
734
+ - Accessible via OpenVINO
735
+
736
+ 3. **What this means for your pipeline**:
737
+
738
+ | Task | Without Intel Arc | With Intel Arc (OpenVINO/Vulkan) |
739
+ |------|-------------------|----------------------------------|
740
+ | RIFE interpolation | 2-5s/frame on CPU | **0.1-0.5s/frame** |
741
+ | MiDaS depth | 5-10s on CPU | **1-3s** |
742
+ | First Order Motion | Not practical on CPU | **5-15 FPS** |
743
+ | Video encoding | Software libx264 | **h264_qsv (3-5x faster)** |
744
+
745
+ ### Setup Commands
746
+
747
+ ```bash
748
+ # Install OpenVINO
749
+ pip install openvino openvino-dev
750
+
751
+ # Install Intel PyTorch Extension
752
+ pip install intel-extension-for-pytorch
753
+
754
+ # RIFE with Vulkan (works on Intel Arc)
755
+ pip install rife-ncnn-vulkan-python
756
+
757
+ # Verify Intel Arc is detected
758
+ python -c "from openvino.runtime import Core; print(Core().available_devices)"
759
+ # Expected: ['CPU', 'GPU', 'NPU']
760
+ ```
761
+
762
+ ---
763
+
764
+ ## 7. IMPLEMENTATION ROADMAP
765
+
766
+ ### Phase 1: Proof of Concept (Week 1-2)
767
+ **Goal**: URL → scene JSON → static image set + narration audio
768
+
769
+ - Set up project structure
770
+ - Build Royal Road scraper
771
+ - Implement NLP pipeline (spaCy + Gemini)
772
+ - Build basic prompt generator
773
+ - Generate images via Pollinations.ai (zero friction, no API key)
774
+ - Generate TTS via Edge TTS
775
+ - **Deliverable**: A folder of images + audio files for one chapter
776
+
777
+ ### Phase 2: First Video (Week 3-4)
778
+ **Goal**: Images + audio → watchable motion comic episode
779
+
780
+ - Implement FFmpeg Ken Burns pipeline
781
+ - Build scene assembler (concat, transitions, subtitles)
782
+ - Add background music mixing
783
+ - Generate SRT subtitles from TTS timestamps
784
+ - **Deliverable**: First complete video episode from a web novel chapter
785
+
786
+ ### Phase 3: Character Consistency (Week 5-6)
787
+ **Goal**: Characters look the same across all scenes
788
+
789
+ - Set up ComfyUI on Kaggle notebook
790
+ - Implement IP-Adapter / PuLID workflow
791
+ - Build character management system
792
+ - Generate character reference sheets
793
+ - Switch from Pollinations to Kaggle/Together for image gen
794
+ - **Deliverable**: Episode with consistent characters
795
+
796
+ ### Phase 4: Animation Enhancement (Week 7-8)
797
+ **Goal**: Add motion beyond Ken Burns
798
+
799
+ - Integrate RIFE interpolation (ncnn-vulkan on Intel Arc)
800
+ - Add SadTalker for talking face close-ups
801
+ - Implement Rhubarb lip sync
802
+ - Add anime effects (speed lines, camera shake, flash)
803
+ - Implement depth-based parallax via MiDaS
804
+ - **Deliverable**: Significantly more animated episodes
805
+
806
+ ### Phase 5: Web Application (Week 9-10)
807
+ **Goal**: Users can access the tool via browser
808
+
809
+ - Build FastAPI backend with task queue (Celery)
810
+ - Build Next.js frontend (paste URL → get video)
811
+ - Add progress tracking
812
+ - Add art style selection
813
+ - Add video player
814
+ - **Deliverable**: Working web app
815
+
816
+ ### Phase 6: Launch & Monetize (Week 11+)
817
+ **Goal**: Get users and start earning
818
+
819
+ - Deploy to cloud (Vercel frontend, Railway backend)
820
+ - Implement user accounts
821
+ - Add Stripe payments (freemium model)
822
+ - Create demo episodes for marketing (TikTok, YouTube Shorts)
823
+ - Submit to ProductHunt, HackerNews, Reddit
824
+ - Author partnership program
825
+
826
+ ---
827
+
828
+ ## 8. MONETIZATION STRATEGY
829
+
830
+ ### Pricing Model (Freemium SaaS)
831
+
832
+ ```
833
+ FREE TIER:
834
+ - 1 chapter/day
835
+ - 720p resolution
836
+ - Watermarked
837
+ - Basic art style
838
+ - 30-second preview
839
+
840
+ PREMIUM ($9.99/month):
841
+ - 10 chapters/day
842
+ - 1080p resolution
843
+ - No watermark
844
+ - Multiple art styles
845
+ - Full episodes
846
+ - Download as MP4
847
+
848
+ PRO ($29.99/month):
849
+ - Unlimited chapters
850
+ - 4K resolution
851
+ - API access
852
+ - Custom character LoRA training
853
+ - Batch processing
854
+ - Commercial usage rights
855
+ ```
856
+
857
+ ### Revenue Projections (Conservative)
858
+
859
+ ```
860
+ Month 6: 10K users, 500 paid → ~$5,000/mo
861
+ Month 12: 50K users, 2,500 paid → ~$25,000/mo
862
+ Month 24: 200K users, 10,000 paid → ~$100,000/mo
863
+ ```
864
+
865
+ ### Market Opportunity
866
+
867
+ - 200M+ monthly web novel readers globally
868
+ - Zero competitors doing exactly this (novel text → anime video)
869
+ - Growing AI video generation market ($500M+ in 2024)
870
+ - Strong social media virality potential (anime clips from popular novels)
871
+
872
+ ---
873
+
874
+ ## 9. FINAL RECOMMENDATION
875
+
876
+ ### The Strategy
877
+
878
+ **Plan A (pure video gen) is not viable for free** -- but it's useful for ~5% of content (hero moments via HuggingFace).
879
+
880
+ **Plan B IS the plan.** And it's not a compromise. Here's why:
881
+
882
+ The "motion comic" style (images + Ken Burns + TTS + effects + interpolation) is:
883
+ 1. Literally how real anime handles dialogue and atmospheric scenes
884
+ 2. Fully achievable on your hardware at $0 cost
885
+ 3. Character-consistent (via IP-Adapter)
886
+ 4. Infinitely controllable (every frame, every angle, every expression)
887
+ 5. Scalable to production quality
888
+
889
+ ### Your Secret Weapons
890
+
891
+ 1. **Edge TTS** -- Unlimited free voice acting with emotion control
892
+ 2. **Pollinations.ai / Together AI / HuggingFace** -- Free anime image generation
893
+ 3. **Intel Arc + OpenVINO** -- Accelerated AI inference most people don't have
894
+ 4. **RIFE + ncnn-vulkan** -- Smooth frame interpolation on Intel Arc
895
+ 5. **Gemini API** -- 1500 free scene parsing calls/day (100+ chapters/day)
896
+ 6. **Kaggle free GPUs** -- 30 hrs/week for heavy image gen with IP-Adapter
897
+
898
+ ### What to Build First
899
+
900
+ ```
901
+ THIS WEEKEND:
902
+ 1. pip install edge-tts spacy google-generativeai requests beautifulsoup4
903
+ 2. Scrape a Royal Road chapter
904
+ 3. Parse it with spaCy + Gemini into scenes
905
+ 4. Generate images via Pollinations.ai
906
+ 5. Generate TTS via Edge TTS
907
+ 6. FFmpeg: images + audio → your first anime episode
908
+
909
+ COST: $0 | TIME: 1 weekend | RESULT: Proof of concept video
910
+ ```
911
+
912
+ ### Key Dependencies to Install
913
+
914
+ ```bash
915
+ # Core
916
+ pip install edge-tts spacy nltk google-generativeai groq
917
+ pip install requests beautifulsoup4 Pillow opencv-python
918
+ pip install moviepy numpy librosa
919
+
920
+ # Intel Arc acceleration
921
+ pip install openvino rife-ncnn-vulkan-python
922
+
923
+ # NLP models
924
+ python -m spacy download en_core_web_lg
925
+ python -m nltk.downloader vader_lexicon punkt
926
+
927
+ # For HuggingFace model access
928
+ pip install huggingface-hub gradio-client
929
+
930
+ # For web novel scraping
931
+ pip install lightnovel-crawler
932
+ ```
933
+
934
+ ---
935
+
936
+ ## APPENDIX: Quick Reference of All Free APIs
937
+
938
+ | Service | What | Free Limit | API Key? | URL |
939
+ |---------|------|-----------|----------|-----|
940
+ | Edge TTS | Voice synthesis | Unlimited | No | `pip install edge-tts` |
941
+ | Pollinations.ai | Image generation (FLUX) | Unlimited | No | image.pollinations.ai |
942
+ | Google Gemini | LLM (scene parsing) | 1500 req/day | Yes (free) | aistudio.google.com |
943
+ | Groq | LLM (fast inference) | 14,400 req/day | Yes (free) | console.groq.com |
944
+ | HuggingFace | Images, video, TTS, LLMs | Rate-limited | Yes (free) | huggingface.co |
945
+ | Together AI | Images (FLUX) + LLMs | $5 free credits | Yes (free) | api.together.xyz |
946
+ | Azure TTS | Voice synthesis + emotion | 500K chars/mo | Yes (free) | azure.microsoft.com |
947
+ | Kaggle | Free T4 GPUs | 30 hrs/week | Account | kaggle.com |
948
+ | Google Colab | Free T4 GPUs | ~4 hrs/day | Account | colab.research.google.com |
949
+ | OpenRouter | Various free LLMs | 200 req/day | Yes (free) | openrouter.ai |
950
+
951
+ ---
952
+
953
+ *This research was conducted across web searches, model documentation, API documentation, and community reports as of February 2026. Verify current free tier limits before relying on them in production.*
TESTED_APIS.md ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Image Generation APIs -- TESTED RESULTS (2026-02-27)
2
+
3
+ ## CONFIRMED WORKING (with your actual API keys)
4
+
5
+ ### 1. NVIDIA NIM -- FLUX.1-dev (BEST QUALITY)
6
+ - **Status**: WORKING
7
+ - **Speed**: 4.8 seconds per image
8
+ - **Resolution**: 1024x768
9
+ - **Anime Quality**: EXCELLENT (professional anime key visual level)
10
+ - **Free Credits**: 1,000 on signup (up to 5,000 with business email + 90-day enterprise license)
11
+ - **API**:
12
+ ```python
13
+ import requests, base64
14
+
15
+ r = requests.post(
16
+ "https://ai.api.nvidia.com/v1/genai/black-forest-labs/flux.1-dev",
17
+ headers={"Authorization": "Bearer YOUR_NVIDIA_KEY", "Accept": "application/json"},
18
+ json={
19
+ "prompt": "anime style, young warrior girl, silver hair, blue eyes, glowing sword, fantasy forest, masterpiece",
20
+ "mode": "base",
21
+ "cfg_scale": 3.5,
22
+ "width": 1024,
23
+ "height": 768,
24
+ "seed": -1,
25
+ "steps": 30
26
+ }
27
+ )
28
+ img_bytes = base64.b64decode(r.json()["artifacts"][0]["base64"])
29
+ with open("output.png", "wb") as f:
30
+ f.write(img_bytes)
31
+ ```
32
+
33
+ ### 2. NVIDIA NIM -- FLUX.1-schnell (FASTEST)
34
+ - **Status**: WORKING
35
+ - **Speed**: 3.4 seconds per image
36
+ - **Resolution**: 1024x1024
37
+ - **Anime Quality**: EXCELLENT
38
+ - **API**: Same as above, endpoint: `black-forest-labs/flux.1-schnell`, steps=4
39
+
40
+ ### 3. HuggingFace -- FLUX.1-schnell (MOST GENEROUS FREE)
41
+ - **Status**: WORKING
42
+ - **Speed**: 1.4-3.8 seconds per image
43
+ - **Resolution**: up to 1024x768
44
+ - **Anime Quality**: EXCELLENT
45
+ - **Free Tier**: Rate-limited but generous (3 rapid requests all succeeded)
46
+ - **API**:
47
+ ```python
48
+ import requests
49
+
50
+ r = requests.post(
51
+ "https://router.huggingface.co/hf-inference/models/black-forest-labs/FLUX.1-schnell",
52
+ headers={"Authorization": "Bearer YOUR_HF_TOKEN", "Content-Type": "application/json"},
53
+ json={
54
+ "inputs": "1girl, silver hair, blue eyes, holding sword, anime style, masterpiece",
55
+ "parameters": {"negative_prompt": "low quality, blurry", "width": 1024, "height": 768}
56
+ }
57
+ )
58
+ with open("output.png", "wb") as f:
59
+ f.write(r.content) # Returns raw image bytes
60
+ ```
61
+
62
+ ### 4. HuggingFace -- SDXL Base (GOOD PAINTERLY STYLE)
63
+ - **Status**: WORKING
64
+ - **Speed**: 9.8 seconds per image
65
+ - **Resolution**: up to 1024x768
66
+ - **Anime Quality**: EXCELLENT (more painterly/artistic style)
67
+ - **API**: Same as above, model: `stabilityai/stable-diffusion-xl-base-1.0`
68
+
69
+ ---
70
+
71
+ ## NOT WORKING / NOT FREE
72
+
73
+ | Service | Status | Reason |
74
+ |---------|--------|--------|
75
+ | OpenRouter | NO free image models | All 29 free models are text-only |
76
+ | Prodia | Needs paid upgrade | Free accounts can't create API tokens |
77
+ | Google Gemini | Quota = 0 on free tier | Image gen removed from free tier entirely |
78
+ | Google Imagen | Paid only | "Imagen 3 is only available on paid plans" |
79
+ | Pollinations | HTTP 530 | Cloudflare blocking (may work from other networks) |
80
+ | Together AI | No free tier (2026) | User confirmed free tier discontinued |
81
+ | DeepInfra | Needs auth/credits | 403 without authentication |
82
+ | fal.ai | Needs auth/credits | 401 without authentication |
83
+ | HF FLUX.1-dev | DEPRECATED | "No longer supported by provider hf-inference" |
84
+ | HF Animagine XL | 404 | Not available on inference API |
85
+ | Stability AI | Needs API key | 401 without key |
86
+
87
+ ---
88
+
89
+ ## RECOMMENDED STRATEGY
90
+
91
+ ### Primary: HuggingFace FLUX.1-schnell
92
+ - Fastest (1.4-3.8s)
93
+ - No visible hard limit yet
94
+ - Excellent anime quality
95
+ - Use for bulk image generation
96
+
97
+ ### Secondary: NVIDIA NIM FLUX.1-dev
98
+ - Best quality (30-step generation)
99
+ - 1,000-5,000 free credits
100
+ - Use for hero/key moment images
101
+ - Save credits for important scenes
102
+
103
+ ### Fallback: HuggingFace SDXL
104
+ - Different artistic style
105
+ - Good for variety
106
+ - 9.8s per image (slower)
107
+
108
+ ### Combined Monthly Estimate
109
+ - HuggingFace: Unknown hard limit, but generous for free tier
110
+ - NVIDIA: 1,000+ credits (unknown images-per-credit ratio)
111
+ - Pollinations: ~1,000 images (if accessible from your network)
112
+ - **Total: Likely 1,000-5,000+ images/month for $0**
113
+
114
+ ### Images Per Chapter
115
+ - A 3-5 minute episode needs ~20-30 keyframe images
116
+ - At 2,000 images/month = ~66-100 chapters/month
117
+ - More than enough for production
118
+
119
+ ---
120
+
121
+ ## QUALITY RANKING (for anime)
122
+
123
+ 1. NVIDIA FLUX.1-dev (30 steps) -- Most detailed, best composition
124
+ 2. NVIDIA FLUX.1-schnell -- Great quality, fastest NVIDIA
125
+ 3. HuggingFace FLUX.1-schnell -- Slightly different style, very fast
126
+ 4. HuggingFace SDXL -- More painterly, artistic feel
127
+
128
+ All four produce professional-grade anime art suitable for our project.
app/__init__.py ADDED
File without changes
app/api/__init__.py ADDED
File without changes
app/api/auth.py ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import APIRouter, Depends, HTTPException, Response, Cookie, status
2
+ from sqlalchemy import select
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.database import get_db
5
+ from app.models.user import User
6
+ from app.schemas.auth import UserCreate, UserLogin, UserResponse, TokenResponse
7
+ from app.services.auth import (
8
+ hash_password, verify_password,
9
+ create_access_token, create_refresh_token, decode_token,
10
+ get_current_user,
11
+ )
12
+ from app.config import get_settings
13
+
14
+ router = APIRouter(prefix="/api/auth", tags=["auth"])
15
+ settings = get_settings()
16
+
17
+
18
+ @router.post("/register", response_model=UserResponse, status_code=201)
19
+ async def register(data: UserCreate, db: AsyncSession = Depends(get_db)):
20
+ # Check existing
21
+ existing = await db.execute(
22
+ select(User).where((User.email == data.email) | (User.username == data.username))
23
+ )
24
+ if existing.scalar_one_or_none():
25
+ raise HTTPException(status_code=409, detail="Email or username already registered")
26
+
27
+ user = User(
28
+ email=data.email,
29
+ username=data.username,
30
+ password_hash=hash_password(data.password),
31
+ )
32
+ db.add(user)
33
+ await db.commit()
34
+ await db.refresh(user)
35
+ return user
36
+
37
+
38
+ @router.post("/login", response_model=TokenResponse)
39
+ async def login(data: UserLogin, response: Response, db: AsyncSession = Depends(get_db)):
40
+ result = await db.execute(select(User).where(User.email == data.email))
41
+ user = result.scalar_one_or_none()
42
+ if not user or not verify_password(data.password, user.password_hash):
43
+ raise HTTPException(status_code=401, detail="Invalid credentials")
44
+
45
+ access_token = create_access_token(user.id)
46
+ refresh_token = create_refresh_token(user.id)
47
+
48
+ response.set_cookie(
49
+ key="refresh_token",
50
+ value=refresh_token,
51
+ httponly=True,
52
+ secure=False, # Set True in production with HTTPS
53
+ samesite="lax",
54
+ max_age=settings.refresh_token_expire_days * 86400,
55
+ path="/api/auth",
56
+ )
57
+
58
+ return TokenResponse(access_token=access_token)
59
+
60
+
61
+ @router.post("/refresh", response_model=TokenResponse)
62
+ async def refresh(
63
+ response: Response,
64
+ refresh_token: str | None = Cookie(default=None),
65
+ db: AsyncSession = Depends(get_db),
66
+ ):
67
+ if not refresh_token:
68
+ raise HTTPException(status_code=401, detail="No refresh token")
69
+
70
+ payload = decode_token(refresh_token)
71
+ if payload.get("type") != "refresh":
72
+ raise HTTPException(status_code=401, detail="Invalid token type")
73
+
74
+ user_id = int(payload["sub"])
75
+ result = await db.execute(select(User).where(User.id == user_id))
76
+ user = result.scalar_one_or_none()
77
+ if not user:
78
+ raise HTTPException(status_code=401, detail="User not found")
79
+
80
+ new_access = create_access_token(user.id)
81
+ new_refresh = create_refresh_token(user.id)
82
+
83
+ response.set_cookie(
84
+ key="refresh_token",
85
+ value=new_refresh,
86
+ httponly=True,
87
+ secure=False,
88
+ samesite="lax",
89
+ max_age=settings.refresh_token_expire_days * 86400,
90
+ path="/api/auth",
91
+ )
92
+
93
+ return TokenResponse(access_token=new_access)
94
+
95
+
96
+ @router.post("/logout")
97
+ async def logout(response: Response):
98
+ response.delete_cookie("refresh_token", path="/api/auth")
99
+ return {"message": "Logged out"}
100
+
101
+
102
+ @router.get("/me", response_model=UserResponse)
103
+ async def me(user: User = Depends(get_current_user)):
104
+ return user
app/api/chapters.py ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import APIRouter, Depends, HTTPException
2
+ from sqlalchemy import select
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.database import get_db
5
+ from app.models.user import User
6
+ from app.models.project import Project
7
+ from app.models.chapter import Chapter
8
+ from app.schemas.chapter import ChapterCreate, ChapterResponse
9
+ from app.services.auth import get_current_user
10
+
11
+ router = APIRouter(prefix="/api/projects/{project_id}/chapters", tags=["chapters"])
12
+
13
+
14
+ async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
15
+ result = await db.execute(
16
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
17
+ )
18
+ project = result.scalar_one_or_none()
19
+ if not project:
20
+ raise HTTPException(status_code=404, detail="Project not found")
21
+ return project
22
+
23
+
24
+ @router.get("", response_model=list[ChapterResponse])
25
+ async def list_chapters(
26
+ project_id: int,
27
+ user: User = Depends(get_current_user),
28
+ db: AsyncSession = Depends(get_db),
29
+ ):
30
+ await _get_user_project(project_id, user, db)
31
+ result = await db.execute(
32
+ select(Chapter)
33
+ .where(Chapter.project_id == project_id)
34
+ .order_by(Chapter.chapter_number)
35
+ )
36
+ return list(result.scalars().all())
37
+
38
+
39
+ @router.post("", response_model=ChapterResponse, status_code=201)
40
+ async def create_chapter(
41
+ project_id: int,
42
+ data: ChapterCreate,
43
+ user: User = Depends(get_current_user),
44
+ db: AsyncSession = Depends(get_db),
45
+ ):
46
+ await _get_user_project(project_id, user, db)
47
+ chapter = Chapter(
48
+ project_id=project_id,
49
+ chapter_number=data.chapter_number,
50
+ title=data.title,
51
+ source_text=data.source_text,
52
+ word_count=len(data.source_text.split()),
53
+ )
54
+ db.add(chapter)
55
+ await db.commit()
56
+ await db.refresh(chapter)
57
+ return chapter
58
+
59
+
60
+ @router.post("/bulk", response_model=list[ChapterResponse], status_code=201)
61
+ async def create_chapters_bulk(
62
+ project_id: int,
63
+ chapters: list[ChapterCreate],
64
+ user: User = Depends(get_current_user),
65
+ db: AsyncSession = Depends(get_db),
66
+ ):
67
+ await _get_user_project(project_id, user, db)
68
+ db_chapters = []
69
+ for data in chapters:
70
+ chapter = Chapter(
71
+ project_id=project_id,
72
+ chapter_number=data.chapter_number,
73
+ title=data.title,
74
+ source_text=data.source_text,
75
+ word_count=len(data.source_text.split()),
76
+ )
77
+ db.add(chapter)
78
+ db_chapters.append(chapter)
79
+ await db.commit()
80
+ for ch in db_chapters:
81
+ await db.refresh(ch)
82
+ return db_chapters
app/api/characters.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import APIRouter, Depends, HTTPException
2
+ from sqlalchemy import select
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.database import get_db
5
+ from app.models.user import User
6
+ from app.models.project import Project
7
+ from app.models.character import Character
8
+ from app.schemas.character import CharacterCreate, CharacterUpdate, CharacterResponse
9
+ from app.services.auth import get_current_user
10
+
11
+ router = APIRouter(prefix="/api/projects/{project_id}/characters", tags=["characters"])
12
+
13
+
14
+ async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
15
+ result = await db.execute(
16
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
17
+ )
18
+ project = result.scalar_one_or_none()
19
+ if not project:
20
+ raise HTTPException(status_code=404, detail="Project not found")
21
+ return project
22
+
23
+
24
+ @router.get("", response_model=list[CharacterResponse])
25
+ async def list_characters(
26
+ project_id: int,
27
+ user: User = Depends(get_current_user),
28
+ db: AsyncSession = Depends(get_db),
29
+ ):
30
+ await _get_user_project(project_id, user, db)
31
+ result = await db.execute(
32
+ select(Character)
33
+ .where(Character.project_id == project_id)
34
+ .order_by(Character.name)
35
+ )
36
+ return list(result.scalars().all())
37
+
38
+
39
+ @router.post("", response_model=CharacterResponse, status_code=201)
40
+ async def create_character(
41
+ project_id: int,
42
+ data: CharacterCreate,
43
+ user: User = Depends(get_current_user),
44
+ db: AsyncSession = Depends(get_db),
45
+ ):
46
+ await _get_user_project(project_id, user, db)
47
+ character = Character(project_id=project_id, **data.model_dump(exclude_none=True))
48
+ db.add(character)
49
+ await db.commit()
50
+ await db.refresh(character)
51
+ return character
52
+
53
+
54
+ @router.patch("/{character_id}", response_model=CharacterResponse)
55
+ async def update_character(
56
+ project_id: int,
57
+ character_id: int,
58
+ data: CharacterUpdate,
59
+ user: User = Depends(get_current_user),
60
+ db: AsyncSession = Depends(get_db),
61
+ ):
62
+ await _get_user_project(project_id, user, db)
63
+ result = await db.execute(
64
+ select(Character).where(Character.id == character_id, Character.project_id == project_id)
65
+ )
66
+ character = result.scalar_one_or_none()
67
+ if not character:
68
+ raise HTTPException(status_code=404, detail="Character not found")
69
+
70
+ for key, value in data.model_dump(exclude_none=True).items():
71
+ setattr(character, key, value)
72
+ await db.commit()
73
+ await db.refresh(character)
74
+ return character
app/api/episodes.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import APIRouter, Depends, HTTPException
2
+ from sqlalchemy import select
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.database import get_db
5
+ from app.models.user import User
6
+ from app.models.project import Project
7
+ from app.models.episode import Episode
8
+ from app.schemas.episode import EpisodeCreate, EpisodeResponse
9
+ from app.services.auth import get_current_user
10
+
11
+ router = APIRouter(prefix="/api/projects/{project_id}/episodes", tags=["episodes"])
12
+
13
+
14
+ async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
15
+ result = await db.execute(
16
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
17
+ )
18
+ project = result.scalar_one_or_none()
19
+ if not project:
20
+ raise HTTPException(status_code=404, detail="Project not found")
21
+ return project
22
+
23
+
24
+ @router.get("", response_model=list[EpisodeResponse])
25
+ async def list_episodes(
26
+ project_id: int,
27
+ user: User = Depends(get_current_user),
28
+ db: AsyncSession = Depends(get_db),
29
+ ):
30
+ await _get_user_project(project_id, user, db)
31
+ result = await db.execute(
32
+ select(Episode)
33
+ .where(Episode.project_id == project_id)
34
+ .order_by(Episode.episode_number)
35
+ )
36
+ return list(result.scalars().all())
37
+
38
+
39
+ @router.post("", response_model=EpisodeResponse, status_code=201)
40
+ async def create_episode(
41
+ project_id: int,
42
+ data: EpisodeCreate,
43
+ user: User = Depends(get_current_user),
44
+ db: AsyncSession = Depends(get_db),
45
+ ):
46
+ await _get_user_project(project_id, user, db)
47
+ episode = Episode(project_id=project_id, **data.model_dump(exclude_none=True))
48
+ db.add(episode)
49
+ await db.commit()
50
+ await db.refresh(episode)
51
+ return episode
52
+
53
+
54
+ @router.get("/{episode_id}", response_model=EpisodeResponse)
55
+ async def get_episode(
56
+ project_id: int,
57
+ episode_id: int,
58
+ user: User = Depends(get_current_user),
59
+ db: AsyncSession = Depends(get_db),
60
+ ):
61
+ await _get_user_project(project_id, user, db)
62
+ result = await db.execute(
63
+ select(Episode).where(Episode.id == episode_id, Episode.project_id == project_id)
64
+ )
65
+ episode = result.scalar_one_or_none()
66
+ if not episode:
67
+ raise HTTPException(status_code=404, detail="Episode not found")
68
+ return episode
app/api/generation.py ADDED
@@ -0,0 +1,241 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import asyncio
2
+ import json
3
+ import logging
4
+ from pathlib import Path
5
+ from fastapi import APIRouter, Depends, HTTPException
6
+ from fastapi.responses import StreamingResponse
7
+ from sqlalchemy import select
8
+ from sqlalchemy.ext.asyncio import AsyncSession
9
+ from app.database import get_db
10
+ from app.models.user import User
11
+ from app.models.project import Project
12
+ from app.models.generation_job import GenerationJob
13
+ from app.schemas.generation import GenerationStart, GenerationJobResponse
14
+ from app.services.auth import get_current_user
15
+ from app.pipeline.orchestrator import run_pipeline
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+ router = APIRouter(prefix="/api/projects/{project_id}", tags=["generation"])
20
+
21
+ # Keep references to running pipeline tasks so they don't get garbage-collected
22
+ _running_tasks: set[asyncio.Task] = set()
23
+
24
+
25
+ def _launch_pipeline(job_id: int, resume: bool = False, chapter_ids: list[int] | None = None):
26
+ """Launch pipeline as background task with error logging."""
27
+ async def _wrapper():
28
+ try:
29
+ await run_pipeline(job_id, resume=resume, chapter_ids=chapter_ids)
30
+ except Exception:
31
+ logger.exception("Pipeline failed for job %s", job_id)
32
+
33
+ task = asyncio.create_task(_wrapper())
34
+ _running_tasks.add(task)
35
+ task.add_done_callback(_running_tasks.discard)
36
+
37
+
38
+ # ── Asset endpoints ──────────────────────────────────────────────────
39
+
40
+ @router.get("/assets/images")
41
+ async def list_images(
42
+ project_id: int,
43
+ user: User = Depends(get_current_user),
44
+ db: AsyncSession = Depends(get_db),
45
+ ):
46
+ """List generated images for a project."""
47
+ result = await db.execute(
48
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
49
+ )
50
+ if not result.scalar_one_or_none():
51
+ raise HTTPException(status_code=404, detail="Project not found")
52
+
53
+ img_dir = Path("workdir/projects") / str(project_id) / "images"
54
+ if not img_dir.exists():
55
+ return []
56
+
57
+ files = sorted(f.name for f in img_dir.iterdir() if f.suffix.lower() in (".png", ".jpg", ".jpeg", ".webp"))
58
+ return files
59
+
60
+
61
+ # ── Generation endpoints ─────────────────────────────────────────────
62
+
63
+ @router.post("/generate", response_model=GenerationJobResponse, status_code=201)
64
+ async def start_generation(
65
+ project_id: int,
66
+ data: GenerationStart,
67
+ user: User = Depends(get_current_user),
68
+ db: AsyncSession = Depends(get_db),
69
+ ):
70
+ result = await db.execute(
71
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
72
+ )
73
+ project = result.scalar_one_or_none()
74
+ if not project:
75
+ raise HTTPException(status_code=404, detail="Project not found")
76
+
77
+ job = GenerationJob(
78
+ project_id=project_id,
79
+ user_id=user.id,
80
+ episode_id=data.episode_id,
81
+ chapter_ids_json=data.chapter_ids,
82
+ status="queued",
83
+ current_stage="ingest",
84
+ progress_pct=0.0,
85
+ )
86
+ db.add(job)
87
+ await db.commit()
88
+ await db.refresh(job)
89
+
90
+ # Launch pipeline in background with error handling
91
+ _launch_pipeline(job.id, chapter_ids=data.chapter_ids)
92
+
93
+ return job
94
+
95
+
96
+ @router.get("/generate/jobs", response_model=list[GenerationJobResponse])
97
+ async def list_jobs(
98
+ project_id: int,
99
+ user: User = Depends(get_current_user),
100
+ db: AsyncSession = Depends(get_db),
101
+ ):
102
+ result = await db.execute(
103
+ select(GenerationJob)
104
+ .where(GenerationJob.project_id == project_id, GenerationJob.user_id == user.id)
105
+ .order_by(GenerationJob.created_at.desc())
106
+ )
107
+ return list(result.scalars().all())
108
+
109
+
110
+ @router.get("/generate/jobs/{job_id}", response_model=GenerationJobResponse)
111
+ async def get_job(
112
+ project_id: int,
113
+ job_id: int,
114
+ user: User = Depends(get_current_user),
115
+ db: AsyncSession = Depends(get_db),
116
+ ):
117
+ result = await db.execute(
118
+ select(GenerationJob).where(
119
+ GenerationJob.id == job_id,
120
+ GenerationJob.project_id == project_id,
121
+ GenerationJob.user_id == user.id,
122
+ )
123
+ )
124
+ job = result.scalar_one_or_none()
125
+ if not job:
126
+ raise HTTPException(status_code=404, detail="Job not found")
127
+ return job
128
+
129
+
130
+ @router.get("/generate/jobs/{job_id}/stream")
131
+ async def stream_progress(
132
+ project_id: int,
133
+ job_id: int,
134
+ user: User = Depends(get_current_user),
135
+ db: AsyncSession = Depends(get_db),
136
+ ):
137
+ """SSE endpoint for real-time generation progress."""
138
+ result = await db.execute(
139
+ select(GenerationJob).where(
140
+ GenerationJob.id == job_id,
141
+ GenerationJob.project_id == project_id,
142
+ GenerationJob.user_id == user.id,
143
+ )
144
+ )
145
+ job = result.scalar_one_or_none()
146
+ if not job:
147
+ raise HTTPException(status_code=404, detail="Job not found")
148
+
149
+ from app.database import async_session as session_factory
150
+
151
+ async def event_stream():
152
+ last_data = None
153
+ while True:
154
+ try:
155
+ async with session_factory() as session:
156
+ result = await session.execute(
157
+ select(GenerationJob).where(GenerationJob.id == job_id)
158
+ )
159
+ current_job = result.scalar_one_or_none()
160
+ if not current_job:
161
+ yield f"data: {json.dumps({'status': 'not_found'})}\n\n"
162
+ break
163
+
164
+ data = {
165
+ "status": current_job.status or "queued",
166
+ "stage": current_job.current_stage or "ingest",
167
+ "progress": current_job.progress_pct or 0,
168
+ "detail": (current_job.progress_detail or "")[:200],
169
+ "error": (current_job.error_message or "")[:500] if current_job.status == "failed" else None,
170
+ }
171
+
172
+ # Always send (even if same) so frontend knows connection is alive
173
+ yield f"data: {json.dumps(data)}\n\n"
174
+ last_data = data
175
+
176
+ if current_job.status in ("completed", "failed", "cancelled"):
177
+ break
178
+ except Exception:
179
+ logger.exception("SSE stream error for job %s", job_id)
180
+ yield f"data: {json.dumps({'status': 'error', 'detail': 'Server error'})}\n\n"
181
+ break
182
+
183
+ await asyncio.sleep(1)
184
+
185
+ return StreamingResponse(
186
+ event_stream(),
187
+ media_type="text/event-stream",
188
+ headers={
189
+ "Cache-Control": "no-cache",
190
+ "Connection": "keep-alive",
191
+ "X-Accel-Buffering": "no",
192
+ },
193
+ )
194
+
195
+
196
+ @router.post("/generate/retry", response_model=GenerationJobResponse, status_code=201)
197
+ async def retry_generation(
198
+ project_id: int,
199
+ user: User = Depends(get_current_user),
200
+ db: AsyncSession = Depends(get_db),
201
+ ):
202
+ """Retry the last failed job for a project, resuming from where it left off."""
203
+ result = await db.execute(
204
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
205
+ )
206
+ project = result.scalar_one_or_none()
207
+ if not project:
208
+ raise HTTPException(status_code=404, detail="Project not found")
209
+
210
+ # Find the last failed job
211
+ result = await db.execute(
212
+ select(GenerationJob)
213
+ .where(
214
+ GenerationJob.project_id == project_id,
215
+ GenerationJob.user_id == user.id,
216
+ GenerationJob.status == "failed",
217
+ )
218
+ .order_by(GenerationJob.created_at.desc())
219
+ .limit(1)
220
+ )
221
+ failed_job = result.scalar_one_or_none()
222
+ if not failed_job:
223
+ raise HTTPException(status_code=404, detail="No failed job to retry")
224
+
225
+ # Create new job, carrying over chapter_ids from failed job
226
+ job = GenerationJob(
227
+ project_id=project_id,
228
+ user_id=user.id,
229
+ episode_id=failed_job.episode_id,
230
+ chapter_ids_json=failed_job.chapter_ids_json,
231
+ status="queued",
232
+ current_stage="ingest",
233
+ progress_pct=0.0,
234
+ )
235
+ db.add(job)
236
+ await db.commit()
237
+ await db.refresh(job)
238
+
239
+ _launch_pipeline(job.id, resume=True, chapter_ids=failed_job.chapter_ids_json)
240
+
241
+ return job
app/api/novels.py ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Novel search, URL resolution, and PDF upload endpoints."""
2
+ import os
3
+ import tempfile
4
+ from fastapi import APIRouter, Depends, HTTPException, UploadFile, File, Query
5
+ from pydantic import BaseModel
6
+ from app.services.auth import get_current_user
7
+ from app.models.user import User
8
+ from app.services.scraper import search_novels, resolve_novel_url, extract_text_from_pdf
9
+
10
+ router = APIRouter(prefix="/api/novels", tags=["novels"])
11
+
12
+
13
+ class SearchResult(BaseModel):
14
+ title: str
15
+ url: str
16
+ source: str
17
+
18
+
19
+ class NovelInfo(BaseModel):
20
+ title: str
21
+ author: str | None = None
22
+ cover_url: str | None = None
23
+ synopsis: str | None = None
24
+ chapters: list[dict] = []
25
+ source: str = ""
26
+
27
+
28
+ @router.get("/search", response_model=list[SearchResult])
29
+ async def search(
30
+ q: str = Query(..., min_length=2),
31
+ user: User = Depends(get_current_user),
32
+ ):
33
+ results = await search_novels(q)
34
+ return [SearchResult(title=r.title, url=r.url, source=r.source) for r in results]
35
+
36
+
37
+ @router.post("/resolve-url", response_model=NovelInfo)
38
+ async def resolve_url(
39
+ data: dict,
40
+ user: User = Depends(get_current_user),
41
+ ):
42
+ url = data.get("url", "").strip()
43
+ if not url:
44
+ raise HTTPException(status_code=400, detail="URL is required")
45
+
46
+ result = await resolve_novel_url(url)
47
+ if not result:
48
+ raise HTTPException(status_code=404, detail="Could not resolve novel from URL")
49
+
50
+ return NovelInfo(
51
+ title=result.title,
52
+ author=result.author,
53
+ cover_url=result.cover_url,
54
+ synopsis=result.synopsis,
55
+ chapters=result.chapters,
56
+ source=result.source,
57
+ )
58
+
59
+
60
+ @router.post("/upload-pdf", response_model=list[dict])
61
+ async def upload_pdf(
62
+ file: UploadFile = File(...),
63
+ user: User = Depends(get_current_user),
64
+ ):
65
+ if not file.filename or not file.filename.lower().endswith(".pdf"):
66
+ raise HTTPException(status_code=400, detail="Only PDF files are supported")
67
+
68
+ # Save to temp file
69
+ with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
70
+ content = await file.read()
71
+ tmp.write(content)
72
+ tmp_path = tmp.name
73
+
74
+ try:
75
+ chapters = await extract_text_from_pdf(tmp_path)
76
+ if not chapters:
77
+ raise HTTPException(status_code=422, detail="Could not extract text from PDF")
78
+ return chapters
79
+ finally:
80
+ os.unlink(tmp_path)
app/api/projects.py ADDED
@@ -0,0 +1,96 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from fastapi import APIRouter, Depends, HTTPException
2
+ from sqlalchemy import select, func
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.database import get_db
5
+ from app.models.user import User
6
+ from app.models.project import Project
7
+ from app.schemas.project import ProjectCreate, ProjectUpdate, ProjectResponse, ProjectListResponse
8
+ from app.services.auth import get_current_user
9
+
10
+ router = APIRouter(prefix="/api/projects", tags=["projects"])
11
+
12
+
13
+ @router.get("", response_model=ProjectListResponse)
14
+ async def list_projects(
15
+ skip: int = 0,
16
+ limit: int = 20,
17
+ user: User = Depends(get_current_user),
18
+ db: AsyncSession = Depends(get_db),
19
+ ):
20
+ result = await db.execute(
21
+ select(Project)
22
+ .where(Project.user_id == user.id)
23
+ .order_by(Project.updated_at.desc())
24
+ .offset(skip).limit(limit)
25
+ )
26
+ projects = list(result.scalars().all())
27
+ count_result = await db.execute(
28
+ select(func.count()).select_from(Project).where(Project.user_id == user.id)
29
+ )
30
+ total = count_result.scalar() or 0
31
+ return ProjectListResponse(projects=projects, total=total)
32
+
33
+
34
+ @router.post("", response_model=ProjectResponse, status_code=201)
35
+ async def create_project(
36
+ data: ProjectCreate,
37
+ user: User = Depends(get_current_user),
38
+ db: AsyncSession = Depends(get_db),
39
+ ):
40
+ project = Project(user_id=user.id, **data.model_dump(exclude_none=True))
41
+ db.add(project)
42
+ await db.commit()
43
+ await db.refresh(project)
44
+ return project
45
+
46
+
47
+ @router.get("/{project_id}", response_model=ProjectResponse)
48
+ async def get_project(
49
+ project_id: int,
50
+ user: User = Depends(get_current_user),
51
+ db: AsyncSession = Depends(get_db),
52
+ ):
53
+ result = await db.execute(
54
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
55
+ )
56
+ project = result.scalar_one_or_none()
57
+ if not project:
58
+ raise HTTPException(status_code=404, detail="Project not found")
59
+ return project
60
+
61
+
62
+ @router.patch("/{project_id}", response_model=ProjectResponse)
63
+ async def update_project(
64
+ project_id: int,
65
+ data: ProjectUpdate,
66
+ user: User = Depends(get_current_user),
67
+ db: AsyncSession = Depends(get_db),
68
+ ):
69
+ result = await db.execute(
70
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
71
+ )
72
+ project = result.scalar_one_or_none()
73
+ if not project:
74
+ raise HTTPException(status_code=404, detail="Project not found")
75
+
76
+ for key, value in data.model_dump(exclude_none=True).items():
77
+ setattr(project, key, value)
78
+ await db.commit()
79
+ await db.refresh(project)
80
+ return project
81
+
82
+
83
+ @router.delete("/{project_id}", status_code=204)
84
+ async def delete_project(
85
+ project_id: int,
86
+ user: User = Depends(get_current_user),
87
+ db: AsyncSession = Depends(get_db),
88
+ ):
89
+ result = await db.execute(
90
+ select(Project).where(Project.id == project_id, Project.user_id == user.id)
91
+ )
92
+ project = result.scalar_one_or_none()
93
+ if not project:
94
+ raise HTTPException(status_code=404, detail="Project not found")
95
+ await db.delete(project)
96
+ await db.commit()
app/config.py ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic_settings import BaseSettings
2
+ from functools import lru_cache
3
+ import json
4
+
5
+
6
+ class Settings(BaseSettings):
7
+ # Auth
8
+ jwt_secret_key: str = "change-this-to-a-random-secret-key-in-production"
9
+ jwt_algorithm: str = "HS256"
10
+ access_token_expire_minutes: int = 30
11
+ refresh_token_expire_days: int = 7
12
+
13
+ # Database
14
+ database_url: str = "sqlite+aiosqlite:///./workdir/anime.db"
15
+
16
+ # API Keys
17
+ hf_api_token: str = ""
18
+ nvidia_api_key: str = ""
19
+ gemini_api_key: str = ""
20
+ openrouter_api_key: str = ""
21
+ pollinations_api_key: str = ""
22
+
23
+ # Server
24
+ port: int = 8000
25
+
26
+ # CORS
27
+ cors_origins: str = '["http://localhost:3000"]'
28
+
29
+ @property
30
+ def cors_origin_list(self) -> list[str]:
31
+ return json.loads(self.cors_origins)
32
+
33
+ model_config = {"env_file": ".env", "env_file_encoding": "utf-8"}
34
+
35
+
36
+ @lru_cache
37
+ def get_settings() -> Settings:
38
+ return Settings()
app/database.py ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy.ext.asyncio import create_async_engine, async_sessionmaker, AsyncSession
2
+ from sqlalchemy.orm import DeclarativeBase
3
+ from app.config import get_settings
4
+
5
+ settings = get_settings()
6
+
7
+ engine = create_async_engine(
8
+ settings.database_url,
9
+ echo=False,
10
+ connect_args={"timeout": 30},
11
+ )
12
+ async_session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
13
+
14
+
15
+ class Base(DeclarativeBase):
16
+ pass
17
+
18
+
19
+ async def get_db():
20
+ async with async_session() as session:
21
+ try:
22
+ yield session
23
+ finally:
24
+ await session.close()
25
+
26
+
27
+ async def init_db():
28
+ async with engine.begin() as conn:
29
+ await conn.run_sync(Base.metadata.create_all)
30
+
31
+ # Migrate existing DBs: add chapter_ids_json to generation_jobs
32
+ async with engine.begin() as conn:
33
+ try:
34
+ await conn.execute(
35
+ __import__("sqlalchemy").text(
36
+ "ALTER TABLE generation_jobs ADD COLUMN chapter_ids_json TEXT"
37
+ )
38
+ )
39
+ except Exception:
40
+ pass # Column already exists
41
+
42
+ # Migrate existing DBs: add recap_summary to episodes
43
+ async with engine.begin() as conn:
44
+ try:
45
+ await conn.execute(
46
+ __import__("sqlalchemy").text(
47
+ "ALTER TABLE episodes ADD COLUMN recap_summary TEXT"
48
+ )
49
+ )
50
+ except Exception:
51
+ pass # Column already exists
app/main.py ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from contextlib import asynccontextmanager
2
+ from fastapi import FastAPI
3
+ from fastapi.middleware.cors import CORSMiddleware
4
+ from fastapi.staticfiles import StaticFiles
5
+ from app.config import get_settings
6
+ from app.database import init_db
7
+ from app.api import auth, projects, chapters, characters, episodes, generation, novels
8
+ import os
9
+
10
+ settings = get_settings()
11
+
12
+
13
+ @asynccontextmanager
14
+ async def lifespan(app: FastAPI):
15
+ # Startup: create tables
16
+ os.makedirs("workdir", exist_ok=True)
17
+ await init_db()
18
+ yield
19
+ # Shutdown
20
+
21
+
22
+ app = FastAPI(
23
+ title="Anime Generator API",
24
+ version="0.1.0",
25
+ description="Web novel to anime video converter",
26
+ lifespan=lifespan,
27
+ )
28
+
29
+ # CORS
30
+ app.add_middleware(
31
+ CORSMiddleware,
32
+ allow_origins=settings.cors_origin_list,
33
+ allow_credentials=True,
34
+ allow_methods=["*"],
35
+ allow_headers=["*"],
36
+ )
37
+
38
+ # Serve generated media files
39
+ os.makedirs("workdir", exist_ok=True)
40
+ app.mount("/media", StaticFiles(directory="workdir"), name="media")
41
+
42
+ # Routes
43
+ app.include_router(auth.router)
44
+ app.include_router(projects.router)
45
+ app.include_router(chapters.router)
46
+ app.include_router(characters.router)
47
+ app.include_router(episodes.router)
48
+ app.include_router(generation.router)
49
+ app.include_router(novels.router)
50
+
51
+
52
+ @app.get("/api/health")
53
+ async def health():
54
+ return {"status": "ok", "version": "0.1.0"}
app/models/__init__.py ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from app.models.user import User
2
+ from app.models.project import Project
3
+ from app.models.chapter import Chapter
4
+ from app.models.character import Character
5
+ from app.models.episode import Episode
6
+ from app.models.generation_job import GenerationJob
7
+ from app.models.critic_result import CriticResult
8
+ from app.models.api_usage import ApiUsage
9
+
10
+ __all__ = [
11
+ "User", "Project", "Chapter", "Character",
12
+ "Episode", "GenerationJob", "CriticResult", "ApiUsage",
13
+ ]
app/models/api_usage.py ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Date, DateTime, func
2
+ from app.database import Base
3
+
4
+
5
+ class ApiUsage(Base):
6
+ __tablename__ = "api_usage"
7
+
8
+ id = Column(Integer, primary_key=True, autoincrement=True)
9
+ service = Column(String(100), nullable=False) # huggingface, nvidia, gemini, edge_tts
10
+ endpoint = Column(String(500), nullable=True)
11
+ request_count = Column(Integer, default=1)
12
+ date = Column(Date, nullable=False)
13
+ created_at = Column(DateTime, server_default=func.now())
app/models/chapter.py ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
2
+ from app.database import Base
3
+ from sqlalchemy.orm import relationship
4
+
5
+
6
+ class Chapter(Base):
7
+ __tablename__ = "chapters"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
11
+ chapter_number = Column(Integer, nullable=False)
12
+ title = Column(String(500), nullable=True)
13
+ source_text = Column(Text, nullable=False)
14
+ word_count = Column(Integer, default=0)
15
+ status = Column(String(50), default="raw") # raw, parsed, processed
16
+ parsed_json = Column(JSON, nullable=True) # storyboard output
17
+ created_at = Column(DateTime, server_default=func.now())
18
+
19
+ project = relationship("Project", back_populates="chapters")
app/models/character.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
2
+ from app.database import Base
3
+ from sqlalchemy.orm import relationship
4
+
5
+
6
+ class Character(Base):
7
+ __tablename__ = "characters"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
11
+ name = Column(String(200), nullable=False)
12
+ aliases_json = Column(JSON, default=list) # ["Su Ling'er", "Ling'er"]
13
+ role = Column(String(50), default="supporting") # protagonist, antagonist, supporting, minor
14
+ gender = Column(String(20), nullable=True)
15
+ age_category = Column(String(20), nullable=True) # child, teen, young_adult, adult, elder
16
+ visual_prompt = Column(Text, nullable=True) # 50-80 word prompt-ready description
17
+ reference_image_path = Column(String(500), nullable=True)
18
+ portrait_url = Column(String(500), nullable=True) # Pollinations media URL for portrait reference
19
+ reference_seed = Column(Integer, nullable=True) # seed pinning for consistency
20
+ voice_config_json = Column(JSON, nullable=True) # {voice_name, rate, pitch, style}
21
+ voice_reference_path = Column(String(500), nullable=True) # path to .safetensors voice state (Pocket TTS)
22
+ voice_source_json = Column(JSON, nullable=True) # {"source": "animevox", "character": "...", "anime": "..."}
23
+ wiki_data_json = Column(JSON, nullable=True) # cached fandom wiki data
24
+ created_at = Column(DateTime, server_default=func.now())
25
+ updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
26
+
27
+ project = relationship("Project", back_populates="characters")
app/models/critic_result.py ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, Float, DateTime, ForeignKey, JSON, func
2
+ from app.database import Base
3
+ from sqlalchemy.orm import relationship
4
+
5
+
6
+ class CriticResult(Base):
7
+ __tablename__ = "critic_results"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ episode_id = Column(Integer, ForeignKey("episodes.id"), nullable=False)
11
+ visual_consistency_score = Column(Float, nullable=True)
12
+ pacing_score = Column(Float, nullable=True)
13
+ audio_sync_score = Column(Float, nullable=True)
14
+ overall_score = Column(Float, nullable=True)
15
+ feedback_json = Column(JSON, nullable=True)
16
+ created_at = Column(DateTime, server_default=func.now())
17
+
18
+ episode = relationship("Episode", back_populates="critic_results")
app/models/episode.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Float, DateTime, ForeignKey, JSON, Text, func
2
+ from app.database import Base
3
+ from sqlalchemy.orm import relationship
4
+
5
+
6
+ class Episode(Base):
7
+ __tablename__ = "episodes"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
11
+ episode_number = Column(Integer, nullable=False)
12
+ title = Column(String(500), nullable=True)
13
+ chapter_ids_json = Column(JSON, default=list)
14
+ storyboard_json = Column(JSON, nullable=True) # full storyboard data
15
+ duration_seconds = Column(Float, nullable=True)
16
+ video_path = Column(String(500), nullable=True)
17
+ thumbnail_path = Column(String(500), nullable=True)
18
+ recap_summary = Column(Text, nullable=True) # narrative recap for continuity
19
+ status = Column(String(50), default="pending") # pending, generating, completed, error
20
+ created_at = Column(DateTime, server_default=func.now())
21
+ updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
22
+
23
+ project = relationship("Project", back_populates="episodes")
24
+ generation_jobs = relationship("GenerationJob", back_populates="episode", cascade="all, delete-orphan")
25
+ critic_results = relationship("CriticResult", back_populates="episode", cascade="all, delete-orphan")
app/models/generation_job.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Float, Text, DateTime, ForeignKey, JSON, func
2
+ from app.database import Base
3
+ from sqlalchemy.orm import relationship
4
+
5
+
6
+ class GenerationJob(Base):
7
+ __tablename__ = "generation_jobs"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
11
+ user_id = Column(Integer, ForeignKey("users.id"), nullable=False)
12
+ episode_id = Column(Integer, ForeignKey("episodes.id"), nullable=True)
13
+ chapter_ids_json = Column(JSON, nullable=True)
14
+ status = Column(String(50), default="queued") # queued, running, completed, failed, cancelled
15
+ current_stage = Column(String(50), nullable=True) # ingest, characters, scene_parse, image_gen, tts, animation, assembly, critic
16
+ progress_pct = Column(Float, default=0.0)
17
+ progress_detail = Column(Text, nullable=True)
18
+ error_message = Column(Text, nullable=True)
19
+ started_at = Column(DateTime, nullable=True)
20
+ completed_at = Column(DateTime, nullable=True)
21
+ created_at = Column(DateTime, server_default=func.now())
22
+
23
+ project = relationship("Project", back_populates="generation_jobs")
24
+ user = relationship("User", back_populates="generation_jobs")
25
+ episode = relationship("Episode", back_populates="generation_jobs")
app/models/project.py ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
2
+ from sqlalchemy.orm import relationship
3
+ from app.database import Base
4
+
5
+
6
+ class Project(Base):
7
+ __tablename__ = "projects"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ user_id = Column(Integer, ForeignKey("users.id"), nullable=False)
11
+ title = Column(String(500), nullable=False)
12
+ novel_url = Column(String(2000), nullable=True)
13
+ cover_image_url = Column(String(2000), nullable=True)
14
+ synopsis = Column(Text, nullable=True)
15
+ settings_json = Column(JSON, default=dict) # art style, episode length, etc.
16
+ status = Column(String(50), default="created") # created, processing, completed, error
17
+ created_at = Column(DateTime, server_default=func.now())
18
+ updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
19
+
20
+ user = relationship("User", back_populates="projects")
21
+ chapters = relationship("Chapter", back_populates="project", cascade="all, delete-orphan")
22
+ characters = relationship("Character", back_populates="project", cascade="all, delete-orphan")
23
+ episodes = relationship("Episode", back_populates="project", cascade="all, delete-orphan")
24
+ generation_jobs = relationship("GenerationJob", back_populates="project", cascade="all, delete-orphan")
app/models/user.py ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import Column, Integer, String, DateTime, func
2
+ from sqlalchemy.orm import relationship
3
+ from app.database import Base
4
+
5
+
6
+ class User(Base):
7
+ __tablename__ = "users"
8
+
9
+ id = Column(Integer, primary_key=True, autoincrement=True)
10
+ email = Column(String(255), unique=True, nullable=False, index=True)
11
+ username = Column(String(100), unique=True, nullable=False, index=True)
12
+ password_hash = Column(String(255), nullable=False)
13
+ role = Column(String(20), default="user") # user, admin
14
+ created_at = Column(DateTime, server_default=func.now())
15
+ updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
16
+
17
+ projects = relationship("Project", back_populates="user", cascade="all, delete-orphan")
18
+ generation_jobs = relationship("GenerationJob", back_populates="user", cascade="all, delete-orphan")
app/pipeline/__init__.py ADDED
File without changes
app/pipeline/orchestrator.py ADDED
@@ -0,0 +1,435 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Pipeline Orchestrator — runs all stages in sequence.
2
+
3
+ Stages:
4
+ 1. Ingest — load chapters from DB
5
+ 2. Characters — extract + enrich characters
6
+ 2b. Portraits — generate reference portraits (Pollinations)
7
+ 2c. Voice Enroll — match characters to AnimeVox voices (Pocket TTS)
8
+ 3. Scene Parse — Gemini → e-konte storyboard JSON
9
+ 4. Image Gen — generate images for all cuts (Pollinations + fallback)
10
+ 5. TTS — generate voice clips (Pocket TTS for main chars, Edge TTS for narration/minor)
11
+ 6. Video Gen — AI video clips (Pollinations grok-video) with Ken Burns fallback
12
+ 7. Assembly — concat clips, scene transitions, subtitles, BGM
13
+ 8. Critic — QA review
14
+ """
15
+ import asyncio
16
+ import json
17
+ import logging
18
+ from datetime import datetime, timezone
19
+ from pathlib import Path
20
+ from sqlalchemy import select
21
+ from sqlalchemy.ext.asyncio import AsyncSession
22
+ from app.database import async_session
23
+ from app.models.project import Project
24
+ from app.models.character import Character
25
+ from app.models.episode import Episode
26
+ from app.models.generation_job import GenerationJob
27
+ from app.models.critic_result import CriticResult
28
+ from app.pipeline.stage_1_ingest import run_ingest
29
+ from app.pipeline.stage_2_characters import run_character_extraction
30
+ from app.pipeline.stage_2b_portraits import run_portrait_generation
31
+ from app.pipeline.stage_2c_voice_enroll import run_voice_enrollment
32
+ from app.pipeline.stage_3_scene_parse import run_scene_parse
33
+ from app.pipeline.stage_4_image_gen import run_image_generation
34
+ from app.pipeline.stage_5_tts import run_tts_generation
35
+ from app.pipeline.stage_6_video_gen import run_video_generation
36
+ from app.pipeline.stage_6_animation import run_animation as run_ken_burns
37
+ from app.pipeline.stage_7_assembly import run_assembly
38
+ from app.pipeline.stage_8_critic import run_critic
39
+ from app.pipeline.recap_generator import generate_episode_recap, build_continuity_context
40
+
41
+ logger = logging.getLogger(__name__)
42
+
43
+ STAGES = [
44
+ ("ingest", 3),
45
+ ("characters", 7),
46
+ ("portraits", 8),
47
+ ("voice_enroll", 4),
48
+ ("scene_parse", 8),
49
+ ("image_gen", 28),
50
+ ("tts", 12),
51
+ ("video_gen", 23),
52
+ ("assembly", 12),
53
+ ("critic", 5),
54
+ ]
55
+
56
+
57
+ async def _update_job(job_id: int, **kwargs):
58
+ """Update a generation job in the database with SQLite retry on lock."""
59
+ for attempt in range(3):
60
+ try:
61
+ async with async_session() as db:
62
+ result = await db.execute(
63
+ select(GenerationJob).where(GenerationJob.id == job_id)
64
+ )
65
+ job = result.scalar_one_or_none()
66
+ if job:
67
+ for k, v in kwargs.items():
68
+ setattr(job, k, v)
69
+ await db.commit()
70
+ return
71
+ except Exception as e:
72
+ if "database is locked" in str(e) and attempt < 2:
73
+ await asyncio.sleep(0.5 * (attempt + 1))
74
+ continue
75
+ if attempt == 2:
76
+ logger.warning("_update_job failed after retries: %s", e)
77
+ else:
78
+ raise
79
+
80
+
81
+ def _make_progress_callback(job_id: int, stage_name: str, stage_base_pct: float, stage_weight: float):
82
+ """Create a progress callback for a specific stage."""
83
+ async def callback(detail: str, stage_pct: float):
84
+ overall_pct = stage_base_pct + (stage_pct / 100.0) * stage_weight
85
+ await _update_job(
86
+ job_id,
87
+ current_stage=stage_name,
88
+ progress_pct=min(overall_pct, 99.0),
89
+ progress_detail=detail,
90
+ )
91
+ return callback
92
+
93
+
94
+ def _check_pollinations_available() -> bool:
95
+ """Check if Pollinations API key is configured."""
96
+ from app.config import get_settings
97
+ settings = get_settings()
98
+ return bool(settings.pollinations_api_key)
99
+
100
+
101
+ async def run_pipeline(job_id: int, resume: bool = False, chapter_ids: list[int] | None = None):
102
+ """Run the complete generation pipeline for a job.
103
+
104
+ If resume=True, skip stages whose outputs already exist on disk/DB.
105
+ If chapter_ids is provided, only ingest those specific chapters.
106
+ """
107
+ async with async_session() as db:
108
+ result = await db.execute(
109
+ select(GenerationJob).where(GenerationJob.id == job_id)
110
+ )
111
+ job = result.scalar_one_or_none()
112
+ if not job:
113
+ return
114
+
115
+ project_id = job.project_id
116
+ # Use chapter_ids from param, falling back to what's stored on the job
117
+ if chapter_ids is None:
118
+ chapter_ids = job.chapter_ids_json
119
+
120
+ # Get project
121
+ proj_result = await db.execute(
122
+ select(Project).where(Project.id == project_id)
123
+ )
124
+ project = proj_result.scalar_one_or_none()
125
+ if not project:
126
+ await _update_job(job_id, status="failed", error_message="Project not found")
127
+ return
128
+
129
+ # Compute episode number early (before any stage runs)
130
+ from sqlalchemy import func as sa_func
131
+ count_result = await db.execute(
132
+ select(sa_func.count()).select_from(Episode).where(Episode.project_id == project_id)
133
+ )
134
+ ep_number = (count_result.scalar() or 0) + 1
135
+
136
+ project_dir = str(Path("workdir/projects") / str(project_id))
137
+ Path(project_dir).mkdir(parents=True, exist_ok=True)
138
+
139
+ # Per-episode directory for episode-specific assets (storyboard, images, audio, clips)
140
+ episode_dir = str(Path(project_dir) / f"ep_{ep_number}")
141
+ Path(episode_dir).mkdir(parents=True, exist_ok=True)
142
+
143
+ use_pollinations = _check_pollinations_available()
144
+
145
+ try:
146
+ logger.info("Pipeline starting for job %s, project %s (resume=%s, pollinations=%s)",
147
+ job_id, project_id, resume, use_pollinations)
148
+ await _update_job(
149
+ job_id,
150
+ status="running",
151
+ started_at=datetime.now(timezone.utc),
152
+ progress_detail="Starting pipeline..." if not resume else "Resuming pipeline...",
153
+ )
154
+
155
+ # Calculate stage base percentages
156
+ total_weight = sum(w for _, w in STAGES)
157
+ cumulative = 0.0
158
+
159
+ # ─── Stage 1: Ingest ───
160
+ logger.info("[Job %s] Stage 1: Ingest (chapter_ids=%s)", job_id, chapter_ids)
161
+ stage_weight = (STAGES[0][1] / total_weight) * 100
162
+ on_progress = _make_progress_callback(job_id, "ingest", cumulative, stage_weight)
163
+ async with async_session() as db:
164
+ chapters = await run_ingest(project_id, db, on_progress, chapter_ids=chapter_ids)
165
+ cumulative += stage_weight
166
+ logger.info("[Job %s] Stage 1 complete: %d chapters", job_id, len(chapters))
167
+
168
+ # ─── Stage 2: Characters ───
169
+ logger.info("[Job %s] Stage 2: Characters", job_id)
170
+ stage_weight = (STAGES[1][1] / total_weight) * 100
171
+ on_progress = _make_progress_callback(job_id, "characters", cumulative, stage_weight)
172
+
173
+ if resume and not chapter_ids:
174
+ async with async_session() as db:
175
+ char_result = await db.execute(
176
+ select(Character).where(Character.project_id == project_id)
177
+ )
178
+ existing_chars = list(char_result.scalars().all())
179
+
180
+ if existing_chars:
181
+ characters = existing_chars
182
+ logger.info("[Job %s] Stage 2 skipped (resume): %d characters exist", job_id, len(characters))
183
+ await on_progress(f"Reusing {len(characters)} existing characters", 100)
184
+ else:
185
+ async with async_session() as db:
186
+ characters = await run_character_extraction(
187
+ project_id, project.title, chapters, db, on_progress
188
+ )
189
+ else:
190
+ async with async_session() as db:
191
+ characters = await run_character_extraction(
192
+ project_id, project.title, chapters, db, on_progress
193
+ )
194
+ cumulative += stage_weight
195
+ logger.info("[Job %s] Stage 2 complete: %d characters", job_id, len(characters))
196
+
197
+ # ─── Stage 2b: Portraits (Pollinations only) ───
198
+ if use_pollinations:
199
+ logger.info("[Job %s] Stage 2b: Portraits", job_id)
200
+ stage_weight = (STAGES[2][1] / total_weight) * 100
201
+ on_progress = _make_progress_callback(job_id, "portraits", cumulative, stage_weight)
202
+
203
+ # Skip if all prominent characters already have portraits
204
+ needs_portraits = any(
205
+ c.role in ("protagonist", "antagonist", "supporting") and not c.portrait_url
206
+ for c in characters
207
+ )
208
+
209
+ if needs_portraits or not resume:
210
+ async with async_session() as db:
211
+ # Re-attach characters to this session
212
+ char_result = await db.execute(
213
+ select(Character).where(Character.project_id == project_id)
214
+ )
215
+ characters = list(char_result.scalars().all())
216
+ characters = await run_portrait_generation(
217
+ project_id, characters, project_dir, db, on_progress
218
+ )
219
+ else:
220
+ await on_progress("All portraits exist", 100)
221
+
222
+ cumulative += stage_weight
223
+ portrait_count = sum(1 for c in characters if c.portrait_url)
224
+ logger.info("[Job %s] Stage 2b complete: %d portraits", job_id, portrait_count)
225
+ else:
226
+ cumulative += (STAGES[2][1] / total_weight) * 100
227
+ logger.info("[Job %s] Stage 2b skipped (no Pollinations API key)", job_id)
228
+
229
+ # ─── Stage 2c: Voice Enrollment ───
230
+ logger.info("[Job %s] Stage 2c: Voice Enrollment", job_id)
231
+ stage_weight = (STAGES[3][1] / total_weight) * 100
232
+ on_progress = _make_progress_callback(job_id, "voice_enroll", cumulative, stage_weight)
233
+
234
+ # Skip if all prominent characters already have voice references
235
+ needs_enrollment = any(
236
+ c.role in ("protagonist", "antagonist", "supporting") and not c.voice_reference_path
237
+ for c in characters
238
+ )
239
+
240
+ if needs_enrollment or not resume:
241
+ async with async_session() as db:
242
+ char_result = await db.execute(
243
+ select(Character).where(Character.project_id == project_id)
244
+ )
245
+ characters = list(char_result.scalars().all())
246
+ characters = await run_voice_enrollment(
247
+ project_id, characters, project_dir, db, on_progress
248
+ )
249
+ else:
250
+ await on_progress("All voices already enrolled", 100)
251
+
252
+ cumulative += stage_weight
253
+ voice_count = sum(1 for c in characters if c.voice_reference_path)
254
+ logger.info("[Job %s] Stage 2c complete: %d voice enrollments", job_id, voice_count)
255
+
256
+ # ─── Narrative Continuity (pre-Stage 3) ───
257
+ continuity_context = None
258
+ if ep_number > 1:
259
+ logger.info("[Job %s] Building narrative continuity for Episode %d", job_id, ep_number)
260
+ try:
261
+ continuity_context = await build_continuity_context(project_id, ep_number)
262
+ if continuity_context:
263
+ logger.info("[Job %s] Continuity context built (%d chars)", job_id, len(continuity_context))
264
+ else:
265
+ logger.info("[Job %s] No prior episodes with recaps found", job_id)
266
+ except Exception as e:
267
+ logger.warning("[Job %s] Failed to build continuity context: %s", job_id, e)
268
+
269
+ # ─── Stage 3: Scene Parse ───
270
+ logger.info("[Job %s] Stage 3: Scene Parse", job_id)
271
+ stage_weight = (STAGES[4][1] / total_weight) * 100
272
+ on_progress = _make_progress_callback(job_id, "scene_parse", cumulative, stage_weight)
273
+
274
+ storyboard_path = Path(episode_dir) / "storyboard.json"
275
+
276
+ if resume and storyboard_path.exists():
277
+ with open(storyboard_path, "r") as f:
278
+ storyboard = json.load(f)
279
+ logger.info("[Job %s] Stage 3 skipped (resume): loaded storyboard from disk", job_id)
280
+ await on_progress("Loaded existing storyboard", 100)
281
+ else:
282
+ storyboard = await run_scene_parse(chapters, characters, on_progress, continuity_context=continuity_context)
283
+ with open(storyboard_path, "w") as f:
284
+ json.dump(storyboard, f)
285
+ logger.info("[Job %s] Stage 3: storyboard saved to disk", job_id)
286
+
287
+ cumulative += stage_weight
288
+ scene_count = len(storyboard.get("scenes", []))
289
+ cut_count = sum(
290
+ len(s.get("cuts", s.get("shots", [])))
291
+ for s in storyboard.get("scenes", [])
292
+ )
293
+ logger.info("[Job %s] Stage 3 complete: %d scenes, %d cuts", job_id, scene_count, cut_count)
294
+
295
+ # ─── Stage 4 + 5: Image Gen + TTS (parallel) ───
296
+ logger.info("[Job %s] Stage 4+5: Image Gen + TTS (parallel)", job_id)
297
+ img_weight = (STAGES[5][1] / total_weight) * 100
298
+ tts_weight = (STAGES[6][1] / total_weight) * 100
299
+ img_progress = _make_progress_callback(job_id, "image_gen", cumulative, img_weight)
300
+ tts_progress = _make_progress_callback(job_id, "tts", cumulative + img_weight, tts_weight)
301
+
302
+ storyboard_img, storyboard_tts = await asyncio.gather(
303
+ run_image_generation(storyboard, characters, episode_dir, img_progress, resume=resume),
304
+ run_tts_generation(storyboard, characters, episode_dir, tts_progress, resume=resume),
305
+ )
306
+
307
+ # Merge TTS data into storyboard (image gen already updated in-place)
308
+ for scene in storyboard.get("scenes", []):
309
+ for cut in scene.get("cuts", scene.get("shots", [])):
310
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
311
+ for tts_scene in storyboard_tts.get("scenes", []):
312
+ for tts_cut in tts_scene.get("cuts", tts_scene.get("shots", [])):
313
+ tts_cut_id = tts_cut.get("cut_id") or tts_cut.get("shot_id")
314
+ if tts_cut_id == cut_id:
315
+ cut["tts_path"] = tts_cut.get("tts_path")
316
+ cut["word_timestamps"] = tts_cut.get("word_timestamps")
317
+ if tts_cut.get("actual_duration_sec"):
318
+ cut["actual_duration_sec"] = tts_cut["actual_duration_sec"]
319
+ cut["duration_sec"] = max(
320
+ tts_cut["actual_duration_sec"] + 0.3,
321
+ cut.get("duration_sec", 2.0)
322
+ )
323
+
324
+ cumulative += img_weight + tts_weight
325
+ logger.info("[Job %s] Stage 4+5 complete", job_id)
326
+
327
+ # Persist updated storyboard (now includes image_path, tts_path)
328
+ with open(Path(episode_dir) / "storyboard.json", "w") as f:
329
+ json.dump(storyboard, f)
330
+
331
+ # ─── Stage 6: Video Generation (or Ken Burns fallback) ───
332
+ logger.info("[Job %s] Stage 6: Video Generation", job_id)
333
+ img_count = sum(
334
+ 1 for sc in storyboard.get("scenes", [])
335
+ for cut in sc.get("cuts", sc.get("shots", []))
336
+ if cut.get("image_path")
337
+ )
338
+ logger.info("[Job %s] Pre-video: %d cuts have image_path", job_id, img_count)
339
+ stage_weight = (STAGES[7][1] / total_weight) * 100
340
+ on_progress = _make_progress_callback(job_id, "video_gen", cumulative, stage_weight)
341
+
342
+ if use_pollinations:
343
+ # Try AI video generation with Ken Burns fallback built-in
344
+ storyboard = await run_video_generation(storyboard, episode_dir, on_progress)
345
+ else:
346
+ # Pure Ken Burns animation (legacy path)
347
+ storyboard = await run_ken_burns(storyboard, episode_dir, on_progress)
348
+
349
+ cumulative += stage_weight
350
+
351
+ # Persist storyboard after video gen (includes clip_path + _video_error for debugging)
352
+ with open(Path(episode_dir) / "storyboard.json", "w") as f:
353
+ json.dump(storyboard, f)
354
+
355
+ clip_count = sum(
356
+ 1 for sc in storyboard.get("scenes", [])
357
+ for cut in sc.get("cuts", sc.get("shots", []))
358
+ if cut.get("clip_path") and Path(cut["clip_path"]).exists()
359
+ )
360
+ logger.info("[Job %s] Stage 6 complete: %d/%d clips produced", job_id, clip_count, cut_count)
361
+
362
+ # ─── Stage 7: Assembly ───
363
+ logger.info("[Job %s] Stage 7: Assembly", job_id)
364
+ stage_weight = (STAGES[8][1] / total_weight) * 100
365
+ on_progress = _make_progress_callback(job_id, "assembly", cumulative, stage_weight)
366
+
367
+ assembly_result = await run_assembly(storyboard, episode_dir, ep_number, on_progress, output_dir=project_dir)
368
+ cumulative += stage_weight
369
+ logger.info("[Job %s] Stage 7 complete: %s", job_id, assembly_result.get("video_path", "?"))
370
+
371
+ # ─── Stage 8: Critic ───
372
+ logger.info("[Job %s] Stage 8: Critic", job_id)
373
+ stage_weight = (STAGES[9][1] / total_weight) * 100
374
+ on_progress = _make_progress_callback(job_id, "critic", cumulative, stage_weight)
375
+ critic_result = await run_critic(storyboard, episode_dir, on_progress)
376
+
377
+ # Save episode and critic result to DB
378
+ async with async_session() as db:
379
+ episode = Episode(
380
+ project_id=project_id,
381
+ episode_number=ep_number,
382
+ title=storyboard.get("episode_title", f"Episode {ep_number}"),
383
+ chapter_ids_json=[ch.id for ch in chapters],
384
+ storyboard_json=storyboard,
385
+ duration_seconds=assembly_result["duration_seconds"],
386
+ video_path=assembly_result["video_path"],
387
+ status="completed",
388
+ )
389
+ db.add(episode)
390
+ await db.commit()
391
+ await db.refresh(episode)
392
+
393
+ critic = CriticResult(
394
+ episode_id=episode.id,
395
+ visual_consistency_score=critic_result.get("visual_consistency_score"),
396
+ pacing_score=critic_result.get("pacing_score"),
397
+ audio_sync_score=critic_result.get("audio_sync_score"),
398
+ overall_score=critic_result.get("overall_score"),
399
+ feedback_json=critic_result.get("feedback"),
400
+ )
401
+ db.add(critic)
402
+ await db.commit()
403
+
404
+ # Generate and save episode recap for future continuity
405
+ try:
406
+ recap = await generate_episode_recap(
407
+ storyboard,
408
+ storyboard.get("episode_title", f"Episode {ep_number}"),
409
+ )
410
+ if recap:
411
+ episode.recap_summary = recap
412
+ await db.commit()
413
+ logger.info("[Job %s] Episode recap saved (%d chars)", job_id, len(recap))
414
+ except Exception as e:
415
+ logger.warning("[Job %s] Failed to generate episode recap: %s (non-fatal)", job_id, e)
416
+
417
+ # Mark job complete
418
+ await _update_job(
419
+ job_id,
420
+ status="completed",
421
+ progress_pct=100.0,
422
+ progress_detail=f"Episode {ep_number} generated successfully",
423
+ current_stage="completed",
424
+ episode_id=episode.id,
425
+ completed_at=datetime.now(timezone.utc),
426
+ )
427
+
428
+ except Exception as e:
429
+ logger.exception("[Job %s] Pipeline failed at stage", job_id)
430
+ await _update_job(
431
+ job_id,
432
+ status="failed",
433
+ error_message=str(e)[:2000],
434
+ progress_detail=f"Failed: {str(e)[:200]}",
435
+ )
app/pipeline/recap_generator.py ADDED
@@ -0,0 +1,191 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Recap generator — episode narrative continuity for multi-episode series.
2
+
3
+ After each episode, generates a 3-5 sentence narrative recap from the storyboard.
4
+ Before generating a new episode, builds a continuity context block from prior recaps.
5
+ """
6
+ import logging
7
+ from sqlalchemy import select
8
+ from sqlalchemy.ext.asyncio import AsyncSession
9
+ from app.database import async_session
10
+ from app.models.episode import Episode
11
+ from app.services.llm import call_gemini
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+ RECAP_PROMPT = """You are summarizing an anime episode for production continuity.
16
+
17
+ Given the storyboard below (dialogue, narration, and scene descriptions), write a 3-5 sentence narrative recap (~150 words).
18
+
19
+ Focus on:
20
+ - Key plot events that happened
21
+ - Important character actions and decisions
22
+ - Relationship changes or reveals
23
+ - How the episode ended (cliffhanger, resolution, etc.)
24
+
25
+ Write in past tense, third person. Be specific about WHAT happened, not vague.
26
+
27
+ EPISODE TITLE: {episode_title}
28
+
29
+ STORYBOARD:
30
+ {storyboard_text}
31
+
32
+ Write ONLY the recap, no headers or labels."""
33
+
34
+ COMPRESS_PROMPT = """Compress these episode recaps into a single "story so far" summary (~200 words).
35
+ Preserve the most important plot points, character arcs, and world-building. Drop minor details.
36
+ Write in past tense, third person.
37
+
38
+ {recaps_text}
39
+
40
+ Write ONLY the compressed summary, no headers or labels."""
41
+
42
+
43
+ def _extract_storyboard_text(storyboard: dict) -> str:
44
+ """Extract readable dialogue/narration from storyboard JSON."""
45
+ lines = []
46
+ for scene in storyboard.get("scenes", []):
47
+ setting = scene.get("setting", "")
48
+ mood = scene.get("mood", "")
49
+ if setting:
50
+ lines.append(f"[Scene: {setting} | Mood: {mood}]")
51
+
52
+ for cut in scene.get("cuts", scene.get("shots", [])):
53
+ dialogue = cut.get("dialogue")
54
+ if not dialogue or not dialogue.get("text"):
55
+ continue
56
+
57
+ speaker = dialogue.get("speaker", "Narrator")
58
+ text = dialogue["text"]
59
+ is_narration = dialogue.get("is_narration", False)
60
+ is_thought = dialogue.get("is_inner_thought", False)
61
+
62
+ if is_narration:
63
+ lines.append(f" (Narration) {text}")
64
+ elif is_thought:
65
+ lines.append(f" ({speaker} thinks) {text}")
66
+ else:
67
+ lines.append(f" {speaker}: \"{text}\"")
68
+
69
+ return "\n".join(lines)
70
+
71
+
72
+ async def generate_episode_recap(storyboard: dict, episode_title: str) -> str:
73
+ """Generate a 3-5 sentence narrative recap from a completed episode's storyboard."""
74
+ storyboard_text = _extract_storyboard_text(storyboard)
75
+ if not storyboard_text.strip():
76
+ return ""
77
+
78
+ prompt = RECAP_PROMPT.format(
79
+ episode_title=episode_title,
80
+ storyboard_text=storyboard_text,
81
+ )
82
+
83
+ recap = await call_gemini(prompt, max_tokens=512, temperature=0.3)
84
+ return recap.strip()
85
+
86
+
87
+ async def build_continuity_context(project_id: int, current_ep_number: int) -> str | None:
88
+ """Build narrative continuity context from prior episodes' recaps.
89
+
90
+ Returns a formatted continuity block string, or None if this is the first episode.
91
+ """
92
+ async with async_session() as db:
93
+ result = await db.execute(
94
+ select(Episode)
95
+ .where(
96
+ Episode.project_id == project_id,
97
+ Episode.episode_number < current_ep_number,
98
+ Episode.status == "completed",
99
+ )
100
+ .order_by(Episode.episode_number)
101
+ )
102
+ prior_episodes = list(result.scalars().all())
103
+
104
+ if not prior_episodes:
105
+ return None
106
+
107
+ # Lazy-generate missing recaps
108
+ for ep in prior_episodes:
109
+ if not ep.recap_summary:
110
+ if ep.storyboard_json:
111
+ logger.info("Lazy-generating recap for Episode %d", ep.episode_number)
112
+ try:
113
+ recap = await generate_episode_recap(
114
+ ep.storyboard_json,
115
+ ep.title or f"Episode {ep.episode_number}",
116
+ )
117
+ if recap:
118
+ async with async_session() as db:
119
+ result = await db.execute(
120
+ select(Episode).where(Episode.id == ep.id)
121
+ )
122
+ db_ep = result.scalar_one()
123
+ db_ep.recap_summary = recap
124
+ await db.commit()
125
+ ep.recap_summary = recap
126
+ except Exception as e:
127
+ logger.warning("Failed to lazy-generate recap for Episode %d: %s",
128
+ ep.episode_number, e)
129
+ else:
130
+ logger.warning("Episode %d has no storyboard_json, skipping recap",
131
+ ep.episode_number)
132
+
133
+ # Collect available recaps
134
+ episodes_with_recaps = [(ep.episode_number, ep.title, ep.recap_summary)
135
+ for ep in prior_episodes if ep.recap_summary]
136
+
137
+ if not episodes_with_recaps:
138
+ return None
139
+
140
+ # Build the continuity block
141
+ if len(episodes_with_recaps) <= 3:
142
+ # Few episodes — include all recaps directly
143
+ recap_lines = []
144
+ for ep_num, title, recap in episodes_with_recaps:
145
+ label = title or f"Episode {ep_num}"
146
+ recap_lines.append(f'EPISODE {ep_num} ("{label}"):\n{recap}')
147
+ all_recaps = "\n\n".join(recap_lines)
148
+
149
+ # Last episode gets special "PREVIOUS EPISODE" treatment
150
+ last_ep_num, last_title, last_recap = episodes_with_recaps[-1]
151
+ if len(episodes_with_recaps) == 1:
152
+ story_so_far = ""
153
+ previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
154
+ else:
155
+ earlier = "\n\n".join(
156
+ f'Episode {n} ("{t or f"Episode {n}"}"): {r}'
157
+ for n, t, r in episodes_with_recaps[:-1]
158
+ )
159
+ story_so_far = f"STORY SO FAR:\n{earlier}\n\n"
160
+ previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
161
+ else:
162
+ # 4+ episodes — compress older ones, keep latest full
163
+ older_recaps = "\n\n".join(
164
+ f'Episode {n} ("{t or f"Episode {n}"}"): {r}'
165
+ for n, t, r in episodes_with_recaps[:-1]
166
+ )
167
+ try:
168
+ compressed = await call_gemini(
169
+ COMPRESS_PROMPT.format(recaps_text=older_recaps),
170
+ max_tokens=512,
171
+ temperature=0.3,
172
+ )
173
+ story_so_far = f"STORY SO FAR:\n{compressed.strip()}\n\n"
174
+ except Exception as e:
175
+ logger.warning("Failed to compress recaps: %s — using last recap only", e)
176
+ story_so_far = f"STORY SO FAR:\n{older_recaps}\n\n"
177
+
178
+ last_ep_num, last_title, last_recap = episodes_with_recaps[-1]
179
+ previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
180
+
181
+ return f"""═══ NARRATIVE CONTINUITY ═══
182
+ This is NOT the first episode. The audience has already watched the previous episode(s).
183
+
184
+ {story_so_far}{previous_block}
185
+
186
+ CONTINUITY RULES:
187
+ - Do NOT re-introduce characters the audience already knows. They can appear immediately.
188
+ - Do NOT repeat exposition from prior episodes.
189
+ - You MAY open with a brief 1-2 cut "recall" moment echoing the previous ending (optional).
190
+ - Reference prior events naturally through dialogue or narration when relevant.
191
+ - Character relationships carry forward — show evolution, not reset."""
app/pipeline/stage_1_ingest.py ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 1: Text Ingestion — get chapter text into the database."""
2
+ from sqlalchemy import select
3
+ from sqlalchemy.ext.asyncio import AsyncSession
4
+ from app.models.chapter import Chapter
5
+ from app.models.project import Project
6
+
7
+
8
+ async def run_ingest(
9
+ project_id: int,
10
+ db: AsyncSession,
11
+ on_progress: callable = None,
12
+ chapter_ids: list[int] | None = None,
13
+ ) -> list[Chapter]:
14
+ """Ensure chapters are loaded for the project. Returns chapter list.
15
+
16
+ If chapter_ids is provided, only load those specific chapters.
17
+ """
18
+ if on_progress:
19
+ await on_progress("Loading chapters...", 0)
20
+
21
+ query = select(Chapter).where(Chapter.project_id == project_id)
22
+ if chapter_ids:
23
+ query = query.where(Chapter.id.in_(chapter_ids))
24
+ query = query.order_by(Chapter.chapter_number)
25
+
26
+ result = await db.execute(query)
27
+ chapters = list(result.scalars().all())
28
+
29
+ if not chapters:
30
+ raise ValueError("No chapters found for this project. Add chapters first.")
31
+
32
+ total_words = sum(len(ch.source_text.split()) for ch in chapters)
33
+ if on_progress:
34
+ await on_progress(
35
+ f"Loaded {len(chapters)} chapter{'s' if len(chapters) != 1 else ''} ({total_words:,} words)",
36
+ 100,
37
+ )
38
+
39
+ return chapters
app/pipeline/stage_2_characters.py ADDED
@@ -0,0 +1,172 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 2: Character Extraction & Enrichment."""
2
+ import json
3
+ from sqlalchemy import select
4
+ from sqlalchemy.ext.asyncio import AsyncSession
5
+ from app.models.character import Character
6
+ from app.models.chapter import Chapter
7
+ from app.services.llm import call_gemini_json
8
+ from app.services.fandom import search_fandom_wiki
9
+ from app.utils.voice_mapper import assign_voice
10
+ from app.utils.prompt_builder import build_character_visual_prompt
11
+
12
+ CHARACTER_EXTRACTION_PROMPT = """You are an expert manhwa/webtoon character designer. Analyze the following novel chapter text and extract ALL named characters with COMPLETE visual designs.
13
+
14
+ CRITICAL RULES:
15
+ 1. NEVER output "unknown" for ANY visual field. If the text doesn't describe a trait, INVENT a distinctive, fitting appearance based on the character's role, personality, and name origin.
16
+ 2. Every character must be VISUALLY DISTINCT from all others — no two characters should share the same hair color + style combination.
17
+ 3. Use SPECIFIC colors, not vague ones: "platinum silver" not "light", "deep crimson" not "red", "midnight blue" not "dark blue".
18
+ 4. Clothing should be detailed: "black high-collar military coat with silver trim and crimson epaulettes" not "dark clothes".
19
+
20
+ For each character, provide:
21
+ - name: The character's primary name
22
+ - aliases: Other names/titles they go by (list)
23
+ - role: protagonist, antagonist, supporting, or minor
24
+ - gender: male, female, or unknown (guess from name/context if possible)
25
+ - age_category: child, teen, young_adult, adult, or elder
26
+ - hair_color: specific color (e.g. "jet black", "platinum silver", "deep crimson", "ash blonde")
27
+ - hair_style: specific style (e.g. "long flowing with side-swept bangs", "short spiky undercut", "waist-length straight with center part")
28
+ - eye_color: specific color (e.g. "golden amber", "deep violet", "ice blue", "emerald green")
29
+ - skin_tone: specific tone (e.g. "fair porcelain", "warm olive", "deep bronze", "pale ivory")
30
+ - height: tall, average, or short
31
+ - build: muscular, athletic, slim, average, or heavyset
32
+ - clothing: detailed outfit description with colors and accessories
33
+ - distinctive_features: scars, tattoos, accessories, jewelry, etc. (invent something unique if none mentioned)
34
+ - color_palette: 3 main colors associated with this character (e.g. ["black", "crimson", "silver"])
35
+ - personality_notes: brief personality traits relevant to voice acting
36
+
37
+ Return a JSON array of character objects. Include EVERY named character that appears, even briefly.
38
+
39
+ Chapter text:
40
+ {chapter_text}"""
41
+
42
+
43
+ CHARACTER_DEDUP_PROMPT = """Review these character visual designs and fix any that look too similar. Two characters should NEVER share the same hair color + style combination.
44
+
45
+ Current characters:
46
+ {characters_json}
47
+
48
+ Rules:
49
+ 1. If two characters have similar hair (same color family AND similar style), change ONE of them to be distinct.
50
+ 2. Ensure color_palette entries don't overlap too much between characters.
51
+ 3. Keep changes minimal — only modify what's needed for visual distinction.
52
+ 4. Return the FULL array with corrections applied.
53
+
54
+ Return a JSON array of all characters (modified where needed)."""
55
+
56
+
57
+ async def run_character_extraction(
58
+ project_id: int,
59
+ project_title: str,
60
+ chapters: list[Chapter],
61
+ db: AsyncSession,
62
+ on_progress: callable = None,
63
+ ) -> list[Character]:
64
+ """Extract characters from chapters, enrich with wiki data, assign voices."""
65
+
66
+ if on_progress:
67
+ await on_progress("Analyzing chapter text for characters...", 0)
68
+
69
+ # Check for existing characters
70
+ existing = await db.execute(
71
+ select(Character).where(Character.project_id == project_id)
72
+ )
73
+ existing_list = existing.scalars().all()
74
+ existing_chars = {c.name.lower(): c for c in existing_list}
75
+ # Also index by aliases for dedup
76
+ for c in list(existing_chars.values()):
77
+ for alias in (c.aliases_json or []):
78
+ existing_chars.setdefault(alias.lower(), c)
79
+
80
+ # Combine chapter texts for extraction (first 3 chapters for efficiency)
81
+ combined_text = "\n\n---\n\n".join(
82
+ ch.source_text[:5000] for ch in chapters[:3]
83
+ )
84
+
85
+ # Call Gemini for character extraction
86
+ prompt = CHARACTER_EXTRACTION_PROMPT.format(chapter_text=combined_text[:15000])
87
+ extracted = await call_gemini_json(prompt)
88
+
89
+ # Gemini may return {"characters": [...]} instead of [...]
90
+ if isinstance(extracted, dict):
91
+ extracted = extracted.get("characters", [extracted])
92
+ if not isinstance(extracted, list):
93
+ extracted = [extracted]
94
+
95
+ # Visual deduplication pass — ensure all characters look distinct
96
+ if len(extracted) > 1:
97
+ if on_progress:
98
+ await on_progress("Ensuring visual distinctness...", 12)
99
+ import json as _json
100
+ dedup_prompt = CHARACTER_DEDUP_PROMPT.format(
101
+ characters_json=_json.dumps(extracted, indent=2)
102
+ )
103
+ try:
104
+ deduped = await call_gemini_json(dedup_prompt)
105
+ if isinstance(deduped, dict):
106
+ deduped = deduped.get("characters", [deduped])
107
+ if isinstance(deduped, list) and len(deduped) == len(extracted):
108
+ extracted = deduped
109
+ except Exception:
110
+ pass # dedup is nice-to-have, not critical
111
+
112
+ total = len(extracted)
113
+ characters = []
114
+
115
+ if on_progress and total:
116
+ await on_progress(f"Discovered {total} characters", 15)
117
+
118
+ for i, char_data in enumerate(extracted):
119
+ name = char_data.get("name", "").strip()
120
+ if not name or name.lower() in existing_chars:
121
+ if name.lower() in existing_chars:
122
+ characters.append(existing_chars[name.lower()])
123
+ continue
124
+
125
+ if on_progress:
126
+ pct = int(20 + (i / max(total, 1)) * 60)
127
+ await on_progress(f"Enriching {name} ({i+1}/{total})", pct)
128
+
129
+ # Try fandom wiki enrichment
130
+ wiki_data = await search_fandom_wiki(project_title, name)
131
+ if wiki_data and wiki_data.get("description"):
132
+ # Merge wiki appearance data (prefer wiki, more detailed)
133
+ char_data["wiki_appearance"] = wiki_data["description"]
134
+
135
+ # Build visual prompt
136
+ visual_prompt = build_character_visual_prompt(char_data)
137
+
138
+ # Assign voice
139
+ voice_config = assign_voice(
140
+ project_id,
141
+ char_data.get("gender"),
142
+ char_data.get("age_category"),
143
+ )
144
+
145
+ # Normalize aliases (Gemini may return a comma-separated string)
146
+ aliases = char_data.get("aliases", [])
147
+ if isinstance(aliases, str):
148
+ aliases = [a.strip() for a in aliases.split(",") if a.strip()]
149
+
150
+ # Create character record
151
+ character = Character(
152
+ project_id=project_id,
153
+ name=name,
154
+ aliases_json=aliases,
155
+ role=char_data.get("role", "supporting"),
156
+ gender=char_data.get("gender"),
157
+ age_category=char_data.get("age_category"),
158
+ visual_prompt=visual_prompt,
159
+ voice_config_json=voice_config,
160
+ wiki_data_json=wiki_data,
161
+ )
162
+ db.add(character)
163
+ characters.append(character)
164
+
165
+ await db.commit()
166
+ for ch in characters:
167
+ await db.refresh(ch)
168
+
169
+ if on_progress:
170
+ await on_progress(f"{len(characters)} characters ready", 100)
171
+
172
+ return characters
app/pipeline/stage_2b_portraits.py ADDED
@@ -0,0 +1,118 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 2b: Character Portrait Generation.
2
+
3
+ Generates a high-quality reference portrait for each prominent character,
4
+ uploads to media.pollinations.ai, and stores the URL in the Character model.
5
+ These portraits are used as visual references during scene image generation
6
+ with vision-capable models (klein, gptimage) for character consistency.
7
+ """
8
+ import asyncio
9
+ import logging
10
+ from pathlib import Path
11
+
12
+ from sqlalchemy.ext.asyncio import AsyncSession
13
+
14
+ from app.models.character import Character
15
+ from app.services.pollinations import generate_image, upload_media, VISION_MODELS
16
+ from app.utils.prompt_builder import MANHWA_STYLE_PREFIX
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ PORTRAIT_PROMPT_TEMPLATE = (
21
+ "{style_prefix}, character portrait sheet, front-facing bust shot, "
22
+ "{visual_prompt}, clean white background, reference sheet style, "
23
+ "sharp details, no background elements, studio lighting, "
24
+ "high detail face and eyes, character design reference"
25
+ )
26
+
27
+ # Only generate portraits for prominent roles
28
+ PROMINENT_ROLES = {"protagonist", "antagonist", "supporting"}
29
+
30
+ # Use klein-large for portraits — scene images also use klein-large with this
31
+ # portrait as reference. Same model = same style space = consistent characters.
32
+ PORTRAIT_MODEL = "klein-large"
33
+ PORTRAIT_SEED = 42
34
+
35
+
36
+ async def run_portrait_generation(
37
+ project_id: int,
38
+ characters: list[Character],
39
+ project_dir: str,
40
+ db: AsyncSession,
41
+ on_progress: callable = None,
42
+ ) -> list[Character]:
43
+ """Generate reference portraits for prominent characters.
44
+
45
+ Skips characters that already have a portrait_url.
46
+ Returns the updated character list.
47
+ """
48
+ # Filter to prominent characters without portraits
49
+ targets = [
50
+ c for c in characters
51
+ if c.role in PROMINENT_ROLES
52
+ and c.visual_prompt
53
+ and not c.portrait_url
54
+ ]
55
+
56
+ if not targets:
57
+ if on_progress:
58
+ await on_progress("All characters already have portraits", 100)
59
+ return characters
60
+
61
+ total = len(targets)
62
+ if on_progress:
63
+ await on_progress(f"Generating {total} character portraits...", 0)
64
+
65
+ portrait_dir = Path(project_dir) / "portraits"
66
+ portrait_dir.mkdir(parents=True, exist_ok=True)
67
+
68
+ completed = 0
69
+
70
+ for char in targets:
71
+ if on_progress:
72
+ pct = int((completed / max(total, 1)) * 90)
73
+ await on_progress(f"[{completed + 1}/{total}] Painting {char.name}...", pct)
74
+
75
+ portrait_path = str(portrait_dir / f"{char.name.replace(' ', '_')}_portrait.png")
76
+
77
+ try:
78
+ # Generate portrait image
79
+ prompt = PORTRAIT_PROMPT_TEMPLATE.format(
80
+ style_prefix=MANHWA_STYLE_PREFIX,
81
+ visual_prompt=char.visual_prompt,
82
+ )
83
+
84
+ await generate_image(
85
+ prompt=prompt,
86
+ output_path=portrait_path,
87
+ model=PORTRAIT_MODEL,
88
+ width=768,
89
+ height=1024, # Portrait orientation
90
+ seed=PORTRAIT_SEED,
91
+ )
92
+
93
+ # Upload to media.pollinations.ai for permanent URL
94
+ portrait_url = await upload_media(portrait_path)
95
+ char.portrait_url = portrait_url
96
+ char.reference_image_path = portrait_path
97
+
98
+ logger.info("Portrait generated for %s: %s", char.name, portrait_url)
99
+
100
+ except Exception as e:
101
+ logger.warning("Portrait generation failed for %s: %s", char.name, e)
102
+ # Non-fatal — character just won't have a portrait reference
103
+
104
+ completed += 1
105
+
106
+ # Persist portrait URLs to database
107
+ try:
108
+ await db.commit()
109
+ for char in characters:
110
+ await db.refresh(char)
111
+ except Exception as e:
112
+ logger.warning("Failed to persist portrait URLs: %s", e)
113
+
114
+ if on_progress:
115
+ success_count = sum(1 for c in targets if c.portrait_url)
116
+ await on_progress(f"{success_count}/{total} portraits generated", 100)
117
+
118
+ return characters
app/pipeline/stage_2c_voice_enroll.py ADDED
@@ -0,0 +1,177 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 2c: Voice Enrollment — assign AnimeVox voices to prominent characters.
2
+
3
+ Runs after Stage 2b (portraits), before Stage 3 (scene parse).
4
+ Matches each protagonist/antagonist/supporting character to an AnimeVox voice
5
+ based on gender, age, and personality traits, then stores the voice state path
6
+ for Pocket TTS to use during TTS generation.
7
+
8
+ Gracefully degrades: if AnimeVox is not set up, all characters fall back to Edge TTS.
9
+ """
10
+ import logging
11
+ from pathlib import Path
12
+
13
+ from sqlalchemy.ext.asyncio import AsyncSession
14
+
15
+ from app.models.character import Character
16
+ from app.utils.voice_catalog import match_voice, is_available
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ PROMINENT_ROLES = {"protagonist", "antagonist", "supporting"}
21
+
22
+ # Manual voice overrides: character name (case-insensitive) → catalog voice ID
23
+ # Use this to force a specific voice when auto-matching picks a bad fit.
24
+ VOICE_OVERRIDES: dict[str, str] = {
25
+ "ye chen": "genshin_diluc",
26
+ }
27
+
28
+
29
+ async def run_voice_enrollment(
30
+ project_id: int,
31
+ characters: list[Character],
32
+ project_dir: str,
33
+ db: AsyncSession,
34
+ on_progress: callable = None,
35
+ ) -> list[Character]:
36
+ """Enroll voices for prominent characters using AnimeVox catalog.
37
+
38
+ For each protagonist/antagonist/supporting without a voice_reference_path:
39
+ 1. Load voice catalog
40
+ 2. Match character to best AnimeVox voice (gender + age + traits)
41
+ 3. Store voice state path in character model
42
+ 4. Persist to DB
43
+
44
+ Returns the updated character list.
45
+ """
46
+ if not is_available():
47
+ if on_progress:
48
+ await on_progress("AnimeVox not set up — using Edge TTS for all characters", 100)
49
+ logger.info("Voice enrollment skipped: AnimeVox catalog not found")
50
+ return characters
51
+
52
+ # Filter to prominent characters that need voice enrollment
53
+ targets = [
54
+ c for c in characters
55
+ if c.role in PROMINENT_ROLES
56
+ and not c.voice_reference_path # don't re-enroll
57
+ ]
58
+
59
+ if not targets:
60
+ if on_progress:
61
+ await on_progress("All characters already have voices assigned", 100)
62
+ return characters
63
+
64
+ total = len(targets)
65
+ if on_progress:
66
+ await on_progress(f"Matching voices for {total} characters...", 0)
67
+
68
+ enrolled = 0
69
+ for i, char in enumerate(targets):
70
+ if on_progress:
71
+ pct = int((i / max(total, 1)) * 90)
72
+ await on_progress(f"[{i+1}/{total}] Matching voice for {char.name}...", pct)
73
+
74
+ # Check manual overrides first
75
+ override_id = VOICE_OVERRIDES.get(char.name.lower())
76
+ if override_id:
77
+ match = _get_override_voice(override_id)
78
+ if match:
79
+ logger.info("Voice override: %s → %s", char.name, match["display_name"])
80
+ else:
81
+ match = None
82
+
83
+ if not match:
84
+ # Extract personality hints from visual_prompt or wiki data
85
+ personality_hints = _extract_personality_hints(char)
86
+
87
+ match = match_voice(
88
+ project_id=project_id,
89
+ gender=char.gender or "male",
90
+ age_category=char.age_category or "young_adult",
91
+ role=char.role,
92
+ personality_hints=personality_hints,
93
+ )
94
+
95
+ # Determine voice reference: .safetensors path or preset name
96
+ voice_ref = None
97
+ if match:
98
+ if match.get("voice_state_path") and Path(match["voice_state_path"]).exists():
99
+ voice_ref = match["voice_state_path"]
100
+ elif match.get("preset_voice"):
101
+ voice_ref = match["preset_voice"]
102
+
103
+ if match and voice_ref:
104
+ char.voice_reference_path = voice_ref
105
+ char.voice_source_json = {
106
+ "source": "animevox",
107
+ "character": match["display_name"],
108
+ "anime": match["source_anime"],
109
+ "voice_id": match["id"],
110
+ "type": "cloned" if voice_ref.endswith(".safetensors") else "preset",
111
+ }
112
+ enrolled += 1
113
+ logger.info(
114
+ "Voice enrolled: %s → %s (%s)",
115
+ char.name, match["display_name"], match["source_anime"],
116
+ )
117
+ else:
118
+ logger.info(
119
+ "No voice match for %s (%s/%s) — will use Edge TTS",
120
+ char.name, char.gender, char.age_category,
121
+ )
122
+
123
+ # Persist to database
124
+ if enrolled > 0:
125
+ try:
126
+ await db.commit()
127
+ for char in characters:
128
+ await db.refresh(char)
129
+ except Exception as e:
130
+ logger.warning("Failed to persist voice enrollment: %s", e)
131
+
132
+ if on_progress:
133
+ await on_progress(
134
+ f"{enrolled}/{total} characters matched to anime voices", 100
135
+ )
136
+
137
+ logger.info(
138
+ "Voice enrollment complete: %d/%d characters enrolled", enrolled, total
139
+ )
140
+ return characters
141
+
142
+
143
+ def _get_override_voice(voice_id: str) -> dict | None:
144
+ """Look up a specific voice by ID from the catalog."""
145
+ from app.utils.voice_catalog import get_catalog
146
+
147
+ for entry in get_catalog():
148
+ if entry.get("id") == voice_id:
149
+ return entry
150
+ logger.warning("Voice override ID '%s' not found in catalog", voice_id)
151
+ return None
152
+
153
+
154
+ def _extract_personality_hints(char: Character) -> list[str]:
155
+ """Extract personality keywords from character data for voice matching."""
156
+ hints = []
157
+
158
+ # From wiki data
159
+ wiki = char.wiki_data_json or {}
160
+ if isinstance(wiki, dict):
161
+ personality = wiki.get("personality", "")
162
+ if personality:
163
+ # Extract adjectives from personality description
164
+ for word in personality.lower().split():
165
+ word = word.strip(".,;:!?()[]")
166
+ if len(word) > 3: # skip short words
167
+ hints.append(word)
168
+
169
+ # From role — add implied traits
170
+ role_hints = {
171
+ "protagonist": ["determined", "heroic"],
172
+ "antagonist": ["cold", "menacing", "calculating"],
173
+ "supporting": ["warm", "loyal"],
174
+ }
175
+ hints.extend(role_hints.get(char.role, []))
176
+
177
+ return hints[:10] # cap to avoid noise
app/pipeline/stage_3_scene_parse.py ADDED
@@ -0,0 +1,346 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 3: Scene Parsing — convert chapter text to e-konte storyboard JSON."""
2
+ import re
3
+ from app.models.chapter import Chapter
4
+ from app.models.character import Character
5
+ from app.services.llm import call_gemini_json
6
+ from app.templates.storyboard_prompt import (
7
+ STORYBOARD_SYSTEM,
8
+ STORYBOARD_PROMPT,
9
+ format_characters_block,
10
+ )
11
+
12
+
13
+ def _estimate_duration(dialogue: dict | None, action: str | None) -> float:
14
+ """Estimate cut duration based on content."""
15
+ if dialogue:
16
+ text = dialogue.get("text") or ""
17
+ if not text:
18
+ return 2.0
19
+ if dialogue.get("is_narration") or dialogue.get("is_inner_thought"):
20
+ dur = len(text) / 12 + 0.5
21
+ else:
22
+ dur = len(text) / 15 + 0.5
23
+ return max(0.5, min(dur, 12.0))
24
+ if action:
25
+ return max(2.0, min(len(action) / 20, 5.0))
26
+ return 2.0
27
+
28
+
29
+ def _normalize_cuts(storyboard: dict) -> dict:
30
+ """Normalize storyboard to use 'cuts' consistently.
31
+
32
+ The new e-konte format uses 'cuts', but the old format used 'shots'.
33
+ This function ensures we always have a 'cuts' key on each scene,
34
+ while keeping 'shots' as an alias for backward compatibility.
35
+ """
36
+ for scene in storyboard.get("scenes", []):
37
+ # If scene has 'shots' but not 'cuts', rename
38
+ if "shots" in scene and "cuts" not in scene:
39
+ scene["cuts"] = scene.pop("shots")
40
+ # If scene has 'cuts', also set 'shots' alias for backward compat
41
+ if "cuts" in scene:
42
+ scene["shots"] = scene["cuts"]
43
+ # Handle cut_id vs shot_id
44
+ for cut in scene.get("cuts", []):
45
+ if "shot_id" in cut and "cut_id" not in cut:
46
+ cut["cut_id"] = cut["shot_id"]
47
+ elif "cut_id" in cut and "shot_id" not in cut:
48
+ cut["shot_id"] = cut["cut_id"]
49
+ return storyboard
50
+
51
+
52
+ def _build_alias_map(characters: list[Character]) -> dict[str, str]:
53
+ """Build a lowercase alias → canonical name map from characters."""
54
+ alias_map = {}
55
+ for c in characters:
56
+ for alias in (c.aliases_json or []):
57
+ alias_map[alias.lower()] = c.name
58
+ return alias_map
59
+
60
+
61
+ def _resolve_name(name: str, character_names: set[str], alias_map: dict[str, str]) -> str | None:
62
+ """Resolve a name to its canonical form via direct match or alias lookup."""
63
+ if name in character_names:
64
+ return name
65
+ canonical = alias_map.get(name.lower())
66
+ if canonical and canonical in character_names:
67
+ return canonical
68
+ return None
69
+
70
+
71
+ def _validate_storyboard(storyboard: dict, characters: list[Character]) -> dict:
72
+ """Post-process storyboard: fix IDs, validate characters, recalc durations."""
73
+ character_names = {c.name for c in characters}
74
+ alias_map = _build_alias_map(characters)
75
+ cut_counter = 0
76
+
77
+ for scene in storyboard.get("scenes", []):
78
+ for cut in scene.get("cuts", scene.get("shots", [])):
79
+ cut_counter += 1
80
+ cut_id = f"C{cut_counter:02d}"
81
+ cut["cut_id"] = cut_id
82
+ cut["shot_id"] = cut_id # Backward compat alias
83
+
84
+ # Coerce characters_present to list and resolve aliases
85
+ chars_present = cut.get("characters_present", [])
86
+ if isinstance(chars_present, str):
87
+ chars_present = [chars_present]
88
+ valid_chars = []
89
+ for c in chars_present:
90
+ resolved = _resolve_name(c, character_names, alias_map)
91
+ if resolved and resolved not in valid_chars:
92
+ valid_chars.append(resolved)
93
+ cut["characters_present"] = valid_chars
94
+
95
+ # Resolve focal_character via alias
96
+ focal = cut.get("focal_character")
97
+ if focal:
98
+ resolved_focal = _resolve_name(focal, character_names, alias_map)
99
+ cut["focal_character"] = resolved_focal if resolved_focal else (valid_chars[0] if valid_chars else None)
100
+ elif focal is not None and not focal:
101
+ cut["focal_character"] = None
102
+
103
+ # Enforce: on-screen speaker should be focal_character
104
+ dialogue = cut.get("dialogue")
105
+ if dialogue and dialogue.get("text"):
106
+ speaker = dialogue.get("speaker", "")
107
+ resolved_speaker = _resolve_name(speaker, character_names, alias_map)
108
+ # Replace alias in dialogue.speaker with canonical name
109
+ if resolved_speaker and resolved_speaker != speaker:
110
+ dialogue["speaker"] = resolved_speaker
111
+ speaker = resolved_speaker
112
+ is_offscreen = dialogue.get("is_offscreen", False)
113
+ is_narration = dialogue.get("is_narration", False)
114
+ is_thought = dialogue.get("is_inner_thought", False)
115
+
116
+ if (speaker != "Narrator"
117
+ and speaker in character_names
118
+ and not is_offscreen
119
+ and not is_narration
120
+ and not is_thought):
121
+ # Speaker is on-screen → must be in characters_present and focal
122
+ if speaker not in valid_chars:
123
+ valid_chars.append(speaker)
124
+ cut["characters_present"] = valid_chars
125
+ if not cut.get("focal_character") or cut["focal_character"] != speaker:
126
+ cut["focal_character"] = speaker
127
+
128
+ # Coerce dialogue: if string, wrap as narration
129
+ dialogue = cut.get("dialogue")
130
+ if isinstance(dialogue, str):
131
+ cut["dialogue"] = {
132
+ "text": dialogue,
133
+ "speaker": "Narrator",
134
+ "is_narration": True,
135
+ }
136
+
137
+ # Coerce duration_sec to float
138
+ try:
139
+ cut["duration_sec"] = float(cut.get("duration_sec", 0))
140
+ except (TypeError, ValueError):
141
+ cut["duration_sec"] = 0
142
+
143
+ # Recalculate duration
144
+ cut["duration_sec"] = _estimate_duration(
145
+ cut.get("dialogue"), cut.get("action_description")
146
+ )
147
+
148
+ # Ensure effects is a list
149
+ if not isinstance(cut.get("effects"), list):
150
+ cut["effects"] = []
151
+
152
+ # Default camera movement
153
+ if not cut.get("camera_movement"):
154
+ cut["camera_movement"] = "FIX"
155
+
156
+ # Ensure video_prompt exists (backfill from action + camera if missing)
157
+ if not cut.get("video_prompt"):
158
+ parts = []
159
+ if cut.get("action_description"):
160
+ parts.append(cut["action_description"])
161
+ camera = cut.get("camera_movement", "FIX")
162
+ if camera != "FIX":
163
+ parts.append(f"camera: {camera}")
164
+ cut["video_prompt"] = ", ".join(parts) if parts else "subtle idle animation, atmospheric motion"
165
+
166
+ # Ensure transition_out exists
167
+ if not cut.get("transition_out"):
168
+ cut["transition_out"] = "cut"
169
+
170
+ # Ensure emotional_beat exists
171
+ if not cut.get("emotional_beat"):
172
+ cut["emotional_beat"] = "setup"
173
+
174
+ # Narration backfill: ensure every cut has dialogue
175
+ for scene in storyboard.get("scenes", []):
176
+ for cut in scene.get("cuts", scene.get("shots", [])):
177
+ dialogue = cut.get("dialogue")
178
+ has_text = dialogue and dialogue.get("text")
179
+ if not has_text:
180
+ action = cut.get("action_description", "")
181
+ if action:
182
+ cut["dialogue"] = {
183
+ "speaker": "Narrator",
184
+ "text": action,
185
+ "is_narration": True,
186
+ "emotion": "neutral",
187
+ "delivery": "normal",
188
+ }
189
+
190
+ return storyboard
191
+
192
+
193
+ def _enforce_cut_limit(storyboard: dict, max_cuts: int = 12) -> dict:
194
+ """If total cut count exceeds max_cuts, keep the highest-scoring cuts."""
195
+ all_cuts = []
196
+ for scene in storyboard.get("scenes", []):
197
+ for cut in scene.get("cuts", scene.get("shots", [])):
198
+ # Score: dialogue=3, establishing=2, narration=1, reaction/empty=0
199
+ # Bonus for climax/tension_peak emotional beats
200
+ score = 0
201
+ dialogue = cut.get("dialogue")
202
+ if dialogue and dialogue.get("text"):
203
+ if dialogue.get("is_narration"):
204
+ score = 1
205
+ else:
206
+ score = 3
207
+ if cut.get("shot_type") == "establishing":
208
+ score = max(score, 2)
209
+ # Emotional beat bonus
210
+ beat = cut.get("emotional_beat", "")
211
+ if beat in ("climax", "tension_peak"):
212
+ score += 2
213
+ elif beat in ("rising_tension", "resolution"):
214
+ score += 1
215
+ all_cuts.append((scene["scene_id"], cut, score))
216
+
217
+ total = len(all_cuts)
218
+ if total <= max_cuts:
219
+ return storyboard
220
+
221
+ # Sort by score descending, keep top max_cuts, then restore original order
222
+ indexed = [(i, sid, cut, score) for i, (sid, cut, score) in enumerate(all_cuts)]
223
+ indexed.sort(key=lambda x: x[3], reverse=True)
224
+ kept = sorted(indexed[:max_cuts], key=lambda x: x[0]) # restore order
225
+
226
+ # Rebuild scenes with only kept cuts
227
+ kept_by_scene: dict[str, list] = {}
228
+ for _, sid, cut, _ in kept:
229
+ kept_by_scene.setdefault(sid, []).append(cut)
230
+
231
+ new_scenes = []
232
+ for scene in storyboard.get("scenes", []):
233
+ sid = scene["scene_id"]
234
+ if sid in kept_by_scene:
235
+ scene["cuts"] = kept_by_scene[sid]
236
+ scene["shots"] = scene["cuts"] # Backward compat
237
+ new_scenes.append(scene)
238
+
239
+ storyboard["scenes"] = new_scenes
240
+
241
+ # Re-number cut IDs
242
+ counter = 0
243
+ for scene in storyboard["scenes"]:
244
+ for cut in scene.get("cuts", scene.get("shots", [])):
245
+ counter += 1
246
+ cut["cut_id"] = f"C{counter:02d}"
247
+ cut["shot_id"] = f"C{counter:02d}"
248
+
249
+ return storyboard
250
+
251
+
252
+ async def run_scene_parse(
253
+ chapters: list[Chapter],
254
+ characters: list[Character],
255
+ on_progress: callable = None,
256
+ continuity_context: str | None = None,
257
+ ) -> dict:
258
+ """Parse chapter text into an e-konte storyboard JSON using Gemini."""
259
+ if on_progress:
260
+ await on_progress("AI Director is composing the e-konte...", 0)
261
+
262
+ # Prepare character data
263
+ char_data = [
264
+ {"name": c.name, "role": c.role, "visual_prompt": c.visual_prompt}
265
+ for c in characters
266
+ ]
267
+ char_names = {c.name for c in characters}
268
+ characters_block = format_characters_block(char_data)
269
+
270
+ # Combine chapter text (limit to ~15k chars for Gemini)
271
+ combined_text = "\n\n".join(ch.source_text for ch in chapters)
272
+ if len(combined_text) > 15000:
273
+ combined_text = combined_text[:15000] + "\n\n[Text truncated for processing]"
274
+
275
+ continuity_block = continuity_context or ""
276
+
277
+ # Count paragraphs so the LLM knows how much content to cover
278
+ paragraphs = [p.strip() for p in combined_text.split("\n") if p.strip()]
279
+ para_hint = f"\n\n[This chapter has {len(paragraphs)} paragraphs. You need at least {max(25, len(paragraphs))} cuts to cover it all.]\n"
280
+
281
+ prompt = STORYBOARD_PROMPT.format(
282
+ characters_block=characters_block,
283
+ continuity_block=continuity_block,
284
+ chapter_text=combined_text + para_hint,
285
+ )
286
+
287
+ if on_progress:
288
+ await on_progress("Composing storyboard...", 20)
289
+
290
+ # Use the stronger model for storyboard — flash-lite truncates large JSON
291
+ storyboard = await call_gemini_json(
292
+ prompt,
293
+ system_instruction=STORYBOARD_SYSTEM,
294
+ max_tokens=65536,
295
+ model="gemini-2.5-flash",
296
+ )
297
+
298
+ # Gemini may return the scenes array directly instead of {"scenes": [...]}
299
+ if not isinstance(storyboard, dict) or "scenes" not in storyboard:
300
+ if isinstance(storyboard, list):
301
+ storyboard = {"scenes": storyboard}
302
+ else:
303
+ raise ValueError("Storyboard must be a dict with 'scenes' key")
304
+
305
+ scene_count = len(storyboard.get("scenes", []))
306
+ raw_cuts = sum(
307
+ len(s.get("cuts", s.get("shots", [])))
308
+ for s in storyboard.get("scenes", [])
309
+ )
310
+
311
+ if on_progress:
312
+ await on_progress(f"Validating {scene_count} scenes, {raw_cuts} cuts...", 80)
313
+
314
+ # Normalize cuts/shots naming
315
+ storyboard = _normalize_cuts(storyboard)
316
+
317
+ # Post-process
318
+ storyboard = _validate_storyboard(storyboard, characters)
319
+ storyboard = _enforce_cut_limit(storyboard, max_cuts=50)
320
+
321
+ total_cuts = sum(
322
+ len(scene.get("cuts", scene.get("shots", [])))
323
+ for scene in storyboard.get("scenes", [])
324
+ )
325
+ total_duration = sum(
326
+ cut.get("duration_sec", 2)
327
+ for scene in storyboard.get("scenes", [])
328
+ for cut in scene.get("cuts", scene.get("shots", []))
329
+ )
330
+
331
+ storyboard["metadata"] = {
332
+ "total_shots": total_cuts,
333
+ "total_cuts": total_cuts,
334
+ "total_duration_estimate": round(total_duration, 1),
335
+ "scene_count": len(storyboard.get("scenes", [])),
336
+ }
337
+
338
+ if on_progress:
339
+ mins = int(total_duration // 60)
340
+ secs = int(total_duration % 60)
341
+ await on_progress(
342
+ f"E-konte ready — {total_cuts} cuts, ~{mins}:{secs:02d}",
343
+ 100,
344
+ )
345
+
346
+ return storyboard
app/pipeline/stage_4_image_gen.py ADDED
@@ -0,0 +1,137 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 4: Image Generation — bulk generate images for all storyboard cuts."""
2
+ import asyncio
3
+ from pathlib import Path
4
+ from app.models.character import Character
5
+ from app.services.image_gen import generate_image
6
+ from app.utils.prompt_builder import build_image_prompt
7
+
8
+
9
+ async def run_image_generation(
10
+ storyboard: dict,
11
+ characters: list[Character],
12
+ project_dir: str,
13
+ on_progress: callable = None,
14
+ resume: bool = False,
15
+ ) -> dict:
16
+ """Generate images for all cuts in the storyboard.
17
+
18
+ Returns updated storyboard with image_path set on each cut.
19
+ If resume=True, skip cuts whose image files already exist on disk.
20
+
21
+ Uses Pollinations as primary provider with vision-capable models
22
+ for characters that have portrait references (for consistency).
23
+ """
24
+ # Support both new "cuts" and legacy "shots" format
25
+ all_cuts = []
26
+ for scene in storyboard.get("scenes", []):
27
+ for cut in scene.get("cuts", scene.get("shots", [])):
28
+ all_cuts.append(cut)
29
+
30
+ total = len(all_cuts)
31
+ if on_progress:
32
+ await on_progress(f"Preparing {total} images...", 0)
33
+
34
+ # Build character prompt map and portrait URL map (with alias support)
35
+ char_prompts = {c.name: c.visual_prompt for c in characters if c.visual_prompt}
36
+ char_portraits = {c.name: c.portrait_url for c in characters if c.portrait_url}
37
+ # Alias map: alias → canonical name
38
+ alias_map: dict[str, str] = {}
39
+ for c in characters:
40
+ for alias in (c.aliases_json or []):
41
+ alias_map[alias.lower()] = c.name
42
+
43
+ completed = 0
44
+ skipped = 0
45
+ errors = []
46
+
47
+ img_dir = Path(project_dir) / "images"
48
+ img_dir.mkdir(parents=True, exist_ok=True)
49
+
50
+ async def generate_one(cut: dict) -> None:
51
+ nonlocal completed, skipped
52
+ cut_id = cut.get("cut_id") or cut.get("shot_id", "S00")
53
+ output_path = str(img_dir / f"{cut_id}.png")
54
+
55
+ # Resume: skip if file already exists on disk
56
+ if resume and Path(output_path).exists():
57
+ cut["image_path"] = output_path
58
+ skipped += 1
59
+ completed += 1
60
+ if on_progress:
61
+ pct = int((completed / max(total, 1)) * 100)
62
+ await on_progress(f"[{completed}/{total}] {cut_id} (cached)", pct)
63
+ return
64
+
65
+ # Build prompt
66
+ prompt = build_image_prompt(cut, char_prompts)
67
+
68
+ # Use klein-large for ALL shots to maintain consistent aesthetics.
69
+ # Same model + same seed = uniform art style across entire episode.
70
+ # grok-imagine only used as fallback if klein-large fails.
71
+ provider = "pollinations"
72
+ pollinations_model = "klein-large"
73
+ reference_url = None
74
+ seed = 42 # Locked seed for style consistency
75
+
76
+ # Pass portrait reference for character shots (with alias fallback)
77
+ focal = cut.get("focal_character")
78
+ resolved_focal = focal
79
+ if focal and focal not in char_portraits:
80
+ resolved_focal = alias_map.get(focal.lower(), focal)
81
+ if resolved_focal and resolved_focal in char_portraits:
82
+ reference_url = char_portraits[resolved_focal]
83
+ else:
84
+ # Wide/establishing: use any present character's portrait as ref
85
+ for c in cut.get("characters_present", []):
86
+ resolved_c = c if c in char_portraits else alias_map.get(c.lower(), c)
87
+ if resolved_c in char_portraits:
88
+ reference_url = char_portraits[resolved_c]
89
+ break
90
+
91
+ try:
92
+ await generate_image(
93
+ prompt=prompt,
94
+ output_path=output_path,
95
+ provider=provider,
96
+ seed=seed,
97
+ pollinations_model=pollinations_model,
98
+ reference_image_url=reference_url,
99
+ )
100
+ cut["image_path"] = output_path
101
+ except Exception as e:
102
+ # Fallback: retry with grok-imagine (no reference) if klein-large failed
103
+ if pollinations_model == "klein-large":
104
+ try:
105
+ await generate_image(
106
+ prompt=prompt,
107
+ output_path=output_path,
108
+ provider=provider,
109
+ seed=seed,
110
+ pollinations_model="grok-imagine",
111
+ reference_image_url=None,
112
+ )
113
+ cut["image_path"] = output_path
114
+ errors.append(f"{cut_id}: klein-large failed, used grok-imagine fallback")
115
+ except Exception as e2:
116
+ errors.append(f"{cut_id}: {e} | fallback also failed: {e2}")
117
+ cut["image_path"] = None
118
+ else:
119
+ errors.append(f"{cut_id}: {e}")
120
+ cut["image_path"] = None
121
+
122
+ completed += 1
123
+ if on_progress:
124
+ pct = int((completed / max(total, 1)) * 100)
125
+ prompt_preview = prompt[:40].rstrip() + "..." if len(prompt) > 40 else prompt
126
+ await on_progress(f"[{completed}/{total}] {prompt_preview}", pct)
127
+
128
+ # Run with concurrency limit (semaphores are in the image_gen service)
129
+ await asyncio.gather(*[generate_one(cut) for cut in all_cuts])
130
+
131
+ if on_progress:
132
+ err_suffix = f" ({len(errors)} errors)" if errors else ""
133
+ skip_suffix = f" ({skipped} cached)" if skipped else ""
134
+ await on_progress(f"{total} images generated{skip_suffix}{err_suffix}", 100)
135
+
136
+ storyboard["_image_errors"] = errors
137
+ return storyboard
app/pipeline/stage_5_tts.py ADDED
@@ -0,0 +1,242 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 5: TTS Generation — generate voice clips for all dialogue/narration.
2
+
3
+ Dual-engine dispatch:
4
+ - Characters with voice_reference_path → Pocket TTS (anime voice cloning)
5
+ - Narration + characters without voice_reference_path → Edge TTS (fast, free)
6
+ """
7
+ import asyncio
8
+ import logging
9
+ from pathlib import Path
10
+ from app.models.character import Character
11
+ from app.services.tts import generate_tts_batch
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+ # Narrator voice: Zhongli via Pocket TTS (deep, wise, authoritative)
16
+ NARRATOR_VOICE_REF = "data/genshin_voices/zhongli/voice_state.safetensors"
17
+
18
+ # Edge TTS fallback for narrator (if Pocket TTS fails)
19
+ NARRATOR_VOICE = {
20
+ "voice_name": "en-US-ChristopherNeural",
21
+ "rate": "-5%",
22
+ "pitch": "-5Hz",
23
+ }
24
+
25
+
26
+ async def _generate_pocket_tts_items(items: list[dict]) -> list[dict]:
27
+ """Generate TTS clips using Pocket TTS (sequential, CPU-bound).
28
+
29
+ Pocket TTS is single-threaded internally so we process one at a time
30
+ via asyncio.to_thread() to avoid blocking the event loop.
31
+ """
32
+ from app.services.pocket_tts_service import PocketTTSService
33
+
34
+ results = []
35
+ for item in items:
36
+ try:
37
+ result = await PocketTTSService.generate(
38
+ text=item["text"],
39
+ voice_ref=item["voice_ref"],
40
+ output_path=item["output_path"],
41
+ )
42
+ result["shot_id"] = item.get("shot_id")
43
+ result["word_timestamps"] = [] # Pocket TTS doesn't provide word timing
44
+ results.append(result)
45
+ except Exception as e:
46
+ logger.warning(
47
+ "Pocket TTS failed for %s: %s — falling back to Edge TTS",
48
+ item.get("shot_id"), e,
49
+ )
50
+ # Fallback: use Edge TTS for this item
51
+ fallback_result = await _edge_tts_fallback(item)
52
+ results.append(fallback_result)
53
+
54
+ return results
55
+
56
+
57
+ async def _edge_tts_fallback(item: dict) -> dict:
58
+ """Generate a single clip via Edge TTS as fallback."""
59
+ from app.services.tts import generate_tts
60
+
61
+ # Change extension to .mp3 for Edge TTS
62
+ output_path = item["output_path"]
63
+ if output_path.endswith(".wav"):
64
+ output_path = output_path[:-4] + ".mp3"
65
+
66
+ vc = item.get("voice_config") or NARRATOR_VOICE
67
+ result = await generate_tts(
68
+ text=item["text"],
69
+ output_path=output_path,
70
+ voice_name=vc.get("voice_name", "en-US-ChristopherNeural"),
71
+ rate=vc.get("rate", "+0%"),
72
+ pitch=vc.get("pitch", "+0Hz"),
73
+ emotion=item.get("emotion", "neutral"),
74
+ )
75
+ result["shot_id"] = item.get("shot_id")
76
+ return result
77
+
78
+
79
+ async def run_tts_generation(
80
+ storyboard: dict,
81
+ characters: list[Character],
82
+ project_dir: str,
83
+ on_progress: callable = None,
84
+ resume: bool = False,
85
+ ) -> dict:
86
+ """Generate TTS for all cuts with dialogue/narration.
87
+
88
+ Dispatches to Pocket TTS (cloned anime voices) or Edge TTS based on
89
+ whether the character has a voice_reference_path set.
90
+
91
+ Returns updated storyboard with tts_path and actual_duration_sec set.
92
+ If resume=True, skip cuts whose audio files already exist on disk.
93
+ """
94
+ if on_progress:
95
+ await on_progress("Preparing voice clips...", 0)
96
+
97
+ # Build character maps
98
+ char_voices = {} # name -> Edge TTS voice config
99
+ char_voice_refs = {} # name -> Pocket TTS voice ref (safetensors path or preset name)
100
+ for c in characters:
101
+ if c.voice_reference_path:
102
+ ref = c.voice_reference_path
103
+ # Accept both file paths (if they exist) and preset names
104
+ if Path(ref).exists() or not ref.endswith((".safetensors", ".wav")):
105
+ char_voice_refs[c.name] = ref
106
+ if c.voice_config_json:
107
+ char_voices[c.name] = c.voice_config_json
108
+
109
+ audio_dir = Path(project_dir) / "audio"
110
+ audio_dir.mkdir(parents=True, exist_ok=True)
111
+
112
+ # Collect TTS tasks, split by engine
113
+ edge_items = []
114
+ pocket_items = []
115
+ skipped = 0
116
+
117
+ for scene in storyboard.get("scenes", []):
118
+ for cut in scene.get("cuts", scene.get("shots", [])):
119
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
120
+ dialogue = cut.get("dialogue")
121
+ if not dialogue or not dialogue.get("text"):
122
+ # Belt-and-suspenders: create narration from action_description
123
+ action = cut.get("action_description", "")
124
+ if not action:
125
+ continue # truly empty — skip
126
+ dialogue = {
127
+ "speaker": "Narrator",
128
+ "text": action,
129
+ "is_narration": True,
130
+ "emotion": "neutral",
131
+ "delivery": "normal",
132
+ }
133
+ cut["dialogue"] = dialogue
134
+
135
+ speaker = dialogue.get("speaker", "Narrator")
136
+ is_narration = dialogue.get("is_narration", False)
137
+
138
+ # Determine engine:
139
+ # - Narrator → Pocket TTS with Zhongli (fallback to Edge TTS)
140
+ # - Characters with voice refs → Pocket TTS
141
+ # - Others → Edge TTS
142
+ is_narrator_line = is_narration or speaker == "Narrator"
143
+ narrator_ref_exists = Path(NARRATOR_VOICE_REF).exists()
144
+ use_pocket = (
145
+ (is_narrator_line and narrator_ref_exists)
146
+ or (not is_narrator_line and speaker in char_voice_refs)
147
+ )
148
+
149
+ # File extension: .wav for Pocket TTS, .mp3 for Edge TTS
150
+ ext = ".wav" if use_pocket else ".mp3"
151
+ output_path = str(audio_dir / f"{cut_id}{ext}")
152
+
153
+ # Resume: skip if audio file already exists (check both extensions)
154
+ if resume:
155
+ wav_path = str(audio_dir / f"{cut_id}.wav")
156
+ mp3_path = str(audio_dir / f"{cut_id}.mp3")
157
+ if Path(wav_path).exists() or Path(mp3_path).exists():
158
+ cut["tts_path"] = wav_path if Path(wav_path).exists() else mp3_path
159
+ skipped += 1
160
+ continue
161
+
162
+ if use_pocket:
163
+ voice_ref = NARRATOR_VOICE_REF if is_narrator_line else char_voice_refs[speaker]
164
+ pocket_items.append({
165
+ "shot_id": cut_id,
166
+ "text": dialogue["text"],
167
+ "output_path": output_path,
168
+ "voice_ref": voice_ref,
169
+ "voice_config": char_voices.get(speaker, NARRATOR_VOICE),
170
+ "emotion": dialogue.get("emotion", "neutral"),
171
+ })
172
+ else:
173
+ voice_config = NARRATOR_VOICE if is_narrator_line else char_voices.get(speaker, NARRATOR_VOICE)
174
+ edge_items.append({
175
+ "shot_id": cut_id,
176
+ "text": dialogue["text"],
177
+ "output_path": output_path,
178
+ "voice_config": voice_config,
179
+ "emotion": dialogue.get("emotion", "neutral"),
180
+ })
181
+
182
+ total_items = len(edge_items) + len(pocket_items)
183
+ if total_items == 0:
184
+ if on_progress:
185
+ msg = "No dialogue found" if skipped == 0 else f"All {skipped} voice clips cached"
186
+ await on_progress(msg, 100)
187
+ return storyboard
188
+
189
+ if on_progress:
190
+ skip_msg = f" ({skipped} cached)" if skipped else ""
191
+ engine_msg = ""
192
+ if pocket_items and edge_items:
193
+ engine_msg = f" [{len(pocket_items)} anime + {len(edge_items)} standard]"
194
+ elif pocket_items:
195
+ engine_msg = f" [{len(pocket_items)} anime voices]"
196
+ await on_progress(f"Recording {total_items} voice clips{engine_msg}{skip_msg}...", 10)
197
+
198
+ # Generate clips — Pocket TTS (sequential) and Edge TTS (concurrent) in parallel
199
+ pocket_results = []
200
+ edge_results = []
201
+
202
+ if pocket_items and edge_items:
203
+ # Run both engines in parallel
204
+ pocket_results, edge_results = await asyncio.gather(
205
+ _generate_pocket_tts_items(pocket_items),
206
+ generate_tts_batch(edge_items),
207
+ )
208
+ elif pocket_items:
209
+ pocket_results = await _generate_pocket_tts_items(pocket_items)
210
+ else:
211
+ edge_results = await generate_tts_batch(edge_items)
212
+
213
+ all_results = pocket_results + edge_results
214
+
215
+ # Map results back to cuts
216
+ result_map = {r["shot_id"]: r for r in all_results if r.get("shot_id")}
217
+
218
+ for scene in storyboard.get("scenes", []):
219
+ for cut in scene.get("cuts", scene.get("shots", [])):
220
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
221
+ if cut_id in result_map:
222
+ r = result_map[cut_id]
223
+ cut["tts_path"] = r["path"]
224
+ cut["word_timestamps"] = r.get("word_timestamps", [])
225
+ # Audio-first: actual TTS duration overrides estimate
226
+ if r.get("duration_sec"):
227
+ cut["actual_duration_sec"] = r["duration_sec"]
228
+ cut["duration_sec"] = max(r["duration_sec"] + 0.3, cut.get("duration_sec", 2.0))
229
+
230
+ if on_progress:
231
+ pocket_count = len(pocket_results)
232
+ edge_count = len(edge_results)
233
+ detail = f"{len(all_results)} voice clips recorded"
234
+ if pocket_count > 0:
235
+ detail += f" ({pocket_count} anime, {edge_count} standard)"
236
+ await on_progress(detail, 100)
237
+
238
+ logger.info(
239
+ "TTS complete: %d pocket + %d edge = %d total (%d cached)",
240
+ len(pocket_results), len(edge_results), len(all_results), skipped,
241
+ )
242
+ return storyboard
app/pipeline/stage_6_animation.py ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 6 (Fallback): Ken Burns Animation on each cut image.
2
+
3
+ This is the fallback animation method when Pollinations video generation
4
+ is unavailable or fails. Applies Ken Burns zoom/pan effects to static images.
5
+ """
6
+ import asyncio
7
+ import logging
8
+ from pathlib import Path
9
+ from app.services.ffmpeg import animate_shot
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ # Map professional camera notation to Ken Burns presets
14
+ CAMERA_TO_KEN_BURNS = {
15
+ "FIX": "static",
16
+ "PAN_LEFT": "pan_left",
17
+ "PAN_RIGHT": "pan_right",
18
+ "TILT_UP": "pan_up",
19
+ "TILT_DOWN": "pan_down",
20
+ "T.U.": "zoom_in_slow",
21
+ "T.B.": "zoom_out",
22
+ "DOLLY_LEFT": "pan_left",
23
+ "DOLLY_RIGHT": "pan_right",
24
+ "FOLLOW": "pan_right",
25
+ "CRANE_UP": "pan_up",
26
+ "CRANE_DOWN": "pan_down",
27
+ "SHAKE": "shake",
28
+ }
29
+
30
+
31
+ async def run_animation(
32
+ storyboard: dict,
33
+ project_dir: str,
34
+ on_progress: callable = None,
35
+ ) -> dict:
36
+ """Apply Ken Burns animation to all cut images.
37
+
38
+ Returns updated storyboard with clip_path set on each cut.
39
+ Supports both new 'cuts' and legacy 'shots' format.
40
+ """
41
+ total_cuts = sum(
42
+ len(s.get("cuts", s.get("shots", [])))
43
+ for s in storyboard.get("scenes", [])
44
+ )
45
+ if on_progress:
46
+ await on_progress(f"Animating {total_cuts} cuts (Ken Burns)...", 0)
47
+
48
+ clips_dir = Path(project_dir) / "clips"
49
+ clips_dir.mkdir(parents=True, exist_ok=True)
50
+
51
+ # Collect cuts with images, annotating scene position for fades
52
+ cuts = []
53
+ for scene in storyboard.get("scenes", []):
54
+ scene_cuts = [
55
+ c for c in scene.get("cuts", scene.get("shots", []))
56
+ if c.get("image_path")
57
+ ]
58
+ for idx, cut in enumerate(scene_cuts):
59
+ is_first = idx == 0
60
+ is_last = idx == len(scene_cuts) - 1
61
+ if is_first and is_last:
62
+ cut["_fade_in"] = 0.4
63
+ cut["_fade_out"] = 0.4
64
+ elif is_first:
65
+ cut["_fade_in"] = 0.4
66
+ cut["_fade_out"] = 0.15
67
+ elif is_last:
68
+ cut["_fade_in"] = 0.15
69
+ cut["_fade_out"] = 0.4
70
+ else:
71
+ cut["_fade_in"] = 0.15
72
+ cut["_fade_out"] = 0.15
73
+ cuts.append(cut)
74
+
75
+ total = len(cuts)
76
+ logger.info("Animation: %d/%d cuts have image_path", total, total_cuts)
77
+ completed = 0
78
+ semaphore = asyncio.Semaphore(2) # Limit concurrent FFmpeg processes
79
+
80
+ async def animate_one(cut: dict) -> None:
81
+ nonlocal completed
82
+ async with semaphore:
83
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
84
+ output = str(clips_dir / f"{cut_id}.mp4")
85
+
86
+ # Use actual TTS duration if available (audio-first)
87
+ duration = cut.get("actual_duration_sec", cut.get("duration_sec", 3.0))
88
+ duration = max(duration, 0.5)
89
+
90
+ # Map professional camera notation to Ken Burns presets
91
+ camera = cut.get("camera_movement", "static")
92
+ kb_camera = CAMERA_TO_KEN_BURNS.get(camera, camera)
93
+
94
+ try:
95
+ await animate_shot(
96
+ image_path=cut["image_path"],
97
+ output_path=output,
98
+ duration_sec=duration,
99
+ camera_movement=kb_camera,
100
+ audio_path=cut.get("tts_path"),
101
+ fade_in=cut.get("_fade_in", 0.0),
102
+ fade_out=cut.get("_fade_out", 0.0),
103
+ )
104
+ cut["clip_path"] = output
105
+ except Exception as e:
106
+ cut["clip_path"] = None
107
+ cut["_animation_error"] = str(e)
108
+ logger.error("Animation failed for %s: %s", cut_id, e)
109
+
110
+ completed += 1
111
+ if on_progress:
112
+ pct = int((completed / max(total, 1)) * 100)
113
+ dur = cut.get("duration_sec", 0)
114
+ await on_progress(f"[{completed}/{total}] {kb_camera} ({dur:.1f}s)", pct)
115
+
116
+ await asyncio.gather(*[animate_one(c) for c in cuts])
117
+
118
+ if on_progress:
119
+ await on_progress(f"{total} cuts animated", 100)
120
+
121
+ return storyboard
app/pipeline/stage_6_video_gen.py ADDED
@@ -0,0 +1,304 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 6 (Video): AI Video Generation via Pollinations grok-video.
2
+
3
+ Replaces Ken Burns animation with actual AI-generated video clips.
4
+ Each keyframe image is uploaded and used as img2vid reference.
5
+
6
+ Falls back to Ken Burns (stage_6_animation.py) on failure.
7
+ """
8
+ import asyncio
9
+ import logging
10
+ import math
11
+ import os
12
+ from pathlib import Path
13
+
14
+ from app.services.pollinations import generate_video, upload_media
15
+ from app.pipeline.stage_6_animation import run_animation as run_ken_burns_fallback
16
+ from app.services.ffmpeg import get_duration, mux_audio, rescale_to_landscape
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ # Sequential video gen — Pollinations throttles concurrent pk_* key requests
21
+ _video_semaphore = asyncio.Semaphore(1)
22
+
23
+ # Delay between video gen requests to avoid throttling
24
+ _VIDEO_DELAY_SEC = 6.0
25
+
26
+ # Camera movement → video prompt mapping for enhanced motion descriptions
27
+ CAMERA_MOTION_MAP = {
28
+ "FIX": "static camera, subtle atmospheric motion",
29
+ "PAN_LEFT": "camera pans smoothly to the left",
30
+ "PAN_RIGHT": "camera pans smoothly to the right",
31
+ "TILT_UP": "camera tilts upward slowly",
32
+ "TILT_DOWN": "camera tilts downward slowly",
33
+ "T.U.": "camera slowly zooms in, tracking toward subject",
34
+ "T.B.": "camera slowly pulls back, revealing more of the scene",
35
+ "DOLLY_LEFT": "camera moves laterally to the left",
36
+ "DOLLY_RIGHT": "camera moves laterally to the right",
37
+ "FOLLOW": "camera follows the character's movement",
38
+ "CRANE_UP": "camera rises upward, bird's eye perspective",
39
+ "CRANE_DOWN": "camera descends from above",
40
+ "SHAKE": "handheld camera shake, intense impact feel",
41
+ # Legacy camera movements (from old storyboard format)
42
+ "static": "subtle idle motion, breathing animation",
43
+ "zoom_in_slow": "camera slowly zooms in",
44
+ "zoom_in_fast": "camera quickly zooms in with urgency",
45
+ "zoom_out": "camera zooms out, revealing the scene",
46
+ "pan_left": "camera pans smoothly to the left",
47
+ "pan_right": "camera pans smoothly to the right",
48
+ "pan_up": "camera tilts upward",
49
+ "pan_down": "camera tilts downward",
50
+ "tilt": "camera tilts with slight rotation",
51
+ "shake": "handheld camera shake, impact",
52
+ "dolly_in": "camera tracks toward subject",
53
+ }
54
+
55
+
56
+ ANIME_VIDEO_PREFIX = "anime cel-shading style, 2D manhwa animation, "
57
+
58
+
59
+ def _build_video_prompt(cut: dict) -> str:
60
+ """Build a video generation prompt from cut data.
61
+
62
+ Combines the explicit video_prompt (from e-konte storyboard) with
63
+ camera movement description and action description.
64
+ Prepends anime style cue to keep grok-video from going realistic.
65
+ """
66
+ parts = []
67
+
68
+ # Use explicit video_prompt if available (new e-konte format)
69
+ video_prompt = cut.get("video_prompt", "")
70
+ if video_prompt:
71
+ parts.append(video_prompt)
72
+
73
+ # Add camera movement context
74
+ camera = cut.get("camera_movement", "static")
75
+ camera_desc = CAMERA_MOTION_MAP.get(camera, "subtle camera motion")
76
+ if camera_desc not in (video_prompt or ""):
77
+ parts.append(camera_desc)
78
+
79
+ # Add action description for additional motion context
80
+ action = cut.get("action_description", "")
81
+ if action and action not in (video_prompt or ""):
82
+ parts.append(action)
83
+
84
+ # If still empty, create a basic motion prompt
85
+ if not parts:
86
+ parts.append("subtle idle animation, slight breathing motion, atmospheric particles floating")
87
+
88
+ return ANIME_VIDEO_PREFIX + ", ".join(parts)
89
+
90
+
91
+ async def _generate_video_clip(
92
+ cut: dict,
93
+ clips_dir: Path,
94
+ image_upload_cache: dict,
95
+ ) -> bool:
96
+ """Generate a video clip for a single cut. Returns True on success."""
97
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
98
+ image_path = cut.get("image_path")
99
+ output_path = str(clips_dir / f"{cut_id}.mp4")
100
+
101
+ if not image_path or not Path(image_path).exists():
102
+ logger.warning("No image for %s, skipping video gen", cut_id)
103
+ return False
104
+
105
+ # Duration: round UP so video is always >= audio length.
106
+ # grok-video accepts integer seconds 1-10. Rounding up ensures
107
+ # -shortest in mux_audio trims to the exact audio duration.
108
+ tts_duration = cut.get("actual_duration_sec")
109
+ storyboard_duration = cut.get("duration_sec", 3.0)
110
+ target_duration = tts_duration or storyboard_duration
111
+ video_duration = max(1, min(math.ceil(target_duration), 10))
112
+
113
+ async with _video_semaphore:
114
+ # Rate-limit delay for pk_* keys
115
+ await asyncio.sleep(_VIDEO_DELAY_SEC)
116
+
117
+ try:
118
+ # Upload image (with caching to avoid re-uploading)
119
+ if image_path not in image_upload_cache:
120
+ image_url = await upload_media(image_path)
121
+ image_upload_cache[image_path] = image_url
122
+ else:
123
+ image_url = image_upload_cache[image_path]
124
+
125
+ # Build video prompt
126
+ video_prompt = _build_video_prompt(cut)
127
+
128
+ # Step 1: Generate silent video from grok-video
129
+ silent_path = str(clips_dir / f"{cut_id}_silent.mp4")
130
+ await generate_video(
131
+ prompt=video_prompt,
132
+ output_path=silent_path,
133
+ model="grok-video",
134
+ duration=video_duration,
135
+ image_url=image_url,
136
+ )
137
+
138
+ # Step 2: Rescale to landscape (grok-video outputs portrait)
139
+ scaled_path = str(clips_dir / f"{cut_id}_scaled.mp4")
140
+ await rescale_to_landscape(silent_path, scaled_path, 1024, 768)
141
+
142
+ # Step 3: Mux TTS audio into the video (or add silent track)
143
+ tts_path = cut.get("tts_path")
144
+ await mux_audio(
145
+ video_path=scaled_path,
146
+ audio_path=tts_path,
147
+ output_path=output_path,
148
+ duration_sec=target_duration,
149
+ )
150
+
151
+ # Clean up temp files
152
+ for tmp in [silent_path, scaled_path]:
153
+ try:
154
+ os.unlink(tmp)
155
+ except OSError:
156
+ pass
157
+
158
+ cut["clip_path"] = output_path
159
+ logger.info(
160
+ "Video generated for %s: %s (video=%ds, audio=%.1fs)",
161
+ cut_id, output_path, video_duration,
162
+ tts_duration or 0,
163
+ )
164
+ return True
165
+
166
+ except Exception as e:
167
+ logger.error("Video generation failed for %s: %r", cut_id, e, exc_info=True)
168
+ cut["_video_error"] = repr(e)
169
+ # Clean up partial files
170
+ for p in [str(clips_dir / f"{cut_id}_silent.mp4"),
171
+ str(clips_dir / f"{cut_id}_scaled.mp4"),
172
+ output_path]:
173
+ try:
174
+ os.unlink(p)
175
+ except OSError:
176
+ pass
177
+ return False
178
+
179
+
180
+ async def run_video_generation(
181
+ storyboard: dict,
182
+ project_dir: str,
183
+ on_progress: callable = None,
184
+ ) -> dict:
185
+ """Generate AI video clips for all cuts in the storyboard.
186
+
187
+ Falls back to Ken Burns animation for any cuts where video generation fails.
188
+ Returns updated storyboard with clip_path set on each cut.
189
+ """
190
+ clips_dir = Path(project_dir) / "clips"
191
+ clips_dir.mkdir(parents=True, exist_ok=True)
192
+
193
+ # Collect all cuts that have images
194
+ all_cuts = []
195
+ for scene in storyboard.get("scenes", []):
196
+ for cut in scene.get("cuts", scene.get("shots", [])):
197
+ if cut.get("image_path"):
198
+ all_cuts.append(cut)
199
+
200
+ total = len(all_cuts)
201
+ if on_progress:
202
+ await on_progress(f"Generating {total} video clips...", 0)
203
+
204
+ if total == 0:
205
+ if on_progress:
206
+ await on_progress("No images available for video generation", 100)
207
+ return storyboard
208
+
209
+ # Shared upload cache to avoid re-uploading the same image
210
+ image_upload_cache = {}
211
+ completed = 0
212
+ succeeded = 0
213
+ failed_cuts = []
214
+
215
+ async def gen_one(cut: dict):
216
+ nonlocal completed, succeeded
217
+ try:
218
+ success = await _generate_video_clip(cut, clips_dir, image_upload_cache)
219
+ except Exception as e:
220
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
221
+ logger.error("Unexpected error in video gen for %s: %s", cut_id, e)
222
+ success = False
223
+ if success:
224
+ succeeded += 1
225
+ else:
226
+ failed_cuts.append(cut)
227
+ completed += 1
228
+ if on_progress:
229
+ pct = int((completed / max(total, 1)) * 80)
230
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
231
+ status = "OK" if success else "FALLBACK"
232
+ await on_progress(f"[{completed}/{total}] {cut_id} ({status})", pct)
233
+
234
+ # Generate videos (limited by semaphore)
235
+ await asyncio.gather(*[gen_one(cut) for cut in all_cuts])
236
+
237
+ # Fallback: Ken Burns for failed cuts
238
+ if failed_cuts:
239
+ if on_progress:
240
+ await on_progress(f"Ken Burns fallback for {len(failed_cuts)} clips...", 85)
241
+ logger.info("Falling back to Ken Burns for %d cuts", len(failed_cuts))
242
+
243
+ # Create a mini-storyboard with just the failed cuts for Ken Burns
244
+ # We need to set up the fade values and pass to the existing Ken Burns code
245
+ from app.services.ffmpeg import animate_shot
246
+ kb_completed = 0
247
+
248
+ for cut in failed_cuts:
249
+ cut_id = cut.get("cut_id") or cut.get("shot_id")
250
+ output = str(clips_dir / f"{cut_id}.mp4")
251
+ duration = cut.get("actual_duration_sec", cut.get("duration_sec", 3.0))
252
+ duration = max(duration, 0.5)
253
+
254
+ # Map new camera notation back to Ken Burns presets
255
+ camera = cut.get("camera_movement", "static")
256
+ kb_camera = _map_camera_to_ken_burns(camera)
257
+
258
+ try:
259
+ await animate_shot(
260
+ image_path=cut["image_path"],
261
+ output_path=output,
262
+ duration_sec=duration,
263
+ camera_movement=kb_camera,
264
+ audio_path=cut.get("tts_path"),
265
+ fade_in=0.15,
266
+ fade_out=0.15,
267
+ )
268
+ cut["clip_path"] = output
269
+ cut["_used_fallback"] = True
270
+ except Exception as e:
271
+ logger.error("Ken Burns fallback also failed for %s: %r", cut_id, e, exc_info=True)
272
+ cut["clip_path"] = None
273
+
274
+ kb_completed += 1
275
+
276
+ if on_progress:
277
+ fallback_count = sum(1 for c in failed_cuts if c.get("clip_path"))
278
+ total_success = succeeded + fallback_count
279
+ await on_progress(
280
+ f"{total_success}/{total} clips ready ({succeeded} video, {fallback_count} Ken Burns)",
281
+ 100,
282
+ )
283
+
284
+ return storyboard
285
+
286
+
287
+ def _map_camera_to_ken_burns(camera: str) -> str:
288
+ """Map professional camera notation to Ken Burns preset names."""
289
+ mapping = {
290
+ "FIX": "static",
291
+ "PAN_LEFT": "pan_left",
292
+ "PAN_RIGHT": "pan_right",
293
+ "TILT_UP": "pan_up",
294
+ "TILT_DOWN": "pan_down",
295
+ "T.U.": "zoom_in_slow",
296
+ "T.B.": "zoom_out",
297
+ "DOLLY_LEFT": "pan_left",
298
+ "DOLLY_RIGHT": "pan_right",
299
+ "FOLLOW": "pan_right",
300
+ "CRANE_UP": "pan_up",
301
+ "CRANE_DOWN": "pan_down",
302
+ "SHAKE": "shake",
303
+ }
304
+ return mapping.get(camera, camera) # Pass through if already a Ken Burns name
app/pipeline/stage_7_assembly.py ADDED
@@ -0,0 +1,278 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 7: Video Assembly — concat clips, scene-aware transitions, subtitles, BGM."""
2
+ import logging
3
+ import os
4
+ from pathlib import Path
5
+ from app.services.ffmpeg import concat_clips, add_subtitles, mix_background_music, get_duration
6
+
7
+ logger = logging.getLogger(__name__)
8
+
9
+ # Map storyboard mood keywords to BGM filenames
10
+ BGM_MOOD_MAP = {
11
+ "tense": "tense",
12
+ "mysterious": "tense",
13
+ "action": "epic",
14
+ "dramatic": "epic",
15
+ "epic": "epic",
16
+ "calm": "calm",
17
+ "romantic": "calm",
18
+ "comedic": "calm",
19
+ "melancholic": "sad",
20
+ "sad": "sad",
21
+ }
22
+ BGM_DIR = Path("data/bgm")
23
+
24
+
25
+ # Overlap durations for xfade transitions (seconds eaten from timeline)
26
+ TRANSITION_OVERLAP = {
27
+ "cut": 0.0,
28
+ "fade": 1.0,
29
+ "fade_black": 1.0,
30
+ "crossfade": 0.5,
31
+ }
32
+
33
+
34
+ def _generate_srt(
35
+ storyboard: dict,
36
+ clip_durations: list[float],
37
+ transitions: list[str],
38
+ ) -> str:
39
+ """Generate SRT subtitle file using actual clip durations from ffprobe.
40
+
41
+ Uses measured clip durations instead of storyboard estimates, and accounts
42
+ for transition overlaps (xfade) that compress the timeline.
43
+ """
44
+ # Collect cuts that have valid clip_paths (same order as clip_durations)
45
+ cuts_with_clips = []
46
+ for scene in storyboard.get("scenes", []):
47
+ for cut in scene.get("cuts", scene.get("shots", [])):
48
+ clip_path = cut.get("clip_path")
49
+ if clip_path and os.path.exists(clip_path):
50
+ cuts_with_clips.append(cut)
51
+
52
+ lines = []
53
+ srt_idx = 1
54
+ elapsed = 0.0
55
+ prev_duration = 0.0
56
+
57
+ for i, (cut, duration) in enumerate(zip(cuts_with_clips, clip_durations)):
58
+ # Advance elapsed time, accounting for transition overlap
59
+ if i > 0:
60
+ overlap = TRANSITION_OVERLAP.get(transitions[i], 0.0) if i < len(transitions) else 0.0
61
+ elapsed += prev_duration - overlap
62
+
63
+ dialogue = cut.get("dialogue")
64
+ if dialogue and dialogue.get("text"):
65
+ start = elapsed
66
+ end = elapsed + duration
67
+ lines.append(str(srt_idx))
68
+ lines.append(f"{_srt_time(start)} --> {_srt_time(end)}")
69
+
70
+ prefix = ""
71
+ if dialogue.get("is_inner_thought"):
72
+ prefix = "\u266a " # Musical note for inner thoughts
73
+ elif dialogue.get("is_narration"):
74
+ prefix = ""
75
+ elif dialogue.get("speaker") and dialogue["speaker"] != "Narrator":
76
+ prefix = f"{dialogue['speaker']}: "
77
+
78
+ text = dialogue["text"]
79
+ # Wrap long lines
80
+ if len(text) > 60:
81
+ mid = len(text) // 2
82
+ space_pos = text.rfind(" ", 0, mid + 10)
83
+ if space_pos > mid - 15:
84
+ text = text[:space_pos] + "\n" + text[space_pos + 1:]
85
+
86
+ lines.append(f"{prefix}{text}")
87
+ lines.append("")
88
+ srt_idx += 1
89
+
90
+ prev_duration = duration
91
+
92
+ return "\n".join(lines)
93
+
94
+
95
+ def _srt_time(seconds: float) -> str:
96
+ """Convert seconds to SRT timestamp format."""
97
+ h = int(seconds // 3600)
98
+ m = int((seconds % 3600) // 60)
99
+ s = int(seconds % 60)
100
+ ms = int((seconds % 1) * 1000)
101
+ return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
102
+
103
+
104
+ def _build_transition_list(storyboard: dict) -> list[str]:
105
+ """Build a list of transitions for each clip boundary.
106
+
107
+ Scene-aware: use the scene's transition_in at scene boundaries,
108
+ 'cut' within scenes for continuous feel.
109
+ """
110
+ transitions = []
111
+ prev_scene_id = None
112
+
113
+ for scene in storyboard.get("scenes", []):
114
+ scene_id = scene.get("scene_id", "")
115
+ scene_transition = scene.get("transition_in", "cut")
116
+ cuts = scene.get("cuts", scene.get("shots", []))
117
+
118
+ for idx, cut in enumerate(cuts):
119
+ if not cut.get("clip_path") or not os.path.exists(cut.get("clip_path", "")):
120
+ continue
121
+
122
+ if idx == 0 and prev_scene_id is not None:
123
+ # Scene boundary — use the scene's transition_in
124
+ transitions.append(scene_transition)
125
+ else:
126
+ # Within scene — use cut's transition_out or default to "cut"
127
+ if idx > 0:
128
+ prev_cut = cuts[idx - 1]
129
+ trans = prev_cut.get("transition_out", "cut")
130
+ transitions.append(trans)
131
+ else:
132
+ transitions.append("cut") # First clip overall
133
+
134
+ prev_scene_id = scene_id
135
+
136
+ return transitions
137
+
138
+
139
+ async def run_assembly(
140
+ storyboard: dict,
141
+ project_dir: str,
142
+ episode_number: int = 1,
143
+ on_progress: callable = None,
144
+ output_dir: str | None = None,
145
+ ) -> dict:
146
+ """Assemble all animated clips into a final episode video.
147
+
148
+ Args:
149
+ project_dir: Directory containing clips (episode_dir in per-episode mode).
150
+ output_dir: If set, final episode goes to output_dir/episodes/ instead.
151
+
152
+ Returns {video_path, duration_seconds, subtitle_path}.
153
+ """
154
+ if on_progress:
155
+ await on_progress("Assembling episode...", 0)
156
+
157
+ episode_dir = Path(output_dir or project_dir) / "episodes"
158
+ episode_dir.mkdir(parents=True, exist_ok=True)
159
+
160
+ # Collect clip paths in order
161
+ clip_paths = []
162
+ for scene in storyboard.get("scenes", []):
163
+ for cut in scene.get("cuts", scene.get("shots", [])):
164
+ clip_path = cut.get("clip_path")
165
+ if clip_path and os.path.exists(clip_path):
166
+ clip_paths.append(clip_path)
167
+
168
+ if not clip_paths:
169
+ raise ValueError("No animated clips to assemble")
170
+
171
+ if on_progress:
172
+ await on_progress(f"Concatenating {len(clip_paths)} clips...", 20)
173
+
174
+ # Build scene-aware transitions
175
+ transitions = _build_transition_list(storyboard)
176
+
177
+ # Measure actual clip durations via ffprobe (ground truth)
178
+ clip_durations = []
179
+ for path in clip_paths:
180
+ dur = await get_duration(path)
181
+ clip_durations.append(dur if dur > 0 else 2.0)
182
+ logger.info("Clip durations (ffprobe): %s", [round(d, 2) for d in clip_durations])
183
+
184
+ # Generate subtitles using actual durations + transition overlaps
185
+ srt_content = _generate_srt(storyboard, clip_durations, transitions)
186
+ srt_path = str(episode_dir / f"episode_{episode_number}.srt")
187
+ with open(srt_path, "w", encoding="utf-8") as f:
188
+ f.write(srt_content)
189
+
190
+ # Concat all clips with scene-aware transitions
191
+ raw_output = str(episode_dir / f"episode_{episode_number}_raw.mp4")
192
+ await concat_clips(clip_paths, raw_output, transitions=transitions)
193
+
194
+ if on_progress:
195
+ await on_progress("Burning subtitles...", 50)
196
+
197
+ # Burn subtitles into video
198
+ subbed_output = str(episode_dir / f"episode_{episode_number}_subbed.mp4")
199
+ try:
200
+ await add_subtitles(raw_output, srt_path, subbed_output)
201
+ # Clean up raw
202
+ if os.path.exists(raw_output):
203
+ os.unlink(raw_output)
204
+ except Exception as e:
205
+ logger.warning("Subtitle burning failed, using raw video: %s", e)
206
+ subbed_output = raw_output # fallback — no crash
207
+
208
+ if on_progress:
209
+ await on_progress("Adding background music...", 70)
210
+
211
+ # Select BGM based on dominant mood
212
+ final_output = str(episode_dir / f"episode_{episode_number}.mp4")
213
+ bgm_path = _select_bgm(storyboard)
214
+
215
+ if bgm_path:
216
+ try:
217
+ await mix_background_music(subbed_output, bgm_path, final_output)
218
+ # Clean up subbed
219
+ if os.path.exists(subbed_output) and subbed_output != final_output:
220
+ os.unlink(subbed_output)
221
+ except Exception as e:
222
+ logger.warning("BGM mixing failed, using video without BGM: %s", e)
223
+ if subbed_output != final_output:
224
+ os.rename(subbed_output, final_output)
225
+ else:
226
+ if subbed_output != final_output:
227
+ os.rename(subbed_output, final_output)
228
+
229
+ # Get final duration
230
+ duration = await get_duration(final_output)
231
+
232
+ if on_progress:
233
+ mins = int(duration // 60)
234
+ secs = int(duration % 60)
235
+ await on_progress(f"Episode assembled \u2014 {mins}:{secs:02d}", 100)
236
+
237
+ return {
238
+ "video_path": final_output,
239
+ "subtitle_path": srt_path,
240
+ "duration_seconds": duration,
241
+ }
242
+
243
+
244
+ def _select_bgm(storyboard: dict) -> str | None:
245
+ """Select a BGM file based on the dominant mood in the storyboard."""
246
+ if not BGM_DIR.exists():
247
+ return None
248
+
249
+ # Count mood occurrences
250
+ mood_counts: dict[str, int] = {}
251
+ for scene in storyboard.get("scenes", []):
252
+ mood = scene.get("mood", "").lower()
253
+ bgm_key = BGM_MOOD_MAP.get(mood)
254
+ if bgm_key:
255
+ mood_counts[bgm_key] = mood_counts.get(bgm_key, 0) + 1
256
+
257
+ if not mood_counts:
258
+ # Default to calm
259
+ mood_counts = {"calm": 1}
260
+
261
+ # Pick the most common mood
262
+ dominant = max(mood_counts, key=mood_counts.get)
263
+
264
+ # Look for matching BGM file
265
+ for ext in (".mp3", ".wav", ".ogg", ".m4a"):
266
+ candidate = BGM_DIR / f"{dominant}{ext}"
267
+ if candidate.exists():
268
+ return str(candidate)
269
+
270
+ # Fallback: try any available BGM file
271
+ try:
272
+ for f in BGM_DIR.iterdir():
273
+ if f.suffix in (".mp3", ".wav", ".ogg", ".m4a"):
274
+ return str(f)
275
+ except Exception:
276
+ pass
277
+
278
+ return None
app/pipeline/stage_8_critic.py ADDED
@@ -0,0 +1,189 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Stage 8: Critic Agent — AI review of generated storyboard/images."""
2
+ import logging
3
+ from pathlib import Path
4
+ from app.services.llm import call_gemini_json, call_openrouter_vision_json
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ CRITIC_PROMPT = """You are an anime quality assurance critic. Review this storyboard for an anime episode and score it.
9
+
10
+ STORYBOARD METADATA:
11
+ - Total shots: {total_shots}
12
+ - Total duration: {total_duration}s
13
+ - Scenes: {scene_count}
14
+
15
+ SCENE BREAKDOWN:
16
+ {scene_summary}
17
+
18
+ Score each dimension from 0.0 to 10.0 and provide specific feedback.
19
+
20
+ Return JSON:
21
+ {{
22
+ "visual_consistency_score": float, // Are character descriptions consistent across shots?
23
+ "pacing_score": float, // Is the pacing good? Not too fast or slow?
24
+ "audio_sync_score": float, // Do dialogue durations match the content?
25
+ "overall_score": float, // Overall quality
26
+ "feedback": {{
27
+ "strengths": ["list of good aspects"],
28
+ "issues": [
29
+ {{"severity": "low/medium/high", "description": "what's wrong", "shot_id": "affected shot or null", "suggestion": "how to fix"}}
30
+ ]
31
+ }}
32
+ }}"""
33
+
34
+ IMAGE_CRITIC_PROMPT = """You are an anime visual quality critic. You are reviewing {image_count} frames from an AI-generated anime episode.
35
+
36
+ Evaluate these images and return a JSON score card:
37
+
38
+ {{
39
+ "art_style_consistency": {{
40
+ "score": float (0-10),
41
+ "notes": "Are all images in a consistent anime art style?"
42
+ }},
43
+ "character_consistency": {{
44
+ "score": float (0-10),
45
+ "notes": "Do characters look the same across different shots?"
46
+ }},
47
+ "quality": {{
48
+ "score": float (0-10),
49
+ "notes": "Are there artifacts, weird anatomy, broken faces, extra fingers?"
50
+ }},
51
+ "anime_adherence": {{
52
+ "score": float (0-10),
53
+ "notes": "Does it look like anime vs photorealistic or other styles?"
54
+ }},
55
+ "overall_visual_score": float (0-10),
56
+ "visual_issues": [
57
+ {{"description": "what's wrong", "severity": "low/medium/high"}}
58
+ ]
59
+ }}"""
60
+
61
+
62
+ def _build_scene_summary(storyboard: dict) -> str:
63
+ """Build a text summary of the storyboard for the critic."""
64
+ lines = []
65
+ for scene in storyboard.get("scenes", []):
66
+ scene_id = scene.get("scene_id", "?")
67
+ lines.append(f"\nScene {scene_id}: {scene.get('setting', 'unknown')} ({scene.get('mood', 'neutral')})")
68
+ for cut in scene.get("cuts", scene.get("shots", [])):
69
+ cut_id = cut.get("cut_id") or cut.get("shot_id", "?")
70
+ st = cut.get("shot_type", "?")
71
+ dur = cut.get("duration_sec", 0)
72
+ chars = ", ".join(cut.get("characters_present", []))
73
+ dialogue = cut.get("dialogue") or {}
74
+ text_preview = (dialogue.get("text", "")[:50] + "...") if dialogue.get("text") else "no dialogue"
75
+ has_image = "yes" if cut.get("image_path") else "NO"
76
+ beat = cut.get("emotional_beat", "")
77
+ beat_str = f" [{beat}]" if beat else ""
78
+ lines.append(f" {cut_id}: {st}, {dur:.1f}s, chars=[{chars}], img={has_image}{beat_str}, \"{text_preview}\"")
79
+ return "\n".join(lines)
80
+
81
+
82
+ def _collect_image_paths(storyboard: dict, project_dir: str) -> list[str]:
83
+ """Collect all existing image paths from storyboard cuts."""
84
+ paths = []
85
+ for scene in storyboard.get("scenes", []):
86
+ for cut in scene.get("cuts", scene.get("shots", [])):
87
+ img_path = cut.get("image_path")
88
+ if img_path:
89
+ p = Path(img_path)
90
+ if p.exists():
91
+ paths.append(str(p.resolve()))
92
+ else:
93
+ # Try relative to project_dir
94
+ full = Path(project_dir) / img_path
95
+ if full.exists():
96
+ paths.append(str(full.resolve()))
97
+ return paths
98
+
99
+
100
+ def _sample_images(paths: list[str], max_count: int = 6) -> list[str]:
101
+ """Evenly sample up to max_count images from the list."""
102
+ if len(paths) <= max_count:
103
+ return paths
104
+ step = len(paths) / max_count
105
+ return [paths[int(i * step)] for i in range(max_count)]
106
+
107
+
108
+ async def _critique_images(storyboard: dict, project_dir: str) -> dict | None:
109
+ """Run visual critique on generated images via OpenRouter vision."""
110
+ all_images = _collect_image_paths(storyboard, project_dir)
111
+ if not all_images:
112
+ logger.info("No images found for visual critique")
113
+ return None
114
+
115
+ sampled = _sample_images(all_images)
116
+ logger.info("Visual critique: %d/%d images sampled", len(sampled), len(all_images))
117
+
118
+ prompt = IMAGE_CRITIC_PROMPT.format(image_count=len(sampled))
119
+ result = await call_openrouter_vision_json(
120
+ prompt,
121
+ sampled,
122
+ system_prompt="You are a professional anime art director reviewing AI-generated frames.",
123
+ )
124
+
125
+ if not isinstance(result, dict):
126
+ return None
127
+
128
+ return result
129
+
130
+
131
+ async def run_critic(
132
+ storyboard: dict,
133
+ project_dir: str,
134
+ on_progress: callable = None,
135
+ ) -> dict:
136
+ """Run AI critic on the storyboard. Returns critic scores and feedback."""
137
+ if on_progress:
138
+ await on_progress("AI critic reviewing the episode...", 0)
139
+
140
+ metadata = storyboard.get("metadata", {})
141
+ scene_summary = _build_scene_summary(storyboard)
142
+
143
+ prompt = CRITIC_PROMPT.format(
144
+ total_shots=metadata.get("total_shots", 0),
145
+ total_duration=metadata.get("total_duration_estimate", 0),
146
+ scene_count=metadata.get("scene_count", 0),
147
+ scene_summary=scene_summary[:5000],
148
+ )
149
+
150
+ if on_progress:
151
+ await on_progress("Checking storyboard quality...", 20)
152
+
153
+ result = await call_gemini_json(prompt)
154
+
155
+ # Type-check the result
156
+ if not isinstance(result, dict):
157
+ result = {"overall_score": 5.0, "feedback": {"strengths": [], "issues": []}}
158
+
159
+ score = result.get("overall_score", 0)
160
+ if isinstance(score, (int, float)):
161
+ score = max(0, min(10, float(score)))
162
+ else:
163
+ score = 5.0
164
+ result["overall_score"] = score
165
+
166
+ # Image critique via OpenRouter vision (non-blocking on failure)
167
+ if on_progress:
168
+ await on_progress("Analyzing generated images...", 50)
169
+
170
+ try:
171
+ image_critique = await _critique_images(storyboard, project_dir)
172
+ if image_critique:
173
+ result["image_critique"] = image_critique
174
+ vis_score = image_critique.get("overall_visual_score")
175
+ if isinstance(vis_score, (int, float)):
176
+ vis_score = max(0, min(10, float(vis_score)))
177
+ result["visual_quality_score"] = vis_score
178
+ # Blend into overall: 60% storyboard, 40% visual
179
+ result["overall_score"] = round(score * 0.6 + vis_score * 0.4, 1)
180
+ score = result["overall_score"]
181
+ logger.info("Image critique complete: visual_score=%.1f", result.get("visual_quality_score", 0))
182
+ except Exception as e:
183
+ logger.warning("Image critique failed (non-fatal): %s", e)
184
+ result["image_critique_error"] = str(e)
185
+
186
+ if on_progress:
187
+ await on_progress(f"Quality score: {score}/10", 100)
188
+
189
+ return result
app/schemas/__init__.py ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from app.schemas.auth import (
2
+ UserCreate, UserLogin, UserResponse, TokenResponse, RefreshRequest,
3
+ )
4
+ from app.schemas.project import (
5
+ ProjectCreate, ProjectUpdate, ProjectResponse, ProjectListResponse,
6
+ )
7
+ from app.schemas.chapter import ChapterCreate, ChapterResponse
8
+ from app.schemas.character import CharacterCreate, CharacterUpdate, CharacterResponse
9
+ from app.schemas.episode import EpisodeCreate, EpisodeResponse
10
+ from app.schemas.generation import GenerationJobResponse, GenerationStart
11
+
12
+ __all__ = [
13
+ "UserCreate", "UserLogin", "UserResponse", "TokenResponse", "RefreshRequest",
14
+ "ProjectCreate", "ProjectUpdate", "ProjectResponse", "ProjectListResponse",
15
+ "ChapterCreate", "ChapterResponse",
16
+ "CharacterCreate", "CharacterUpdate", "CharacterResponse",
17
+ "EpisodeCreate", "EpisodeResponse",
18
+ "GenerationJobResponse", "GenerationStart",
19
+ ]
app/schemas/auth.py ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel, EmailStr
2
+ from datetime import datetime
3
+
4
+
5
+ class UserCreate(BaseModel):
6
+ email: EmailStr
7
+ username: str
8
+ password: str
9
+
10
+
11
+ class UserLogin(BaseModel):
12
+ email: EmailStr
13
+ password: str
14
+
15
+
16
+ class UserResponse(BaseModel):
17
+ id: int
18
+ email: str
19
+ username: str
20
+ role: str
21
+ created_at: datetime
22
+
23
+ model_config = {"from_attributes": True}
24
+
25
+
26
+ class TokenResponse(BaseModel):
27
+ access_token: str
28
+ token_type: str = "bearer"
29
+
30
+
31
+ class RefreshRequest(BaseModel):
32
+ refresh_token: str
app/schemas/chapter.py ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel
2
+ from datetime import datetime
3
+
4
+
5
+ class ChapterCreate(BaseModel):
6
+ chapter_number: int
7
+ title: str | None = None
8
+ source_text: str
9
+
10
+
11
+ class ChapterResponse(BaseModel):
12
+ id: int
13
+ project_id: int
14
+ chapter_number: int
15
+ title: str | None
16
+ word_count: int
17
+ status: str
18
+ created_at: datetime
19
+
20
+ model_config = {"from_attributes": True}
app/schemas/character.py ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel
2
+ from datetime import datetime
3
+
4
+
5
+ class CharacterCreate(BaseModel):
6
+ name: str
7
+ aliases_json: list[str] | None = None
8
+ role: str = "supporting"
9
+ gender: str | None = None
10
+ age_category: str | None = None
11
+ visual_prompt: str | None = None
12
+ voice_config_json: dict | None = None
13
+
14
+
15
+ class CharacterUpdate(BaseModel):
16
+ name: str | None = None
17
+ aliases_json: list[str] | None = None
18
+ role: str | None = None
19
+ gender: str | None = None
20
+ age_category: str | None = None
21
+ visual_prompt: str | None = None
22
+ voice_config_json: dict | None = None
23
+
24
+
25
+ class CharacterResponse(BaseModel):
26
+ id: int
27
+ project_id: int
28
+ name: str
29
+ aliases_json: list | None
30
+ role: str
31
+ gender: str | None
32
+ age_category: str | None
33
+ visual_prompt: str | None
34
+ reference_image_path: str | None
35
+ voice_config_json: dict | None
36
+ created_at: datetime
37
+
38
+ model_config = {"from_attributes": True}
app/schemas/episode.py ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel
2
+ from datetime import datetime
3
+
4
+
5
+ class EpisodeCreate(BaseModel):
6
+ episode_number: int
7
+ title: str | None = None
8
+ chapter_ids_json: list[int] | None = None
9
+
10
+
11
+ class EpisodeResponse(BaseModel):
12
+ id: int
13
+ project_id: int
14
+ episode_number: int
15
+ title: str | None
16
+ chapter_ids_json: list | None
17
+ duration_seconds: float | None
18
+ video_path: str | None
19
+ status: str
20
+ created_at: datetime
21
+
22
+ model_config = {"from_attributes": True}
app/schemas/generation.py ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel
2
+ from datetime import datetime
3
+
4
+
5
+ class GenerationStart(BaseModel):
6
+ episode_id: int | None = None
7
+ chapter_ids: list[int] | None = None
8
+
9
+
10
+ class GenerationJobResponse(BaseModel):
11
+ id: int
12
+ project_id: int
13
+ episode_id: int | None
14
+ status: str
15
+ current_stage: str | None
16
+ progress_pct: float
17
+ progress_detail: str | None
18
+ error_message: str | None
19
+ started_at: datetime | None
20
+ completed_at: datetime | None
21
+ created_at: datetime
22
+
23
+ model_config = {"from_attributes": True}
app/schemas/project.py ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic import BaseModel
2
+ from datetime import datetime
3
+
4
+
5
+ class ProjectCreate(BaseModel):
6
+ title: str
7
+ novel_url: str | None = None
8
+ cover_image_url: str | None = None
9
+ synopsis: str | None = None
10
+ settings_json: dict | None = None
11
+
12
+
13
+ class ProjectUpdate(BaseModel):
14
+ title: str | None = None
15
+ novel_url: str | None = None
16
+ cover_image_url: str | None = None
17
+ synopsis: str | None = None
18
+ settings_json: dict | None = None
19
+ status: str | None = None
20
+
21
+
22
+ class ProjectResponse(BaseModel):
23
+ id: int
24
+ user_id: int
25
+ title: str
26
+ novel_url: str | None
27
+ cover_image_url: str | None
28
+ synopsis: str | None
29
+ settings_json: dict | None
30
+ status: str
31
+ created_at: datetime
32
+ updated_at: datetime | None
33
+
34
+ model_config = {"from_attributes": True}
35
+
36
+
37
+ class ProjectListResponse(BaseModel):
38
+ projects: list[ProjectResponse]
39
+ total: int
app/services/__init__.py ADDED
File without changes
app/services/auth.py ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from datetime import datetime, timedelta, timezone
2
+ import bcrypt
3
+ from jose import jwt, JWTError
4
+ from fastapi import Depends, HTTPException, status
5
+ from fastapi.security import OAuth2PasswordBearer
6
+ from sqlalchemy import select
7
+ from sqlalchemy.ext.asyncio import AsyncSession
8
+ from app.config import get_settings
9
+ from app.database import get_db
10
+ from app.models.user import User
11
+
12
+ settings = get_settings()
13
+ oauth2_scheme = OAuth2PasswordBearer(tokenUrl="/api/auth/login")
14
+
15
+
16
+ def hash_password(password: str) -> str:
17
+ return bcrypt.hashpw(password.encode(), bcrypt.gensalt()).decode()
18
+
19
+
20
+ def verify_password(plain: str, hashed: str) -> bool:
21
+ return bcrypt.checkpw(plain.encode(), hashed.encode())
22
+
23
+
24
+ def create_token(data: dict, expires_delta: timedelta) -> str:
25
+ to_encode = data.copy()
26
+ to_encode["exp"] = datetime.now(timezone.utc) + expires_delta
27
+ return jwt.encode(to_encode, settings.jwt_secret_key, algorithm=settings.jwt_algorithm)
28
+
29
+
30
+ def create_access_token(user_id: int) -> str:
31
+ return create_token(
32
+ {"sub": str(user_id), "type": "access"},
33
+ timedelta(minutes=settings.access_token_expire_minutes),
34
+ )
35
+
36
+
37
+ def create_refresh_token(user_id: int) -> str:
38
+ return create_token(
39
+ {"sub": str(user_id), "type": "refresh"},
40
+ timedelta(days=settings.refresh_token_expire_days),
41
+ )
42
+
43
+
44
+ def decode_token(token: str) -> dict:
45
+ try:
46
+ return jwt.decode(token, settings.jwt_secret_key, algorithms=[settings.jwt_algorithm])
47
+ except JWTError:
48
+ raise HTTPException(
49
+ status_code=status.HTTP_401_UNAUTHORIZED,
50
+ detail="Invalid or expired token",
51
+ )
52
+
53
+
54
+ async def get_current_user(
55
+ token: str = Depends(oauth2_scheme),
56
+ db: AsyncSession = Depends(get_db),
57
+ ) -> User:
58
+ payload = decode_token(token)
59
+ if payload.get("type") != "access":
60
+ raise HTTPException(status_code=401, detail="Invalid token type")
61
+ user_id = int(payload["sub"])
62
+ result = await db.execute(select(User).where(User.id == user_id))
63
+ user = result.scalar_one_or_none()
64
+ if not user:
65
+ raise HTTPException(status_code=401, detail="User not found")
66
+ return user
app/services/fandom.py ADDED
@@ -0,0 +1,107 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Fandom/MediaWiki API client for character enrichment."""
2
+ import re
3
+ import httpx
4
+
5
+
6
+ async def search_fandom_wiki(
7
+ novel_title: str,
8
+ character_name: str,
9
+ ) -> dict | None:
10
+ """Search Fandom wikis for character appearance data.
11
+ Returns {description, image_url} or None."""
12
+
13
+ # Generate wiki subdomain from novel title
14
+ slug = re.sub(r'[^a-zA-Z0-9]', '', novel_title.lower().replace(' ', ''))
15
+ wiki_url = f"https://{slug}.fandom.com"
16
+
17
+ try:
18
+ async with httpx.AsyncClient(timeout=15.0) as client:
19
+ # Search for character page
20
+ resp = await client.get(
21
+ f"{wiki_url}/api.php",
22
+ params={
23
+ "action": "query",
24
+ "list": "search",
25
+ "srsearch": character_name,
26
+ "srnamespace": "0",
27
+ "srlimit": "3",
28
+ "format": "json",
29
+ },
30
+ )
31
+ if resp.status_code != 200:
32
+ return None
33
+
34
+ data = resp.json()
35
+ results = data.get("query", {}).get("search", [])
36
+ if not results:
37
+ return None
38
+
39
+ page_title = results[0]["title"]
40
+
41
+ # Get page content (parse for appearance section)
42
+ resp2 = await client.get(
43
+ f"{wiki_url}/api.php",
44
+ params={
45
+ "action": "parse",
46
+ "page": page_title,
47
+ "prop": "wikitext|images",
48
+ "format": "json",
49
+ },
50
+ )
51
+ if resp2.status_code != 200:
52
+ return None
53
+
54
+ parse_data = resp2.json().get("parse", {})
55
+ wikitext = parse_data.get("wikitext", {}).get("*", "")
56
+ images = parse_data.get("images", [])
57
+
58
+ # Extract appearance section
59
+ appearance = _extract_section(wikitext, ["Appearance", "Physical Description", "Description"])
60
+
61
+ # Get first image URL
62
+ image_url = None
63
+ if images:
64
+ img_name = images[0]
65
+ img_resp = await client.get(
66
+ f"{wiki_url}/api.php",
67
+ params={
68
+ "action": "query",
69
+ "titles": f"File:{img_name}",
70
+ "prop": "imageinfo",
71
+ "iiprop": "url",
72
+ "format": "json",
73
+ },
74
+ )
75
+ if img_resp.status_code == 200:
76
+ pages = img_resp.json().get("query", {}).get("pages", {})
77
+ for page in pages.values():
78
+ ii = page.get("imageinfo", [])
79
+ if ii:
80
+ image_url = ii[0].get("url")
81
+
82
+ return {
83
+ "description": _clean_wikitext(appearance) if appearance else None,
84
+ "image_url": image_url,
85
+ }
86
+ except Exception:
87
+ return None
88
+
89
+
90
+ def _extract_section(wikitext: str, headings: list[str]) -> str | None:
91
+ """Extract content under a wiki section heading."""
92
+ for heading in headings:
93
+ pattern = rf'==+\s*{re.escape(heading)}\s*==+(.*?)(?===|\Z)'
94
+ match = re.search(pattern, wikitext, re.DOTALL | re.IGNORECASE)
95
+ if match:
96
+ return match.group(1).strip()
97
+ return None
98
+
99
+
100
+ def _clean_wikitext(text: str) -> str:
101
+ """Remove wiki markup, leaving plain text."""
102
+ text = re.sub(r'\[\[(?:[^|\]]*\|)?([^\]]*)\]\]', r'\1', text) # [[link|text]] → text
103
+ text = re.sub(r'\{\{[^}]*\}\}', '', text) # {{templates}}
104
+ text = re.sub(r"'''?", '', text) # bold/italic
105
+ text = re.sub(r'<[^>]+>', '', text) # HTML tags
106
+ text = re.sub(r'\n{3,}', '\n\n', text)
107
+ return text.strip()