Spaces:
Sleeping
Sleeping
Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +4 -0
- .gitignore +31 -0
- Dockerfile +51 -0
- RESEARCH.md +953 -0
- TESTED_APIS.md +128 -0
- app/__init__.py +0 -0
- app/api/__init__.py +0 -0
- app/api/auth.py +104 -0
- app/api/chapters.py +82 -0
- app/api/characters.py +74 -0
- app/api/episodes.py +68 -0
- app/api/generation.py +241 -0
- app/api/novels.py +80 -0
- app/api/projects.py +96 -0
- app/config.py +38 -0
- app/database.py +51 -0
- app/main.py +54 -0
- app/models/__init__.py +13 -0
- app/models/api_usage.py +13 -0
- app/models/chapter.py +19 -0
- app/models/character.py +27 -0
- app/models/critic_result.py +18 -0
- app/models/episode.py +25 -0
- app/models/generation_job.py +25 -0
- app/models/project.py +24 -0
- app/models/user.py +18 -0
- app/pipeline/__init__.py +0 -0
- app/pipeline/orchestrator.py +435 -0
- app/pipeline/recap_generator.py +191 -0
- app/pipeline/stage_1_ingest.py +39 -0
- app/pipeline/stage_2_characters.py +172 -0
- app/pipeline/stage_2b_portraits.py +118 -0
- app/pipeline/stage_2c_voice_enroll.py +177 -0
- app/pipeline/stage_3_scene_parse.py +346 -0
- app/pipeline/stage_4_image_gen.py +137 -0
- app/pipeline/stage_5_tts.py +242 -0
- app/pipeline/stage_6_animation.py +121 -0
- app/pipeline/stage_6_video_gen.py +304 -0
- app/pipeline/stage_7_assembly.py +278 -0
- app/pipeline/stage_8_critic.py +189 -0
- app/schemas/__init__.py +19 -0
- app/schemas/auth.py +32 -0
- app/schemas/chapter.py +20 -0
- app/schemas/character.py +38 -0
- app/schemas/episode.py +22 -0
- app/schemas/generation.py +23 -0
- app/schemas/project.py +39 -0
- app/services/__init__.py +0 -0
- app/services/auth.py +66 -0
- app/services/fandom.py +107 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
data/bgm/calm.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
data/bgm/epic.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
data/bgm/sad.mp3 filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
data/bgm/tense.mp3 filter=lfs diff=lfs merge=lfs -text
|
.gitignore
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Python
|
| 2 |
+
__pycache__/
|
| 3 |
+
*.py[cod]
|
| 4 |
+
*.egg-info/
|
| 5 |
+
dist/
|
| 6 |
+
.venv/
|
| 7 |
+
venv/
|
| 8 |
+
|
| 9 |
+
# Environment
|
| 10 |
+
.env
|
| 11 |
+
|
| 12 |
+
# Workdir (runtime output)
|
| 13 |
+
workdir/
|
| 14 |
+
|
| 15 |
+
# Test outputs (generated images, videos, audio — large binary files)
|
| 16 |
+
test_*_output/
|
| 17 |
+
test_tts_calibration/
|
| 18 |
+
test_*.png
|
| 19 |
+
|
| 20 |
+
# IDE / Claude
|
| 21 |
+
.vscode/
|
| 22 |
+
.idea/
|
| 23 |
+
.claude/
|
| 24 |
+
|
| 25 |
+
# Node
|
| 26 |
+
node_modules/
|
| 27 |
+
.next/
|
| 28 |
+
|
| 29 |
+
# OS
|
| 30 |
+
.DS_Store
|
| 31 |
+
Thumbs.db
|
Dockerfile
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.12-slim
|
| 2 |
+
|
| 3 |
+
# Install FFmpeg + git
|
| 4 |
+
RUN apt-get update && \
|
| 5 |
+
apt-get install -y --no-install-recommends ffmpeg git && \
|
| 6 |
+
rm -rf /var/lib/apt/lists/*
|
| 7 |
+
|
| 8 |
+
# Create non-root user (HF Spaces requirement)
|
| 9 |
+
RUN useradd -m -u 1000 user
|
| 10 |
+
WORKDIR /app
|
| 11 |
+
|
| 12 |
+
# Install Python dependencies (cache layer)
|
| 13 |
+
RUN pip install --no-cache-dir \
|
| 14 |
+
"fastapi[standard]>=0.115.0" \
|
| 15 |
+
"uvicorn[standard]>=0.32.0" \
|
| 16 |
+
"sqlalchemy[asyncio]>=2.0.0" \
|
| 17 |
+
"aiosqlite>=0.20.0" \
|
| 18 |
+
"pydantic>=2.0.0" \
|
| 19 |
+
"pydantic-settings>=2.0.0" \
|
| 20 |
+
"python-jose[cryptography]>=3.3.0" \
|
| 21 |
+
"passlib[bcrypt]>=1.7.4" \
|
| 22 |
+
"python-multipart>=0.0.9" \
|
| 23 |
+
"httpx>=0.27.0" \
|
| 24 |
+
"edge-tts>=6.1.0" \
|
| 25 |
+
"Pillow>=10.0.0" \
|
| 26 |
+
"pymupdf>=1.24.0" \
|
| 27 |
+
"opencv-python-headless>=4.9.0" \
|
| 28 |
+
"numpy>=1.26.0" \
|
| 29 |
+
"pocket-tts>=0.1.0" \
|
| 30 |
+
"huggingface-hub>=0.20.0" \
|
| 31 |
+
"scipy>=1.12.0" \
|
| 32 |
+
"safetensors>=0.4.0" \
|
| 33 |
+
"pyarrow>=15.0.0" \
|
| 34 |
+
"soundfile>=0.12.0"
|
| 35 |
+
|
| 36 |
+
# Copy application code
|
| 37 |
+
COPY app/ app/
|
| 38 |
+
COPY data/*.json data/
|
| 39 |
+
COPY scripts/ scripts/
|
| 40 |
+
COPY deploy/start.sh .
|
| 41 |
+
RUN chmod +x start.sh
|
| 42 |
+
|
| 43 |
+
# Create directories with correct permissions
|
| 44 |
+
RUN mkdir -p /data/workdir/projects /data/genshin_voices /data/animevox_voices && \
|
| 45 |
+
chown -R user:user /app /data
|
| 46 |
+
|
| 47 |
+
USER user
|
| 48 |
+
|
| 49 |
+
EXPOSE 7860
|
| 50 |
+
|
| 51 |
+
CMD ["./start.sh"]
|
RESEARCH.md
ADDED
|
@@ -0,0 +1,953 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Web Novel to Anime -- Complete Research Report
|
| 2 |
+
|
| 3 |
+
> Generated: 2026-02-27 | Status: Comprehensive research across 5 domains
|
| 4 |
+
|
| 5 |
+
---
|
| 6 |
+
|
| 7 |
+
## TABLE OF CONTENTS
|
| 8 |
+
|
| 9 |
+
1. [Executive Summary & Verdict](#1-executive-summary--verdict)
|
| 10 |
+
2. [PLAN A: Free Video Generation APIs](#2-plan-a-free-video-generation-apis)
|
| 11 |
+
3. [PLAN B: The Hybrid Pipeline](#3-plan-b-the-hybrid-pipeline)
|
| 12 |
+
- 3.1 [Image Generation (Free APIs)](#31-image-generation-free-apis)
|
| 13 |
+
- 3.2 [Character Consistency Solutions](#32-character-consistency-solutions)
|
| 14 |
+
- 3.3 [TTS / Voice Acting](#33-tts--voice-acting)
|
| 15 |
+
- 3.4 [Interpolation & Animation](#34-interpolation--animation)
|
| 16 |
+
- 3.5 [Video Assembly](#35-video-assembly)
|
| 17 |
+
4. [NLP & Intelligence Layer](#4-nlp--intelligence-layer)
|
| 18 |
+
5. [System Architecture](#5-system-architecture)
|
| 19 |
+
6. [Your Hardware Advantage: Intel Arc + OpenVINO](#6-your-hardware-advantage)
|
| 20 |
+
7. [Implementation Roadmap](#7-implementation-roadmap)
|
| 21 |
+
8. [Monetization Strategy](#8-monetization-strategy)
|
| 22 |
+
9. [Final Recommendation](#9-final-recommendation)
|
| 23 |
+
|
| 24 |
+
---
|
| 25 |
+
|
| 26 |
+
## 1. EXECUTIVE SUMMARY & VERDICT
|
| 27 |
+
|
| 28 |
+
### The Honest Truth About Plan A (Pure Video Generation)
|
| 29 |
+
|
| 30 |
+
**Plan A is NOT viable today for free, automated, anime-quality video generation.** Here's why:
|
| 31 |
+
|
| 32 |
+
| Service | Free API? | Anime Quality? | Programmatic? | Verdict |
|
| 33 |
+
|---------|-----------|----------------|---------------|---------|
|
| 34 |
+
| Wan 2.1 (Alibaba) | HuggingFace free inference (slow, queued) | Good anime style | Yes via HF API | **Best free option but too slow/unreliable for production** |
|
| 35 |
+
| CogVideoX (Tsinghua) | HuggingFace free inference | Decent | Yes via HF API | Queue times make it impractical |
|
| 36 |
+
| Kling AI | Free web tier, NO free API | Good | No API on free tier | Unusable for automation |
|
| 37 |
+
| Runway Gen-3 | No free API | Good | Paid API only | Too expensive |
|
| 38 |
+
| Pika Labs | No free API | Decent | No API access | Unusable |
|
| 39 |
+
| Minimax (Hailuo) | Limited free API credits | Good | Yes | Credits run out fast |
|
| 40 |
+
| Luma Dream Machine | No free API | Decent | Paid only | Unusable |
|
| 41 |
+
| Stable Video Diffusion | Open source | Basic (image-to-4sec) | Yes locally | Too short, needs GPU |
|
| 42 |
+
| AnimateDiff | Open source | Good anime | Yes locally | **Needs 8GB+ VRAM** |
|
| 43 |
+
| Google Veo | No public API | Unknown | No | Not available |
|
| 44 |
+
|
| 45 |
+
**The core problem**: Video generation models need massive GPU power (12-24GB VRAM minimum). Free API tiers either:
|
| 46 |
+
- Don't exist (Runway, Pika, Luma)
|
| 47 |
+
- Have no programmatic access (Kling free tier)
|
| 48 |
+
- Are too slow/queued on HuggingFace (Wan, CogVideoX -- minutes of queue time per 4-second clip)
|
| 49 |
+
- Produce only 3-6 second clips (not episodes)
|
| 50 |
+
|
| 51 |
+
**However**, there ARE some usable free options for short video clips via HuggingFace that could supplement Plan B (see section 2 below).
|
| 52 |
+
|
| 53 |
+
### Why Plan B Is Actually BETTER
|
| 54 |
+
|
| 55 |
+
Plan B (Image + TTS + Interpolation) is not just a fallback -- it's the **superior approach** because:
|
| 56 |
+
|
| 57 |
+
1. **Character consistency** -- Image gen with IP-Adapter keeps characters looking the same. Video gen cannot do this yet.
|
| 58 |
+
2. **Controllability** -- You control every frame, every camera angle, every expression. Video gen is a black box.
|
| 59 |
+
3. **This IS how anime works** -- Real anime uses 8-12 unique drawings per second with camera movements. That's literally Plan B.
|
| 60 |
+
4. **Free and unlimited** -- Multiple free image gen APIs exist with generous limits.
|
| 61 |
+
5. **Your Intel Arc GPU helps** -- OpenVINO acceleration makes interpolation and animation feasible on your hardware.
|
| 62 |
+
|
| 63 |
+
---
|
| 64 |
+
|
| 65 |
+
## 2. PLAN A: FREE VIDEO GENERATION APIs
|
| 66 |
+
|
| 67 |
+
Despite video gen not being viable as the primary approach, here's every free option found:
|
| 68 |
+
|
| 69 |
+
### 2.1 Wan 2.1 (Best Free Video Gen Available)
|
| 70 |
+
|
| 71 |
+
- **Source**: Alibaba's open-source video generation model
|
| 72 |
+
- **Access**: HuggingFace Inference API (free, queued) + HuggingFace Spaces
|
| 73 |
+
- **Quality**: Excellent anime-style output, one of the best for anime
|
| 74 |
+
- **Clip Length**: 4-8 seconds
|
| 75 |
+
- **Resolution**: Up to 720p
|
| 76 |
+
- **Limitations**: Queue wait times of 2-15 minutes per clip. Rate limited. Not reliable for batch processing.
|
| 77 |
+
- **Programmatic access**:
|
| 78 |
+
```python
|
| 79 |
+
from huggingface_hub import InferenceClient
|
| 80 |
+
client = InferenceClient(token="YOUR_HF_TOKEN")
|
| 81 |
+
|
| 82 |
+
# Text-to-video
|
| 83 |
+
video = client.text_to_video(
|
| 84 |
+
"anime girl with silver hair walking through a fantasy forest, detailed anime style",
|
| 85 |
+
model="Wan-AI/Wan2.1-T2V-14B"
|
| 86 |
+
)
|
| 87 |
+
with open("output.mp4", "wb") as f:
|
| 88 |
+
f.write(video)
|
| 89 |
+
```
|
| 90 |
+
- **Or via Gradio Spaces**:
|
| 91 |
+
```python
|
| 92 |
+
from gradio_client import Client
|
| 93 |
+
client = Client("Wan-AI/Wan2.1-T2V-14B")
|
| 94 |
+
result = client.predict(prompt="...", api_name="/generate")
|
| 95 |
+
```
|
| 96 |
+
- **Verdict**: Usable for generating occasional hero clips (opening shots, climactic moments). NOT for generating full episodes.
|
| 97 |
+
|
| 98 |
+
### 2.2 CogVideoX (Tsinghua)
|
| 99 |
+
|
| 100 |
+
- **Access**: HuggingFace Inference API + Spaces
|
| 101 |
+
- **Quality**: Good, less anime-specific than Wan
|
| 102 |
+
- **Clip Length**: 6 seconds
|
| 103 |
+
- **Programmatic access**: Same pattern as Wan via HF
|
| 104 |
+
- **Verdict**: Backup to Wan 2.1
|
| 105 |
+
|
| 106 |
+
### 2.3 Stable Video Diffusion (Stability AI)
|
| 107 |
+
|
| 108 |
+
- **Access**: Open source, can run locally
|
| 109 |
+
- **What it does**: Image-to-video (takes a still image, adds subtle motion)
|
| 110 |
+
- **Clip Length**: 3-4 seconds
|
| 111 |
+
- **Hardware**: Needs ~6GB VRAM minimum. **Might work on Intel Arc via OpenVINO** (untested but theoretically possible)
|
| 112 |
+
- **Verdict**: Could complement Plan B by adding subtle motion to generated keyframes. Worth testing on Intel Arc.
|
| 113 |
+
|
| 114 |
+
### 2.4 Free Cloud GPU Options for Running Video Models
|
| 115 |
+
|
| 116 |
+
| Platform | Free GPU | GPU Type | Time Limit | Video Gen Feasible? |
|
| 117 |
+
|----------|----------|----------|------------|-------------------|
|
| 118 |
+
| Google Colab | Yes | T4 (15GB VRAM) | ~4hrs/day | **Yes** -- can run AnimateDiff, Wan 2.1 small |
|
| 119 |
+
| Kaggle | Yes | T4x2 or P100 | 30hrs/week | **Yes** -- best free option |
|
| 120 |
+
| Lightning.ai | Yes | T4 | 22hrs/month | Marginal |
|
| 121 |
+
| Paperspace | Free tier | M4000 (8GB) | Limited | Tight but possible |
|
| 122 |
+
|
| 123 |
+
**Google Colab + Kaggle Strategy**: Use Kaggle's 30 hours/week to batch-generate video clips. AnimateDiff on a T4 can produce ~4 seconds of anime video per 30-60 seconds of compute. That's about 120-240 4-second clips per week = ~8-16 minutes of video content.
|
| 124 |
+
|
| 125 |
+
### 2.5 Recommended Use of Video Gen (Hybrid with Plan B)
|
| 126 |
+
|
| 127 |
+
```
|
| 128 |
+
Plan B generates 95% of the episode (images + camera + TTS + interpolation)
|
| 129 |
+
|
|
| 130 |
+
v
|
| 131 |
+
For KEY MOMENTS (5% of episode):
|
| 132 |
+
- Opening/ending sequence → Wan 2.1 via HuggingFace
|
| 133 |
+
- Dramatic reveals → AnimateDiff via Kaggle/Colab
|
| 134 |
+
- Action climaxes → CogVideoX via HuggingFace
|
| 135 |
+
|
|
| 136 |
+
v
|
| 137 |
+
These premium clips are spliced into the Plan B video
|
| 138 |
+
```
|
| 139 |
+
|
| 140 |
+
---
|
| 141 |
+
|
| 142 |
+
## 3. PLAN B: THE HYBRID PIPELINE
|
| 143 |
+
|
| 144 |
+
### 3.1 Image Generation (Free APIs)
|
| 145 |
+
|
| 146 |
+
#### TOP TIER: Truly Free with Good Anime Quality
|
| 147 |
+
|
| 148 |
+
**1. Together AI -- FLUX.1 (Best Overall)**
|
| 149 |
+
- **Free**: $5 credits on signup (~500 images). Periodically refreshed for active users.
|
| 150 |
+
- **Quality**: FLUX produces excellent anime art with strong prompt adherence
|
| 151 |
+
- **API**:
|
| 152 |
+
```python
|
| 153 |
+
from together import Together
|
| 154 |
+
client = Together(api_key="YOUR_KEY")
|
| 155 |
+
|
| 156 |
+
response = client.images.generate(
|
| 157 |
+
model="black-forest-labs/FLUX.1-schnell-Free", # Free model
|
| 158 |
+
prompt="anime girl, silver hair, blue eyes, holding a sword, fantasy forest background, detailed anime illustration",
|
| 159 |
+
width=1024, height=768, n=1
|
| 160 |
+
)
|
| 161 |
+
image_url = response.data[0].url
|
| 162 |
+
```
|
| 163 |
+
- **Anime-specific**: FLUX.1-schnell is fast and free. Excellent prompt following.
|
| 164 |
+
|
| 165 |
+
**2. HuggingFace Inference API -- Stable Diffusion XL / FLUX**
|
| 166 |
+
- **Free**: Rate-limited (rough estimate: 100-1000 requests/day depending on model load)
|
| 167 |
+
- **Models available**: SDXL, FLUX, SD 1.5 anime finetunes
|
| 168 |
+
- **API**:
|
| 169 |
+
```python
|
| 170 |
+
from huggingface_hub import InferenceClient
|
| 171 |
+
client = InferenceClient(token="YOUR_HF_TOKEN")
|
| 172 |
+
|
| 173 |
+
image = client.text_to_image(
|
| 174 |
+
"1girl, silver hair, blue eyes, anime style, masterpiece, best quality",
|
| 175 |
+
model="stabilityai/stable-diffusion-xl-base-1.0"
|
| 176 |
+
)
|
| 177 |
+
image.save("output.png")
|
| 178 |
+
|
| 179 |
+
# Anime-specific model:
|
| 180 |
+
image = client.text_to_image(
|
| 181 |
+
"1girl, silver hair, detailed anime illustration",
|
| 182 |
+
model="cagliostrolab/animagine-xl-3.1" # Top anime SDXL model
|
| 183 |
+
)
|
| 184 |
+
```
|
| 185 |
+
- **Key anime models on HF**: `animagine-xl-3.1`, `CounterfeitXL`, `BluePencilXL`
|
| 186 |
+
|
| 187 |
+
**3. Stability AI Free Tier**
|
| 188 |
+
- **Free**: Limited free credits on signup
|
| 189 |
+
- **Models**: SDXL, SD3, Stable Image Core
|
| 190 |
+
- **API**:
|
| 191 |
+
```python
|
| 192 |
+
import requests
|
| 193 |
+
|
| 194 |
+
response = requests.post(
|
| 195 |
+
"https://api.stability.ai/v2beta/stable-image/generate/sd3",
|
| 196 |
+
headers={"Authorization": f"Bearer {api_key}"},
|
| 197 |
+
files={"none": ""},
|
| 198 |
+
data={
|
| 199 |
+
"prompt": "anime style illustration, young warrior...",
|
| 200 |
+
"output_format": "png",
|
| 201 |
+
"model": "sd3.5-large",
|
| 202 |
+
}
|
| 203 |
+
)
|
| 204 |
+
```
|
| 205 |
+
|
| 206 |
+
**4. Pollinations.ai (Completely Free, No API Key)**
|
| 207 |
+
- **Free**: 100% free, no signup, no API key needed
|
| 208 |
+
- **Quality**: Uses FLUX under the hood
|
| 209 |
+
- **API**:
|
| 210 |
+
```python
|
| 211 |
+
import requests
|
| 212 |
+
import urllib.parse
|
| 213 |
+
|
| 214 |
+
prompt = urllib.parse.quote("anime girl, silver hair, fantasy setting, detailed illustration")
|
| 215 |
+
url = f"https://image.pollinations.ai/prompt/{prompt}?width=1024&height=768&model=flux"
|
| 216 |
+
response = requests.get(url)
|
| 217 |
+
with open("output.png", "wb") as f:
|
| 218 |
+
f.write(response.content)
|
| 219 |
+
```
|
| 220 |
+
- **Limits**: No hard limits documented, but be respectful with rate
|
| 221 |
+
- **Verdict**: **Best for prototyping and early development. Zero friction.**
|
| 222 |
+
|
| 223 |
+
**5. Google Colab / Kaggle -- Run Your Own**
|
| 224 |
+
- Run SDXL, FLUX, or anime-specialized models on free T4 GPUs
|
| 225 |
+
- Full control over model selection, LoRA, IP-Adapter, ControlNet
|
| 226 |
+
- 30 hrs/week on Kaggle = ~3,000-5,000 images per week
|
| 227 |
+
- This is the **production-grade free option**
|
| 228 |
+
|
| 229 |
+
#### Anime-Specialized Models (All Free/Open Source)
|
| 230 |
+
|
| 231 |
+
| Model | Base | Anime Quality | Best For |
|
| 232 |
+
|-------|------|---------------|----------|
|
| 233 |
+
| Animagine XL 3.1 | SDXL | Excellent | General anime |
|
| 234 |
+
| CounterfeitXL | SDXL | Excellent | Detailed anime |
|
| 235 |
+
| BluePencilXL | SDXL | Very Good | Soft anime style |
|
| 236 |
+
| Anything V5 | SD 1.5 | Good | Classic anime |
|
| 237 |
+
| NovelAI anime model | SD 1.5 | Excellent | Light novel style |
|
| 238 |
+
| Pony Diffusion XL | SDXL | Good | Stylized anime |
|
| 239 |
+
|
| 240 |
+
### 3.2 Character Consistency Solutions
|
| 241 |
+
|
| 242 |
+
This is **THE hardest problem** in the entire pipeline. Every image gen call produces a different-looking character.
|
| 243 |
+
|
| 244 |
+
#### Solution 1: IP-Adapter (Recommended Primary)
|
| 245 |
+
|
| 246 |
+
IP-Adapter takes a reference image and forces new generations to match the character's appearance.
|
| 247 |
+
|
| 248 |
+
- **How**: Generate one "canonical" image of each character, then use it as reference for ALL subsequent images
|
| 249 |
+
- **Where to run**: Kaggle/Colab with ComfyUI, or via API services that support it
|
| 250 |
+
- **Quality**: ~70-85% consistency (face may drift slightly)
|
| 251 |
+
- **ComfyUI workflow on Kaggle**:
|
| 252 |
+
```python
|
| 253 |
+
# In a Kaggle notebook:
|
| 254 |
+
# 1. Install ComfyUI
|
| 255 |
+
# 2. Load SDXL + IP-Adapter
|
| 256 |
+
# 3. Provide reference image + new pose/scene prompt
|
| 257 |
+
# 4. Get consistent character in new situation
|
| 258 |
+
```
|
| 259 |
+
|
| 260 |
+
#### Solution 2: PuLID / InstantID (Best for Face Consistency)
|
| 261 |
+
|
| 262 |
+
- Specifically designed to maintain facial identity across images
|
| 263 |
+
- Takes a single face reference and locks facial features
|
| 264 |
+
- Available as ComfyUI nodes
|
| 265 |
+
- Run on Kaggle/Colab
|
| 266 |
+
|
| 267 |
+
#### Solution 3: Character Sheet + Reference Technique
|
| 268 |
+
|
| 269 |
+
1. First, generate a character reference sheet (front/side/3-quarter views)
|
| 270 |
+
2. Use this sheet as IP-Adapter reference for all subsequent generations
|
| 271 |
+
3. This is how professional anime production works
|
| 272 |
+
|
| 273 |
+
```python
|
| 274 |
+
CHARACTER_SHEET_PROMPT = """
|
| 275 |
+
character reference sheet, multiple views, front view, side view,
|
| 276 |
+
three-quarter view, same character, {character_description},
|
| 277 |
+
anime style, white background, character turnaround,
|
| 278 |
+
masterpiece, best quality
|
| 279 |
+
"""
|
| 280 |
+
```
|
| 281 |
+
|
| 282 |
+
#### Solution 4: LoRA Training (Most Consistent, More Work)
|
| 283 |
+
|
| 284 |
+
- Train a small LoRA model on 10-20 images of a specific character design
|
| 285 |
+
- Then use this LoRA every time you generate that character
|
| 286 |
+
- Can be done for free on Kaggle (LoRA training takes ~20-30 minutes on T4)
|
| 287 |
+
- Most consistent results but requires upfront work per character
|
| 288 |
+
|
| 289 |
+
#### Solution 5: StoryDiffusion (Emerging Research)
|
| 290 |
+
|
| 291 |
+
- Research project specifically designed for consistent characters across comic/manga panels
|
| 292 |
+
- Maintains character identity without LoRA training
|
| 293 |
+
- Still experimental but very promising for this exact use case
|
| 294 |
+
|
| 295 |
+
#### Recommended Strategy:
|
| 296 |
+
|
| 297 |
+
```
|
| 298 |
+
1. Generate character reference sheet (one-time per character)
|
| 299 |
+
2. Use IP-Adapter + PuLID for all subsequent images
|
| 300 |
+
3. For protagonist/main characters: train a quick LoRA on Kaggle
|
| 301 |
+
4. Store all reference images in character database
|
| 302 |
+
5. Include character visual description in EVERY prompt
|
| 303 |
+
```
|
| 304 |
+
|
| 305 |
+
### 3.3 TTS / Voice Acting
|
| 306 |
+
|
| 307 |
+
#### PRIMARY: Edge TTS (Unlimited, Free, Great Quality)
|
| 308 |
+
|
| 309 |
+
```python
|
| 310 |
+
pip install edge-tts
|
| 311 |
+
```
|
| 312 |
+
|
| 313 |
+
- **Cost**: Completely free, no API key, no account
|
| 314 |
+
- **Limits**: Effectively unlimited
|
| 315 |
+
- **Emotion support**: SSML with styles: `cheerful`, `sad`, `angry`, `excited`, `terrified`, `whispering`, `shouting`, and more
|
| 316 |
+
- **Voices**: 300+ voices, including Japanese voices for anime feel
|
| 317 |
+
- **Code**:
|
| 318 |
+
```python
|
| 319 |
+
import edge_tts
|
| 320 |
+
import asyncio
|
| 321 |
+
|
| 322 |
+
async def speak(text, voice="en-US-AriaNeural", style="cheerful", output="out.mp3"):
|
| 323 |
+
if style:
|
| 324 |
+
ssml = f"""
|
| 325 |
+
<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis'
|
| 326 |
+
xmlns:mstts='https://www.w3.org/2001/mstts' xml:lang='en-US'>
|
| 327 |
+
<voice name='{voice}'>
|
| 328 |
+
<mstts:express-as style='{style}' styledegree='1.5'>
|
| 329 |
+
{text}
|
| 330 |
+
</mstts:express-as>
|
| 331 |
+
</voice>
|
| 332 |
+
</speak>"""
|
| 333 |
+
communicate = edge_tts.Communicate(ssml)
|
| 334 |
+
else:
|
| 335 |
+
communicate = edge_tts.Communicate(text, voice)
|
| 336 |
+
await communicate.save(output)
|
| 337 |
+
|
| 338 |
+
asyncio.run(speak("I won't let you down!", style="excited"))
|
| 339 |
+
```
|
| 340 |
+
|
| 341 |
+
- **Recommended voice assignments**:
|
| 342 |
+
```python
|
| 343 |
+
CHARACTER_VOICES = {
|
| 344 |
+
"narrator": {"voice": "en-US-AriaNeural", "style": "narration-professional"},
|
| 345 |
+
"male_hero": {"voice": "en-US-GuyNeural", "style": "chat"},
|
| 346 |
+
"female_lead": {"voice": "en-US-JennyNeural", "style": "cheerful"},
|
| 347 |
+
"villain": {"voice": "en-US-DavisNeural", "style": "angry"},
|
| 348 |
+
"old_mentor": {"voice": "en-US-TonyNeural", "style": "calm"},
|
| 349 |
+
"child": {"voice": "en-US-SaraNeural", "style": "friendly"},
|
| 350 |
+
# Japanese voices for anime feel:
|
| 351 |
+
"jp_female": {"voice": "ja-JP-NanamiNeural", "style": None},
|
| 352 |
+
"jp_male": {"voice": "ja-JP-KeitaNeural", "style": None},
|
| 353 |
+
}
|
| 354 |
+
|
| 355 |
+
EMOTION_TO_STYLE = {
|
| 356 |
+
"happy": "cheerful", "sad": "sad", "angry": "angry",
|
| 357 |
+
"scared": "terrified", "excited": "excited", "calm": "calm",
|
| 358 |
+
"whisper": "whispering", "shout": "shouting", "love": "affectionate",
|
| 359 |
+
"nervous": "embarrassed",
|
| 360 |
+
}
|
| 361 |
+
```
|
| 362 |
+
|
| 363 |
+
#### SECONDARY: Azure TTS (500K chars/month free, BEST emotion system)
|
| 364 |
+
|
| 365 |
+
- Same voices as Edge TTS but with full API control
|
| 366 |
+
- `styledegree` (0.01 to 2.0) for emotion intensity
|
| 367 |
+
- Role tags: `Girl`, `Boy`, `YoungAdultFemale`, `OlderAdultMale`, `SeniorFemale`
|
| 368 |
+
- Requires Azure account (free tier, credit card on file but charges nothing)
|
| 369 |
+
|
| 370 |
+
#### SPECIAL: Bark by Suno (Local, CPU, Sound Effects)
|
| 371 |
+
|
| 372 |
+
- Open source, runs on CPU (slow: ~30-90s per 10s of audio)
|
| 373 |
+
- **Unique feature**: Inline sound effect tags:
|
| 374 |
+
- `[laughs]`, `[sighs]`, `[gasps]`, `[clears throat]`
|
| 375 |
+
- Great for moments Edge TTS can't handle
|
| 376 |
+
- Use for 5-10% of lines that need sound effects
|
| 377 |
+
- Batch process overnight
|
| 378 |
+
|
| 379 |
+
```python
|
| 380 |
+
os.environ["SUNO_USE_SMALL_MODELS"] = "True"
|
| 381 |
+
os.environ["SUNO_OFFLOAD_CPU"] = "True"
|
| 382 |
+
|
| 383 |
+
from bark import generate_audio, SAMPLE_RATE
|
| 384 |
+
audio = generate_audio("[sighs] I thought we were friends... [laughs nervously]",
|
| 385 |
+
history_prompt="v2/en_speaker_6")
|
| 386 |
+
```
|
| 387 |
+
|
| 388 |
+
#### BACKUP: Piper TTS (Local, CPU, Real-Time)
|
| 389 |
+
|
| 390 |
+
- Designed specifically for CPU -- runs in real-time on your hardware
|
| 391 |
+
- No emotion control but perfect for bulk narration
|
| 392 |
+
- Use as offline fallback when internet is unavailable
|
| 393 |
+
|
| 394 |
+
#### Total Free Monthly Capacity:
|
| 395 |
+
|
| 396 |
+
| Service | Free Chars/Month | Emotion | Speed |
|
| 397 |
+
|---------|-----------------|---------|-------|
|
| 398 |
+
| Edge TTS | **Unlimited** | Good SSML | Fast |
|
| 399 |
+
| Azure TTS | **500,000** | **Best** | Fast |
|
| 400 |
+
| Bark (local) | **Unlimited** | **Excellent** ([laughs] etc.) | Slow |
|
| 401 |
+
| Piper (local) | **Unlimited** | None | Real-time |
|
| 402 |
+
|
| 403 |
+
### 3.4 Interpolation & Animation
|
| 404 |
+
|
| 405 |
+
#### Frame Interpolation: RIFE (Primary Tool)
|
| 406 |
+
|
| 407 |
+
Takes 2 keyframe images and generates smooth intermediate frames.
|
| 408 |
+
|
| 409 |
+
```
|
| 410 |
+
Frame A → [RIFE generates 7 in-between frames] → Frame B
|
| 411 |
+
Result: 9 frames of smooth animation from just 2 images
|
| 412 |
+
```
|
| 413 |
+
|
| 414 |
+
- **Best option for your hardware**: `rife-ncnn-vulkan-python`
|
| 415 |
+
- Uses Vulkan backend -- **works natively on Intel Arc**
|
| 416 |
+
- No OpenVINO conversion needed
|
| 417 |
+
- Fast: near real-time at 512x512
|
| 418 |
+
|
| 419 |
+
```bash
|
| 420 |
+
pip install rife-ncnn-vulkan-python
|
| 421 |
+
```
|
| 422 |
+
```python
|
| 423 |
+
from rife_ncnn_vulkan_python import Rife
|
| 424 |
+
rife = Rife(gpuid=0) # Uses Vulkan on Intel Arc
|
| 425 |
+
mid_frame = rife.process(img0, img1) # Returns interpolated frame
|
| 426 |
+
```
|
| 427 |
+
|
| 428 |
+
- **Alternative**: RIFE via OpenVINO (potentially faster but requires model conversion)
|
| 429 |
+
- **Fallback**: FILM by Google (better for large motion gaps, runs on CPU via TensorFlow)
|
| 430 |
+
|
| 431 |
+
#### Image Animation: SadTalker (For Talking Faces)
|
| 432 |
+
|
| 433 |
+
Takes character image + TTS audio → outputs video of character talking with lip sync.
|
| 434 |
+
|
| 435 |
+
- **CPU mode**: `--cpu` flag, ~30-60s for 10s video at 256x256
|
| 436 |
+
- **Free on HuggingFace Spaces**: Can call remotely via `gradio_client`
|
| 437 |
+
|
| 438 |
+
```python
|
| 439 |
+
from gradio_client import Client
|
| 440 |
+
client = Client("vinthony/SadTalker")
|
| 441 |
+
result = client.predict("character.png", "speech.wav", "crop", True, True)
|
| 442 |
+
```
|
| 443 |
+
|
| 444 |
+
#### Image Animation: LivePortrait (For Character Motion)
|
| 445 |
+
|
| 446 |
+
Animates portrait with expressions and head movements.
|
| 447 |
+
|
| 448 |
+
- Has OpenVINO support -- can use Intel Arc GPU
|
| 449 |
+
- ~15-30 FPS with OpenVINO on Intel Arc
|
| 450 |
+
|
| 451 |
+
#### Lip Sync: Rhubarb Lip Sync (CPU, Real-Time)
|
| 452 |
+
|
| 453 |
+
Analyzes audio and outputs timed mouth shape data. Anime only uses ~5-7 mouth shapes.
|
| 454 |
+
|
| 455 |
+
```python
|
| 456 |
+
# Rhubarb outputs: {"mouthCues": [{"start": 0.00, "end": 0.05, "value": "A"}, ...]}
|
| 457 |
+
# Map to anime mouth shapes: Closed, Small, Medium, Open, Wide
|
| 458 |
+
```
|
| 459 |
+
|
| 460 |
+
#### Ken Burns Effects: FFmpeg (ZERO AI, Instant)
|
| 461 |
+
|
| 462 |
+
This should be **70-80% of your animation**. Real anime uses this constantly.
|
| 463 |
+
|
| 464 |
+
```python
|
| 465 |
+
# Slow zoom into image
|
| 466 |
+
ffmpeg -loop 1 -i image.png -vf "zoompan=z='min(zoom+0.001,1.3)':s=1920x1080:fps=24:d=120" \
|
| 467 |
+
-t 5 -c:v libx264 output.mp4
|
| 468 |
+
|
| 469 |
+
# Pan left to right
|
| 470 |
+
ffmpeg -loop 1 -i image.png -vf "zoompan=z=1.2:x='iw/2-(iw/zoom/2)+(-0.3+on/120*0.6)*iw/zoom':..." \
|
| 471 |
+
-t 5 output.mp4
|
| 472 |
+
```
|
| 473 |
+
|
| 474 |
+
- **Performance**: Instant. FFmpeg is heavily optimized.
|
| 475 |
+
- **Intel QSV**: Use `h264_qsv` encoder for 3-5x faster encoding on Intel Arc
|
| 476 |
+
|
| 477 |
+
#### Anime-Specific Effects (CPU, OpenCV)
|
| 478 |
+
|
| 479 |
+
All of these are trivial on CPU:
|
| 480 |
+
|
| 481 |
+
| Effect | Implementation | Use Case |
|
| 482 |
+
|--------|---------------|----------|
|
| 483 |
+
| Speed lines | OpenCV line drawing overlay | Action scenes |
|
| 484 |
+
| Camera shake | Random frame offset | Impact moments |
|
| 485 |
+
| White flash | Alpha blend with white | Explosions, reveals |
|
| 486 |
+
| Dramatic zoom lines | Radial lines from center | Power-up moments |
|
| 487 |
+
| Vignette | Radial gradient overlay | Moody scenes |
|
| 488 |
+
| Screen shake | Sine wave displacement | Earthquakes, impacts |
|
| 489 |
+
|
| 490 |
+
#### Parallax / 2.5D Effect
|
| 491 |
+
|
| 492 |
+
Use MiDaS depth estimation (has OpenVINO support!) to create depth maps, then move foreground/background at different speeds.
|
| 493 |
+
|
| 494 |
+
```python
|
| 495 |
+
from openvino.runtime import Core
|
| 496 |
+
core = Core()
|
| 497 |
+
model = core.compile_model("midas_v21_small.xml", "GPU") # Intel Arc
|
| 498 |
+
depth_map = model([image])[0]
|
| 499 |
+
# Separate into layers based on depth → animate at different speeds
|
| 500 |
+
```
|
| 501 |
+
|
| 502 |
+
### 3.5 Video Assembly
|
| 503 |
+
|
| 504 |
+
#### FFmpeg Pipeline (Core)
|
| 505 |
+
|
| 506 |
+
```python
|
| 507 |
+
class AnimeVideoAssembler:
|
| 508 |
+
def images_to_video(self, images, fps=24):
|
| 509 |
+
"""Number images → video"""
|
| 510 |
+
|
| 511 |
+
def add_audio(self, video, tts_audio):
|
| 512 |
+
"""Combine video + TTS"""
|
| 513 |
+
|
| 514 |
+
def mix_audio(self, tts, bg_music, music_vol=0.15):
|
| 515 |
+
"""Mix narration + background music"""
|
| 516 |
+
|
| 517 |
+
def add_subtitles(self, video, srt_file):
|
| 518 |
+
"""Burn anime-style subtitles"""
|
| 519 |
+
|
| 520 |
+
def crossfade(self, clip1, clip2, duration=0.5):
|
| 521 |
+
"""Scene transitions"""
|
| 522 |
+
|
| 523 |
+
def concat(self, clips):
|
| 524 |
+
"""Join all scenes into episode"""
|
| 525 |
+
|
| 526 |
+
def encode_qsv(self, input, output):
|
| 527 |
+
"""Intel Arc hardware-accelerated encoding"""
|
| 528 |
+
# ffmpeg -i input -c:v h264_qsv -global_quality 23 output.mp4
|
| 529 |
+
```
|
| 530 |
+
|
| 531 |
+
#### MoviePy (Python-Friendly Alternative)
|
| 532 |
+
|
| 533 |
+
```python
|
| 534 |
+
from moviepy.editor import *
|
| 535 |
+
|
| 536 |
+
# Per scene: image + Ken Burns + audio + subtitle
|
| 537 |
+
clip = ImageClip("scene.png", duration=5)
|
| 538 |
+
clip = clip.resize(lambda t: 1 + 0.04*t) # Zoom effect
|
| 539 |
+
clip = clip.set_audio(AudioFileClip("tts.mp3"))
|
| 540 |
+
txt = TextClip("Dialogue here", fontsize=36, color='white', stroke_color='black')
|
| 541 |
+
clip = CompositeVideoClip([clip, txt.set_position(('center','bottom'))])
|
| 542 |
+
```
|
| 543 |
+
|
| 544 |
+
---
|
| 545 |
+
|
| 546 |
+
## 4. NLP & INTELLIGENCE LAYER
|
| 547 |
+
|
| 548 |
+
### 4.1 Free LLM APIs (for Scene Parsing, Prompt Generation)
|
| 549 |
+
|
| 550 |
+
| Provider | Free Limits | Best Model | Context | Best For |
|
| 551 |
+
|----------|------------|------------|---------|----------|
|
| 552 |
+
| **Google Gemini** | 1500 req/day, 1M TPM | Gemini 2.0 Flash | **1M tokens** | Scene parsing (fit whole chapter in one call) |
|
| 553 |
+
| **Groq** | 14,400 req/day | Llama 3.3 70B | 128K | Batch emotion analysis (fastest) |
|
| 554 |
+
| **Together AI** | $5 free credits | Llama 3.3 70B | 128K | Backup + image gen (FLUX) |
|
| 555 |
+
| **OpenRouter** | 200 req/day | Various free models | Varies | Unified fallback |
|
| 556 |
+
| **HuggingFace** | ~100 req/hour | Specialized NLP models | Varies | Emotion classification |
|
| 557 |
+
| **Cohere** | 1000 calls/month | Command-R | 128K | Structured extraction backup |
|
| 558 |
+
| **SambaNova** | Free during beta | Llama 3.1 405B | 128K | Strongest reasoning |
|
| 559 |
+
|
| 560 |
+
**Strategy**: Gemini as primary (1M context = entire chapter in one call), Groq as fast secondary, others as fallbacks.
|
| 561 |
+
|
| 562 |
+
### 4.2 Text → Scene Pipeline
|
| 563 |
+
|
| 564 |
+
```
|
| 565 |
+
Raw Chapter Text
|
| 566 |
+
↓
|
| 567 |
+
[spaCy + NLTK] -- Fast local NLP (FREE)
|
| 568 |
+
- Sentence tokenization
|
| 569 |
+
- Dialogue extraction (regex)
|
| 570 |
+
- Named Entity Recognition
|
| 571 |
+
- Scene break detection (--- markers, time skips)
|
| 572 |
+
↓
|
| 573 |
+
[Gemini API] -- Semantic understanding (FREE)
|
| 574 |
+
- Scene segmentation
|
| 575 |
+
- Emotion per line
|
| 576 |
+
- Action extraction
|
| 577 |
+
- Setting descriptions
|
| 578 |
+
↓
|
| 579 |
+
[Prompt Generator] -- Convert to image prompts
|
| 580 |
+
- Character visual descriptions (from database)
|
| 581 |
+
- Camera angle selection
|
| 582 |
+
- Art style tags
|
| 583 |
+
↓
|
| 584 |
+
Structured Scene JSON
|
| 585 |
+
```
|
| 586 |
+
|
| 587 |
+
### 4.3 Key Libraries
|
| 588 |
+
|
| 589 |
+
```bash
|
| 590 |
+
pip install spacy nltk google-generativeai groq huggingface-hub
|
| 591 |
+
python -m spacy download en_core_web_lg
|
| 592 |
+
```
|
| 593 |
+
|
| 594 |
+
### 4.4 Character Management
|
| 595 |
+
|
| 596 |
+
```python
|
| 597 |
+
# Each character gets:
|
| 598 |
+
# - Canonical name + aliases
|
| 599 |
+
# - Visual description (50 words, prompt-ready, NO name)
|
| 600 |
+
# - Reference image paths (for IP-Adapter)
|
| 601 |
+
# - Voice assignment (Edge TTS voice + default emotion)
|
| 602 |
+
# - Outfit tracking per chapter
|
| 603 |
+
|
| 604 |
+
# Example:
|
| 605 |
+
{
|
| 606 |
+
"name": "Sarah",
|
| 607 |
+
"aliases": ["the silver-haired girl", "the mage"],
|
| 608 |
+
"visual_prompt": "young woman, long silver hair, blue eyes, slender build, wearing blue mage robes with gold trim, carrying wooden staff",
|
| 609 |
+
"reference_images": ["characters/sarah_ref1.png", "characters/sarah_ref2.png"],
|
| 610 |
+
"voice": {"engine": "edge-tts", "voice": "en-US-JennyNeural", "default_style": "cheerful"},
|
| 611 |
+
}
|
| 612 |
+
```
|
| 613 |
+
|
| 614 |
+
### 4.5 Web Novel Scraping
|
| 615 |
+
|
| 616 |
+
- **Primary target**: Royal Road (easy, clean HTML, minimal anti-scraping)
|
| 617 |
+
- **Best existing library**: `lightnovel-crawler` (supports 100+ sites)
|
| 618 |
+
- **Custom scraper**: BeautifulSoup + requests with 2s delay between pages
|
| 619 |
+
- **Legal**: Author opt-in system for commercial use. Public domain novels for demos.
|
| 620 |
+
|
| 621 |
+
---
|
| 622 |
+
|
| 623 |
+
## 5. SYSTEM ARCHITECTURE
|
| 624 |
+
|
| 625 |
+
### 5.1 Full Pipeline Overview
|
| 626 |
+
|
| 627 |
+
```
|
| 628 |
+
USER INPUT (URL or pasted text)
|
| 629 |
+
│
|
| 630 |
+
▼
|
| 631 |
+
┌─────────────────────┐
|
| 632 |
+
│ 1. TEXT INGESTION │ Scraper or paste → raw chapter text
|
| 633 |
+
│ (2 seconds) │ lightnovel-crawler / BeautifulSoup
|
| 634 |
+
└────────┬────────────┘
|
| 635 |
+
│
|
| 636 |
+
▼
|
| 637 |
+
┌─────────────────────┐
|
| 638 |
+
│ 2. NLP ANALYSIS │ spaCy + NLTK → entities, dialogue, breaks
|
| 639 |
+
│ (5 seconds) │ Gemini API → scenes, emotions, actions
|
| 640 |
+
└────────┬────────────┘
|
| 641 |
+
│
|
| 642 |
+
▼
|
| 643 |
+
┌─────────────────────┐
|
| 644 |
+
│ 3. VISUAL SCRIPT │ Scene → shots → image prompts
|
| 645 |
+
│ (10 seconds) │ Character DB → consistent descriptions
|
| 646 |
+
│ │ Camera angle selection
|
| 647 |
+
└────────┬────────────┘
|
| 648 |
+
│
|
| 649 |
+
▼
|
| 650 |
+
┌─────────────────────────────────────────┐
|
| 651 |
+
│ 4. ASSET GENERATION (parallel) │
|
| 652 |
+
│ │
|
| 653 |
+
│ ┌──────────────┐ ┌──────────────────┐ │
|
| 654 |
+
│ │ Image Gen │ │ TTS Generation │ │
|
| 655 |
+
│ │ 15-30 images │ │ All dialogue + │ │
|
| 656 |
+
│ │ via FLUX/SDXL│ │ narration via │ │
|
| 657 |
+
│ │ + IP-Adapter │ │ Edge TTS │ │
|
| 658 |
+
│ │ (2-5 min) │ │ (1-2 min) │ │
|
| 659 |
+
│ └──────┬───────┘ └──────┬───────────┘ │
|
| 660 |
+
└─────────┼─────────────────┼─────────────┘
|
| 661 |
+
│ │
|
| 662 |
+
▼ ▼
|
| 663 |
+
┌─────────────────────────────────────────┐
|
| 664 |
+
│ 5. ANIMATION LAYER │
|
| 665 |
+
│ │
|
| 666 |
+
│ • Ken Burns on all scenes (instant) │
|
| 667 |
+
│ • RIFE interpolation between keyframes │
|
| 668 |
+
│ • SadTalker for key dialogue faces │
|
| 669 |
+
│ • Lip sync via Rhubarb │
|
| 670 |
+
│ • Camera shake / effects for action │
|
| 671 |
+
│ • Depth parallax for atmosphere │
|
| 672 |
+
│ (1-3 min) │
|
| 673 |
+
└────────┬────────────────────────────────┘
|
| 674 |
+
│
|
| 675 |
+
▼
|
| 676 |
+
┌─────────────────────┐
|
| 677 |
+
│ 6. VIDEO ASSEMBLY │ FFmpeg: concat scenes + add subtitles
|
| 678 |
+
│ (30 seconds) │ Mix TTS + background music
|
| 679 |
+
│ │ Encode with Intel QSV
|
| 680 |
+
└────────┬────────────┘
|
| 681 |
+
│
|
| 682 |
+
▼
|
| 683 |
+
FINAL MP4 EPISODE
|
| 684 |
+
(3-10 minutes per chapter)
|
| 685 |
+
```
|
| 686 |
+
|
| 687 |
+
### 5.2 Tech Stack
|
| 688 |
+
|
| 689 |
+
```
|
| 690 |
+
Backend: Python 3.11+, FastAPI, Celery + Redis (task queue)
|
| 691 |
+
NLP: spaCy (en_core_web_lg) + NLTK + Gemini API
|
| 692 |
+
Image Gen: FLUX/SDXL via Together AI, HuggingFace, Pollinations, or Kaggle
|
| 693 |
+
Animation: RIFE (rife-ncnn-vulkan), SadTalker, FFmpeg, OpenCV
|
| 694 |
+
TTS: Edge TTS (primary), Bark (effects), Piper (offline fallback)
|
| 695 |
+
Video: FFmpeg + MoviePy, Intel QSV encoding
|
| 696 |
+
Frontend: Next.js + Tailwind CSS + Video.js player
|
| 697 |
+
Storage: Local filesystem → Cloudflare R2 (production)
|
| 698 |
+
Database: SQLite (start) → PostgreSQL (scale)
|
| 699 |
+
```
|
| 700 |
+
|
| 701 |
+
### 5.3 Performance Estimates (Your Hardware)
|
| 702 |
+
|
| 703 |
+
| Task | Time | Tool |
|
| 704 |
+
|------|------|------|
|
| 705 |
+
| Scrape 1 chapter | 3s | requests/beautifulsoup |
|
| 706 |
+
| NLP analysis | 5s | spaCy local |
|
| 707 |
+
| LLM scene parsing | 10s | Gemini API |
|
| 708 |
+
| Generate 1 image (API) | 5-15s | FLUX via Together/HF/Pollinations |
|
| 709 |
+
| Generate 1 image (Kaggle) | 10-30s | SDXL + IP-Adapter |
|
| 710 |
+
| RIFE interpolation (1 pair) | 0.5-2s | ncnn-vulkan on Intel Arc |
|
| 711 |
+
| Ken Burns 5s clip | <1s | FFmpeg |
|
| 712 |
+
| TTS for 1 line | 1-3s | Edge TTS |
|
| 713 |
+
| SadTalker 10s video | 30-60s | CPU mode |
|
| 714 |
+
| Depth map | 1-3s | MiDaS + OpenVINO on Intel Arc |
|
| 715 |
+
| Final encoding | 5-10s/min | FFmpeg + QSV |
|
| 716 |
+
| **FULL CHAPTER** | **~5-15 minutes** | **Complete pipeline** |
|
| 717 |
+
|
| 718 |
+
---
|
| 719 |
+
|
| 720 |
+
## 6. YOUR HARDWARE ADVANTAGE
|
| 721 |
+
|
| 722 |
+
### Intel Core Ultra 5 + Intel Arc Graphics
|
| 723 |
+
|
| 724 |
+
Your hardware is actually better than you think for AI:
|
| 725 |
+
|
| 726 |
+
1. **Intel Arc GPU supports**:
|
| 727 |
+
- OpenVINO (Intel's AI inference toolkit)
|
| 728 |
+
- Vulkan (for ncnn-based models like RIFE)
|
| 729 |
+
- Intel QSV (hardware video encoding -- 3-5x faster)
|
| 730 |
+
- DirectML (Windows AI acceleration)
|
| 731 |
+
|
| 732 |
+
2. **NPU (Neural Processing Unit)** in Core Ultra:
|
| 733 |
+
- Can run lightweight models (face detection, classification)
|
| 734 |
+
- Accessible via OpenVINO
|
| 735 |
+
|
| 736 |
+
3. **What this means for your pipeline**:
|
| 737 |
+
|
| 738 |
+
| Task | Without Intel Arc | With Intel Arc (OpenVINO/Vulkan) |
|
| 739 |
+
|------|-------------------|----------------------------------|
|
| 740 |
+
| RIFE interpolation | 2-5s/frame on CPU | **0.1-0.5s/frame** |
|
| 741 |
+
| MiDaS depth | 5-10s on CPU | **1-3s** |
|
| 742 |
+
| First Order Motion | Not practical on CPU | **5-15 FPS** |
|
| 743 |
+
| Video encoding | Software libx264 | **h264_qsv (3-5x faster)** |
|
| 744 |
+
|
| 745 |
+
### Setup Commands
|
| 746 |
+
|
| 747 |
+
```bash
|
| 748 |
+
# Install OpenVINO
|
| 749 |
+
pip install openvino openvino-dev
|
| 750 |
+
|
| 751 |
+
# Install Intel PyTorch Extension
|
| 752 |
+
pip install intel-extension-for-pytorch
|
| 753 |
+
|
| 754 |
+
# RIFE with Vulkan (works on Intel Arc)
|
| 755 |
+
pip install rife-ncnn-vulkan-python
|
| 756 |
+
|
| 757 |
+
# Verify Intel Arc is detected
|
| 758 |
+
python -c "from openvino.runtime import Core; print(Core().available_devices)"
|
| 759 |
+
# Expected: ['CPU', 'GPU', 'NPU']
|
| 760 |
+
```
|
| 761 |
+
|
| 762 |
+
---
|
| 763 |
+
|
| 764 |
+
## 7. IMPLEMENTATION ROADMAP
|
| 765 |
+
|
| 766 |
+
### Phase 1: Proof of Concept (Week 1-2)
|
| 767 |
+
**Goal**: URL → scene JSON → static image set + narration audio
|
| 768 |
+
|
| 769 |
+
- Set up project structure
|
| 770 |
+
- Build Royal Road scraper
|
| 771 |
+
- Implement NLP pipeline (spaCy + Gemini)
|
| 772 |
+
- Build basic prompt generator
|
| 773 |
+
- Generate images via Pollinations.ai (zero friction, no API key)
|
| 774 |
+
- Generate TTS via Edge TTS
|
| 775 |
+
- **Deliverable**: A folder of images + audio files for one chapter
|
| 776 |
+
|
| 777 |
+
### Phase 2: First Video (Week 3-4)
|
| 778 |
+
**Goal**: Images + audio → watchable motion comic episode
|
| 779 |
+
|
| 780 |
+
- Implement FFmpeg Ken Burns pipeline
|
| 781 |
+
- Build scene assembler (concat, transitions, subtitles)
|
| 782 |
+
- Add background music mixing
|
| 783 |
+
- Generate SRT subtitles from TTS timestamps
|
| 784 |
+
- **Deliverable**: First complete video episode from a web novel chapter
|
| 785 |
+
|
| 786 |
+
### Phase 3: Character Consistency (Week 5-6)
|
| 787 |
+
**Goal**: Characters look the same across all scenes
|
| 788 |
+
|
| 789 |
+
- Set up ComfyUI on Kaggle notebook
|
| 790 |
+
- Implement IP-Adapter / PuLID workflow
|
| 791 |
+
- Build character management system
|
| 792 |
+
- Generate character reference sheets
|
| 793 |
+
- Switch from Pollinations to Kaggle/Together for image gen
|
| 794 |
+
- **Deliverable**: Episode with consistent characters
|
| 795 |
+
|
| 796 |
+
### Phase 4: Animation Enhancement (Week 7-8)
|
| 797 |
+
**Goal**: Add motion beyond Ken Burns
|
| 798 |
+
|
| 799 |
+
- Integrate RIFE interpolation (ncnn-vulkan on Intel Arc)
|
| 800 |
+
- Add SadTalker for talking face close-ups
|
| 801 |
+
- Implement Rhubarb lip sync
|
| 802 |
+
- Add anime effects (speed lines, camera shake, flash)
|
| 803 |
+
- Implement depth-based parallax via MiDaS
|
| 804 |
+
- **Deliverable**: Significantly more animated episodes
|
| 805 |
+
|
| 806 |
+
### Phase 5: Web Application (Week 9-10)
|
| 807 |
+
**Goal**: Users can access the tool via browser
|
| 808 |
+
|
| 809 |
+
- Build FastAPI backend with task queue (Celery)
|
| 810 |
+
- Build Next.js frontend (paste URL → get video)
|
| 811 |
+
- Add progress tracking
|
| 812 |
+
- Add art style selection
|
| 813 |
+
- Add video player
|
| 814 |
+
- **Deliverable**: Working web app
|
| 815 |
+
|
| 816 |
+
### Phase 6: Launch & Monetize (Week 11+)
|
| 817 |
+
**Goal**: Get users and start earning
|
| 818 |
+
|
| 819 |
+
- Deploy to cloud (Vercel frontend, Railway backend)
|
| 820 |
+
- Implement user accounts
|
| 821 |
+
- Add Stripe payments (freemium model)
|
| 822 |
+
- Create demo episodes for marketing (TikTok, YouTube Shorts)
|
| 823 |
+
- Submit to ProductHunt, HackerNews, Reddit
|
| 824 |
+
- Author partnership program
|
| 825 |
+
|
| 826 |
+
---
|
| 827 |
+
|
| 828 |
+
## 8. MONETIZATION STRATEGY
|
| 829 |
+
|
| 830 |
+
### Pricing Model (Freemium SaaS)
|
| 831 |
+
|
| 832 |
+
```
|
| 833 |
+
FREE TIER:
|
| 834 |
+
- 1 chapter/day
|
| 835 |
+
- 720p resolution
|
| 836 |
+
- Watermarked
|
| 837 |
+
- Basic art style
|
| 838 |
+
- 30-second preview
|
| 839 |
+
|
| 840 |
+
PREMIUM ($9.99/month):
|
| 841 |
+
- 10 chapters/day
|
| 842 |
+
- 1080p resolution
|
| 843 |
+
- No watermark
|
| 844 |
+
- Multiple art styles
|
| 845 |
+
- Full episodes
|
| 846 |
+
- Download as MP4
|
| 847 |
+
|
| 848 |
+
PRO ($29.99/month):
|
| 849 |
+
- Unlimited chapters
|
| 850 |
+
- 4K resolution
|
| 851 |
+
- API access
|
| 852 |
+
- Custom character LoRA training
|
| 853 |
+
- Batch processing
|
| 854 |
+
- Commercial usage rights
|
| 855 |
+
```
|
| 856 |
+
|
| 857 |
+
### Revenue Projections (Conservative)
|
| 858 |
+
|
| 859 |
+
```
|
| 860 |
+
Month 6: 10K users, 500 paid → ~$5,000/mo
|
| 861 |
+
Month 12: 50K users, 2,500 paid → ~$25,000/mo
|
| 862 |
+
Month 24: 200K users, 10,000 paid → ~$100,000/mo
|
| 863 |
+
```
|
| 864 |
+
|
| 865 |
+
### Market Opportunity
|
| 866 |
+
|
| 867 |
+
- 200M+ monthly web novel readers globally
|
| 868 |
+
- Zero competitors doing exactly this (novel text → anime video)
|
| 869 |
+
- Growing AI video generation market ($500M+ in 2024)
|
| 870 |
+
- Strong social media virality potential (anime clips from popular novels)
|
| 871 |
+
|
| 872 |
+
---
|
| 873 |
+
|
| 874 |
+
## 9. FINAL RECOMMENDATION
|
| 875 |
+
|
| 876 |
+
### The Strategy
|
| 877 |
+
|
| 878 |
+
**Plan A (pure video gen) is not viable for free** -- but it's useful for ~5% of content (hero moments via HuggingFace).
|
| 879 |
+
|
| 880 |
+
**Plan B IS the plan.** And it's not a compromise. Here's why:
|
| 881 |
+
|
| 882 |
+
The "motion comic" style (images + Ken Burns + TTS + effects + interpolation) is:
|
| 883 |
+
1. Literally how real anime handles dialogue and atmospheric scenes
|
| 884 |
+
2. Fully achievable on your hardware at $0 cost
|
| 885 |
+
3. Character-consistent (via IP-Adapter)
|
| 886 |
+
4. Infinitely controllable (every frame, every angle, every expression)
|
| 887 |
+
5. Scalable to production quality
|
| 888 |
+
|
| 889 |
+
### Your Secret Weapons
|
| 890 |
+
|
| 891 |
+
1. **Edge TTS** -- Unlimited free voice acting with emotion control
|
| 892 |
+
2. **Pollinations.ai / Together AI / HuggingFace** -- Free anime image generation
|
| 893 |
+
3. **Intel Arc + OpenVINO** -- Accelerated AI inference most people don't have
|
| 894 |
+
4. **RIFE + ncnn-vulkan** -- Smooth frame interpolation on Intel Arc
|
| 895 |
+
5. **Gemini API** -- 1500 free scene parsing calls/day (100+ chapters/day)
|
| 896 |
+
6. **Kaggle free GPUs** -- 30 hrs/week for heavy image gen with IP-Adapter
|
| 897 |
+
|
| 898 |
+
### What to Build First
|
| 899 |
+
|
| 900 |
+
```
|
| 901 |
+
THIS WEEKEND:
|
| 902 |
+
1. pip install edge-tts spacy google-generativeai requests beautifulsoup4
|
| 903 |
+
2. Scrape a Royal Road chapter
|
| 904 |
+
3. Parse it with spaCy + Gemini into scenes
|
| 905 |
+
4. Generate images via Pollinations.ai
|
| 906 |
+
5. Generate TTS via Edge TTS
|
| 907 |
+
6. FFmpeg: images + audio → your first anime episode
|
| 908 |
+
|
| 909 |
+
COST: $0 | TIME: 1 weekend | RESULT: Proof of concept video
|
| 910 |
+
```
|
| 911 |
+
|
| 912 |
+
### Key Dependencies to Install
|
| 913 |
+
|
| 914 |
+
```bash
|
| 915 |
+
# Core
|
| 916 |
+
pip install edge-tts spacy nltk google-generativeai groq
|
| 917 |
+
pip install requests beautifulsoup4 Pillow opencv-python
|
| 918 |
+
pip install moviepy numpy librosa
|
| 919 |
+
|
| 920 |
+
# Intel Arc acceleration
|
| 921 |
+
pip install openvino rife-ncnn-vulkan-python
|
| 922 |
+
|
| 923 |
+
# NLP models
|
| 924 |
+
python -m spacy download en_core_web_lg
|
| 925 |
+
python -m nltk.downloader vader_lexicon punkt
|
| 926 |
+
|
| 927 |
+
# For HuggingFace model access
|
| 928 |
+
pip install huggingface-hub gradio-client
|
| 929 |
+
|
| 930 |
+
# For web novel scraping
|
| 931 |
+
pip install lightnovel-crawler
|
| 932 |
+
```
|
| 933 |
+
|
| 934 |
+
---
|
| 935 |
+
|
| 936 |
+
## APPENDIX: Quick Reference of All Free APIs
|
| 937 |
+
|
| 938 |
+
| Service | What | Free Limit | API Key? | URL |
|
| 939 |
+
|---------|------|-----------|----------|-----|
|
| 940 |
+
| Edge TTS | Voice synthesis | Unlimited | No | `pip install edge-tts` |
|
| 941 |
+
| Pollinations.ai | Image generation (FLUX) | Unlimited | No | image.pollinations.ai |
|
| 942 |
+
| Google Gemini | LLM (scene parsing) | 1500 req/day | Yes (free) | aistudio.google.com |
|
| 943 |
+
| Groq | LLM (fast inference) | 14,400 req/day | Yes (free) | console.groq.com |
|
| 944 |
+
| HuggingFace | Images, video, TTS, LLMs | Rate-limited | Yes (free) | huggingface.co |
|
| 945 |
+
| Together AI | Images (FLUX) + LLMs | $5 free credits | Yes (free) | api.together.xyz |
|
| 946 |
+
| Azure TTS | Voice synthesis + emotion | 500K chars/mo | Yes (free) | azure.microsoft.com |
|
| 947 |
+
| Kaggle | Free T4 GPUs | 30 hrs/week | Account | kaggle.com |
|
| 948 |
+
| Google Colab | Free T4 GPUs | ~4 hrs/day | Account | colab.research.google.com |
|
| 949 |
+
| OpenRouter | Various free LLMs | 200 req/day | Yes (free) | openrouter.ai |
|
| 950 |
+
|
| 951 |
+
---
|
| 952 |
+
|
| 953 |
+
*This research was conducted across web searches, model documentation, API documentation, and community reports as of February 2026. Verify current free tier limits before relying on them in production.*
|
TESTED_APIS.md
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Image Generation APIs -- TESTED RESULTS (2026-02-27)
|
| 2 |
+
|
| 3 |
+
## CONFIRMED WORKING (with your actual API keys)
|
| 4 |
+
|
| 5 |
+
### 1. NVIDIA NIM -- FLUX.1-dev (BEST QUALITY)
|
| 6 |
+
- **Status**: WORKING
|
| 7 |
+
- **Speed**: 4.8 seconds per image
|
| 8 |
+
- **Resolution**: 1024x768
|
| 9 |
+
- **Anime Quality**: EXCELLENT (professional anime key visual level)
|
| 10 |
+
- **Free Credits**: 1,000 on signup (up to 5,000 with business email + 90-day enterprise license)
|
| 11 |
+
- **API**:
|
| 12 |
+
```python
|
| 13 |
+
import requests, base64
|
| 14 |
+
|
| 15 |
+
r = requests.post(
|
| 16 |
+
"https://ai.api.nvidia.com/v1/genai/black-forest-labs/flux.1-dev",
|
| 17 |
+
headers={"Authorization": "Bearer YOUR_NVIDIA_KEY", "Accept": "application/json"},
|
| 18 |
+
json={
|
| 19 |
+
"prompt": "anime style, young warrior girl, silver hair, blue eyes, glowing sword, fantasy forest, masterpiece",
|
| 20 |
+
"mode": "base",
|
| 21 |
+
"cfg_scale": 3.5,
|
| 22 |
+
"width": 1024,
|
| 23 |
+
"height": 768,
|
| 24 |
+
"seed": -1,
|
| 25 |
+
"steps": 30
|
| 26 |
+
}
|
| 27 |
+
)
|
| 28 |
+
img_bytes = base64.b64decode(r.json()["artifacts"][0]["base64"])
|
| 29 |
+
with open("output.png", "wb") as f:
|
| 30 |
+
f.write(img_bytes)
|
| 31 |
+
```
|
| 32 |
+
|
| 33 |
+
### 2. NVIDIA NIM -- FLUX.1-schnell (FASTEST)
|
| 34 |
+
- **Status**: WORKING
|
| 35 |
+
- **Speed**: 3.4 seconds per image
|
| 36 |
+
- **Resolution**: 1024x1024
|
| 37 |
+
- **Anime Quality**: EXCELLENT
|
| 38 |
+
- **API**: Same as above, endpoint: `black-forest-labs/flux.1-schnell`, steps=4
|
| 39 |
+
|
| 40 |
+
### 3. HuggingFace -- FLUX.1-schnell (MOST GENEROUS FREE)
|
| 41 |
+
- **Status**: WORKING
|
| 42 |
+
- **Speed**: 1.4-3.8 seconds per image
|
| 43 |
+
- **Resolution**: up to 1024x768
|
| 44 |
+
- **Anime Quality**: EXCELLENT
|
| 45 |
+
- **Free Tier**: Rate-limited but generous (3 rapid requests all succeeded)
|
| 46 |
+
- **API**:
|
| 47 |
+
```python
|
| 48 |
+
import requests
|
| 49 |
+
|
| 50 |
+
r = requests.post(
|
| 51 |
+
"https://router.huggingface.co/hf-inference/models/black-forest-labs/FLUX.1-schnell",
|
| 52 |
+
headers={"Authorization": "Bearer YOUR_HF_TOKEN", "Content-Type": "application/json"},
|
| 53 |
+
json={
|
| 54 |
+
"inputs": "1girl, silver hair, blue eyes, holding sword, anime style, masterpiece",
|
| 55 |
+
"parameters": {"negative_prompt": "low quality, blurry", "width": 1024, "height": 768}
|
| 56 |
+
}
|
| 57 |
+
)
|
| 58 |
+
with open("output.png", "wb") as f:
|
| 59 |
+
f.write(r.content) # Returns raw image bytes
|
| 60 |
+
```
|
| 61 |
+
|
| 62 |
+
### 4. HuggingFace -- SDXL Base (GOOD PAINTERLY STYLE)
|
| 63 |
+
- **Status**: WORKING
|
| 64 |
+
- **Speed**: 9.8 seconds per image
|
| 65 |
+
- **Resolution**: up to 1024x768
|
| 66 |
+
- **Anime Quality**: EXCELLENT (more painterly/artistic style)
|
| 67 |
+
- **API**: Same as above, model: `stabilityai/stable-diffusion-xl-base-1.0`
|
| 68 |
+
|
| 69 |
+
---
|
| 70 |
+
|
| 71 |
+
## NOT WORKING / NOT FREE
|
| 72 |
+
|
| 73 |
+
| Service | Status | Reason |
|
| 74 |
+
|---------|--------|--------|
|
| 75 |
+
| OpenRouter | NO free image models | All 29 free models are text-only |
|
| 76 |
+
| Prodia | Needs paid upgrade | Free accounts can't create API tokens |
|
| 77 |
+
| Google Gemini | Quota = 0 on free tier | Image gen removed from free tier entirely |
|
| 78 |
+
| Google Imagen | Paid only | "Imagen 3 is only available on paid plans" |
|
| 79 |
+
| Pollinations | HTTP 530 | Cloudflare blocking (may work from other networks) |
|
| 80 |
+
| Together AI | No free tier (2026) | User confirmed free tier discontinued |
|
| 81 |
+
| DeepInfra | Needs auth/credits | 403 without authentication |
|
| 82 |
+
| fal.ai | Needs auth/credits | 401 without authentication |
|
| 83 |
+
| HF FLUX.1-dev | DEPRECATED | "No longer supported by provider hf-inference" |
|
| 84 |
+
| HF Animagine XL | 404 | Not available on inference API |
|
| 85 |
+
| Stability AI | Needs API key | 401 without key |
|
| 86 |
+
|
| 87 |
+
---
|
| 88 |
+
|
| 89 |
+
## RECOMMENDED STRATEGY
|
| 90 |
+
|
| 91 |
+
### Primary: HuggingFace FLUX.1-schnell
|
| 92 |
+
- Fastest (1.4-3.8s)
|
| 93 |
+
- No visible hard limit yet
|
| 94 |
+
- Excellent anime quality
|
| 95 |
+
- Use for bulk image generation
|
| 96 |
+
|
| 97 |
+
### Secondary: NVIDIA NIM FLUX.1-dev
|
| 98 |
+
- Best quality (30-step generation)
|
| 99 |
+
- 1,000-5,000 free credits
|
| 100 |
+
- Use for hero/key moment images
|
| 101 |
+
- Save credits for important scenes
|
| 102 |
+
|
| 103 |
+
### Fallback: HuggingFace SDXL
|
| 104 |
+
- Different artistic style
|
| 105 |
+
- Good for variety
|
| 106 |
+
- 9.8s per image (slower)
|
| 107 |
+
|
| 108 |
+
### Combined Monthly Estimate
|
| 109 |
+
- HuggingFace: Unknown hard limit, but generous for free tier
|
| 110 |
+
- NVIDIA: 1,000+ credits (unknown images-per-credit ratio)
|
| 111 |
+
- Pollinations: ~1,000 images (if accessible from your network)
|
| 112 |
+
- **Total: Likely 1,000-5,000+ images/month for $0**
|
| 113 |
+
|
| 114 |
+
### Images Per Chapter
|
| 115 |
+
- A 3-5 minute episode needs ~20-30 keyframe images
|
| 116 |
+
- At 2,000 images/month = ~66-100 chapters/month
|
| 117 |
+
- More than enough for production
|
| 118 |
+
|
| 119 |
+
---
|
| 120 |
+
|
| 121 |
+
## QUALITY RANKING (for anime)
|
| 122 |
+
|
| 123 |
+
1. NVIDIA FLUX.1-dev (30 steps) -- Most detailed, best composition
|
| 124 |
+
2. NVIDIA FLUX.1-schnell -- Great quality, fastest NVIDIA
|
| 125 |
+
3. HuggingFace FLUX.1-schnell -- Slightly different style, very fast
|
| 126 |
+
4. HuggingFace SDXL -- More painterly, artistic feel
|
| 127 |
+
|
| 128 |
+
All four produce professional-grade anime art suitable for our project.
|
app/__init__.py
ADDED
|
File without changes
|
app/api/__init__.py
ADDED
|
File without changes
|
app/api/auth.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import APIRouter, Depends, HTTPException, Response, Cookie, status
|
| 2 |
+
from sqlalchemy import select
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.database import get_db
|
| 5 |
+
from app.models.user import User
|
| 6 |
+
from app.schemas.auth import UserCreate, UserLogin, UserResponse, TokenResponse
|
| 7 |
+
from app.services.auth import (
|
| 8 |
+
hash_password, verify_password,
|
| 9 |
+
create_access_token, create_refresh_token, decode_token,
|
| 10 |
+
get_current_user,
|
| 11 |
+
)
|
| 12 |
+
from app.config import get_settings
|
| 13 |
+
|
| 14 |
+
router = APIRouter(prefix="/api/auth", tags=["auth"])
|
| 15 |
+
settings = get_settings()
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
@router.post("/register", response_model=UserResponse, status_code=201)
|
| 19 |
+
async def register(data: UserCreate, db: AsyncSession = Depends(get_db)):
|
| 20 |
+
# Check existing
|
| 21 |
+
existing = await db.execute(
|
| 22 |
+
select(User).where((User.email == data.email) | (User.username == data.username))
|
| 23 |
+
)
|
| 24 |
+
if existing.scalar_one_or_none():
|
| 25 |
+
raise HTTPException(status_code=409, detail="Email or username already registered")
|
| 26 |
+
|
| 27 |
+
user = User(
|
| 28 |
+
email=data.email,
|
| 29 |
+
username=data.username,
|
| 30 |
+
password_hash=hash_password(data.password),
|
| 31 |
+
)
|
| 32 |
+
db.add(user)
|
| 33 |
+
await db.commit()
|
| 34 |
+
await db.refresh(user)
|
| 35 |
+
return user
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
@router.post("/login", response_model=TokenResponse)
|
| 39 |
+
async def login(data: UserLogin, response: Response, db: AsyncSession = Depends(get_db)):
|
| 40 |
+
result = await db.execute(select(User).where(User.email == data.email))
|
| 41 |
+
user = result.scalar_one_or_none()
|
| 42 |
+
if not user or not verify_password(data.password, user.password_hash):
|
| 43 |
+
raise HTTPException(status_code=401, detail="Invalid credentials")
|
| 44 |
+
|
| 45 |
+
access_token = create_access_token(user.id)
|
| 46 |
+
refresh_token = create_refresh_token(user.id)
|
| 47 |
+
|
| 48 |
+
response.set_cookie(
|
| 49 |
+
key="refresh_token",
|
| 50 |
+
value=refresh_token,
|
| 51 |
+
httponly=True,
|
| 52 |
+
secure=False, # Set True in production with HTTPS
|
| 53 |
+
samesite="lax",
|
| 54 |
+
max_age=settings.refresh_token_expire_days * 86400,
|
| 55 |
+
path="/api/auth",
|
| 56 |
+
)
|
| 57 |
+
|
| 58 |
+
return TokenResponse(access_token=access_token)
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
@router.post("/refresh", response_model=TokenResponse)
|
| 62 |
+
async def refresh(
|
| 63 |
+
response: Response,
|
| 64 |
+
refresh_token: str | None = Cookie(default=None),
|
| 65 |
+
db: AsyncSession = Depends(get_db),
|
| 66 |
+
):
|
| 67 |
+
if not refresh_token:
|
| 68 |
+
raise HTTPException(status_code=401, detail="No refresh token")
|
| 69 |
+
|
| 70 |
+
payload = decode_token(refresh_token)
|
| 71 |
+
if payload.get("type") != "refresh":
|
| 72 |
+
raise HTTPException(status_code=401, detail="Invalid token type")
|
| 73 |
+
|
| 74 |
+
user_id = int(payload["sub"])
|
| 75 |
+
result = await db.execute(select(User).where(User.id == user_id))
|
| 76 |
+
user = result.scalar_one_or_none()
|
| 77 |
+
if not user:
|
| 78 |
+
raise HTTPException(status_code=401, detail="User not found")
|
| 79 |
+
|
| 80 |
+
new_access = create_access_token(user.id)
|
| 81 |
+
new_refresh = create_refresh_token(user.id)
|
| 82 |
+
|
| 83 |
+
response.set_cookie(
|
| 84 |
+
key="refresh_token",
|
| 85 |
+
value=new_refresh,
|
| 86 |
+
httponly=True,
|
| 87 |
+
secure=False,
|
| 88 |
+
samesite="lax",
|
| 89 |
+
max_age=settings.refresh_token_expire_days * 86400,
|
| 90 |
+
path="/api/auth",
|
| 91 |
+
)
|
| 92 |
+
|
| 93 |
+
return TokenResponse(access_token=new_access)
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
@router.post("/logout")
|
| 97 |
+
async def logout(response: Response):
|
| 98 |
+
response.delete_cookie("refresh_token", path="/api/auth")
|
| 99 |
+
return {"message": "Logged out"}
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
@router.get("/me", response_model=UserResponse)
|
| 103 |
+
async def me(user: User = Depends(get_current_user)):
|
| 104 |
+
return user
|
app/api/chapters.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import APIRouter, Depends, HTTPException
|
| 2 |
+
from sqlalchemy import select
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.database import get_db
|
| 5 |
+
from app.models.user import User
|
| 6 |
+
from app.models.project import Project
|
| 7 |
+
from app.models.chapter import Chapter
|
| 8 |
+
from app.schemas.chapter import ChapterCreate, ChapterResponse
|
| 9 |
+
from app.services.auth import get_current_user
|
| 10 |
+
|
| 11 |
+
router = APIRouter(prefix="/api/projects/{project_id}/chapters", tags=["chapters"])
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
|
| 15 |
+
result = await db.execute(
|
| 16 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 17 |
+
)
|
| 18 |
+
project = result.scalar_one_or_none()
|
| 19 |
+
if not project:
|
| 20 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 21 |
+
return project
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
@router.get("", response_model=list[ChapterResponse])
|
| 25 |
+
async def list_chapters(
|
| 26 |
+
project_id: int,
|
| 27 |
+
user: User = Depends(get_current_user),
|
| 28 |
+
db: AsyncSession = Depends(get_db),
|
| 29 |
+
):
|
| 30 |
+
await _get_user_project(project_id, user, db)
|
| 31 |
+
result = await db.execute(
|
| 32 |
+
select(Chapter)
|
| 33 |
+
.where(Chapter.project_id == project_id)
|
| 34 |
+
.order_by(Chapter.chapter_number)
|
| 35 |
+
)
|
| 36 |
+
return list(result.scalars().all())
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
@router.post("", response_model=ChapterResponse, status_code=201)
|
| 40 |
+
async def create_chapter(
|
| 41 |
+
project_id: int,
|
| 42 |
+
data: ChapterCreate,
|
| 43 |
+
user: User = Depends(get_current_user),
|
| 44 |
+
db: AsyncSession = Depends(get_db),
|
| 45 |
+
):
|
| 46 |
+
await _get_user_project(project_id, user, db)
|
| 47 |
+
chapter = Chapter(
|
| 48 |
+
project_id=project_id,
|
| 49 |
+
chapter_number=data.chapter_number,
|
| 50 |
+
title=data.title,
|
| 51 |
+
source_text=data.source_text,
|
| 52 |
+
word_count=len(data.source_text.split()),
|
| 53 |
+
)
|
| 54 |
+
db.add(chapter)
|
| 55 |
+
await db.commit()
|
| 56 |
+
await db.refresh(chapter)
|
| 57 |
+
return chapter
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
@router.post("/bulk", response_model=list[ChapterResponse], status_code=201)
|
| 61 |
+
async def create_chapters_bulk(
|
| 62 |
+
project_id: int,
|
| 63 |
+
chapters: list[ChapterCreate],
|
| 64 |
+
user: User = Depends(get_current_user),
|
| 65 |
+
db: AsyncSession = Depends(get_db),
|
| 66 |
+
):
|
| 67 |
+
await _get_user_project(project_id, user, db)
|
| 68 |
+
db_chapters = []
|
| 69 |
+
for data in chapters:
|
| 70 |
+
chapter = Chapter(
|
| 71 |
+
project_id=project_id,
|
| 72 |
+
chapter_number=data.chapter_number,
|
| 73 |
+
title=data.title,
|
| 74 |
+
source_text=data.source_text,
|
| 75 |
+
word_count=len(data.source_text.split()),
|
| 76 |
+
)
|
| 77 |
+
db.add(chapter)
|
| 78 |
+
db_chapters.append(chapter)
|
| 79 |
+
await db.commit()
|
| 80 |
+
for ch in db_chapters:
|
| 81 |
+
await db.refresh(ch)
|
| 82 |
+
return db_chapters
|
app/api/characters.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import APIRouter, Depends, HTTPException
|
| 2 |
+
from sqlalchemy import select
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.database import get_db
|
| 5 |
+
from app.models.user import User
|
| 6 |
+
from app.models.project import Project
|
| 7 |
+
from app.models.character import Character
|
| 8 |
+
from app.schemas.character import CharacterCreate, CharacterUpdate, CharacterResponse
|
| 9 |
+
from app.services.auth import get_current_user
|
| 10 |
+
|
| 11 |
+
router = APIRouter(prefix="/api/projects/{project_id}/characters", tags=["characters"])
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
|
| 15 |
+
result = await db.execute(
|
| 16 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 17 |
+
)
|
| 18 |
+
project = result.scalar_one_or_none()
|
| 19 |
+
if not project:
|
| 20 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 21 |
+
return project
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
@router.get("", response_model=list[CharacterResponse])
|
| 25 |
+
async def list_characters(
|
| 26 |
+
project_id: int,
|
| 27 |
+
user: User = Depends(get_current_user),
|
| 28 |
+
db: AsyncSession = Depends(get_db),
|
| 29 |
+
):
|
| 30 |
+
await _get_user_project(project_id, user, db)
|
| 31 |
+
result = await db.execute(
|
| 32 |
+
select(Character)
|
| 33 |
+
.where(Character.project_id == project_id)
|
| 34 |
+
.order_by(Character.name)
|
| 35 |
+
)
|
| 36 |
+
return list(result.scalars().all())
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
@router.post("", response_model=CharacterResponse, status_code=201)
|
| 40 |
+
async def create_character(
|
| 41 |
+
project_id: int,
|
| 42 |
+
data: CharacterCreate,
|
| 43 |
+
user: User = Depends(get_current_user),
|
| 44 |
+
db: AsyncSession = Depends(get_db),
|
| 45 |
+
):
|
| 46 |
+
await _get_user_project(project_id, user, db)
|
| 47 |
+
character = Character(project_id=project_id, **data.model_dump(exclude_none=True))
|
| 48 |
+
db.add(character)
|
| 49 |
+
await db.commit()
|
| 50 |
+
await db.refresh(character)
|
| 51 |
+
return character
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
@router.patch("/{character_id}", response_model=CharacterResponse)
|
| 55 |
+
async def update_character(
|
| 56 |
+
project_id: int,
|
| 57 |
+
character_id: int,
|
| 58 |
+
data: CharacterUpdate,
|
| 59 |
+
user: User = Depends(get_current_user),
|
| 60 |
+
db: AsyncSession = Depends(get_db),
|
| 61 |
+
):
|
| 62 |
+
await _get_user_project(project_id, user, db)
|
| 63 |
+
result = await db.execute(
|
| 64 |
+
select(Character).where(Character.id == character_id, Character.project_id == project_id)
|
| 65 |
+
)
|
| 66 |
+
character = result.scalar_one_or_none()
|
| 67 |
+
if not character:
|
| 68 |
+
raise HTTPException(status_code=404, detail="Character not found")
|
| 69 |
+
|
| 70 |
+
for key, value in data.model_dump(exclude_none=True).items():
|
| 71 |
+
setattr(character, key, value)
|
| 72 |
+
await db.commit()
|
| 73 |
+
await db.refresh(character)
|
| 74 |
+
return character
|
app/api/episodes.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import APIRouter, Depends, HTTPException
|
| 2 |
+
from sqlalchemy import select
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.database import get_db
|
| 5 |
+
from app.models.user import User
|
| 6 |
+
from app.models.project import Project
|
| 7 |
+
from app.models.episode import Episode
|
| 8 |
+
from app.schemas.episode import EpisodeCreate, EpisodeResponse
|
| 9 |
+
from app.services.auth import get_current_user
|
| 10 |
+
|
| 11 |
+
router = APIRouter(prefix="/api/projects/{project_id}/episodes", tags=["episodes"])
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
async def _get_user_project(project_id: int, user: User, db: AsyncSession) -> Project:
|
| 15 |
+
result = await db.execute(
|
| 16 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 17 |
+
)
|
| 18 |
+
project = result.scalar_one_or_none()
|
| 19 |
+
if not project:
|
| 20 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 21 |
+
return project
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
@router.get("", response_model=list[EpisodeResponse])
|
| 25 |
+
async def list_episodes(
|
| 26 |
+
project_id: int,
|
| 27 |
+
user: User = Depends(get_current_user),
|
| 28 |
+
db: AsyncSession = Depends(get_db),
|
| 29 |
+
):
|
| 30 |
+
await _get_user_project(project_id, user, db)
|
| 31 |
+
result = await db.execute(
|
| 32 |
+
select(Episode)
|
| 33 |
+
.where(Episode.project_id == project_id)
|
| 34 |
+
.order_by(Episode.episode_number)
|
| 35 |
+
)
|
| 36 |
+
return list(result.scalars().all())
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
@router.post("", response_model=EpisodeResponse, status_code=201)
|
| 40 |
+
async def create_episode(
|
| 41 |
+
project_id: int,
|
| 42 |
+
data: EpisodeCreate,
|
| 43 |
+
user: User = Depends(get_current_user),
|
| 44 |
+
db: AsyncSession = Depends(get_db),
|
| 45 |
+
):
|
| 46 |
+
await _get_user_project(project_id, user, db)
|
| 47 |
+
episode = Episode(project_id=project_id, **data.model_dump(exclude_none=True))
|
| 48 |
+
db.add(episode)
|
| 49 |
+
await db.commit()
|
| 50 |
+
await db.refresh(episode)
|
| 51 |
+
return episode
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
@router.get("/{episode_id}", response_model=EpisodeResponse)
|
| 55 |
+
async def get_episode(
|
| 56 |
+
project_id: int,
|
| 57 |
+
episode_id: int,
|
| 58 |
+
user: User = Depends(get_current_user),
|
| 59 |
+
db: AsyncSession = Depends(get_db),
|
| 60 |
+
):
|
| 61 |
+
await _get_user_project(project_id, user, db)
|
| 62 |
+
result = await db.execute(
|
| 63 |
+
select(Episode).where(Episode.id == episode_id, Episode.project_id == project_id)
|
| 64 |
+
)
|
| 65 |
+
episode = result.scalar_one_or_none()
|
| 66 |
+
if not episode:
|
| 67 |
+
raise HTTPException(status_code=404, detail="Episode not found")
|
| 68 |
+
return episode
|
app/api/generation.py
ADDED
|
@@ -0,0 +1,241 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import asyncio
|
| 2 |
+
import json
|
| 3 |
+
import logging
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
from fastapi import APIRouter, Depends, HTTPException
|
| 6 |
+
from fastapi.responses import StreamingResponse
|
| 7 |
+
from sqlalchemy import select
|
| 8 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 9 |
+
from app.database import get_db
|
| 10 |
+
from app.models.user import User
|
| 11 |
+
from app.models.project import Project
|
| 12 |
+
from app.models.generation_job import GenerationJob
|
| 13 |
+
from app.schemas.generation import GenerationStart, GenerationJobResponse
|
| 14 |
+
from app.services.auth import get_current_user
|
| 15 |
+
from app.pipeline.orchestrator import run_pipeline
|
| 16 |
+
|
| 17 |
+
logger = logging.getLogger(__name__)
|
| 18 |
+
|
| 19 |
+
router = APIRouter(prefix="/api/projects/{project_id}", tags=["generation"])
|
| 20 |
+
|
| 21 |
+
# Keep references to running pipeline tasks so they don't get garbage-collected
|
| 22 |
+
_running_tasks: set[asyncio.Task] = set()
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def _launch_pipeline(job_id: int, resume: bool = False, chapter_ids: list[int] | None = None):
|
| 26 |
+
"""Launch pipeline as background task with error logging."""
|
| 27 |
+
async def _wrapper():
|
| 28 |
+
try:
|
| 29 |
+
await run_pipeline(job_id, resume=resume, chapter_ids=chapter_ids)
|
| 30 |
+
except Exception:
|
| 31 |
+
logger.exception("Pipeline failed for job %s", job_id)
|
| 32 |
+
|
| 33 |
+
task = asyncio.create_task(_wrapper())
|
| 34 |
+
_running_tasks.add(task)
|
| 35 |
+
task.add_done_callback(_running_tasks.discard)
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
# ── Asset endpoints ──────────────────────────────────────────────────
|
| 39 |
+
|
| 40 |
+
@router.get("/assets/images")
|
| 41 |
+
async def list_images(
|
| 42 |
+
project_id: int,
|
| 43 |
+
user: User = Depends(get_current_user),
|
| 44 |
+
db: AsyncSession = Depends(get_db),
|
| 45 |
+
):
|
| 46 |
+
"""List generated images for a project."""
|
| 47 |
+
result = await db.execute(
|
| 48 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 49 |
+
)
|
| 50 |
+
if not result.scalar_one_or_none():
|
| 51 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 52 |
+
|
| 53 |
+
img_dir = Path("workdir/projects") / str(project_id) / "images"
|
| 54 |
+
if not img_dir.exists():
|
| 55 |
+
return []
|
| 56 |
+
|
| 57 |
+
files = sorted(f.name for f in img_dir.iterdir() if f.suffix.lower() in (".png", ".jpg", ".jpeg", ".webp"))
|
| 58 |
+
return files
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
# ── Generation endpoints ─────────────────────────────────────────────
|
| 62 |
+
|
| 63 |
+
@router.post("/generate", response_model=GenerationJobResponse, status_code=201)
|
| 64 |
+
async def start_generation(
|
| 65 |
+
project_id: int,
|
| 66 |
+
data: GenerationStart,
|
| 67 |
+
user: User = Depends(get_current_user),
|
| 68 |
+
db: AsyncSession = Depends(get_db),
|
| 69 |
+
):
|
| 70 |
+
result = await db.execute(
|
| 71 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 72 |
+
)
|
| 73 |
+
project = result.scalar_one_or_none()
|
| 74 |
+
if not project:
|
| 75 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 76 |
+
|
| 77 |
+
job = GenerationJob(
|
| 78 |
+
project_id=project_id,
|
| 79 |
+
user_id=user.id,
|
| 80 |
+
episode_id=data.episode_id,
|
| 81 |
+
chapter_ids_json=data.chapter_ids,
|
| 82 |
+
status="queued",
|
| 83 |
+
current_stage="ingest",
|
| 84 |
+
progress_pct=0.0,
|
| 85 |
+
)
|
| 86 |
+
db.add(job)
|
| 87 |
+
await db.commit()
|
| 88 |
+
await db.refresh(job)
|
| 89 |
+
|
| 90 |
+
# Launch pipeline in background with error handling
|
| 91 |
+
_launch_pipeline(job.id, chapter_ids=data.chapter_ids)
|
| 92 |
+
|
| 93 |
+
return job
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
@router.get("/generate/jobs", response_model=list[GenerationJobResponse])
|
| 97 |
+
async def list_jobs(
|
| 98 |
+
project_id: int,
|
| 99 |
+
user: User = Depends(get_current_user),
|
| 100 |
+
db: AsyncSession = Depends(get_db),
|
| 101 |
+
):
|
| 102 |
+
result = await db.execute(
|
| 103 |
+
select(GenerationJob)
|
| 104 |
+
.where(GenerationJob.project_id == project_id, GenerationJob.user_id == user.id)
|
| 105 |
+
.order_by(GenerationJob.created_at.desc())
|
| 106 |
+
)
|
| 107 |
+
return list(result.scalars().all())
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
@router.get("/generate/jobs/{job_id}", response_model=GenerationJobResponse)
|
| 111 |
+
async def get_job(
|
| 112 |
+
project_id: int,
|
| 113 |
+
job_id: int,
|
| 114 |
+
user: User = Depends(get_current_user),
|
| 115 |
+
db: AsyncSession = Depends(get_db),
|
| 116 |
+
):
|
| 117 |
+
result = await db.execute(
|
| 118 |
+
select(GenerationJob).where(
|
| 119 |
+
GenerationJob.id == job_id,
|
| 120 |
+
GenerationJob.project_id == project_id,
|
| 121 |
+
GenerationJob.user_id == user.id,
|
| 122 |
+
)
|
| 123 |
+
)
|
| 124 |
+
job = result.scalar_one_or_none()
|
| 125 |
+
if not job:
|
| 126 |
+
raise HTTPException(status_code=404, detail="Job not found")
|
| 127 |
+
return job
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
@router.get("/generate/jobs/{job_id}/stream")
|
| 131 |
+
async def stream_progress(
|
| 132 |
+
project_id: int,
|
| 133 |
+
job_id: int,
|
| 134 |
+
user: User = Depends(get_current_user),
|
| 135 |
+
db: AsyncSession = Depends(get_db),
|
| 136 |
+
):
|
| 137 |
+
"""SSE endpoint for real-time generation progress."""
|
| 138 |
+
result = await db.execute(
|
| 139 |
+
select(GenerationJob).where(
|
| 140 |
+
GenerationJob.id == job_id,
|
| 141 |
+
GenerationJob.project_id == project_id,
|
| 142 |
+
GenerationJob.user_id == user.id,
|
| 143 |
+
)
|
| 144 |
+
)
|
| 145 |
+
job = result.scalar_one_or_none()
|
| 146 |
+
if not job:
|
| 147 |
+
raise HTTPException(status_code=404, detail="Job not found")
|
| 148 |
+
|
| 149 |
+
from app.database import async_session as session_factory
|
| 150 |
+
|
| 151 |
+
async def event_stream():
|
| 152 |
+
last_data = None
|
| 153 |
+
while True:
|
| 154 |
+
try:
|
| 155 |
+
async with session_factory() as session:
|
| 156 |
+
result = await session.execute(
|
| 157 |
+
select(GenerationJob).where(GenerationJob.id == job_id)
|
| 158 |
+
)
|
| 159 |
+
current_job = result.scalar_one_or_none()
|
| 160 |
+
if not current_job:
|
| 161 |
+
yield f"data: {json.dumps({'status': 'not_found'})}\n\n"
|
| 162 |
+
break
|
| 163 |
+
|
| 164 |
+
data = {
|
| 165 |
+
"status": current_job.status or "queued",
|
| 166 |
+
"stage": current_job.current_stage or "ingest",
|
| 167 |
+
"progress": current_job.progress_pct or 0,
|
| 168 |
+
"detail": (current_job.progress_detail or "")[:200],
|
| 169 |
+
"error": (current_job.error_message or "")[:500] if current_job.status == "failed" else None,
|
| 170 |
+
}
|
| 171 |
+
|
| 172 |
+
# Always send (even if same) so frontend knows connection is alive
|
| 173 |
+
yield f"data: {json.dumps(data)}\n\n"
|
| 174 |
+
last_data = data
|
| 175 |
+
|
| 176 |
+
if current_job.status in ("completed", "failed", "cancelled"):
|
| 177 |
+
break
|
| 178 |
+
except Exception:
|
| 179 |
+
logger.exception("SSE stream error for job %s", job_id)
|
| 180 |
+
yield f"data: {json.dumps({'status': 'error', 'detail': 'Server error'})}\n\n"
|
| 181 |
+
break
|
| 182 |
+
|
| 183 |
+
await asyncio.sleep(1)
|
| 184 |
+
|
| 185 |
+
return StreamingResponse(
|
| 186 |
+
event_stream(),
|
| 187 |
+
media_type="text/event-stream",
|
| 188 |
+
headers={
|
| 189 |
+
"Cache-Control": "no-cache",
|
| 190 |
+
"Connection": "keep-alive",
|
| 191 |
+
"X-Accel-Buffering": "no",
|
| 192 |
+
},
|
| 193 |
+
)
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
@router.post("/generate/retry", response_model=GenerationJobResponse, status_code=201)
|
| 197 |
+
async def retry_generation(
|
| 198 |
+
project_id: int,
|
| 199 |
+
user: User = Depends(get_current_user),
|
| 200 |
+
db: AsyncSession = Depends(get_db),
|
| 201 |
+
):
|
| 202 |
+
"""Retry the last failed job for a project, resuming from where it left off."""
|
| 203 |
+
result = await db.execute(
|
| 204 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 205 |
+
)
|
| 206 |
+
project = result.scalar_one_or_none()
|
| 207 |
+
if not project:
|
| 208 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 209 |
+
|
| 210 |
+
# Find the last failed job
|
| 211 |
+
result = await db.execute(
|
| 212 |
+
select(GenerationJob)
|
| 213 |
+
.where(
|
| 214 |
+
GenerationJob.project_id == project_id,
|
| 215 |
+
GenerationJob.user_id == user.id,
|
| 216 |
+
GenerationJob.status == "failed",
|
| 217 |
+
)
|
| 218 |
+
.order_by(GenerationJob.created_at.desc())
|
| 219 |
+
.limit(1)
|
| 220 |
+
)
|
| 221 |
+
failed_job = result.scalar_one_or_none()
|
| 222 |
+
if not failed_job:
|
| 223 |
+
raise HTTPException(status_code=404, detail="No failed job to retry")
|
| 224 |
+
|
| 225 |
+
# Create new job, carrying over chapter_ids from failed job
|
| 226 |
+
job = GenerationJob(
|
| 227 |
+
project_id=project_id,
|
| 228 |
+
user_id=user.id,
|
| 229 |
+
episode_id=failed_job.episode_id,
|
| 230 |
+
chapter_ids_json=failed_job.chapter_ids_json,
|
| 231 |
+
status="queued",
|
| 232 |
+
current_stage="ingest",
|
| 233 |
+
progress_pct=0.0,
|
| 234 |
+
)
|
| 235 |
+
db.add(job)
|
| 236 |
+
await db.commit()
|
| 237 |
+
await db.refresh(job)
|
| 238 |
+
|
| 239 |
+
_launch_pipeline(job.id, resume=True, chapter_ids=failed_job.chapter_ids_json)
|
| 240 |
+
|
| 241 |
+
return job
|
app/api/novels.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Novel search, URL resolution, and PDF upload endpoints."""
|
| 2 |
+
import os
|
| 3 |
+
import tempfile
|
| 4 |
+
from fastapi import APIRouter, Depends, HTTPException, UploadFile, File, Query
|
| 5 |
+
from pydantic import BaseModel
|
| 6 |
+
from app.services.auth import get_current_user
|
| 7 |
+
from app.models.user import User
|
| 8 |
+
from app.services.scraper import search_novels, resolve_novel_url, extract_text_from_pdf
|
| 9 |
+
|
| 10 |
+
router = APIRouter(prefix="/api/novels", tags=["novels"])
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
class SearchResult(BaseModel):
|
| 14 |
+
title: str
|
| 15 |
+
url: str
|
| 16 |
+
source: str
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
class NovelInfo(BaseModel):
|
| 20 |
+
title: str
|
| 21 |
+
author: str | None = None
|
| 22 |
+
cover_url: str | None = None
|
| 23 |
+
synopsis: str | None = None
|
| 24 |
+
chapters: list[dict] = []
|
| 25 |
+
source: str = ""
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@router.get("/search", response_model=list[SearchResult])
|
| 29 |
+
async def search(
|
| 30 |
+
q: str = Query(..., min_length=2),
|
| 31 |
+
user: User = Depends(get_current_user),
|
| 32 |
+
):
|
| 33 |
+
results = await search_novels(q)
|
| 34 |
+
return [SearchResult(title=r.title, url=r.url, source=r.source) for r in results]
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
@router.post("/resolve-url", response_model=NovelInfo)
|
| 38 |
+
async def resolve_url(
|
| 39 |
+
data: dict,
|
| 40 |
+
user: User = Depends(get_current_user),
|
| 41 |
+
):
|
| 42 |
+
url = data.get("url", "").strip()
|
| 43 |
+
if not url:
|
| 44 |
+
raise HTTPException(status_code=400, detail="URL is required")
|
| 45 |
+
|
| 46 |
+
result = await resolve_novel_url(url)
|
| 47 |
+
if not result:
|
| 48 |
+
raise HTTPException(status_code=404, detail="Could not resolve novel from URL")
|
| 49 |
+
|
| 50 |
+
return NovelInfo(
|
| 51 |
+
title=result.title,
|
| 52 |
+
author=result.author,
|
| 53 |
+
cover_url=result.cover_url,
|
| 54 |
+
synopsis=result.synopsis,
|
| 55 |
+
chapters=result.chapters,
|
| 56 |
+
source=result.source,
|
| 57 |
+
)
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
@router.post("/upload-pdf", response_model=list[dict])
|
| 61 |
+
async def upload_pdf(
|
| 62 |
+
file: UploadFile = File(...),
|
| 63 |
+
user: User = Depends(get_current_user),
|
| 64 |
+
):
|
| 65 |
+
if not file.filename or not file.filename.lower().endswith(".pdf"):
|
| 66 |
+
raise HTTPException(status_code=400, detail="Only PDF files are supported")
|
| 67 |
+
|
| 68 |
+
# Save to temp file
|
| 69 |
+
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as tmp:
|
| 70 |
+
content = await file.read()
|
| 71 |
+
tmp.write(content)
|
| 72 |
+
tmp_path = tmp.name
|
| 73 |
+
|
| 74 |
+
try:
|
| 75 |
+
chapters = await extract_text_from_pdf(tmp_path)
|
| 76 |
+
if not chapters:
|
| 77 |
+
raise HTTPException(status_code=422, detail="Could not extract text from PDF")
|
| 78 |
+
return chapters
|
| 79 |
+
finally:
|
| 80 |
+
os.unlink(tmp_path)
|
app/api/projects.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import APIRouter, Depends, HTTPException
|
| 2 |
+
from sqlalchemy import select, func
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.database import get_db
|
| 5 |
+
from app.models.user import User
|
| 6 |
+
from app.models.project import Project
|
| 7 |
+
from app.schemas.project import ProjectCreate, ProjectUpdate, ProjectResponse, ProjectListResponse
|
| 8 |
+
from app.services.auth import get_current_user
|
| 9 |
+
|
| 10 |
+
router = APIRouter(prefix="/api/projects", tags=["projects"])
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
@router.get("", response_model=ProjectListResponse)
|
| 14 |
+
async def list_projects(
|
| 15 |
+
skip: int = 0,
|
| 16 |
+
limit: int = 20,
|
| 17 |
+
user: User = Depends(get_current_user),
|
| 18 |
+
db: AsyncSession = Depends(get_db),
|
| 19 |
+
):
|
| 20 |
+
result = await db.execute(
|
| 21 |
+
select(Project)
|
| 22 |
+
.where(Project.user_id == user.id)
|
| 23 |
+
.order_by(Project.updated_at.desc())
|
| 24 |
+
.offset(skip).limit(limit)
|
| 25 |
+
)
|
| 26 |
+
projects = list(result.scalars().all())
|
| 27 |
+
count_result = await db.execute(
|
| 28 |
+
select(func.count()).select_from(Project).where(Project.user_id == user.id)
|
| 29 |
+
)
|
| 30 |
+
total = count_result.scalar() or 0
|
| 31 |
+
return ProjectListResponse(projects=projects, total=total)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
@router.post("", response_model=ProjectResponse, status_code=201)
|
| 35 |
+
async def create_project(
|
| 36 |
+
data: ProjectCreate,
|
| 37 |
+
user: User = Depends(get_current_user),
|
| 38 |
+
db: AsyncSession = Depends(get_db),
|
| 39 |
+
):
|
| 40 |
+
project = Project(user_id=user.id, **data.model_dump(exclude_none=True))
|
| 41 |
+
db.add(project)
|
| 42 |
+
await db.commit()
|
| 43 |
+
await db.refresh(project)
|
| 44 |
+
return project
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
@router.get("/{project_id}", response_model=ProjectResponse)
|
| 48 |
+
async def get_project(
|
| 49 |
+
project_id: int,
|
| 50 |
+
user: User = Depends(get_current_user),
|
| 51 |
+
db: AsyncSession = Depends(get_db),
|
| 52 |
+
):
|
| 53 |
+
result = await db.execute(
|
| 54 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 55 |
+
)
|
| 56 |
+
project = result.scalar_one_or_none()
|
| 57 |
+
if not project:
|
| 58 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 59 |
+
return project
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
@router.patch("/{project_id}", response_model=ProjectResponse)
|
| 63 |
+
async def update_project(
|
| 64 |
+
project_id: int,
|
| 65 |
+
data: ProjectUpdate,
|
| 66 |
+
user: User = Depends(get_current_user),
|
| 67 |
+
db: AsyncSession = Depends(get_db),
|
| 68 |
+
):
|
| 69 |
+
result = await db.execute(
|
| 70 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 71 |
+
)
|
| 72 |
+
project = result.scalar_one_or_none()
|
| 73 |
+
if not project:
|
| 74 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 75 |
+
|
| 76 |
+
for key, value in data.model_dump(exclude_none=True).items():
|
| 77 |
+
setattr(project, key, value)
|
| 78 |
+
await db.commit()
|
| 79 |
+
await db.refresh(project)
|
| 80 |
+
return project
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
@router.delete("/{project_id}", status_code=204)
|
| 84 |
+
async def delete_project(
|
| 85 |
+
project_id: int,
|
| 86 |
+
user: User = Depends(get_current_user),
|
| 87 |
+
db: AsyncSession = Depends(get_db),
|
| 88 |
+
):
|
| 89 |
+
result = await db.execute(
|
| 90 |
+
select(Project).where(Project.id == project_id, Project.user_id == user.id)
|
| 91 |
+
)
|
| 92 |
+
project = result.scalar_one_or_none()
|
| 93 |
+
if not project:
|
| 94 |
+
raise HTTPException(status_code=404, detail="Project not found")
|
| 95 |
+
await db.delete(project)
|
| 96 |
+
await db.commit()
|
app/config.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic_settings import BaseSettings
|
| 2 |
+
from functools import lru_cache
|
| 3 |
+
import json
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class Settings(BaseSettings):
|
| 7 |
+
# Auth
|
| 8 |
+
jwt_secret_key: str = "change-this-to-a-random-secret-key-in-production"
|
| 9 |
+
jwt_algorithm: str = "HS256"
|
| 10 |
+
access_token_expire_minutes: int = 30
|
| 11 |
+
refresh_token_expire_days: int = 7
|
| 12 |
+
|
| 13 |
+
# Database
|
| 14 |
+
database_url: str = "sqlite+aiosqlite:///./workdir/anime.db"
|
| 15 |
+
|
| 16 |
+
# API Keys
|
| 17 |
+
hf_api_token: str = ""
|
| 18 |
+
nvidia_api_key: str = ""
|
| 19 |
+
gemini_api_key: str = ""
|
| 20 |
+
openrouter_api_key: str = ""
|
| 21 |
+
pollinations_api_key: str = ""
|
| 22 |
+
|
| 23 |
+
# Server
|
| 24 |
+
port: int = 8000
|
| 25 |
+
|
| 26 |
+
# CORS
|
| 27 |
+
cors_origins: str = '["http://localhost:3000"]'
|
| 28 |
+
|
| 29 |
+
@property
|
| 30 |
+
def cors_origin_list(self) -> list[str]:
|
| 31 |
+
return json.loads(self.cors_origins)
|
| 32 |
+
|
| 33 |
+
model_config = {"env_file": ".env", "env_file_encoding": "utf-8"}
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
@lru_cache
|
| 37 |
+
def get_settings() -> Settings:
|
| 38 |
+
return Settings()
|
app/database.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy.ext.asyncio import create_async_engine, async_sessionmaker, AsyncSession
|
| 2 |
+
from sqlalchemy.orm import DeclarativeBase
|
| 3 |
+
from app.config import get_settings
|
| 4 |
+
|
| 5 |
+
settings = get_settings()
|
| 6 |
+
|
| 7 |
+
engine = create_async_engine(
|
| 8 |
+
settings.database_url,
|
| 9 |
+
echo=False,
|
| 10 |
+
connect_args={"timeout": 30},
|
| 11 |
+
)
|
| 12 |
+
async_session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class Base(DeclarativeBase):
|
| 16 |
+
pass
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
async def get_db():
|
| 20 |
+
async with async_session() as session:
|
| 21 |
+
try:
|
| 22 |
+
yield session
|
| 23 |
+
finally:
|
| 24 |
+
await session.close()
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
async def init_db():
|
| 28 |
+
async with engine.begin() as conn:
|
| 29 |
+
await conn.run_sync(Base.metadata.create_all)
|
| 30 |
+
|
| 31 |
+
# Migrate existing DBs: add chapter_ids_json to generation_jobs
|
| 32 |
+
async with engine.begin() as conn:
|
| 33 |
+
try:
|
| 34 |
+
await conn.execute(
|
| 35 |
+
__import__("sqlalchemy").text(
|
| 36 |
+
"ALTER TABLE generation_jobs ADD COLUMN chapter_ids_json TEXT"
|
| 37 |
+
)
|
| 38 |
+
)
|
| 39 |
+
except Exception:
|
| 40 |
+
pass # Column already exists
|
| 41 |
+
|
| 42 |
+
# Migrate existing DBs: add recap_summary to episodes
|
| 43 |
+
async with engine.begin() as conn:
|
| 44 |
+
try:
|
| 45 |
+
await conn.execute(
|
| 46 |
+
__import__("sqlalchemy").text(
|
| 47 |
+
"ALTER TABLE episodes ADD COLUMN recap_summary TEXT"
|
| 48 |
+
)
|
| 49 |
+
)
|
| 50 |
+
except Exception:
|
| 51 |
+
pass # Column already exists
|
app/main.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from contextlib import asynccontextmanager
|
| 2 |
+
from fastapi import FastAPI
|
| 3 |
+
from fastapi.middleware.cors import CORSMiddleware
|
| 4 |
+
from fastapi.staticfiles import StaticFiles
|
| 5 |
+
from app.config import get_settings
|
| 6 |
+
from app.database import init_db
|
| 7 |
+
from app.api import auth, projects, chapters, characters, episodes, generation, novels
|
| 8 |
+
import os
|
| 9 |
+
|
| 10 |
+
settings = get_settings()
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
@asynccontextmanager
|
| 14 |
+
async def lifespan(app: FastAPI):
|
| 15 |
+
# Startup: create tables
|
| 16 |
+
os.makedirs("workdir", exist_ok=True)
|
| 17 |
+
await init_db()
|
| 18 |
+
yield
|
| 19 |
+
# Shutdown
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
app = FastAPI(
|
| 23 |
+
title="Anime Generator API",
|
| 24 |
+
version="0.1.0",
|
| 25 |
+
description="Web novel to anime video converter",
|
| 26 |
+
lifespan=lifespan,
|
| 27 |
+
)
|
| 28 |
+
|
| 29 |
+
# CORS
|
| 30 |
+
app.add_middleware(
|
| 31 |
+
CORSMiddleware,
|
| 32 |
+
allow_origins=settings.cors_origin_list,
|
| 33 |
+
allow_credentials=True,
|
| 34 |
+
allow_methods=["*"],
|
| 35 |
+
allow_headers=["*"],
|
| 36 |
+
)
|
| 37 |
+
|
| 38 |
+
# Serve generated media files
|
| 39 |
+
os.makedirs("workdir", exist_ok=True)
|
| 40 |
+
app.mount("/media", StaticFiles(directory="workdir"), name="media")
|
| 41 |
+
|
| 42 |
+
# Routes
|
| 43 |
+
app.include_router(auth.router)
|
| 44 |
+
app.include_router(projects.router)
|
| 45 |
+
app.include_router(chapters.router)
|
| 46 |
+
app.include_router(characters.router)
|
| 47 |
+
app.include_router(episodes.router)
|
| 48 |
+
app.include_router(generation.router)
|
| 49 |
+
app.include_router(novels.router)
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
@app.get("/api/health")
|
| 53 |
+
async def health():
|
| 54 |
+
return {"status": "ok", "version": "0.1.0"}
|
app/models/__init__.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from app.models.user import User
|
| 2 |
+
from app.models.project import Project
|
| 3 |
+
from app.models.chapter import Chapter
|
| 4 |
+
from app.models.character import Character
|
| 5 |
+
from app.models.episode import Episode
|
| 6 |
+
from app.models.generation_job import GenerationJob
|
| 7 |
+
from app.models.critic_result import CriticResult
|
| 8 |
+
from app.models.api_usage import ApiUsage
|
| 9 |
+
|
| 10 |
+
__all__ = [
|
| 11 |
+
"User", "Project", "Chapter", "Character",
|
| 12 |
+
"Episode", "GenerationJob", "CriticResult", "ApiUsage",
|
| 13 |
+
]
|
app/models/api_usage.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Date, DateTime, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class ApiUsage(Base):
|
| 6 |
+
__tablename__ = "api_usage"
|
| 7 |
+
|
| 8 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 9 |
+
service = Column(String(100), nullable=False) # huggingface, nvidia, gemini, edge_tts
|
| 10 |
+
endpoint = Column(String(500), nullable=True)
|
| 11 |
+
request_count = Column(Integer, default=1)
|
| 12 |
+
date = Column(Date, nullable=False)
|
| 13 |
+
created_at = Column(DateTime, server_default=func.now())
|
app/models/chapter.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
from sqlalchemy.orm import relationship
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class Chapter(Base):
|
| 7 |
+
__tablename__ = "chapters"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
|
| 11 |
+
chapter_number = Column(Integer, nullable=False)
|
| 12 |
+
title = Column(String(500), nullable=True)
|
| 13 |
+
source_text = Column(Text, nullable=False)
|
| 14 |
+
word_count = Column(Integer, default=0)
|
| 15 |
+
status = Column(String(50), default="raw") # raw, parsed, processed
|
| 16 |
+
parsed_json = Column(JSON, nullable=True) # storyboard output
|
| 17 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 18 |
+
|
| 19 |
+
project = relationship("Project", back_populates="chapters")
|
app/models/character.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
from sqlalchemy.orm import relationship
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class Character(Base):
|
| 7 |
+
__tablename__ = "characters"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
|
| 11 |
+
name = Column(String(200), nullable=False)
|
| 12 |
+
aliases_json = Column(JSON, default=list) # ["Su Ling'er", "Ling'er"]
|
| 13 |
+
role = Column(String(50), default="supporting") # protagonist, antagonist, supporting, minor
|
| 14 |
+
gender = Column(String(20), nullable=True)
|
| 15 |
+
age_category = Column(String(20), nullable=True) # child, teen, young_adult, adult, elder
|
| 16 |
+
visual_prompt = Column(Text, nullable=True) # 50-80 word prompt-ready description
|
| 17 |
+
reference_image_path = Column(String(500), nullable=True)
|
| 18 |
+
portrait_url = Column(String(500), nullable=True) # Pollinations media URL for portrait reference
|
| 19 |
+
reference_seed = Column(Integer, nullable=True) # seed pinning for consistency
|
| 20 |
+
voice_config_json = Column(JSON, nullable=True) # {voice_name, rate, pitch, style}
|
| 21 |
+
voice_reference_path = Column(String(500), nullable=True) # path to .safetensors voice state (Pocket TTS)
|
| 22 |
+
voice_source_json = Column(JSON, nullable=True) # {"source": "animevox", "character": "...", "anime": "..."}
|
| 23 |
+
wiki_data_json = Column(JSON, nullable=True) # cached fandom wiki data
|
| 24 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 25 |
+
updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
|
| 26 |
+
|
| 27 |
+
project = relationship("Project", back_populates="characters")
|
app/models/critic_result.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, Float, DateTime, ForeignKey, JSON, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
from sqlalchemy.orm import relationship
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class CriticResult(Base):
|
| 7 |
+
__tablename__ = "critic_results"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
episode_id = Column(Integer, ForeignKey("episodes.id"), nullable=False)
|
| 11 |
+
visual_consistency_score = Column(Float, nullable=True)
|
| 12 |
+
pacing_score = Column(Float, nullable=True)
|
| 13 |
+
audio_sync_score = Column(Float, nullable=True)
|
| 14 |
+
overall_score = Column(Float, nullable=True)
|
| 15 |
+
feedback_json = Column(JSON, nullable=True)
|
| 16 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 17 |
+
|
| 18 |
+
episode = relationship("Episode", back_populates="critic_results")
|
app/models/episode.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Float, DateTime, ForeignKey, JSON, Text, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
from sqlalchemy.orm import relationship
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class Episode(Base):
|
| 7 |
+
__tablename__ = "episodes"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
|
| 11 |
+
episode_number = Column(Integer, nullable=False)
|
| 12 |
+
title = Column(String(500), nullable=True)
|
| 13 |
+
chapter_ids_json = Column(JSON, default=list)
|
| 14 |
+
storyboard_json = Column(JSON, nullable=True) # full storyboard data
|
| 15 |
+
duration_seconds = Column(Float, nullable=True)
|
| 16 |
+
video_path = Column(String(500), nullable=True)
|
| 17 |
+
thumbnail_path = Column(String(500), nullable=True)
|
| 18 |
+
recap_summary = Column(Text, nullable=True) # narrative recap for continuity
|
| 19 |
+
status = Column(String(50), default="pending") # pending, generating, completed, error
|
| 20 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 21 |
+
updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
|
| 22 |
+
|
| 23 |
+
project = relationship("Project", back_populates="episodes")
|
| 24 |
+
generation_jobs = relationship("GenerationJob", back_populates="episode", cascade="all, delete-orphan")
|
| 25 |
+
critic_results = relationship("CriticResult", back_populates="episode", cascade="all, delete-orphan")
|
app/models/generation_job.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Float, Text, DateTime, ForeignKey, JSON, func
|
| 2 |
+
from app.database import Base
|
| 3 |
+
from sqlalchemy.orm import relationship
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class GenerationJob(Base):
|
| 7 |
+
__tablename__ = "generation_jobs"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
project_id = Column(Integer, ForeignKey("projects.id"), nullable=False)
|
| 11 |
+
user_id = Column(Integer, ForeignKey("users.id"), nullable=False)
|
| 12 |
+
episode_id = Column(Integer, ForeignKey("episodes.id"), nullable=True)
|
| 13 |
+
chapter_ids_json = Column(JSON, nullable=True)
|
| 14 |
+
status = Column(String(50), default="queued") # queued, running, completed, failed, cancelled
|
| 15 |
+
current_stage = Column(String(50), nullable=True) # ingest, characters, scene_parse, image_gen, tts, animation, assembly, critic
|
| 16 |
+
progress_pct = Column(Float, default=0.0)
|
| 17 |
+
progress_detail = Column(Text, nullable=True)
|
| 18 |
+
error_message = Column(Text, nullable=True)
|
| 19 |
+
started_at = Column(DateTime, nullable=True)
|
| 20 |
+
completed_at = Column(DateTime, nullable=True)
|
| 21 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 22 |
+
|
| 23 |
+
project = relationship("Project", back_populates="generation_jobs")
|
| 24 |
+
user = relationship("User", back_populates="generation_jobs")
|
| 25 |
+
episode = relationship("Episode", back_populates="generation_jobs")
|
app/models/project.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, Text, DateTime, ForeignKey, JSON, func
|
| 2 |
+
from sqlalchemy.orm import relationship
|
| 3 |
+
from app.database import Base
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class Project(Base):
|
| 7 |
+
__tablename__ = "projects"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
user_id = Column(Integer, ForeignKey("users.id"), nullable=False)
|
| 11 |
+
title = Column(String(500), nullable=False)
|
| 12 |
+
novel_url = Column(String(2000), nullable=True)
|
| 13 |
+
cover_image_url = Column(String(2000), nullable=True)
|
| 14 |
+
synopsis = Column(Text, nullable=True)
|
| 15 |
+
settings_json = Column(JSON, default=dict) # art style, episode length, etc.
|
| 16 |
+
status = Column(String(50), default="created") # created, processing, completed, error
|
| 17 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 18 |
+
updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
|
| 19 |
+
|
| 20 |
+
user = relationship("User", back_populates="projects")
|
| 21 |
+
chapters = relationship("Chapter", back_populates="project", cascade="all, delete-orphan")
|
| 22 |
+
characters = relationship("Character", back_populates="project", cascade="all, delete-orphan")
|
| 23 |
+
episodes = relationship("Episode", back_populates="project", cascade="all, delete-orphan")
|
| 24 |
+
generation_jobs = relationship("GenerationJob", back_populates="project", cascade="all, delete-orphan")
|
app/models/user.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from sqlalchemy import Column, Integer, String, DateTime, func
|
| 2 |
+
from sqlalchemy.orm import relationship
|
| 3 |
+
from app.database import Base
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
class User(Base):
|
| 7 |
+
__tablename__ = "users"
|
| 8 |
+
|
| 9 |
+
id = Column(Integer, primary_key=True, autoincrement=True)
|
| 10 |
+
email = Column(String(255), unique=True, nullable=False, index=True)
|
| 11 |
+
username = Column(String(100), unique=True, nullable=False, index=True)
|
| 12 |
+
password_hash = Column(String(255), nullable=False)
|
| 13 |
+
role = Column(String(20), default="user") # user, admin
|
| 14 |
+
created_at = Column(DateTime, server_default=func.now())
|
| 15 |
+
updated_at = Column(DateTime, server_default=func.now(), onupdate=func.now())
|
| 16 |
+
|
| 17 |
+
projects = relationship("Project", back_populates="user", cascade="all, delete-orphan")
|
| 18 |
+
generation_jobs = relationship("GenerationJob", back_populates="user", cascade="all, delete-orphan")
|
app/pipeline/__init__.py
ADDED
|
File without changes
|
app/pipeline/orchestrator.py
ADDED
|
@@ -0,0 +1,435 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Pipeline Orchestrator — runs all stages in sequence.
|
| 2 |
+
|
| 3 |
+
Stages:
|
| 4 |
+
1. Ingest — load chapters from DB
|
| 5 |
+
2. Characters — extract + enrich characters
|
| 6 |
+
2b. Portraits — generate reference portraits (Pollinations)
|
| 7 |
+
2c. Voice Enroll — match characters to AnimeVox voices (Pocket TTS)
|
| 8 |
+
3. Scene Parse — Gemini → e-konte storyboard JSON
|
| 9 |
+
4. Image Gen — generate images for all cuts (Pollinations + fallback)
|
| 10 |
+
5. TTS — generate voice clips (Pocket TTS for main chars, Edge TTS for narration/minor)
|
| 11 |
+
6. Video Gen — AI video clips (Pollinations grok-video) with Ken Burns fallback
|
| 12 |
+
7. Assembly — concat clips, scene transitions, subtitles, BGM
|
| 13 |
+
8. Critic — QA review
|
| 14 |
+
"""
|
| 15 |
+
import asyncio
|
| 16 |
+
import json
|
| 17 |
+
import logging
|
| 18 |
+
from datetime import datetime, timezone
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
from sqlalchemy import select
|
| 21 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 22 |
+
from app.database import async_session
|
| 23 |
+
from app.models.project import Project
|
| 24 |
+
from app.models.character import Character
|
| 25 |
+
from app.models.episode import Episode
|
| 26 |
+
from app.models.generation_job import GenerationJob
|
| 27 |
+
from app.models.critic_result import CriticResult
|
| 28 |
+
from app.pipeline.stage_1_ingest import run_ingest
|
| 29 |
+
from app.pipeline.stage_2_characters import run_character_extraction
|
| 30 |
+
from app.pipeline.stage_2b_portraits import run_portrait_generation
|
| 31 |
+
from app.pipeline.stage_2c_voice_enroll import run_voice_enrollment
|
| 32 |
+
from app.pipeline.stage_3_scene_parse import run_scene_parse
|
| 33 |
+
from app.pipeline.stage_4_image_gen import run_image_generation
|
| 34 |
+
from app.pipeline.stage_5_tts import run_tts_generation
|
| 35 |
+
from app.pipeline.stage_6_video_gen import run_video_generation
|
| 36 |
+
from app.pipeline.stage_6_animation import run_animation as run_ken_burns
|
| 37 |
+
from app.pipeline.stage_7_assembly import run_assembly
|
| 38 |
+
from app.pipeline.stage_8_critic import run_critic
|
| 39 |
+
from app.pipeline.recap_generator import generate_episode_recap, build_continuity_context
|
| 40 |
+
|
| 41 |
+
logger = logging.getLogger(__name__)
|
| 42 |
+
|
| 43 |
+
STAGES = [
|
| 44 |
+
("ingest", 3),
|
| 45 |
+
("characters", 7),
|
| 46 |
+
("portraits", 8),
|
| 47 |
+
("voice_enroll", 4),
|
| 48 |
+
("scene_parse", 8),
|
| 49 |
+
("image_gen", 28),
|
| 50 |
+
("tts", 12),
|
| 51 |
+
("video_gen", 23),
|
| 52 |
+
("assembly", 12),
|
| 53 |
+
("critic", 5),
|
| 54 |
+
]
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
async def _update_job(job_id: int, **kwargs):
|
| 58 |
+
"""Update a generation job in the database with SQLite retry on lock."""
|
| 59 |
+
for attempt in range(3):
|
| 60 |
+
try:
|
| 61 |
+
async with async_session() as db:
|
| 62 |
+
result = await db.execute(
|
| 63 |
+
select(GenerationJob).where(GenerationJob.id == job_id)
|
| 64 |
+
)
|
| 65 |
+
job = result.scalar_one_or_none()
|
| 66 |
+
if job:
|
| 67 |
+
for k, v in kwargs.items():
|
| 68 |
+
setattr(job, k, v)
|
| 69 |
+
await db.commit()
|
| 70 |
+
return
|
| 71 |
+
except Exception as e:
|
| 72 |
+
if "database is locked" in str(e) and attempt < 2:
|
| 73 |
+
await asyncio.sleep(0.5 * (attempt + 1))
|
| 74 |
+
continue
|
| 75 |
+
if attempt == 2:
|
| 76 |
+
logger.warning("_update_job failed after retries: %s", e)
|
| 77 |
+
else:
|
| 78 |
+
raise
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def _make_progress_callback(job_id: int, stage_name: str, stage_base_pct: float, stage_weight: float):
|
| 82 |
+
"""Create a progress callback for a specific stage."""
|
| 83 |
+
async def callback(detail: str, stage_pct: float):
|
| 84 |
+
overall_pct = stage_base_pct + (stage_pct / 100.0) * stage_weight
|
| 85 |
+
await _update_job(
|
| 86 |
+
job_id,
|
| 87 |
+
current_stage=stage_name,
|
| 88 |
+
progress_pct=min(overall_pct, 99.0),
|
| 89 |
+
progress_detail=detail,
|
| 90 |
+
)
|
| 91 |
+
return callback
|
| 92 |
+
|
| 93 |
+
|
| 94 |
+
def _check_pollinations_available() -> bool:
|
| 95 |
+
"""Check if Pollinations API key is configured."""
|
| 96 |
+
from app.config import get_settings
|
| 97 |
+
settings = get_settings()
|
| 98 |
+
return bool(settings.pollinations_api_key)
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
async def run_pipeline(job_id: int, resume: bool = False, chapter_ids: list[int] | None = None):
|
| 102 |
+
"""Run the complete generation pipeline for a job.
|
| 103 |
+
|
| 104 |
+
If resume=True, skip stages whose outputs already exist on disk/DB.
|
| 105 |
+
If chapter_ids is provided, only ingest those specific chapters.
|
| 106 |
+
"""
|
| 107 |
+
async with async_session() as db:
|
| 108 |
+
result = await db.execute(
|
| 109 |
+
select(GenerationJob).where(GenerationJob.id == job_id)
|
| 110 |
+
)
|
| 111 |
+
job = result.scalar_one_or_none()
|
| 112 |
+
if not job:
|
| 113 |
+
return
|
| 114 |
+
|
| 115 |
+
project_id = job.project_id
|
| 116 |
+
# Use chapter_ids from param, falling back to what's stored on the job
|
| 117 |
+
if chapter_ids is None:
|
| 118 |
+
chapter_ids = job.chapter_ids_json
|
| 119 |
+
|
| 120 |
+
# Get project
|
| 121 |
+
proj_result = await db.execute(
|
| 122 |
+
select(Project).where(Project.id == project_id)
|
| 123 |
+
)
|
| 124 |
+
project = proj_result.scalar_one_or_none()
|
| 125 |
+
if not project:
|
| 126 |
+
await _update_job(job_id, status="failed", error_message="Project not found")
|
| 127 |
+
return
|
| 128 |
+
|
| 129 |
+
# Compute episode number early (before any stage runs)
|
| 130 |
+
from sqlalchemy import func as sa_func
|
| 131 |
+
count_result = await db.execute(
|
| 132 |
+
select(sa_func.count()).select_from(Episode).where(Episode.project_id == project_id)
|
| 133 |
+
)
|
| 134 |
+
ep_number = (count_result.scalar() or 0) + 1
|
| 135 |
+
|
| 136 |
+
project_dir = str(Path("workdir/projects") / str(project_id))
|
| 137 |
+
Path(project_dir).mkdir(parents=True, exist_ok=True)
|
| 138 |
+
|
| 139 |
+
# Per-episode directory for episode-specific assets (storyboard, images, audio, clips)
|
| 140 |
+
episode_dir = str(Path(project_dir) / f"ep_{ep_number}")
|
| 141 |
+
Path(episode_dir).mkdir(parents=True, exist_ok=True)
|
| 142 |
+
|
| 143 |
+
use_pollinations = _check_pollinations_available()
|
| 144 |
+
|
| 145 |
+
try:
|
| 146 |
+
logger.info("Pipeline starting for job %s, project %s (resume=%s, pollinations=%s)",
|
| 147 |
+
job_id, project_id, resume, use_pollinations)
|
| 148 |
+
await _update_job(
|
| 149 |
+
job_id,
|
| 150 |
+
status="running",
|
| 151 |
+
started_at=datetime.now(timezone.utc),
|
| 152 |
+
progress_detail="Starting pipeline..." if not resume else "Resuming pipeline...",
|
| 153 |
+
)
|
| 154 |
+
|
| 155 |
+
# Calculate stage base percentages
|
| 156 |
+
total_weight = sum(w for _, w in STAGES)
|
| 157 |
+
cumulative = 0.0
|
| 158 |
+
|
| 159 |
+
# ─── Stage 1: Ingest ───
|
| 160 |
+
logger.info("[Job %s] Stage 1: Ingest (chapter_ids=%s)", job_id, chapter_ids)
|
| 161 |
+
stage_weight = (STAGES[0][1] / total_weight) * 100
|
| 162 |
+
on_progress = _make_progress_callback(job_id, "ingest", cumulative, stage_weight)
|
| 163 |
+
async with async_session() as db:
|
| 164 |
+
chapters = await run_ingest(project_id, db, on_progress, chapter_ids=chapter_ids)
|
| 165 |
+
cumulative += stage_weight
|
| 166 |
+
logger.info("[Job %s] Stage 1 complete: %d chapters", job_id, len(chapters))
|
| 167 |
+
|
| 168 |
+
# ─── Stage 2: Characters ───
|
| 169 |
+
logger.info("[Job %s] Stage 2: Characters", job_id)
|
| 170 |
+
stage_weight = (STAGES[1][1] / total_weight) * 100
|
| 171 |
+
on_progress = _make_progress_callback(job_id, "characters", cumulative, stage_weight)
|
| 172 |
+
|
| 173 |
+
if resume and not chapter_ids:
|
| 174 |
+
async with async_session() as db:
|
| 175 |
+
char_result = await db.execute(
|
| 176 |
+
select(Character).where(Character.project_id == project_id)
|
| 177 |
+
)
|
| 178 |
+
existing_chars = list(char_result.scalars().all())
|
| 179 |
+
|
| 180 |
+
if existing_chars:
|
| 181 |
+
characters = existing_chars
|
| 182 |
+
logger.info("[Job %s] Stage 2 skipped (resume): %d characters exist", job_id, len(characters))
|
| 183 |
+
await on_progress(f"Reusing {len(characters)} existing characters", 100)
|
| 184 |
+
else:
|
| 185 |
+
async with async_session() as db:
|
| 186 |
+
characters = await run_character_extraction(
|
| 187 |
+
project_id, project.title, chapters, db, on_progress
|
| 188 |
+
)
|
| 189 |
+
else:
|
| 190 |
+
async with async_session() as db:
|
| 191 |
+
characters = await run_character_extraction(
|
| 192 |
+
project_id, project.title, chapters, db, on_progress
|
| 193 |
+
)
|
| 194 |
+
cumulative += stage_weight
|
| 195 |
+
logger.info("[Job %s] Stage 2 complete: %d characters", job_id, len(characters))
|
| 196 |
+
|
| 197 |
+
# ─── Stage 2b: Portraits (Pollinations only) ───
|
| 198 |
+
if use_pollinations:
|
| 199 |
+
logger.info("[Job %s] Stage 2b: Portraits", job_id)
|
| 200 |
+
stage_weight = (STAGES[2][1] / total_weight) * 100
|
| 201 |
+
on_progress = _make_progress_callback(job_id, "portraits", cumulative, stage_weight)
|
| 202 |
+
|
| 203 |
+
# Skip if all prominent characters already have portraits
|
| 204 |
+
needs_portraits = any(
|
| 205 |
+
c.role in ("protagonist", "antagonist", "supporting") and not c.portrait_url
|
| 206 |
+
for c in characters
|
| 207 |
+
)
|
| 208 |
+
|
| 209 |
+
if needs_portraits or not resume:
|
| 210 |
+
async with async_session() as db:
|
| 211 |
+
# Re-attach characters to this session
|
| 212 |
+
char_result = await db.execute(
|
| 213 |
+
select(Character).where(Character.project_id == project_id)
|
| 214 |
+
)
|
| 215 |
+
characters = list(char_result.scalars().all())
|
| 216 |
+
characters = await run_portrait_generation(
|
| 217 |
+
project_id, characters, project_dir, db, on_progress
|
| 218 |
+
)
|
| 219 |
+
else:
|
| 220 |
+
await on_progress("All portraits exist", 100)
|
| 221 |
+
|
| 222 |
+
cumulative += stage_weight
|
| 223 |
+
portrait_count = sum(1 for c in characters if c.portrait_url)
|
| 224 |
+
logger.info("[Job %s] Stage 2b complete: %d portraits", job_id, portrait_count)
|
| 225 |
+
else:
|
| 226 |
+
cumulative += (STAGES[2][1] / total_weight) * 100
|
| 227 |
+
logger.info("[Job %s] Stage 2b skipped (no Pollinations API key)", job_id)
|
| 228 |
+
|
| 229 |
+
# ─── Stage 2c: Voice Enrollment ───
|
| 230 |
+
logger.info("[Job %s] Stage 2c: Voice Enrollment", job_id)
|
| 231 |
+
stage_weight = (STAGES[3][1] / total_weight) * 100
|
| 232 |
+
on_progress = _make_progress_callback(job_id, "voice_enroll", cumulative, stage_weight)
|
| 233 |
+
|
| 234 |
+
# Skip if all prominent characters already have voice references
|
| 235 |
+
needs_enrollment = any(
|
| 236 |
+
c.role in ("protagonist", "antagonist", "supporting") and not c.voice_reference_path
|
| 237 |
+
for c in characters
|
| 238 |
+
)
|
| 239 |
+
|
| 240 |
+
if needs_enrollment or not resume:
|
| 241 |
+
async with async_session() as db:
|
| 242 |
+
char_result = await db.execute(
|
| 243 |
+
select(Character).where(Character.project_id == project_id)
|
| 244 |
+
)
|
| 245 |
+
characters = list(char_result.scalars().all())
|
| 246 |
+
characters = await run_voice_enrollment(
|
| 247 |
+
project_id, characters, project_dir, db, on_progress
|
| 248 |
+
)
|
| 249 |
+
else:
|
| 250 |
+
await on_progress("All voices already enrolled", 100)
|
| 251 |
+
|
| 252 |
+
cumulative += stage_weight
|
| 253 |
+
voice_count = sum(1 for c in characters if c.voice_reference_path)
|
| 254 |
+
logger.info("[Job %s] Stage 2c complete: %d voice enrollments", job_id, voice_count)
|
| 255 |
+
|
| 256 |
+
# ─── Narrative Continuity (pre-Stage 3) ───
|
| 257 |
+
continuity_context = None
|
| 258 |
+
if ep_number > 1:
|
| 259 |
+
logger.info("[Job %s] Building narrative continuity for Episode %d", job_id, ep_number)
|
| 260 |
+
try:
|
| 261 |
+
continuity_context = await build_continuity_context(project_id, ep_number)
|
| 262 |
+
if continuity_context:
|
| 263 |
+
logger.info("[Job %s] Continuity context built (%d chars)", job_id, len(continuity_context))
|
| 264 |
+
else:
|
| 265 |
+
logger.info("[Job %s] No prior episodes with recaps found", job_id)
|
| 266 |
+
except Exception as e:
|
| 267 |
+
logger.warning("[Job %s] Failed to build continuity context: %s", job_id, e)
|
| 268 |
+
|
| 269 |
+
# ─── Stage 3: Scene Parse ───
|
| 270 |
+
logger.info("[Job %s] Stage 3: Scene Parse", job_id)
|
| 271 |
+
stage_weight = (STAGES[4][1] / total_weight) * 100
|
| 272 |
+
on_progress = _make_progress_callback(job_id, "scene_parse", cumulative, stage_weight)
|
| 273 |
+
|
| 274 |
+
storyboard_path = Path(episode_dir) / "storyboard.json"
|
| 275 |
+
|
| 276 |
+
if resume and storyboard_path.exists():
|
| 277 |
+
with open(storyboard_path, "r") as f:
|
| 278 |
+
storyboard = json.load(f)
|
| 279 |
+
logger.info("[Job %s] Stage 3 skipped (resume): loaded storyboard from disk", job_id)
|
| 280 |
+
await on_progress("Loaded existing storyboard", 100)
|
| 281 |
+
else:
|
| 282 |
+
storyboard = await run_scene_parse(chapters, characters, on_progress, continuity_context=continuity_context)
|
| 283 |
+
with open(storyboard_path, "w") as f:
|
| 284 |
+
json.dump(storyboard, f)
|
| 285 |
+
logger.info("[Job %s] Stage 3: storyboard saved to disk", job_id)
|
| 286 |
+
|
| 287 |
+
cumulative += stage_weight
|
| 288 |
+
scene_count = len(storyboard.get("scenes", []))
|
| 289 |
+
cut_count = sum(
|
| 290 |
+
len(s.get("cuts", s.get("shots", [])))
|
| 291 |
+
for s in storyboard.get("scenes", [])
|
| 292 |
+
)
|
| 293 |
+
logger.info("[Job %s] Stage 3 complete: %d scenes, %d cuts", job_id, scene_count, cut_count)
|
| 294 |
+
|
| 295 |
+
# ─── Stage 4 + 5: Image Gen + TTS (parallel) ───
|
| 296 |
+
logger.info("[Job %s] Stage 4+5: Image Gen + TTS (parallel)", job_id)
|
| 297 |
+
img_weight = (STAGES[5][1] / total_weight) * 100
|
| 298 |
+
tts_weight = (STAGES[6][1] / total_weight) * 100
|
| 299 |
+
img_progress = _make_progress_callback(job_id, "image_gen", cumulative, img_weight)
|
| 300 |
+
tts_progress = _make_progress_callback(job_id, "tts", cumulative + img_weight, tts_weight)
|
| 301 |
+
|
| 302 |
+
storyboard_img, storyboard_tts = await asyncio.gather(
|
| 303 |
+
run_image_generation(storyboard, characters, episode_dir, img_progress, resume=resume),
|
| 304 |
+
run_tts_generation(storyboard, characters, episode_dir, tts_progress, resume=resume),
|
| 305 |
+
)
|
| 306 |
+
|
| 307 |
+
# Merge TTS data into storyboard (image gen already updated in-place)
|
| 308 |
+
for scene in storyboard.get("scenes", []):
|
| 309 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 310 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 311 |
+
for tts_scene in storyboard_tts.get("scenes", []):
|
| 312 |
+
for tts_cut in tts_scene.get("cuts", tts_scene.get("shots", [])):
|
| 313 |
+
tts_cut_id = tts_cut.get("cut_id") or tts_cut.get("shot_id")
|
| 314 |
+
if tts_cut_id == cut_id:
|
| 315 |
+
cut["tts_path"] = tts_cut.get("tts_path")
|
| 316 |
+
cut["word_timestamps"] = tts_cut.get("word_timestamps")
|
| 317 |
+
if tts_cut.get("actual_duration_sec"):
|
| 318 |
+
cut["actual_duration_sec"] = tts_cut["actual_duration_sec"]
|
| 319 |
+
cut["duration_sec"] = max(
|
| 320 |
+
tts_cut["actual_duration_sec"] + 0.3,
|
| 321 |
+
cut.get("duration_sec", 2.0)
|
| 322 |
+
)
|
| 323 |
+
|
| 324 |
+
cumulative += img_weight + tts_weight
|
| 325 |
+
logger.info("[Job %s] Stage 4+5 complete", job_id)
|
| 326 |
+
|
| 327 |
+
# Persist updated storyboard (now includes image_path, tts_path)
|
| 328 |
+
with open(Path(episode_dir) / "storyboard.json", "w") as f:
|
| 329 |
+
json.dump(storyboard, f)
|
| 330 |
+
|
| 331 |
+
# ─── Stage 6: Video Generation (or Ken Burns fallback) ───
|
| 332 |
+
logger.info("[Job %s] Stage 6: Video Generation", job_id)
|
| 333 |
+
img_count = sum(
|
| 334 |
+
1 for sc in storyboard.get("scenes", [])
|
| 335 |
+
for cut in sc.get("cuts", sc.get("shots", []))
|
| 336 |
+
if cut.get("image_path")
|
| 337 |
+
)
|
| 338 |
+
logger.info("[Job %s] Pre-video: %d cuts have image_path", job_id, img_count)
|
| 339 |
+
stage_weight = (STAGES[7][1] / total_weight) * 100
|
| 340 |
+
on_progress = _make_progress_callback(job_id, "video_gen", cumulative, stage_weight)
|
| 341 |
+
|
| 342 |
+
if use_pollinations:
|
| 343 |
+
# Try AI video generation with Ken Burns fallback built-in
|
| 344 |
+
storyboard = await run_video_generation(storyboard, episode_dir, on_progress)
|
| 345 |
+
else:
|
| 346 |
+
# Pure Ken Burns animation (legacy path)
|
| 347 |
+
storyboard = await run_ken_burns(storyboard, episode_dir, on_progress)
|
| 348 |
+
|
| 349 |
+
cumulative += stage_weight
|
| 350 |
+
|
| 351 |
+
# Persist storyboard after video gen (includes clip_path + _video_error for debugging)
|
| 352 |
+
with open(Path(episode_dir) / "storyboard.json", "w") as f:
|
| 353 |
+
json.dump(storyboard, f)
|
| 354 |
+
|
| 355 |
+
clip_count = sum(
|
| 356 |
+
1 for sc in storyboard.get("scenes", [])
|
| 357 |
+
for cut in sc.get("cuts", sc.get("shots", []))
|
| 358 |
+
if cut.get("clip_path") and Path(cut["clip_path"]).exists()
|
| 359 |
+
)
|
| 360 |
+
logger.info("[Job %s] Stage 6 complete: %d/%d clips produced", job_id, clip_count, cut_count)
|
| 361 |
+
|
| 362 |
+
# ─── Stage 7: Assembly ───
|
| 363 |
+
logger.info("[Job %s] Stage 7: Assembly", job_id)
|
| 364 |
+
stage_weight = (STAGES[8][1] / total_weight) * 100
|
| 365 |
+
on_progress = _make_progress_callback(job_id, "assembly", cumulative, stage_weight)
|
| 366 |
+
|
| 367 |
+
assembly_result = await run_assembly(storyboard, episode_dir, ep_number, on_progress, output_dir=project_dir)
|
| 368 |
+
cumulative += stage_weight
|
| 369 |
+
logger.info("[Job %s] Stage 7 complete: %s", job_id, assembly_result.get("video_path", "?"))
|
| 370 |
+
|
| 371 |
+
# ─── Stage 8: Critic ───
|
| 372 |
+
logger.info("[Job %s] Stage 8: Critic", job_id)
|
| 373 |
+
stage_weight = (STAGES[9][1] / total_weight) * 100
|
| 374 |
+
on_progress = _make_progress_callback(job_id, "critic", cumulative, stage_weight)
|
| 375 |
+
critic_result = await run_critic(storyboard, episode_dir, on_progress)
|
| 376 |
+
|
| 377 |
+
# Save episode and critic result to DB
|
| 378 |
+
async with async_session() as db:
|
| 379 |
+
episode = Episode(
|
| 380 |
+
project_id=project_id,
|
| 381 |
+
episode_number=ep_number,
|
| 382 |
+
title=storyboard.get("episode_title", f"Episode {ep_number}"),
|
| 383 |
+
chapter_ids_json=[ch.id for ch in chapters],
|
| 384 |
+
storyboard_json=storyboard,
|
| 385 |
+
duration_seconds=assembly_result["duration_seconds"],
|
| 386 |
+
video_path=assembly_result["video_path"],
|
| 387 |
+
status="completed",
|
| 388 |
+
)
|
| 389 |
+
db.add(episode)
|
| 390 |
+
await db.commit()
|
| 391 |
+
await db.refresh(episode)
|
| 392 |
+
|
| 393 |
+
critic = CriticResult(
|
| 394 |
+
episode_id=episode.id,
|
| 395 |
+
visual_consistency_score=critic_result.get("visual_consistency_score"),
|
| 396 |
+
pacing_score=critic_result.get("pacing_score"),
|
| 397 |
+
audio_sync_score=critic_result.get("audio_sync_score"),
|
| 398 |
+
overall_score=critic_result.get("overall_score"),
|
| 399 |
+
feedback_json=critic_result.get("feedback"),
|
| 400 |
+
)
|
| 401 |
+
db.add(critic)
|
| 402 |
+
await db.commit()
|
| 403 |
+
|
| 404 |
+
# Generate and save episode recap for future continuity
|
| 405 |
+
try:
|
| 406 |
+
recap = await generate_episode_recap(
|
| 407 |
+
storyboard,
|
| 408 |
+
storyboard.get("episode_title", f"Episode {ep_number}"),
|
| 409 |
+
)
|
| 410 |
+
if recap:
|
| 411 |
+
episode.recap_summary = recap
|
| 412 |
+
await db.commit()
|
| 413 |
+
logger.info("[Job %s] Episode recap saved (%d chars)", job_id, len(recap))
|
| 414 |
+
except Exception as e:
|
| 415 |
+
logger.warning("[Job %s] Failed to generate episode recap: %s (non-fatal)", job_id, e)
|
| 416 |
+
|
| 417 |
+
# Mark job complete
|
| 418 |
+
await _update_job(
|
| 419 |
+
job_id,
|
| 420 |
+
status="completed",
|
| 421 |
+
progress_pct=100.0,
|
| 422 |
+
progress_detail=f"Episode {ep_number} generated successfully",
|
| 423 |
+
current_stage="completed",
|
| 424 |
+
episode_id=episode.id,
|
| 425 |
+
completed_at=datetime.now(timezone.utc),
|
| 426 |
+
)
|
| 427 |
+
|
| 428 |
+
except Exception as e:
|
| 429 |
+
logger.exception("[Job %s] Pipeline failed at stage", job_id)
|
| 430 |
+
await _update_job(
|
| 431 |
+
job_id,
|
| 432 |
+
status="failed",
|
| 433 |
+
error_message=str(e)[:2000],
|
| 434 |
+
progress_detail=f"Failed: {str(e)[:200]}",
|
| 435 |
+
)
|
app/pipeline/recap_generator.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Recap generator — episode narrative continuity for multi-episode series.
|
| 2 |
+
|
| 3 |
+
After each episode, generates a 3-5 sentence narrative recap from the storyboard.
|
| 4 |
+
Before generating a new episode, builds a continuity context block from prior recaps.
|
| 5 |
+
"""
|
| 6 |
+
import logging
|
| 7 |
+
from sqlalchemy import select
|
| 8 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 9 |
+
from app.database import async_session
|
| 10 |
+
from app.models.episode import Episode
|
| 11 |
+
from app.services.llm import call_gemini
|
| 12 |
+
|
| 13 |
+
logger = logging.getLogger(__name__)
|
| 14 |
+
|
| 15 |
+
RECAP_PROMPT = """You are summarizing an anime episode for production continuity.
|
| 16 |
+
|
| 17 |
+
Given the storyboard below (dialogue, narration, and scene descriptions), write a 3-5 sentence narrative recap (~150 words).
|
| 18 |
+
|
| 19 |
+
Focus on:
|
| 20 |
+
- Key plot events that happened
|
| 21 |
+
- Important character actions and decisions
|
| 22 |
+
- Relationship changes or reveals
|
| 23 |
+
- How the episode ended (cliffhanger, resolution, etc.)
|
| 24 |
+
|
| 25 |
+
Write in past tense, third person. Be specific about WHAT happened, not vague.
|
| 26 |
+
|
| 27 |
+
EPISODE TITLE: {episode_title}
|
| 28 |
+
|
| 29 |
+
STORYBOARD:
|
| 30 |
+
{storyboard_text}
|
| 31 |
+
|
| 32 |
+
Write ONLY the recap, no headers or labels."""
|
| 33 |
+
|
| 34 |
+
COMPRESS_PROMPT = """Compress these episode recaps into a single "story so far" summary (~200 words).
|
| 35 |
+
Preserve the most important plot points, character arcs, and world-building. Drop minor details.
|
| 36 |
+
Write in past tense, third person.
|
| 37 |
+
|
| 38 |
+
{recaps_text}
|
| 39 |
+
|
| 40 |
+
Write ONLY the compressed summary, no headers or labels."""
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def _extract_storyboard_text(storyboard: dict) -> str:
|
| 44 |
+
"""Extract readable dialogue/narration from storyboard JSON."""
|
| 45 |
+
lines = []
|
| 46 |
+
for scene in storyboard.get("scenes", []):
|
| 47 |
+
setting = scene.get("setting", "")
|
| 48 |
+
mood = scene.get("mood", "")
|
| 49 |
+
if setting:
|
| 50 |
+
lines.append(f"[Scene: {setting} | Mood: {mood}]")
|
| 51 |
+
|
| 52 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 53 |
+
dialogue = cut.get("dialogue")
|
| 54 |
+
if not dialogue or not dialogue.get("text"):
|
| 55 |
+
continue
|
| 56 |
+
|
| 57 |
+
speaker = dialogue.get("speaker", "Narrator")
|
| 58 |
+
text = dialogue["text"]
|
| 59 |
+
is_narration = dialogue.get("is_narration", False)
|
| 60 |
+
is_thought = dialogue.get("is_inner_thought", False)
|
| 61 |
+
|
| 62 |
+
if is_narration:
|
| 63 |
+
lines.append(f" (Narration) {text}")
|
| 64 |
+
elif is_thought:
|
| 65 |
+
lines.append(f" ({speaker} thinks) {text}")
|
| 66 |
+
else:
|
| 67 |
+
lines.append(f" {speaker}: \"{text}\"")
|
| 68 |
+
|
| 69 |
+
return "\n".join(lines)
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
async def generate_episode_recap(storyboard: dict, episode_title: str) -> str:
|
| 73 |
+
"""Generate a 3-5 sentence narrative recap from a completed episode's storyboard."""
|
| 74 |
+
storyboard_text = _extract_storyboard_text(storyboard)
|
| 75 |
+
if not storyboard_text.strip():
|
| 76 |
+
return ""
|
| 77 |
+
|
| 78 |
+
prompt = RECAP_PROMPT.format(
|
| 79 |
+
episode_title=episode_title,
|
| 80 |
+
storyboard_text=storyboard_text,
|
| 81 |
+
)
|
| 82 |
+
|
| 83 |
+
recap = await call_gemini(prompt, max_tokens=512, temperature=0.3)
|
| 84 |
+
return recap.strip()
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
async def build_continuity_context(project_id: int, current_ep_number: int) -> str | None:
|
| 88 |
+
"""Build narrative continuity context from prior episodes' recaps.
|
| 89 |
+
|
| 90 |
+
Returns a formatted continuity block string, or None if this is the first episode.
|
| 91 |
+
"""
|
| 92 |
+
async with async_session() as db:
|
| 93 |
+
result = await db.execute(
|
| 94 |
+
select(Episode)
|
| 95 |
+
.where(
|
| 96 |
+
Episode.project_id == project_id,
|
| 97 |
+
Episode.episode_number < current_ep_number,
|
| 98 |
+
Episode.status == "completed",
|
| 99 |
+
)
|
| 100 |
+
.order_by(Episode.episode_number)
|
| 101 |
+
)
|
| 102 |
+
prior_episodes = list(result.scalars().all())
|
| 103 |
+
|
| 104 |
+
if not prior_episodes:
|
| 105 |
+
return None
|
| 106 |
+
|
| 107 |
+
# Lazy-generate missing recaps
|
| 108 |
+
for ep in prior_episodes:
|
| 109 |
+
if not ep.recap_summary:
|
| 110 |
+
if ep.storyboard_json:
|
| 111 |
+
logger.info("Lazy-generating recap for Episode %d", ep.episode_number)
|
| 112 |
+
try:
|
| 113 |
+
recap = await generate_episode_recap(
|
| 114 |
+
ep.storyboard_json,
|
| 115 |
+
ep.title or f"Episode {ep.episode_number}",
|
| 116 |
+
)
|
| 117 |
+
if recap:
|
| 118 |
+
async with async_session() as db:
|
| 119 |
+
result = await db.execute(
|
| 120 |
+
select(Episode).where(Episode.id == ep.id)
|
| 121 |
+
)
|
| 122 |
+
db_ep = result.scalar_one()
|
| 123 |
+
db_ep.recap_summary = recap
|
| 124 |
+
await db.commit()
|
| 125 |
+
ep.recap_summary = recap
|
| 126 |
+
except Exception as e:
|
| 127 |
+
logger.warning("Failed to lazy-generate recap for Episode %d: %s",
|
| 128 |
+
ep.episode_number, e)
|
| 129 |
+
else:
|
| 130 |
+
logger.warning("Episode %d has no storyboard_json, skipping recap",
|
| 131 |
+
ep.episode_number)
|
| 132 |
+
|
| 133 |
+
# Collect available recaps
|
| 134 |
+
episodes_with_recaps = [(ep.episode_number, ep.title, ep.recap_summary)
|
| 135 |
+
for ep in prior_episodes if ep.recap_summary]
|
| 136 |
+
|
| 137 |
+
if not episodes_with_recaps:
|
| 138 |
+
return None
|
| 139 |
+
|
| 140 |
+
# Build the continuity block
|
| 141 |
+
if len(episodes_with_recaps) <= 3:
|
| 142 |
+
# Few episodes — include all recaps directly
|
| 143 |
+
recap_lines = []
|
| 144 |
+
for ep_num, title, recap in episodes_with_recaps:
|
| 145 |
+
label = title or f"Episode {ep_num}"
|
| 146 |
+
recap_lines.append(f'EPISODE {ep_num} ("{label}"):\n{recap}')
|
| 147 |
+
all_recaps = "\n\n".join(recap_lines)
|
| 148 |
+
|
| 149 |
+
# Last episode gets special "PREVIOUS EPISODE" treatment
|
| 150 |
+
last_ep_num, last_title, last_recap = episodes_with_recaps[-1]
|
| 151 |
+
if len(episodes_with_recaps) == 1:
|
| 152 |
+
story_so_far = ""
|
| 153 |
+
previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
|
| 154 |
+
else:
|
| 155 |
+
earlier = "\n\n".join(
|
| 156 |
+
f'Episode {n} ("{t or f"Episode {n}"}"): {r}'
|
| 157 |
+
for n, t, r in episodes_with_recaps[:-1]
|
| 158 |
+
)
|
| 159 |
+
story_so_far = f"STORY SO FAR:\n{earlier}\n\n"
|
| 160 |
+
previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
|
| 161 |
+
else:
|
| 162 |
+
# 4+ episodes — compress older ones, keep latest full
|
| 163 |
+
older_recaps = "\n\n".join(
|
| 164 |
+
f'Episode {n} ("{t or f"Episode {n}"}"): {r}'
|
| 165 |
+
for n, t, r in episodes_with_recaps[:-1]
|
| 166 |
+
)
|
| 167 |
+
try:
|
| 168 |
+
compressed = await call_gemini(
|
| 169 |
+
COMPRESS_PROMPT.format(recaps_text=older_recaps),
|
| 170 |
+
max_tokens=512,
|
| 171 |
+
temperature=0.3,
|
| 172 |
+
)
|
| 173 |
+
story_so_far = f"STORY SO FAR:\n{compressed.strip()}\n\n"
|
| 174 |
+
except Exception as e:
|
| 175 |
+
logger.warning("Failed to compress recaps: %s — using last recap only", e)
|
| 176 |
+
story_so_far = f"STORY SO FAR:\n{older_recaps}\n\n"
|
| 177 |
+
|
| 178 |
+
last_ep_num, last_title, last_recap = episodes_with_recaps[-1]
|
| 179 |
+
previous_block = f'PREVIOUS EPISODE ("{last_title or f"Episode {last_ep_num}"}"):\n{last_recap}'
|
| 180 |
+
|
| 181 |
+
return f"""═══ NARRATIVE CONTINUITY ═══
|
| 182 |
+
This is NOT the first episode. The audience has already watched the previous episode(s).
|
| 183 |
+
|
| 184 |
+
{story_so_far}{previous_block}
|
| 185 |
+
|
| 186 |
+
CONTINUITY RULES:
|
| 187 |
+
- Do NOT re-introduce characters the audience already knows. They can appear immediately.
|
| 188 |
+
- Do NOT repeat exposition from prior episodes.
|
| 189 |
+
- You MAY open with a brief 1-2 cut "recall" moment echoing the previous ending (optional).
|
| 190 |
+
- Reference prior events naturally through dialogue or narration when relevant.
|
| 191 |
+
- Character relationships carry forward — show evolution, not reset."""
|
app/pipeline/stage_1_ingest.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 1: Text Ingestion — get chapter text into the database."""
|
| 2 |
+
from sqlalchemy import select
|
| 3 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 4 |
+
from app.models.chapter import Chapter
|
| 5 |
+
from app.models.project import Project
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
async def run_ingest(
|
| 9 |
+
project_id: int,
|
| 10 |
+
db: AsyncSession,
|
| 11 |
+
on_progress: callable = None,
|
| 12 |
+
chapter_ids: list[int] | None = None,
|
| 13 |
+
) -> list[Chapter]:
|
| 14 |
+
"""Ensure chapters are loaded for the project. Returns chapter list.
|
| 15 |
+
|
| 16 |
+
If chapter_ids is provided, only load those specific chapters.
|
| 17 |
+
"""
|
| 18 |
+
if on_progress:
|
| 19 |
+
await on_progress("Loading chapters...", 0)
|
| 20 |
+
|
| 21 |
+
query = select(Chapter).where(Chapter.project_id == project_id)
|
| 22 |
+
if chapter_ids:
|
| 23 |
+
query = query.where(Chapter.id.in_(chapter_ids))
|
| 24 |
+
query = query.order_by(Chapter.chapter_number)
|
| 25 |
+
|
| 26 |
+
result = await db.execute(query)
|
| 27 |
+
chapters = list(result.scalars().all())
|
| 28 |
+
|
| 29 |
+
if not chapters:
|
| 30 |
+
raise ValueError("No chapters found for this project. Add chapters first.")
|
| 31 |
+
|
| 32 |
+
total_words = sum(len(ch.source_text.split()) for ch in chapters)
|
| 33 |
+
if on_progress:
|
| 34 |
+
await on_progress(
|
| 35 |
+
f"Loaded {len(chapters)} chapter{'s' if len(chapters) != 1 else ''} ({total_words:,} words)",
|
| 36 |
+
100,
|
| 37 |
+
)
|
| 38 |
+
|
| 39 |
+
return chapters
|
app/pipeline/stage_2_characters.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 2: Character Extraction & Enrichment."""
|
| 2 |
+
import json
|
| 3 |
+
from sqlalchemy import select
|
| 4 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 5 |
+
from app.models.character import Character
|
| 6 |
+
from app.models.chapter import Chapter
|
| 7 |
+
from app.services.llm import call_gemini_json
|
| 8 |
+
from app.services.fandom import search_fandom_wiki
|
| 9 |
+
from app.utils.voice_mapper import assign_voice
|
| 10 |
+
from app.utils.prompt_builder import build_character_visual_prompt
|
| 11 |
+
|
| 12 |
+
CHARACTER_EXTRACTION_PROMPT = """You are an expert manhwa/webtoon character designer. Analyze the following novel chapter text and extract ALL named characters with COMPLETE visual designs.
|
| 13 |
+
|
| 14 |
+
CRITICAL RULES:
|
| 15 |
+
1. NEVER output "unknown" for ANY visual field. If the text doesn't describe a trait, INVENT a distinctive, fitting appearance based on the character's role, personality, and name origin.
|
| 16 |
+
2. Every character must be VISUALLY DISTINCT from all others — no two characters should share the same hair color + style combination.
|
| 17 |
+
3. Use SPECIFIC colors, not vague ones: "platinum silver" not "light", "deep crimson" not "red", "midnight blue" not "dark blue".
|
| 18 |
+
4. Clothing should be detailed: "black high-collar military coat with silver trim and crimson epaulettes" not "dark clothes".
|
| 19 |
+
|
| 20 |
+
For each character, provide:
|
| 21 |
+
- name: The character's primary name
|
| 22 |
+
- aliases: Other names/titles they go by (list)
|
| 23 |
+
- role: protagonist, antagonist, supporting, or minor
|
| 24 |
+
- gender: male, female, or unknown (guess from name/context if possible)
|
| 25 |
+
- age_category: child, teen, young_adult, adult, or elder
|
| 26 |
+
- hair_color: specific color (e.g. "jet black", "platinum silver", "deep crimson", "ash blonde")
|
| 27 |
+
- hair_style: specific style (e.g. "long flowing with side-swept bangs", "short spiky undercut", "waist-length straight with center part")
|
| 28 |
+
- eye_color: specific color (e.g. "golden amber", "deep violet", "ice blue", "emerald green")
|
| 29 |
+
- skin_tone: specific tone (e.g. "fair porcelain", "warm olive", "deep bronze", "pale ivory")
|
| 30 |
+
- height: tall, average, or short
|
| 31 |
+
- build: muscular, athletic, slim, average, or heavyset
|
| 32 |
+
- clothing: detailed outfit description with colors and accessories
|
| 33 |
+
- distinctive_features: scars, tattoos, accessories, jewelry, etc. (invent something unique if none mentioned)
|
| 34 |
+
- color_palette: 3 main colors associated with this character (e.g. ["black", "crimson", "silver"])
|
| 35 |
+
- personality_notes: brief personality traits relevant to voice acting
|
| 36 |
+
|
| 37 |
+
Return a JSON array of character objects. Include EVERY named character that appears, even briefly.
|
| 38 |
+
|
| 39 |
+
Chapter text:
|
| 40 |
+
{chapter_text}"""
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
CHARACTER_DEDUP_PROMPT = """Review these character visual designs and fix any that look too similar. Two characters should NEVER share the same hair color + style combination.
|
| 44 |
+
|
| 45 |
+
Current characters:
|
| 46 |
+
{characters_json}
|
| 47 |
+
|
| 48 |
+
Rules:
|
| 49 |
+
1. If two characters have similar hair (same color family AND similar style), change ONE of them to be distinct.
|
| 50 |
+
2. Ensure color_palette entries don't overlap too much between characters.
|
| 51 |
+
3. Keep changes minimal — only modify what's needed for visual distinction.
|
| 52 |
+
4. Return the FULL array with corrections applied.
|
| 53 |
+
|
| 54 |
+
Return a JSON array of all characters (modified where needed)."""
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
async def run_character_extraction(
|
| 58 |
+
project_id: int,
|
| 59 |
+
project_title: str,
|
| 60 |
+
chapters: list[Chapter],
|
| 61 |
+
db: AsyncSession,
|
| 62 |
+
on_progress: callable = None,
|
| 63 |
+
) -> list[Character]:
|
| 64 |
+
"""Extract characters from chapters, enrich with wiki data, assign voices."""
|
| 65 |
+
|
| 66 |
+
if on_progress:
|
| 67 |
+
await on_progress("Analyzing chapter text for characters...", 0)
|
| 68 |
+
|
| 69 |
+
# Check for existing characters
|
| 70 |
+
existing = await db.execute(
|
| 71 |
+
select(Character).where(Character.project_id == project_id)
|
| 72 |
+
)
|
| 73 |
+
existing_list = existing.scalars().all()
|
| 74 |
+
existing_chars = {c.name.lower(): c for c in existing_list}
|
| 75 |
+
# Also index by aliases for dedup
|
| 76 |
+
for c in list(existing_chars.values()):
|
| 77 |
+
for alias in (c.aliases_json or []):
|
| 78 |
+
existing_chars.setdefault(alias.lower(), c)
|
| 79 |
+
|
| 80 |
+
# Combine chapter texts for extraction (first 3 chapters for efficiency)
|
| 81 |
+
combined_text = "\n\n---\n\n".join(
|
| 82 |
+
ch.source_text[:5000] for ch in chapters[:3]
|
| 83 |
+
)
|
| 84 |
+
|
| 85 |
+
# Call Gemini for character extraction
|
| 86 |
+
prompt = CHARACTER_EXTRACTION_PROMPT.format(chapter_text=combined_text[:15000])
|
| 87 |
+
extracted = await call_gemini_json(prompt)
|
| 88 |
+
|
| 89 |
+
# Gemini may return {"characters": [...]} instead of [...]
|
| 90 |
+
if isinstance(extracted, dict):
|
| 91 |
+
extracted = extracted.get("characters", [extracted])
|
| 92 |
+
if not isinstance(extracted, list):
|
| 93 |
+
extracted = [extracted]
|
| 94 |
+
|
| 95 |
+
# Visual deduplication pass — ensure all characters look distinct
|
| 96 |
+
if len(extracted) > 1:
|
| 97 |
+
if on_progress:
|
| 98 |
+
await on_progress("Ensuring visual distinctness...", 12)
|
| 99 |
+
import json as _json
|
| 100 |
+
dedup_prompt = CHARACTER_DEDUP_PROMPT.format(
|
| 101 |
+
characters_json=_json.dumps(extracted, indent=2)
|
| 102 |
+
)
|
| 103 |
+
try:
|
| 104 |
+
deduped = await call_gemini_json(dedup_prompt)
|
| 105 |
+
if isinstance(deduped, dict):
|
| 106 |
+
deduped = deduped.get("characters", [deduped])
|
| 107 |
+
if isinstance(deduped, list) and len(deduped) == len(extracted):
|
| 108 |
+
extracted = deduped
|
| 109 |
+
except Exception:
|
| 110 |
+
pass # dedup is nice-to-have, not critical
|
| 111 |
+
|
| 112 |
+
total = len(extracted)
|
| 113 |
+
characters = []
|
| 114 |
+
|
| 115 |
+
if on_progress and total:
|
| 116 |
+
await on_progress(f"Discovered {total} characters", 15)
|
| 117 |
+
|
| 118 |
+
for i, char_data in enumerate(extracted):
|
| 119 |
+
name = char_data.get("name", "").strip()
|
| 120 |
+
if not name or name.lower() in existing_chars:
|
| 121 |
+
if name.lower() in existing_chars:
|
| 122 |
+
characters.append(existing_chars[name.lower()])
|
| 123 |
+
continue
|
| 124 |
+
|
| 125 |
+
if on_progress:
|
| 126 |
+
pct = int(20 + (i / max(total, 1)) * 60)
|
| 127 |
+
await on_progress(f"Enriching {name} ({i+1}/{total})", pct)
|
| 128 |
+
|
| 129 |
+
# Try fandom wiki enrichment
|
| 130 |
+
wiki_data = await search_fandom_wiki(project_title, name)
|
| 131 |
+
if wiki_data and wiki_data.get("description"):
|
| 132 |
+
# Merge wiki appearance data (prefer wiki, more detailed)
|
| 133 |
+
char_data["wiki_appearance"] = wiki_data["description"]
|
| 134 |
+
|
| 135 |
+
# Build visual prompt
|
| 136 |
+
visual_prompt = build_character_visual_prompt(char_data)
|
| 137 |
+
|
| 138 |
+
# Assign voice
|
| 139 |
+
voice_config = assign_voice(
|
| 140 |
+
project_id,
|
| 141 |
+
char_data.get("gender"),
|
| 142 |
+
char_data.get("age_category"),
|
| 143 |
+
)
|
| 144 |
+
|
| 145 |
+
# Normalize aliases (Gemini may return a comma-separated string)
|
| 146 |
+
aliases = char_data.get("aliases", [])
|
| 147 |
+
if isinstance(aliases, str):
|
| 148 |
+
aliases = [a.strip() for a in aliases.split(",") if a.strip()]
|
| 149 |
+
|
| 150 |
+
# Create character record
|
| 151 |
+
character = Character(
|
| 152 |
+
project_id=project_id,
|
| 153 |
+
name=name,
|
| 154 |
+
aliases_json=aliases,
|
| 155 |
+
role=char_data.get("role", "supporting"),
|
| 156 |
+
gender=char_data.get("gender"),
|
| 157 |
+
age_category=char_data.get("age_category"),
|
| 158 |
+
visual_prompt=visual_prompt,
|
| 159 |
+
voice_config_json=voice_config,
|
| 160 |
+
wiki_data_json=wiki_data,
|
| 161 |
+
)
|
| 162 |
+
db.add(character)
|
| 163 |
+
characters.append(character)
|
| 164 |
+
|
| 165 |
+
await db.commit()
|
| 166 |
+
for ch in characters:
|
| 167 |
+
await db.refresh(ch)
|
| 168 |
+
|
| 169 |
+
if on_progress:
|
| 170 |
+
await on_progress(f"{len(characters)} characters ready", 100)
|
| 171 |
+
|
| 172 |
+
return characters
|
app/pipeline/stage_2b_portraits.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 2b: Character Portrait Generation.
|
| 2 |
+
|
| 3 |
+
Generates a high-quality reference portrait for each prominent character,
|
| 4 |
+
uploads to media.pollinations.ai, and stores the URL in the Character model.
|
| 5 |
+
These portraits are used as visual references during scene image generation
|
| 6 |
+
with vision-capable models (klein, gptimage) for character consistency.
|
| 7 |
+
"""
|
| 8 |
+
import asyncio
|
| 9 |
+
import logging
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 13 |
+
|
| 14 |
+
from app.models.character import Character
|
| 15 |
+
from app.services.pollinations import generate_image, upload_media, VISION_MODELS
|
| 16 |
+
from app.utils.prompt_builder import MANHWA_STYLE_PREFIX
|
| 17 |
+
|
| 18 |
+
logger = logging.getLogger(__name__)
|
| 19 |
+
|
| 20 |
+
PORTRAIT_PROMPT_TEMPLATE = (
|
| 21 |
+
"{style_prefix}, character portrait sheet, front-facing bust shot, "
|
| 22 |
+
"{visual_prompt}, clean white background, reference sheet style, "
|
| 23 |
+
"sharp details, no background elements, studio lighting, "
|
| 24 |
+
"high detail face and eyes, character design reference"
|
| 25 |
+
)
|
| 26 |
+
|
| 27 |
+
# Only generate portraits for prominent roles
|
| 28 |
+
PROMINENT_ROLES = {"protagonist", "antagonist", "supporting"}
|
| 29 |
+
|
| 30 |
+
# Use klein-large for portraits — scene images also use klein-large with this
|
| 31 |
+
# portrait as reference. Same model = same style space = consistent characters.
|
| 32 |
+
PORTRAIT_MODEL = "klein-large"
|
| 33 |
+
PORTRAIT_SEED = 42
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
async def run_portrait_generation(
|
| 37 |
+
project_id: int,
|
| 38 |
+
characters: list[Character],
|
| 39 |
+
project_dir: str,
|
| 40 |
+
db: AsyncSession,
|
| 41 |
+
on_progress: callable = None,
|
| 42 |
+
) -> list[Character]:
|
| 43 |
+
"""Generate reference portraits for prominent characters.
|
| 44 |
+
|
| 45 |
+
Skips characters that already have a portrait_url.
|
| 46 |
+
Returns the updated character list.
|
| 47 |
+
"""
|
| 48 |
+
# Filter to prominent characters without portraits
|
| 49 |
+
targets = [
|
| 50 |
+
c for c in characters
|
| 51 |
+
if c.role in PROMINENT_ROLES
|
| 52 |
+
and c.visual_prompt
|
| 53 |
+
and not c.portrait_url
|
| 54 |
+
]
|
| 55 |
+
|
| 56 |
+
if not targets:
|
| 57 |
+
if on_progress:
|
| 58 |
+
await on_progress("All characters already have portraits", 100)
|
| 59 |
+
return characters
|
| 60 |
+
|
| 61 |
+
total = len(targets)
|
| 62 |
+
if on_progress:
|
| 63 |
+
await on_progress(f"Generating {total} character portraits...", 0)
|
| 64 |
+
|
| 65 |
+
portrait_dir = Path(project_dir) / "portraits"
|
| 66 |
+
portrait_dir.mkdir(parents=True, exist_ok=True)
|
| 67 |
+
|
| 68 |
+
completed = 0
|
| 69 |
+
|
| 70 |
+
for char in targets:
|
| 71 |
+
if on_progress:
|
| 72 |
+
pct = int((completed / max(total, 1)) * 90)
|
| 73 |
+
await on_progress(f"[{completed + 1}/{total}] Painting {char.name}...", pct)
|
| 74 |
+
|
| 75 |
+
portrait_path = str(portrait_dir / f"{char.name.replace(' ', '_')}_portrait.png")
|
| 76 |
+
|
| 77 |
+
try:
|
| 78 |
+
# Generate portrait image
|
| 79 |
+
prompt = PORTRAIT_PROMPT_TEMPLATE.format(
|
| 80 |
+
style_prefix=MANHWA_STYLE_PREFIX,
|
| 81 |
+
visual_prompt=char.visual_prompt,
|
| 82 |
+
)
|
| 83 |
+
|
| 84 |
+
await generate_image(
|
| 85 |
+
prompt=prompt,
|
| 86 |
+
output_path=portrait_path,
|
| 87 |
+
model=PORTRAIT_MODEL,
|
| 88 |
+
width=768,
|
| 89 |
+
height=1024, # Portrait orientation
|
| 90 |
+
seed=PORTRAIT_SEED,
|
| 91 |
+
)
|
| 92 |
+
|
| 93 |
+
# Upload to media.pollinations.ai for permanent URL
|
| 94 |
+
portrait_url = await upload_media(portrait_path)
|
| 95 |
+
char.portrait_url = portrait_url
|
| 96 |
+
char.reference_image_path = portrait_path
|
| 97 |
+
|
| 98 |
+
logger.info("Portrait generated for %s: %s", char.name, portrait_url)
|
| 99 |
+
|
| 100 |
+
except Exception as e:
|
| 101 |
+
logger.warning("Portrait generation failed for %s: %s", char.name, e)
|
| 102 |
+
# Non-fatal — character just won't have a portrait reference
|
| 103 |
+
|
| 104 |
+
completed += 1
|
| 105 |
+
|
| 106 |
+
# Persist portrait URLs to database
|
| 107 |
+
try:
|
| 108 |
+
await db.commit()
|
| 109 |
+
for char in characters:
|
| 110 |
+
await db.refresh(char)
|
| 111 |
+
except Exception as e:
|
| 112 |
+
logger.warning("Failed to persist portrait URLs: %s", e)
|
| 113 |
+
|
| 114 |
+
if on_progress:
|
| 115 |
+
success_count = sum(1 for c in targets if c.portrait_url)
|
| 116 |
+
await on_progress(f"{success_count}/{total} portraits generated", 100)
|
| 117 |
+
|
| 118 |
+
return characters
|
app/pipeline/stage_2c_voice_enroll.py
ADDED
|
@@ -0,0 +1,177 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 2c: Voice Enrollment — assign AnimeVox voices to prominent characters.
|
| 2 |
+
|
| 3 |
+
Runs after Stage 2b (portraits), before Stage 3 (scene parse).
|
| 4 |
+
Matches each protagonist/antagonist/supporting character to an AnimeVox voice
|
| 5 |
+
based on gender, age, and personality traits, then stores the voice state path
|
| 6 |
+
for Pocket TTS to use during TTS generation.
|
| 7 |
+
|
| 8 |
+
Gracefully degrades: if AnimeVox is not set up, all characters fall back to Edge TTS.
|
| 9 |
+
"""
|
| 10 |
+
import logging
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
|
| 13 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 14 |
+
|
| 15 |
+
from app.models.character import Character
|
| 16 |
+
from app.utils.voice_catalog import match_voice, is_available
|
| 17 |
+
|
| 18 |
+
logger = logging.getLogger(__name__)
|
| 19 |
+
|
| 20 |
+
PROMINENT_ROLES = {"protagonist", "antagonist", "supporting"}
|
| 21 |
+
|
| 22 |
+
# Manual voice overrides: character name (case-insensitive) → catalog voice ID
|
| 23 |
+
# Use this to force a specific voice when auto-matching picks a bad fit.
|
| 24 |
+
VOICE_OVERRIDES: dict[str, str] = {
|
| 25 |
+
"ye chen": "genshin_diluc",
|
| 26 |
+
}
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
async def run_voice_enrollment(
|
| 30 |
+
project_id: int,
|
| 31 |
+
characters: list[Character],
|
| 32 |
+
project_dir: str,
|
| 33 |
+
db: AsyncSession,
|
| 34 |
+
on_progress: callable = None,
|
| 35 |
+
) -> list[Character]:
|
| 36 |
+
"""Enroll voices for prominent characters using AnimeVox catalog.
|
| 37 |
+
|
| 38 |
+
For each protagonist/antagonist/supporting without a voice_reference_path:
|
| 39 |
+
1. Load voice catalog
|
| 40 |
+
2. Match character to best AnimeVox voice (gender + age + traits)
|
| 41 |
+
3. Store voice state path in character model
|
| 42 |
+
4. Persist to DB
|
| 43 |
+
|
| 44 |
+
Returns the updated character list.
|
| 45 |
+
"""
|
| 46 |
+
if not is_available():
|
| 47 |
+
if on_progress:
|
| 48 |
+
await on_progress("AnimeVox not set up — using Edge TTS for all characters", 100)
|
| 49 |
+
logger.info("Voice enrollment skipped: AnimeVox catalog not found")
|
| 50 |
+
return characters
|
| 51 |
+
|
| 52 |
+
# Filter to prominent characters that need voice enrollment
|
| 53 |
+
targets = [
|
| 54 |
+
c for c in characters
|
| 55 |
+
if c.role in PROMINENT_ROLES
|
| 56 |
+
and not c.voice_reference_path # don't re-enroll
|
| 57 |
+
]
|
| 58 |
+
|
| 59 |
+
if not targets:
|
| 60 |
+
if on_progress:
|
| 61 |
+
await on_progress("All characters already have voices assigned", 100)
|
| 62 |
+
return characters
|
| 63 |
+
|
| 64 |
+
total = len(targets)
|
| 65 |
+
if on_progress:
|
| 66 |
+
await on_progress(f"Matching voices for {total} characters...", 0)
|
| 67 |
+
|
| 68 |
+
enrolled = 0
|
| 69 |
+
for i, char in enumerate(targets):
|
| 70 |
+
if on_progress:
|
| 71 |
+
pct = int((i / max(total, 1)) * 90)
|
| 72 |
+
await on_progress(f"[{i+1}/{total}] Matching voice for {char.name}...", pct)
|
| 73 |
+
|
| 74 |
+
# Check manual overrides first
|
| 75 |
+
override_id = VOICE_OVERRIDES.get(char.name.lower())
|
| 76 |
+
if override_id:
|
| 77 |
+
match = _get_override_voice(override_id)
|
| 78 |
+
if match:
|
| 79 |
+
logger.info("Voice override: %s → %s", char.name, match["display_name"])
|
| 80 |
+
else:
|
| 81 |
+
match = None
|
| 82 |
+
|
| 83 |
+
if not match:
|
| 84 |
+
# Extract personality hints from visual_prompt or wiki data
|
| 85 |
+
personality_hints = _extract_personality_hints(char)
|
| 86 |
+
|
| 87 |
+
match = match_voice(
|
| 88 |
+
project_id=project_id,
|
| 89 |
+
gender=char.gender or "male",
|
| 90 |
+
age_category=char.age_category or "young_adult",
|
| 91 |
+
role=char.role,
|
| 92 |
+
personality_hints=personality_hints,
|
| 93 |
+
)
|
| 94 |
+
|
| 95 |
+
# Determine voice reference: .safetensors path or preset name
|
| 96 |
+
voice_ref = None
|
| 97 |
+
if match:
|
| 98 |
+
if match.get("voice_state_path") and Path(match["voice_state_path"]).exists():
|
| 99 |
+
voice_ref = match["voice_state_path"]
|
| 100 |
+
elif match.get("preset_voice"):
|
| 101 |
+
voice_ref = match["preset_voice"]
|
| 102 |
+
|
| 103 |
+
if match and voice_ref:
|
| 104 |
+
char.voice_reference_path = voice_ref
|
| 105 |
+
char.voice_source_json = {
|
| 106 |
+
"source": "animevox",
|
| 107 |
+
"character": match["display_name"],
|
| 108 |
+
"anime": match["source_anime"],
|
| 109 |
+
"voice_id": match["id"],
|
| 110 |
+
"type": "cloned" if voice_ref.endswith(".safetensors") else "preset",
|
| 111 |
+
}
|
| 112 |
+
enrolled += 1
|
| 113 |
+
logger.info(
|
| 114 |
+
"Voice enrolled: %s → %s (%s)",
|
| 115 |
+
char.name, match["display_name"], match["source_anime"],
|
| 116 |
+
)
|
| 117 |
+
else:
|
| 118 |
+
logger.info(
|
| 119 |
+
"No voice match for %s (%s/%s) — will use Edge TTS",
|
| 120 |
+
char.name, char.gender, char.age_category,
|
| 121 |
+
)
|
| 122 |
+
|
| 123 |
+
# Persist to database
|
| 124 |
+
if enrolled > 0:
|
| 125 |
+
try:
|
| 126 |
+
await db.commit()
|
| 127 |
+
for char in characters:
|
| 128 |
+
await db.refresh(char)
|
| 129 |
+
except Exception as e:
|
| 130 |
+
logger.warning("Failed to persist voice enrollment: %s", e)
|
| 131 |
+
|
| 132 |
+
if on_progress:
|
| 133 |
+
await on_progress(
|
| 134 |
+
f"{enrolled}/{total} characters matched to anime voices", 100
|
| 135 |
+
)
|
| 136 |
+
|
| 137 |
+
logger.info(
|
| 138 |
+
"Voice enrollment complete: %d/%d characters enrolled", enrolled, total
|
| 139 |
+
)
|
| 140 |
+
return characters
|
| 141 |
+
|
| 142 |
+
|
| 143 |
+
def _get_override_voice(voice_id: str) -> dict | None:
|
| 144 |
+
"""Look up a specific voice by ID from the catalog."""
|
| 145 |
+
from app.utils.voice_catalog import get_catalog
|
| 146 |
+
|
| 147 |
+
for entry in get_catalog():
|
| 148 |
+
if entry.get("id") == voice_id:
|
| 149 |
+
return entry
|
| 150 |
+
logger.warning("Voice override ID '%s' not found in catalog", voice_id)
|
| 151 |
+
return None
|
| 152 |
+
|
| 153 |
+
|
| 154 |
+
def _extract_personality_hints(char: Character) -> list[str]:
|
| 155 |
+
"""Extract personality keywords from character data for voice matching."""
|
| 156 |
+
hints = []
|
| 157 |
+
|
| 158 |
+
# From wiki data
|
| 159 |
+
wiki = char.wiki_data_json or {}
|
| 160 |
+
if isinstance(wiki, dict):
|
| 161 |
+
personality = wiki.get("personality", "")
|
| 162 |
+
if personality:
|
| 163 |
+
# Extract adjectives from personality description
|
| 164 |
+
for word in personality.lower().split():
|
| 165 |
+
word = word.strip(".,;:!?()[]")
|
| 166 |
+
if len(word) > 3: # skip short words
|
| 167 |
+
hints.append(word)
|
| 168 |
+
|
| 169 |
+
# From role — add implied traits
|
| 170 |
+
role_hints = {
|
| 171 |
+
"protagonist": ["determined", "heroic"],
|
| 172 |
+
"antagonist": ["cold", "menacing", "calculating"],
|
| 173 |
+
"supporting": ["warm", "loyal"],
|
| 174 |
+
}
|
| 175 |
+
hints.extend(role_hints.get(char.role, []))
|
| 176 |
+
|
| 177 |
+
return hints[:10] # cap to avoid noise
|
app/pipeline/stage_3_scene_parse.py
ADDED
|
@@ -0,0 +1,346 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 3: Scene Parsing — convert chapter text to e-konte storyboard JSON."""
|
| 2 |
+
import re
|
| 3 |
+
from app.models.chapter import Chapter
|
| 4 |
+
from app.models.character import Character
|
| 5 |
+
from app.services.llm import call_gemini_json
|
| 6 |
+
from app.templates.storyboard_prompt import (
|
| 7 |
+
STORYBOARD_SYSTEM,
|
| 8 |
+
STORYBOARD_PROMPT,
|
| 9 |
+
format_characters_block,
|
| 10 |
+
)
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def _estimate_duration(dialogue: dict | None, action: str | None) -> float:
|
| 14 |
+
"""Estimate cut duration based on content."""
|
| 15 |
+
if dialogue:
|
| 16 |
+
text = dialogue.get("text") or ""
|
| 17 |
+
if not text:
|
| 18 |
+
return 2.0
|
| 19 |
+
if dialogue.get("is_narration") or dialogue.get("is_inner_thought"):
|
| 20 |
+
dur = len(text) / 12 + 0.5
|
| 21 |
+
else:
|
| 22 |
+
dur = len(text) / 15 + 0.5
|
| 23 |
+
return max(0.5, min(dur, 12.0))
|
| 24 |
+
if action:
|
| 25 |
+
return max(2.0, min(len(action) / 20, 5.0))
|
| 26 |
+
return 2.0
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def _normalize_cuts(storyboard: dict) -> dict:
|
| 30 |
+
"""Normalize storyboard to use 'cuts' consistently.
|
| 31 |
+
|
| 32 |
+
The new e-konte format uses 'cuts', but the old format used 'shots'.
|
| 33 |
+
This function ensures we always have a 'cuts' key on each scene,
|
| 34 |
+
while keeping 'shots' as an alias for backward compatibility.
|
| 35 |
+
"""
|
| 36 |
+
for scene in storyboard.get("scenes", []):
|
| 37 |
+
# If scene has 'shots' but not 'cuts', rename
|
| 38 |
+
if "shots" in scene and "cuts" not in scene:
|
| 39 |
+
scene["cuts"] = scene.pop("shots")
|
| 40 |
+
# If scene has 'cuts', also set 'shots' alias for backward compat
|
| 41 |
+
if "cuts" in scene:
|
| 42 |
+
scene["shots"] = scene["cuts"]
|
| 43 |
+
# Handle cut_id vs shot_id
|
| 44 |
+
for cut in scene.get("cuts", []):
|
| 45 |
+
if "shot_id" in cut and "cut_id" not in cut:
|
| 46 |
+
cut["cut_id"] = cut["shot_id"]
|
| 47 |
+
elif "cut_id" in cut and "shot_id" not in cut:
|
| 48 |
+
cut["shot_id"] = cut["cut_id"]
|
| 49 |
+
return storyboard
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _build_alias_map(characters: list[Character]) -> dict[str, str]:
|
| 53 |
+
"""Build a lowercase alias → canonical name map from characters."""
|
| 54 |
+
alias_map = {}
|
| 55 |
+
for c in characters:
|
| 56 |
+
for alias in (c.aliases_json or []):
|
| 57 |
+
alias_map[alias.lower()] = c.name
|
| 58 |
+
return alias_map
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
def _resolve_name(name: str, character_names: set[str], alias_map: dict[str, str]) -> str | None:
|
| 62 |
+
"""Resolve a name to its canonical form via direct match or alias lookup."""
|
| 63 |
+
if name in character_names:
|
| 64 |
+
return name
|
| 65 |
+
canonical = alias_map.get(name.lower())
|
| 66 |
+
if canonical and canonical in character_names:
|
| 67 |
+
return canonical
|
| 68 |
+
return None
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def _validate_storyboard(storyboard: dict, characters: list[Character]) -> dict:
|
| 72 |
+
"""Post-process storyboard: fix IDs, validate characters, recalc durations."""
|
| 73 |
+
character_names = {c.name for c in characters}
|
| 74 |
+
alias_map = _build_alias_map(characters)
|
| 75 |
+
cut_counter = 0
|
| 76 |
+
|
| 77 |
+
for scene in storyboard.get("scenes", []):
|
| 78 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 79 |
+
cut_counter += 1
|
| 80 |
+
cut_id = f"C{cut_counter:02d}"
|
| 81 |
+
cut["cut_id"] = cut_id
|
| 82 |
+
cut["shot_id"] = cut_id # Backward compat alias
|
| 83 |
+
|
| 84 |
+
# Coerce characters_present to list and resolve aliases
|
| 85 |
+
chars_present = cut.get("characters_present", [])
|
| 86 |
+
if isinstance(chars_present, str):
|
| 87 |
+
chars_present = [chars_present]
|
| 88 |
+
valid_chars = []
|
| 89 |
+
for c in chars_present:
|
| 90 |
+
resolved = _resolve_name(c, character_names, alias_map)
|
| 91 |
+
if resolved and resolved not in valid_chars:
|
| 92 |
+
valid_chars.append(resolved)
|
| 93 |
+
cut["characters_present"] = valid_chars
|
| 94 |
+
|
| 95 |
+
# Resolve focal_character via alias
|
| 96 |
+
focal = cut.get("focal_character")
|
| 97 |
+
if focal:
|
| 98 |
+
resolved_focal = _resolve_name(focal, character_names, alias_map)
|
| 99 |
+
cut["focal_character"] = resolved_focal if resolved_focal else (valid_chars[0] if valid_chars else None)
|
| 100 |
+
elif focal is not None and not focal:
|
| 101 |
+
cut["focal_character"] = None
|
| 102 |
+
|
| 103 |
+
# Enforce: on-screen speaker should be focal_character
|
| 104 |
+
dialogue = cut.get("dialogue")
|
| 105 |
+
if dialogue and dialogue.get("text"):
|
| 106 |
+
speaker = dialogue.get("speaker", "")
|
| 107 |
+
resolved_speaker = _resolve_name(speaker, character_names, alias_map)
|
| 108 |
+
# Replace alias in dialogue.speaker with canonical name
|
| 109 |
+
if resolved_speaker and resolved_speaker != speaker:
|
| 110 |
+
dialogue["speaker"] = resolved_speaker
|
| 111 |
+
speaker = resolved_speaker
|
| 112 |
+
is_offscreen = dialogue.get("is_offscreen", False)
|
| 113 |
+
is_narration = dialogue.get("is_narration", False)
|
| 114 |
+
is_thought = dialogue.get("is_inner_thought", False)
|
| 115 |
+
|
| 116 |
+
if (speaker != "Narrator"
|
| 117 |
+
and speaker in character_names
|
| 118 |
+
and not is_offscreen
|
| 119 |
+
and not is_narration
|
| 120 |
+
and not is_thought):
|
| 121 |
+
# Speaker is on-screen → must be in characters_present and focal
|
| 122 |
+
if speaker not in valid_chars:
|
| 123 |
+
valid_chars.append(speaker)
|
| 124 |
+
cut["characters_present"] = valid_chars
|
| 125 |
+
if not cut.get("focal_character") or cut["focal_character"] != speaker:
|
| 126 |
+
cut["focal_character"] = speaker
|
| 127 |
+
|
| 128 |
+
# Coerce dialogue: if string, wrap as narration
|
| 129 |
+
dialogue = cut.get("dialogue")
|
| 130 |
+
if isinstance(dialogue, str):
|
| 131 |
+
cut["dialogue"] = {
|
| 132 |
+
"text": dialogue,
|
| 133 |
+
"speaker": "Narrator",
|
| 134 |
+
"is_narration": True,
|
| 135 |
+
}
|
| 136 |
+
|
| 137 |
+
# Coerce duration_sec to float
|
| 138 |
+
try:
|
| 139 |
+
cut["duration_sec"] = float(cut.get("duration_sec", 0))
|
| 140 |
+
except (TypeError, ValueError):
|
| 141 |
+
cut["duration_sec"] = 0
|
| 142 |
+
|
| 143 |
+
# Recalculate duration
|
| 144 |
+
cut["duration_sec"] = _estimate_duration(
|
| 145 |
+
cut.get("dialogue"), cut.get("action_description")
|
| 146 |
+
)
|
| 147 |
+
|
| 148 |
+
# Ensure effects is a list
|
| 149 |
+
if not isinstance(cut.get("effects"), list):
|
| 150 |
+
cut["effects"] = []
|
| 151 |
+
|
| 152 |
+
# Default camera movement
|
| 153 |
+
if not cut.get("camera_movement"):
|
| 154 |
+
cut["camera_movement"] = "FIX"
|
| 155 |
+
|
| 156 |
+
# Ensure video_prompt exists (backfill from action + camera if missing)
|
| 157 |
+
if not cut.get("video_prompt"):
|
| 158 |
+
parts = []
|
| 159 |
+
if cut.get("action_description"):
|
| 160 |
+
parts.append(cut["action_description"])
|
| 161 |
+
camera = cut.get("camera_movement", "FIX")
|
| 162 |
+
if camera != "FIX":
|
| 163 |
+
parts.append(f"camera: {camera}")
|
| 164 |
+
cut["video_prompt"] = ", ".join(parts) if parts else "subtle idle animation, atmospheric motion"
|
| 165 |
+
|
| 166 |
+
# Ensure transition_out exists
|
| 167 |
+
if not cut.get("transition_out"):
|
| 168 |
+
cut["transition_out"] = "cut"
|
| 169 |
+
|
| 170 |
+
# Ensure emotional_beat exists
|
| 171 |
+
if not cut.get("emotional_beat"):
|
| 172 |
+
cut["emotional_beat"] = "setup"
|
| 173 |
+
|
| 174 |
+
# Narration backfill: ensure every cut has dialogue
|
| 175 |
+
for scene in storyboard.get("scenes", []):
|
| 176 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 177 |
+
dialogue = cut.get("dialogue")
|
| 178 |
+
has_text = dialogue and dialogue.get("text")
|
| 179 |
+
if not has_text:
|
| 180 |
+
action = cut.get("action_description", "")
|
| 181 |
+
if action:
|
| 182 |
+
cut["dialogue"] = {
|
| 183 |
+
"speaker": "Narrator",
|
| 184 |
+
"text": action,
|
| 185 |
+
"is_narration": True,
|
| 186 |
+
"emotion": "neutral",
|
| 187 |
+
"delivery": "normal",
|
| 188 |
+
}
|
| 189 |
+
|
| 190 |
+
return storyboard
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
def _enforce_cut_limit(storyboard: dict, max_cuts: int = 12) -> dict:
|
| 194 |
+
"""If total cut count exceeds max_cuts, keep the highest-scoring cuts."""
|
| 195 |
+
all_cuts = []
|
| 196 |
+
for scene in storyboard.get("scenes", []):
|
| 197 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 198 |
+
# Score: dialogue=3, establishing=2, narration=1, reaction/empty=0
|
| 199 |
+
# Bonus for climax/tension_peak emotional beats
|
| 200 |
+
score = 0
|
| 201 |
+
dialogue = cut.get("dialogue")
|
| 202 |
+
if dialogue and dialogue.get("text"):
|
| 203 |
+
if dialogue.get("is_narration"):
|
| 204 |
+
score = 1
|
| 205 |
+
else:
|
| 206 |
+
score = 3
|
| 207 |
+
if cut.get("shot_type") == "establishing":
|
| 208 |
+
score = max(score, 2)
|
| 209 |
+
# Emotional beat bonus
|
| 210 |
+
beat = cut.get("emotional_beat", "")
|
| 211 |
+
if beat in ("climax", "tension_peak"):
|
| 212 |
+
score += 2
|
| 213 |
+
elif beat in ("rising_tension", "resolution"):
|
| 214 |
+
score += 1
|
| 215 |
+
all_cuts.append((scene["scene_id"], cut, score))
|
| 216 |
+
|
| 217 |
+
total = len(all_cuts)
|
| 218 |
+
if total <= max_cuts:
|
| 219 |
+
return storyboard
|
| 220 |
+
|
| 221 |
+
# Sort by score descending, keep top max_cuts, then restore original order
|
| 222 |
+
indexed = [(i, sid, cut, score) for i, (sid, cut, score) in enumerate(all_cuts)]
|
| 223 |
+
indexed.sort(key=lambda x: x[3], reverse=True)
|
| 224 |
+
kept = sorted(indexed[:max_cuts], key=lambda x: x[0]) # restore order
|
| 225 |
+
|
| 226 |
+
# Rebuild scenes with only kept cuts
|
| 227 |
+
kept_by_scene: dict[str, list] = {}
|
| 228 |
+
for _, sid, cut, _ in kept:
|
| 229 |
+
kept_by_scene.setdefault(sid, []).append(cut)
|
| 230 |
+
|
| 231 |
+
new_scenes = []
|
| 232 |
+
for scene in storyboard.get("scenes", []):
|
| 233 |
+
sid = scene["scene_id"]
|
| 234 |
+
if sid in kept_by_scene:
|
| 235 |
+
scene["cuts"] = kept_by_scene[sid]
|
| 236 |
+
scene["shots"] = scene["cuts"] # Backward compat
|
| 237 |
+
new_scenes.append(scene)
|
| 238 |
+
|
| 239 |
+
storyboard["scenes"] = new_scenes
|
| 240 |
+
|
| 241 |
+
# Re-number cut IDs
|
| 242 |
+
counter = 0
|
| 243 |
+
for scene in storyboard["scenes"]:
|
| 244 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 245 |
+
counter += 1
|
| 246 |
+
cut["cut_id"] = f"C{counter:02d}"
|
| 247 |
+
cut["shot_id"] = f"C{counter:02d}"
|
| 248 |
+
|
| 249 |
+
return storyboard
|
| 250 |
+
|
| 251 |
+
|
| 252 |
+
async def run_scene_parse(
|
| 253 |
+
chapters: list[Chapter],
|
| 254 |
+
characters: list[Character],
|
| 255 |
+
on_progress: callable = None,
|
| 256 |
+
continuity_context: str | None = None,
|
| 257 |
+
) -> dict:
|
| 258 |
+
"""Parse chapter text into an e-konte storyboard JSON using Gemini."""
|
| 259 |
+
if on_progress:
|
| 260 |
+
await on_progress("AI Director is composing the e-konte...", 0)
|
| 261 |
+
|
| 262 |
+
# Prepare character data
|
| 263 |
+
char_data = [
|
| 264 |
+
{"name": c.name, "role": c.role, "visual_prompt": c.visual_prompt}
|
| 265 |
+
for c in characters
|
| 266 |
+
]
|
| 267 |
+
char_names = {c.name for c in characters}
|
| 268 |
+
characters_block = format_characters_block(char_data)
|
| 269 |
+
|
| 270 |
+
# Combine chapter text (limit to ~15k chars for Gemini)
|
| 271 |
+
combined_text = "\n\n".join(ch.source_text for ch in chapters)
|
| 272 |
+
if len(combined_text) > 15000:
|
| 273 |
+
combined_text = combined_text[:15000] + "\n\n[Text truncated for processing]"
|
| 274 |
+
|
| 275 |
+
continuity_block = continuity_context or ""
|
| 276 |
+
|
| 277 |
+
# Count paragraphs so the LLM knows how much content to cover
|
| 278 |
+
paragraphs = [p.strip() for p in combined_text.split("\n") if p.strip()]
|
| 279 |
+
para_hint = f"\n\n[This chapter has {len(paragraphs)} paragraphs. You need at least {max(25, len(paragraphs))} cuts to cover it all.]\n"
|
| 280 |
+
|
| 281 |
+
prompt = STORYBOARD_PROMPT.format(
|
| 282 |
+
characters_block=characters_block,
|
| 283 |
+
continuity_block=continuity_block,
|
| 284 |
+
chapter_text=combined_text + para_hint,
|
| 285 |
+
)
|
| 286 |
+
|
| 287 |
+
if on_progress:
|
| 288 |
+
await on_progress("Composing storyboard...", 20)
|
| 289 |
+
|
| 290 |
+
# Use the stronger model for storyboard — flash-lite truncates large JSON
|
| 291 |
+
storyboard = await call_gemini_json(
|
| 292 |
+
prompt,
|
| 293 |
+
system_instruction=STORYBOARD_SYSTEM,
|
| 294 |
+
max_tokens=65536,
|
| 295 |
+
model="gemini-2.5-flash",
|
| 296 |
+
)
|
| 297 |
+
|
| 298 |
+
# Gemini may return the scenes array directly instead of {"scenes": [...]}
|
| 299 |
+
if not isinstance(storyboard, dict) or "scenes" not in storyboard:
|
| 300 |
+
if isinstance(storyboard, list):
|
| 301 |
+
storyboard = {"scenes": storyboard}
|
| 302 |
+
else:
|
| 303 |
+
raise ValueError("Storyboard must be a dict with 'scenes' key")
|
| 304 |
+
|
| 305 |
+
scene_count = len(storyboard.get("scenes", []))
|
| 306 |
+
raw_cuts = sum(
|
| 307 |
+
len(s.get("cuts", s.get("shots", [])))
|
| 308 |
+
for s in storyboard.get("scenes", [])
|
| 309 |
+
)
|
| 310 |
+
|
| 311 |
+
if on_progress:
|
| 312 |
+
await on_progress(f"Validating {scene_count} scenes, {raw_cuts} cuts...", 80)
|
| 313 |
+
|
| 314 |
+
# Normalize cuts/shots naming
|
| 315 |
+
storyboard = _normalize_cuts(storyboard)
|
| 316 |
+
|
| 317 |
+
# Post-process
|
| 318 |
+
storyboard = _validate_storyboard(storyboard, characters)
|
| 319 |
+
storyboard = _enforce_cut_limit(storyboard, max_cuts=50)
|
| 320 |
+
|
| 321 |
+
total_cuts = sum(
|
| 322 |
+
len(scene.get("cuts", scene.get("shots", [])))
|
| 323 |
+
for scene in storyboard.get("scenes", [])
|
| 324 |
+
)
|
| 325 |
+
total_duration = sum(
|
| 326 |
+
cut.get("duration_sec", 2)
|
| 327 |
+
for scene in storyboard.get("scenes", [])
|
| 328 |
+
for cut in scene.get("cuts", scene.get("shots", []))
|
| 329 |
+
)
|
| 330 |
+
|
| 331 |
+
storyboard["metadata"] = {
|
| 332 |
+
"total_shots": total_cuts,
|
| 333 |
+
"total_cuts": total_cuts,
|
| 334 |
+
"total_duration_estimate": round(total_duration, 1),
|
| 335 |
+
"scene_count": len(storyboard.get("scenes", [])),
|
| 336 |
+
}
|
| 337 |
+
|
| 338 |
+
if on_progress:
|
| 339 |
+
mins = int(total_duration // 60)
|
| 340 |
+
secs = int(total_duration % 60)
|
| 341 |
+
await on_progress(
|
| 342 |
+
f"E-konte ready — {total_cuts} cuts, ~{mins}:{secs:02d}",
|
| 343 |
+
100,
|
| 344 |
+
)
|
| 345 |
+
|
| 346 |
+
return storyboard
|
app/pipeline/stage_4_image_gen.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 4: Image Generation — bulk generate images for all storyboard cuts."""
|
| 2 |
+
import asyncio
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
from app.models.character import Character
|
| 5 |
+
from app.services.image_gen import generate_image
|
| 6 |
+
from app.utils.prompt_builder import build_image_prompt
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
async def run_image_generation(
|
| 10 |
+
storyboard: dict,
|
| 11 |
+
characters: list[Character],
|
| 12 |
+
project_dir: str,
|
| 13 |
+
on_progress: callable = None,
|
| 14 |
+
resume: bool = False,
|
| 15 |
+
) -> dict:
|
| 16 |
+
"""Generate images for all cuts in the storyboard.
|
| 17 |
+
|
| 18 |
+
Returns updated storyboard with image_path set on each cut.
|
| 19 |
+
If resume=True, skip cuts whose image files already exist on disk.
|
| 20 |
+
|
| 21 |
+
Uses Pollinations as primary provider with vision-capable models
|
| 22 |
+
for characters that have portrait references (for consistency).
|
| 23 |
+
"""
|
| 24 |
+
# Support both new "cuts" and legacy "shots" format
|
| 25 |
+
all_cuts = []
|
| 26 |
+
for scene in storyboard.get("scenes", []):
|
| 27 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 28 |
+
all_cuts.append(cut)
|
| 29 |
+
|
| 30 |
+
total = len(all_cuts)
|
| 31 |
+
if on_progress:
|
| 32 |
+
await on_progress(f"Preparing {total} images...", 0)
|
| 33 |
+
|
| 34 |
+
# Build character prompt map and portrait URL map (with alias support)
|
| 35 |
+
char_prompts = {c.name: c.visual_prompt for c in characters if c.visual_prompt}
|
| 36 |
+
char_portraits = {c.name: c.portrait_url for c in characters if c.portrait_url}
|
| 37 |
+
# Alias map: alias → canonical name
|
| 38 |
+
alias_map: dict[str, str] = {}
|
| 39 |
+
for c in characters:
|
| 40 |
+
for alias in (c.aliases_json or []):
|
| 41 |
+
alias_map[alias.lower()] = c.name
|
| 42 |
+
|
| 43 |
+
completed = 0
|
| 44 |
+
skipped = 0
|
| 45 |
+
errors = []
|
| 46 |
+
|
| 47 |
+
img_dir = Path(project_dir) / "images"
|
| 48 |
+
img_dir.mkdir(parents=True, exist_ok=True)
|
| 49 |
+
|
| 50 |
+
async def generate_one(cut: dict) -> None:
|
| 51 |
+
nonlocal completed, skipped
|
| 52 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id", "S00")
|
| 53 |
+
output_path = str(img_dir / f"{cut_id}.png")
|
| 54 |
+
|
| 55 |
+
# Resume: skip if file already exists on disk
|
| 56 |
+
if resume and Path(output_path).exists():
|
| 57 |
+
cut["image_path"] = output_path
|
| 58 |
+
skipped += 1
|
| 59 |
+
completed += 1
|
| 60 |
+
if on_progress:
|
| 61 |
+
pct = int((completed / max(total, 1)) * 100)
|
| 62 |
+
await on_progress(f"[{completed}/{total}] {cut_id} (cached)", pct)
|
| 63 |
+
return
|
| 64 |
+
|
| 65 |
+
# Build prompt
|
| 66 |
+
prompt = build_image_prompt(cut, char_prompts)
|
| 67 |
+
|
| 68 |
+
# Use klein-large for ALL shots to maintain consistent aesthetics.
|
| 69 |
+
# Same model + same seed = uniform art style across entire episode.
|
| 70 |
+
# grok-imagine only used as fallback if klein-large fails.
|
| 71 |
+
provider = "pollinations"
|
| 72 |
+
pollinations_model = "klein-large"
|
| 73 |
+
reference_url = None
|
| 74 |
+
seed = 42 # Locked seed for style consistency
|
| 75 |
+
|
| 76 |
+
# Pass portrait reference for character shots (with alias fallback)
|
| 77 |
+
focal = cut.get("focal_character")
|
| 78 |
+
resolved_focal = focal
|
| 79 |
+
if focal and focal not in char_portraits:
|
| 80 |
+
resolved_focal = alias_map.get(focal.lower(), focal)
|
| 81 |
+
if resolved_focal and resolved_focal in char_portraits:
|
| 82 |
+
reference_url = char_portraits[resolved_focal]
|
| 83 |
+
else:
|
| 84 |
+
# Wide/establishing: use any present character's portrait as ref
|
| 85 |
+
for c in cut.get("characters_present", []):
|
| 86 |
+
resolved_c = c if c in char_portraits else alias_map.get(c.lower(), c)
|
| 87 |
+
if resolved_c in char_portraits:
|
| 88 |
+
reference_url = char_portraits[resolved_c]
|
| 89 |
+
break
|
| 90 |
+
|
| 91 |
+
try:
|
| 92 |
+
await generate_image(
|
| 93 |
+
prompt=prompt,
|
| 94 |
+
output_path=output_path,
|
| 95 |
+
provider=provider,
|
| 96 |
+
seed=seed,
|
| 97 |
+
pollinations_model=pollinations_model,
|
| 98 |
+
reference_image_url=reference_url,
|
| 99 |
+
)
|
| 100 |
+
cut["image_path"] = output_path
|
| 101 |
+
except Exception as e:
|
| 102 |
+
# Fallback: retry with grok-imagine (no reference) if klein-large failed
|
| 103 |
+
if pollinations_model == "klein-large":
|
| 104 |
+
try:
|
| 105 |
+
await generate_image(
|
| 106 |
+
prompt=prompt,
|
| 107 |
+
output_path=output_path,
|
| 108 |
+
provider=provider,
|
| 109 |
+
seed=seed,
|
| 110 |
+
pollinations_model="grok-imagine",
|
| 111 |
+
reference_image_url=None,
|
| 112 |
+
)
|
| 113 |
+
cut["image_path"] = output_path
|
| 114 |
+
errors.append(f"{cut_id}: klein-large failed, used grok-imagine fallback")
|
| 115 |
+
except Exception as e2:
|
| 116 |
+
errors.append(f"{cut_id}: {e} | fallback also failed: {e2}")
|
| 117 |
+
cut["image_path"] = None
|
| 118 |
+
else:
|
| 119 |
+
errors.append(f"{cut_id}: {e}")
|
| 120 |
+
cut["image_path"] = None
|
| 121 |
+
|
| 122 |
+
completed += 1
|
| 123 |
+
if on_progress:
|
| 124 |
+
pct = int((completed / max(total, 1)) * 100)
|
| 125 |
+
prompt_preview = prompt[:40].rstrip() + "..." if len(prompt) > 40 else prompt
|
| 126 |
+
await on_progress(f"[{completed}/{total}] {prompt_preview}", pct)
|
| 127 |
+
|
| 128 |
+
# Run with concurrency limit (semaphores are in the image_gen service)
|
| 129 |
+
await asyncio.gather(*[generate_one(cut) for cut in all_cuts])
|
| 130 |
+
|
| 131 |
+
if on_progress:
|
| 132 |
+
err_suffix = f" ({len(errors)} errors)" if errors else ""
|
| 133 |
+
skip_suffix = f" ({skipped} cached)" if skipped else ""
|
| 134 |
+
await on_progress(f"{total} images generated{skip_suffix}{err_suffix}", 100)
|
| 135 |
+
|
| 136 |
+
storyboard["_image_errors"] = errors
|
| 137 |
+
return storyboard
|
app/pipeline/stage_5_tts.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 5: TTS Generation — generate voice clips for all dialogue/narration.
|
| 2 |
+
|
| 3 |
+
Dual-engine dispatch:
|
| 4 |
+
- Characters with voice_reference_path → Pocket TTS (anime voice cloning)
|
| 5 |
+
- Narration + characters without voice_reference_path → Edge TTS (fast, free)
|
| 6 |
+
"""
|
| 7 |
+
import asyncio
|
| 8 |
+
import logging
|
| 9 |
+
from pathlib import Path
|
| 10 |
+
from app.models.character import Character
|
| 11 |
+
from app.services.tts import generate_tts_batch
|
| 12 |
+
|
| 13 |
+
logger = logging.getLogger(__name__)
|
| 14 |
+
|
| 15 |
+
# Narrator voice: Zhongli via Pocket TTS (deep, wise, authoritative)
|
| 16 |
+
NARRATOR_VOICE_REF = "data/genshin_voices/zhongli/voice_state.safetensors"
|
| 17 |
+
|
| 18 |
+
# Edge TTS fallback for narrator (if Pocket TTS fails)
|
| 19 |
+
NARRATOR_VOICE = {
|
| 20 |
+
"voice_name": "en-US-ChristopherNeural",
|
| 21 |
+
"rate": "-5%",
|
| 22 |
+
"pitch": "-5Hz",
|
| 23 |
+
}
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
async def _generate_pocket_tts_items(items: list[dict]) -> list[dict]:
|
| 27 |
+
"""Generate TTS clips using Pocket TTS (sequential, CPU-bound).
|
| 28 |
+
|
| 29 |
+
Pocket TTS is single-threaded internally so we process one at a time
|
| 30 |
+
via asyncio.to_thread() to avoid blocking the event loop.
|
| 31 |
+
"""
|
| 32 |
+
from app.services.pocket_tts_service import PocketTTSService
|
| 33 |
+
|
| 34 |
+
results = []
|
| 35 |
+
for item in items:
|
| 36 |
+
try:
|
| 37 |
+
result = await PocketTTSService.generate(
|
| 38 |
+
text=item["text"],
|
| 39 |
+
voice_ref=item["voice_ref"],
|
| 40 |
+
output_path=item["output_path"],
|
| 41 |
+
)
|
| 42 |
+
result["shot_id"] = item.get("shot_id")
|
| 43 |
+
result["word_timestamps"] = [] # Pocket TTS doesn't provide word timing
|
| 44 |
+
results.append(result)
|
| 45 |
+
except Exception as e:
|
| 46 |
+
logger.warning(
|
| 47 |
+
"Pocket TTS failed for %s: %s — falling back to Edge TTS",
|
| 48 |
+
item.get("shot_id"), e,
|
| 49 |
+
)
|
| 50 |
+
# Fallback: use Edge TTS for this item
|
| 51 |
+
fallback_result = await _edge_tts_fallback(item)
|
| 52 |
+
results.append(fallback_result)
|
| 53 |
+
|
| 54 |
+
return results
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
async def _edge_tts_fallback(item: dict) -> dict:
|
| 58 |
+
"""Generate a single clip via Edge TTS as fallback."""
|
| 59 |
+
from app.services.tts import generate_tts
|
| 60 |
+
|
| 61 |
+
# Change extension to .mp3 for Edge TTS
|
| 62 |
+
output_path = item["output_path"]
|
| 63 |
+
if output_path.endswith(".wav"):
|
| 64 |
+
output_path = output_path[:-4] + ".mp3"
|
| 65 |
+
|
| 66 |
+
vc = item.get("voice_config") or NARRATOR_VOICE
|
| 67 |
+
result = await generate_tts(
|
| 68 |
+
text=item["text"],
|
| 69 |
+
output_path=output_path,
|
| 70 |
+
voice_name=vc.get("voice_name", "en-US-ChristopherNeural"),
|
| 71 |
+
rate=vc.get("rate", "+0%"),
|
| 72 |
+
pitch=vc.get("pitch", "+0Hz"),
|
| 73 |
+
emotion=item.get("emotion", "neutral"),
|
| 74 |
+
)
|
| 75 |
+
result["shot_id"] = item.get("shot_id")
|
| 76 |
+
return result
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
async def run_tts_generation(
|
| 80 |
+
storyboard: dict,
|
| 81 |
+
characters: list[Character],
|
| 82 |
+
project_dir: str,
|
| 83 |
+
on_progress: callable = None,
|
| 84 |
+
resume: bool = False,
|
| 85 |
+
) -> dict:
|
| 86 |
+
"""Generate TTS for all cuts with dialogue/narration.
|
| 87 |
+
|
| 88 |
+
Dispatches to Pocket TTS (cloned anime voices) or Edge TTS based on
|
| 89 |
+
whether the character has a voice_reference_path set.
|
| 90 |
+
|
| 91 |
+
Returns updated storyboard with tts_path and actual_duration_sec set.
|
| 92 |
+
If resume=True, skip cuts whose audio files already exist on disk.
|
| 93 |
+
"""
|
| 94 |
+
if on_progress:
|
| 95 |
+
await on_progress("Preparing voice clips...", 0)
|
| 96 |
+
|
| 97 |
+
# Build character maps
|
| 98 |
+
char_voices = {} # name -> Edge TTS voice config
|
| 99 |
+
char_voice_refs = {} # name -> Pocket TTS voice ref (safetensors path or preset name)
|
| 100 |
+
for c in characters:
|
| 101 |
+
if c.voice_reference_path:
|
| 102 |
+
ref = c.voice_reference_path
|
| 103 |
+
# Accept both file paths (if they exist) and preset names
|
| 104 |
+
if Path(ref).exists() or not ref.endswith((".safetensors", ".wav")):
|
| 105 |
+
char_voice_refs[c.name] = ref
|
| 106 |
+
if c.voice_config_json:
|
| 107 |
+
char_voices[c.name] = c.voice_config_json
|
| 108 |
+
|
| 109 |
+
audio_dir = Path(project_dir) / "audio"
|
| 110 |
+
audio_dir.mkdir(parents=True, exist_ok=True)
|
| 111 |
+
|
| 112 |
+
# Collect TTS tasks, split by engine
|
| 113 |
+
edge_items = []
|
| 114 |
+
pocket_items = []
|
| 115 |
+
skipped = 0
|
| 116 |
+
|
| 117 |
+
for scene in storyboard.get("scenes", []):
|
| 118 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 119 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 120 |
+
dialogue = cut.get("dialogue")
|
| 121 |
+
if not dialogue or not dialogue.get("text"):
|
| 122 |
+
# Belt-and-suspenders: create narration from action_description
|
| 123 |
+
action = cut.get("action_description", "")
|
| 124 |
+
if not action:
|
| 125 |
+
continue # truly empty — skip
|
| 126 |
+
dialogue = {
|
| 127 |
+
"speaker": "Narrator",
|
| 128 |
+
"text": action,
|
| 129 |
+
"is_narration": True,
|
| 130 |
+
"emotion": "neutral",
|
| 131 |
+
"delivery": "normal",
|
| 132 |
+
}
|
| 133 |
+
cut["dialogue"] = dialogue
|
| 134 |
+
|
| 135 |
+
speaker = dialogue.get("speaker", "Narrator")
|
| 136 |
+
is_narration = dialogue.get("is_narration", False)
|
| 137 |
+
|
| 138 |
+
# Determine engine:
|
| 139 |
+
# - Narrator → Pocket TTS with Zhongli (fallback to Edge TTS)
|
| 140 |
+
# - Characters with voice refs → Pocket TTS
|
| 141 |
+
# - Others → Edge TTS
|
| 142 |
+
is_narrator_line = is_narration or speaker == "Narrator"
|
| 143 |
+
narrator_ref_exists = Path(NARRATOR_VOICE_REF).exists()
|
| 144 |
+
use_pocket = (
|
| 145 |
+
(is_narrator_line and narrator_ref_exists)
|
| 146 |
+
or (not is_narrator_line and speaker in char_voice_refs)
|
| 147 |
+
)
|
| 148 |
+
|
| 149 |
+
# File extension: .wav for Pocket TTS, .mp3 for Edge TTS
|
| 150 |
+
ext = ".wav" if use_pocket else ".mp3"
|
| 151 |
+
output_path = str(audio_dir / f"{cut_id}{ext}")
|
| 152 |
+
|
| 153 |
+
# Resume: skip if audio file already exists (check both extensions)
|
| 154 |
+
if resume:
|
| 155 |
+
wav_path = str(audio_dir / f"{cut_id}.wav")
|
| 156 |
+
mp3_path = str(audio_dir / f"{cut_id}.mp3")
|
| 157 |
+
if Path(wav_path).exists() or Path(mp3_path).exists():
|
| 158 |
+
cut["tts_path"] = wav_path if Path(wav_path).exists() else mp3_path
|
| 159 |
+
skipped += 1
|
| 160 |
+
continue
|
| 161 |
+
|
| 162 |
+
if use_pocket:
|
| 163 |
+
voice_ref = NARRATOR_VOICE_REF if is_narrator_line else char_voice_refs[speaker]
|
| 164 |
+
pocket_items.append({
|
| 165 |
+
"shot_id": cut_id,
|
| 166 |
+
"text": dialogue["text"],
|
| 167 |
+
"output_path": output_path,
|
| 168 |
+
"voice_ref": voice_ref,
|
| 169 |
+
"voice_config": char_voices.get(speaker, NARRATOR_VOICE),
|
| 170 |
+
"emotion": dialogue.get("emotion", "neutral"),
|
| 171 |
+
})
|
| 172 |
+
else:
|
| 173 |
+
voice_config = NARRATOR_VOICE if is_narrator_line else char_voices.get(speaker, NARRATOR_VOICE)
|
| 174 |
+
edge_items.append({
|
| 175 |
+
"shot_id": cut_id,
|
| 176 |
+
"text": dialogue["text"],
|
| 177 |
+
"output_path": output_path,
|
| 178 |
+
"voice_config": voice_config,
|
| 179 |
+
"emotion": dialogue.get("emotion", "neutral"),
|
| 180 |
+
})
|
| 181 |
+
|
| 182 |
+
total_items = len(edge_items) + len(pocket_items)
|
| 183 |
+
if total_items == 0:
|
| 184 |
+
if on_progress:
|
| 185 |
+
msg = "No dialogue found" if skipped == 0 else f"All {skipped} voice clips cached"
|
| 186 |
+
await on_progress(msg, 100)
|
| 187 |
+
return storyboard
|
| 188 |
+
|
| 189 |
+
if on_progress:
|
| 190 |
+
skip_msg = f" ({skipped} cached)" if skipped else ""
|
| 191 |
+
engine_msg = ""
|
| 192 |
+
if pocket_items and edge_items:
|
| 193 |
+
engine_msg = f" [{len(pocket_items)} anime + {len(edge_items)} standard]"
|
| 194 |
+
elif pocket_items:
|
| 195 |
+
engine_msg = f" [{len(pocket_items)} anime voices]"
|
| 196 |
+
await on_progress(f"Recording {total_items} voice clips{engine_msg}{skip_msg}...", 10)
|
| 197 |
+
|
| 198 |
+
# Generate clips — Pocket TTS (sequential) and Edge TTS (concurrent) in parallel
|
| 199 |
+
pocket_results = []
|
| 200 |
+
edge_results = []
|
| 201 |
+
|
| 202 |
+
if pocket_items and edge_items:
|
| 203 |
+
# Run both engines in parallel
|
| 204 |
+
pocket_results, edge_results = await asyncio.gather(
|
| 205 |
+
_generate_pocket_tts_items(pocket_items),
|
| 206 |
+
generate_tts_batch(edge_items),
|
| 207 |
+
)
|
| 208 |
+
elif pocket_items:
|
| 209 |
+
pocket_results = await _generate_pocket_tts_items(pocket_items)
|
| 210 |
+
else:
|
| 211 |
+
edge_results = await generate_tts_batch(edge_items)
|
| 212 |
+
|
| 213 |
+
all_results = pocket_results + edge_results
|
| 214 |
+
|
| 215 |
+
# Map results back to cuts
|
| 216 |
+
result_map = {r["shot_id"]: r for r in all_results if r.get("shot_id")}
|
| 217 |
+
|
| 218 |
+
for scene in storyboard.get("scenes", []):
|
| 219 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 220 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 221 |
+
if cut_id in result_map:
|
| 222 |
+
r = result_map[cut_id]
|
| 223 |
+
cut["tts_path"] = r["path"]
|
| 224 |
+
cut["word_timestamps"] = r.get("word_timestamps", [])
|
| 225 |
+
# Audio-first: actual TTS duration overrides estimate
|
| 226 |
+
if r.get("duration_sec"):
|
| 227 |
+
cut["actual_duration_sec"] = r["duration_sec"]
|
| 228 |
+
cut["duration_sec"] = max(r["duration_sec"] + 0.3, cut.get("duration_sec", 2.0))
|
| 229 |
+
|
| 230 |
+
if on_progress:
|
| 231 |
+
pocket_count = len(pocket_results)
|
| 232 |
+
edge_count = len(edge_results)
|
| 233 |
+
detail = f"{len(all_results)} voice clips recorded"
|
| 234 |
+
if pocket_count > 0:
|
| 235 |
+
detail += f" ({pocket_count} anime, {edge_count} standard)"
|
| 236 |
+
await on_progress(detail, 100)
|
| 237 |
+
|
| 238 |
+
logger.info(
|
| 239 |
+
"TTS complete: %d pocket + %d edge = %d total (%d cached)",
|
| 240 |
+
len(pocket_results), len(edge_results), len(all_results), skipped,
|
| 241 |
+
)
|
| 242 |
+
return storyboard
|
app/pipeline/stage_6_animation.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 6 (Fallback): Ken Burns Animation on each cut image.
|
| 2 |
+
|
| 3 |
+
This is the fallback animation method when Pollinations video generation
|
| 4 |
+
is unavailable or fails. Applies Ken Burns zoom/pan effects to static images.
|
| 5 |
+
"""
|
| 6 |
+
import asyncio
|
| 7 |
+
import logging
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
from app.services.ffmpeg import animate_shot
|
| 10 |
+
|
| 11 |
+
logger = logging.getLogger(__name__)
|
| 12 |
+
|
| 13 |
+
# Map professional camera notation to Ken Burns presets
|
| 14 |
+
CAMERA_TO_KEN_BURNS = {
|
| 15 |
+
"FIX": "static",
|
| 16 |
+
"PAN_LEFT": "pan_left",
|
| 17 |
+
"PAN_RIGHT": "pan_right",
|
| 18 |
+
"TILT_UP": "pan_up",
|
| 19 |
+
"TILT_DOWN": "pan_down",
|
| 20 |
+
"T.U.": "zoom_in_slow",
|
| 21 |
+
"T.B.": "zoom_out",
|
| 22 |
+
"DOLLY_LEFT": "pan_left",
|
| 23 |
+
"DOLLY_RIGHT": "pan_right",
|
| 24 |
+
"FOLLOW": "pan_right",
|
| 25 |
+
"CRANE_UP": "pan_up",
|
| 26 |
+
"CRANE_DOWN": "pan_down",
|
| 27 |
+
"SHAKE": "shake",
|
| 28 |
+
}
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
async def run_animation(
|
| 32 |
+
storyboard: dict,
|
| 33 |
+
project_dir: str,
|
| 34 |
+
on_progress: callable = None,
|
| 35 |
+
) -> dict:
|
| 36 |
+
"""Apply Ken Burns animation to all cut images.
|
| 37 |
+
|
| 38 |
+
Returns updated storyboard with clip_path set on each cut.
|
| 39 |
+
Supports both new 'cuts' and legacy 'shots' format.
|
| 40 |
+
"""
|
| 41 |
+
total_cuts = sum(
|
| 42 |
+
len(s.get("cuts", s.get("shots", [])))
|
| 43 |
+
for s in storyboard.get("scenes", [])
|
| 44 |
+
)
|
| 45 |
+
if on_progress:
|
| 46 |
+
await on_progress(f"Animating {total_cuts} cuts (Ken Burns)...", 0)
|
| 47 |
+
|
| 48 |
+
clips_dir = Path(project_dir) / "clips"
|
| 49 |
+
clips_dir.mkdir(parents=True, exist_ok=True)
|
| 50 |
+
|
| 51 |
+
# Collect cuts with images, annotating scene position for fades
|
| 52 |
+
cuts = []
|
| 53 |
+
for scene in storyboard.get("scenes", []):
|
| 54 |
+
scene_cuts = [
|
| 55 |
+
c for c in scene.get("cuts", scene.get("shots", []))
|
| 56 |
+
if c.get("image_path")
|
| 57 |
+
]
|
| 58 |
+
for idx, cut in enumerate(scene_cuts):
|
| 59 |
+
is_first = idx == 0
|
| 60 |
+
is_last = idx == len(scene_cuts) - 1
|
| 61 |
+
if is_first and is_last:
|
| 62 |
+
cut["_fade_in"] = 0.4
|
| 63 |
+
cut["_fade_out"] = 0.4
|
| 64 |
+
elif is_first:
|
| 65 |
+
cut["_fade_in"] = 0.4
|
| 66 |
+
cut["_fade_out"] = 0.15
|
| 67 |
+
elif is_last:
|
| 68 |
+
cut["_fade_in"] = 0.15
|
| 69 |
+
cut["_fade_out"] = 0.4
|
| 70 |
+
else:
|
| 71 |
+
cut["_fade_in"] = 0.15
|
| 72 |
+
cut["_fade_out"] = 0.15
|
| 73 |
+
cuts.append(cut)
|
| 74 |
+
|
| 75 |
+
total = len(cuts)
|
| 76 |
+
logger.info("Animation: %d/%d cuts have image_path", total, total_cuts)
|
| 77 |
+
completed = 0
|
| 78 |
+
semaphore = asyncio.Semaphore(2) # Limit concurrent FFmpeg processes
|
| 79 |
+
|
| 80 |
+
async def animate_one(cut: dict) -> None:
|
| 81 |
+
nonlocal completed
|
| 82 |
+
async with semaphore:
|
| 83 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 84 |
+
output = str(clips_dir / f"{cut_id}.mp4")
|
| 85 |
+
|
| 86 |
+
# Use actual TTS duration if available (audio-first)
|
| 87 |
+
duration = cut.get("actual_duration_sec", cut.get("duration_sec", 3.0))
|
| 88 |
+
duration = max(duration, 0.5)
|
| 89 |
+
|
| 90 |
+
# Map professional camera notation to Ken Burns presets
|
| 91 |
+
camera = cut.get("camera_movement", "static")
|
| 92 |
+
kb_camera = CAMERA_TO_KEN_BURNS.get(camera, camera)
|
| 93 |
+
|
| 94 |
+
try:
|
| 95 |
+
await animate_shot(
|
| 96 |
+
image_path=cut["image_path"],
|
| 97 |
+
output_path=output,
|
| 98 |
+
duration_sec=duration,
|
| 99 |
+
camera_movement=kb_camera,
|
| 100 |
+
audio_path=cut.get("tts_path"),
|
| 101 |
+
fade_in=cut.get("_fade_in", 0.0),
|
| 102 |
+
fade_out=cut.get("_fade_out", 0.0),
|
| 103 |
+
)
|
| 104 |
+
cut["clip_path"] = output
|
| 105 |
+
except Exception as e:
|
| 106 |
+
cut["clip_path"] = None
|
| 107 |
+
cut["_animation_error"] = str(e)
|
| 108 |
+
logger.error("Animation failed for %s: %s", cut_id, e)
|
| 109 |
+
|
| 110 |
+
completed += 1
|
| 111 |
+
if on_progress:
|
| 112 |
+
pct = int((completed / max(total, 1)) * 100)
|
| 113 |
+
dur = cut.get("duration_sec", 0)
|
| 114 |
+
await on_progress(f"[{completed}/{total}] {kb_camera} ({dur:.1f}s)", pct)
|
| 115 |
+
|
| 116 |
+
await asyncio.gather(*[animate_one(c) for c in cuts])
|
| 117 |
+
|
| 118 |
+
if on_progress:
|
| 119 |
+
await on_progress(f"{total} cuts animated", 100)
|
| 120 |
+
|
| 121 |
+
return storyboard
|
app/pipeline/stage_6_video_gen.py
ADDED
|
@@ -0,0 +1,304 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 6 (Video): AI Video Generation via Pollinations grok-video.
|
| 2 |
+
|
| 3 |
+
Replaces Ken Burns animation with actual AI-generated video clips.
|
| 4 |
+
Each keyframe image is uploaded and used as img2vid reference.
|
| 5 |
+
|
| 6 |
+
Falls back to Ken Burns (stage_6_animation.py) on failure.
|
| 7 |
+
"""
|
| 8 |
+
import asyncio
|
| 9 |
+
import logging
|
| 10 |
+
import math
|
| 11 |
+
import os
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
|
| 14 |
+
from app.services.pollinations import generate_video, upload_media
|
| 15 |
+
from app.pipeline.stage_6_animation import run_animation as run_ken_burns_fallback
|
| 16 |
+
from app.services.ffmpeg import get_duration, mux_audio, rescale_to_landscape
|
| 17 |
+
|
| 18 |
+
logger = logging.getLogger(__name__)
|
| 19 |
+
|
| 20 |
+
# Sequential video gen — Pollinations throttles concurrent pk_* key requests
|
| 21 |
+
_video_semaphore = asyncio.Semaphore(1)
|
| 22 |
+
|
| 23 |
+
# Delay between video gen requests to avoid throttling
|
| 24 |
+
_VIDEO_DELAY_SEC = 6.0
|
| 25 |
+
|
| 26 |
+
# Camera movement → video prompt mapping for enhanced motion descriptions
|
| 27 |
+
CAMERA_MOTION_MAP = {
|
| 28 |
+
"FIX": "static camera, subtle atmospheric motion",
|
| 29 |
+
"PAN_LEFT": "camera pans smoothly to the left",
|
| 30 |
+
"PAN_RIGHT": "camera pans smoothly to the right",
|
| 31 |
+
"TILT_UP": "camera tilts upward slowly",
|
| 32 |
+
"TILT_DOWN": "camera tilts downward slowly",
|
| 33 |
+
"T.U.": "camera slowly zooms in, tracking toward subject",
|
| 34 |
+
"T.B.": "camera slowly pulls back, revealing more of the scene",
|
| 35 |
+
"DOLLY_LEFT": "camera moves laterally to the left",
|
| 36 |
+
"DOLLY_RIGHT": "camera moves laterally to the right",
|
| 37 |
+
"FOLLOW": "camera follows the character's movement",
|
| 38 |
+
"CRANE_UP": "camera rises upward, bird's eye perspective",
|
| 39 |
+
"CRANE_DOWN": "camera descends from above",
|
| 40 |
+
"SHAKE": "handheld camera shake, intense impact feel",
|
| 41 |
+
# Legacy camera movements (from old storyboard format)
|
| 42 |
+
"static": "subtle idle motion, breathing animation",
|
| 43 |
+
"zoom_in_slow": "camera slowly zooms in",
|
| 44 |
+
"zoom_in_fast": "camera quickly zooms in with urgency",
|
| 45 |
+
"zoom_out": "camera zooms out, revealing the scene",
|
| 46 |
+
"pan_left": "camera pans smoothly to the left",
|
| 47 |
+
"pan_right": "camera pans smoothly to the right",
|
| 48 |
+
"pan_up": "camera tilts upward",
|
| 49 |
+
"pan_down": "camera tilts downward",
|
| 50 |
+
"tilt": "camera tilts with slight rotation",
|
| 51 |
+
"shake": "handheld camera shake, impact",
|
| 52 |
+
"dolly_in": "camera tracks toward subject",
|
| 53 |
+
}
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
ANIME_VIDEO_PREFIX = "anime cel-shading style, 2D manhwa animation, "
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def _build_video_prompt(cut: dict) -> str:
|
| 60 |
+
"""Build a video generation prompt from cut data.
|
| 61 |
+
|
| 62 |
+
Combines the explicit video_prompt (from e-konte storyboard) with
|
| 63 |
+
camera movement description and action description.
|
| 64 |
+
Prepends anime style cue to keep grok-video from going realistic.
|
| 65 |
+
"""
|
| 66 |
+
parts = []
|
| 67 |
+
|
| 68 |
+
# Use explicit video_prompt if available (new e-konte format)
|
| 69 |
+
video_prompt = cut.get("video_prompt", "")
|
| 70 |
+
if video_prompt:
|
| 71 |
+
parts.append(video_prompt)
|
| 72 |
+
|
| 73 |
+
# Add camera movement context
|
| 74 |
+
camera = cut.get("camera_movement", "static")
|
| 75 |
+
camera_desc = CAMERA_MOTION_MAP.get(camera, "subtle camera motion")
|
| 76 |
+
if camera_desc not in (video_prompt or ""):
|
| 77 |
+
parts.append(camera_desc)
|
| 78 |
+
|
| 79 |
+
# Add action description for additional motion context
|
| 80 |
+
action = cut.get("action_description", "")
|
| 81 |
+
if action and action not in (video_prompt or ""):
|
| 82 |
+
parts.append(action)
|
| 83 |
+
|
| 84 |
+
# If still empty, create a basic motion prompt
|
| 85 |
+
if not parts:
|
| 86 |
+
parts.append("subtle idle animation, slight breathing motion, atmospheric particles floating")
|
| 87 |
+
|
| 88 |
+
return ANIME_VIDEO_PREFIX + ", ".join(parts)
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
async def _generate_video_clip(
|
| 92 |
+
cut: dict,
|
| 93 |
+
clips_dir: Path,
|
| 94 |
+
image_upload_cache: dict,
|
| 95 |
+
) -> bool:
|
| 96 |
+
"""Generate a video clip for a single cut. Returns True on success."""
|
| 97 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 98 |
+
image_path = cut.get("image_path")
|
| 99 |
+
output_path = str(clips_dir / f"{cut_id}.mp4")
|
| 100 |
+
|
| 101 |
+
if not image_path or not Path(image_path).exists():
|
| 102 |
+
logger.warning("No image for %s, skipping video gen", cut_id)
|
| 103 |
+
return False
|
| 104 |
+
|
| 105 |
+
# Duration: round UP so video is always >= audio length.
|
| 106 |
+
# grok-video accepts integer seconds 1-10. Rounding up ensures
|
| 107 |
+
# -shortest in mux_audio trims to the exact audio duration.
|
| 108 |
+
tts_duration = cut.get("actual_duration_sec")
|
| 109 |
+
storyboard_duration = cut.get("duration_sec", 3.0)
|
| 110 |
+
target_duration = tts_duration or storyboard_duration
|
| 111 |
+
video_duration = max(1, min(math.ceil(target_duration), 10))
|
| 112 |
+
|
| 113 |
+
async with _video_semaphore:
|
| 114 |
+
# Rate-limit delay for pk_* keys
|
| 115 |
+
await asyncio.sleep(_VIDEO_DELAY_SEC)
|
| 116 |
+
|
| 117 |
+
try:
|
| 118 |
+
# Upload image (with caching to avoid re-uploading)
|
| 119 |
+
if image_path not in image_upload_cache:
|
| 120 |
+
image_url = await upload_media(image_path)
|
| 121 |
+
image_upload_cache[image_path] = image_url
|
| 122 |
+
else:
|
| 123 |
+
image_url = image_upload_cache[image_path]
|
| 124 |
+
|
| 125 |
+
# Build video prompt
|
| 126 |
+
video_prompt = _build_video_prompt(cut)
|
| 127 |
+
|
| 128 |
+
# Step 1: Generate silent video from grok-video
|
| 129 |
+
silent_path = str(clips_dir / f"{cut_id}_silent.mp4")
|
| 130 |
+
await generate_video(
|
| 131 |
+
prompt=video_prompt,
|
| 132 |
+
output_path=silent_path,
|
| 133 |
+
model="grok-video",
|
| 134 |
+
duration=video_duration,
|
| 135 |
+
image_url=image_url,
|
| 136 |
+
)
|
| 137 |
+
|
| 138 |
+
# Step 2: Rescale to landscape (grok-video outputs portrait)
|
| 139 |
+
scaled_path = str(clips_dir / f"{cut_id}_scaled.mp4")
|
| 140 |
+
await rescale_to_landscape(silent_path, scaled_path, 1024, 768)
|
| 141 |
+
|
| 142 |
+
# Step 3: Mux TTS audio into the video (or add silent track)
|
| 143 |
+
tts_path = cut.get("tts_path")
|
| 144 |
+
await mux_audio(
|
| 145 |
+
video_path=scaled_path,
|
| 146 |
+
audio_path=tts_path,
|
| 147 |
+
output_path=output_path,
|
| 148 |
+
duration_sec=target_duration,
|
| 149 |
+
)
|
| 150 |
+
|
| 151 |
+
# Clean up temp files
|
| 152 |
+
for tmp in [silent_path, scaled_path]:
|
| 153 |
+
try:
|
| 154 |
+
os.unlink(tmp)
|
| 155 |
+
except OSError:
|
| 156 |
+
pass
|
| 157 |
+
|
| 158 |
+
cut["clip_path"] = output_path
|
| 159 |
+
logger.info(
|
| 160 |
+
"Video generated for %s: %s (video=%ds, audio=%.1fs)",
|
| 161 |
+
cut_id, output_path, video_duration,
|
| 162 |
+
tts_duration or 0,
|
| 163 |
+
)
|
| 164 |
+
return True
|
| 165 |
+
|
| 166 |
+
except Exception as e:
|
| 167 |
+
logger.error("Video generation failed for %s: %r", cut_id, e, exc_info=True)
|
| 168 |
+
cut["_video_error"] = repr(e)
|
| 169 |
+
# Clean up partial files
|
| 170 |
+
for p in [str(clips_dir / f"{cut_id}_silent.mp4"),
|
| 171 |
+
str(clips_dir / f"{cut_id}_scaled.mp4"),
|
| 172 |
+
output_path]:
|
| 173 |
+
try:
|
| 174 |
+
os.unlink(p)
|
| 175 |
+
except OSError:
|
| 176 |
+
pass
|
| 177 |
+
return False
|
| 178 |
+
|
| 179 |
+
|
| 180 |
+
async def run_video_generation(
|
| 181 |
+
storyboard: dict,
|
| 182 |
+
project_dir: str,
|
| 183 |
+
on_progress: callable = None,
|
| 184 |
+
) -> dict:
|
| 185 |
+
"""Generate AI video clips for all cuts in the storyboard.
|
| 186 |
+
|
| 187 |
+
Falls back to Ken Burns animation for any cuts where video generation fails.
|
| 188 |
+
Returns updated storyboard with clip_path set on each cut.
|
| 189 |
+
"""
|
| 190 |
+
clips_dir = Path(project_dir) / "clips"
|
| 191 |
+
clips_dir.mkdir(parents=True, exist_ok=True)
|
| 192 |
+
|
| 193 |
+
# Collect all cuts that have images
|
| 194 |
+
all_cuts = []
|
| 195 |
+
for scene in storyboard.get("scenes", []):
|
| 196 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 197 |
+
if cut.get("image_path"):
|
| 198 |
+
all_cuts.append(cut)
|
| 199 |
+
|
| 200 |
+
total = len(all_cuts)
|
| 201 |
+
if on_progress:
|
| 202 |
+
await on_progress(f"Generating {total} video clips...", 0)
|
| 203 |
+
|
| 204 |
+
if total == 0:
|
| 205 |
+
if on_progress:
|
| 206 |
+
await on_progress("No images available for video generation", 100)
|
| 207 |
+
return storyboard
|
| 208 |
+
|
| 209 |
+
# Shared upload cache to avoid re-uploading the same image
|
| 210 |
+
image_upload_cache = {}
|
| 211 |
+
completed = 0
|
| 212 |
+
succeeded = 0
|
| 213 |
+
failed_cuts = []
|
| 214 |
+
|
| 215 |
+
async def gen_one(cut: dict):
|
| 216 |
+
nonlocal completed, succeeded
|
| 217 |
+
try:
|
| 218 |
+
success = await _generate_video_clip(cut, clips_dir, image_upload_cache)
|
| 219 |
+
except Exception as e:
|
| 220 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 221 |
+
logger.error("Unexpected error in video gen for %s: %s", cut_id, e)
|
| 222 |
+
success = False
|
| 223 |
+
if success:
|
| 224 |
+
succeeded += 1
|
| 225 |
+
else:
|
| 226 |
+
failed_cuts.append(cut)
|
| 227 |
+
completed += 1
|
| 228 |
+
if on_progress:
|
| 229 |
+
pct = int((completed / max(total, 1)) * 80)
|
| 230 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 231 |
+
status = "OK" if success else "FALLBACK"
|
| 232 |
+
await on_progress(f"[{completed}/{total}] {cut_id} ({status})", pct)
|
| 233 |
+
|
| 234 |
+
# Generate videos (limited by semaphore)
|
| 235 |
+
await asyncio.gather(*[gen_one(cut) for cut in all_cuts])
|
| 236 |
+
|
| 237 |
+
# Fallback: Ken Burns for failed cuts
|
| 238 |
+
if failed_cuts:
|
| 239 |
+
if on_progress:
|
| 240 |
+
await on_progress(f"Ken Burns fallback for {len(failed_cuts)} clips...", 85)
|
| 241 |
+
logger.info("Falling back to Ken Burns for %d cuts", len(failed_cuts))
|
| 242 |
+
|
| 243 |
+
# Create a mini-storyboard with just the failed cuts for Ken Burns
|
| 244 |
+
# We need to set up the fade values and pass to the existing Ken Burns code
|
| 245 |
+
from app.services.ffmpeg import animate_shot
|
| 246 |
+
kb_completed = 0
|
| 247 |
+
|
| 248 |
+
for cut in failed_cuts:
|
| 249 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id")
|
| 250 |
+
output = str(clips_dir / f"{cut_id}.mp4")
|
| 251 |
+
duration = cut.get("actual_duration_sec", cut.get("duration_sec", 3.0))
|
| 252 |
+
duration = max(duration, 0.5)
|
| 253 |
+
|
| 254 |
+
# Map new camera notation back to Ken Burns presets
|
| 255 |
+
camera = cut.get("camera_movement", "static")
|
| 256 |
+
kb_camera = _map_camera_to_ken_burns(camera)
|
| 257 |
+
|
| 258 |
+
try:
|
| 259 |
+
await animate_shot(
|
| 260 |
+
image_path=cut["image_path"],
|
| 261 |
+
output_path=output,
|
| 262 |
+
duration_sec=duration,
|
| 263 |
+
camera_movement=kb_camera,
|
| 264 |
+
audio_path=cut.get("tts_path"),
|
| 265 |
+
fade_in=0.15,
|
| 266 |
+
fade_out=0.15,
|
| 267 |
+
)
|
| 268 |
+
cut["clip_path"] = output
|
| 269 |
+
cut["_used_fallback"] = True
|
| 270 |
+
except Exception as e:
|
| 271 |
+
logger.error("Ken Burns fallback also failed for %s: %r", cut_id, e, exc_info=True)
|
| 272 |
+
cut["clip_path"] = None
|
| 273 |
+
|
| 274 |
+
kb_completed += 1
|
| 275 |
+
|
| 276 |
+
if on_progress:
|
| 277 |
+
fallback_count = sum(1 for c in failed_cuts if c.get("clip_path"))
|
| 278 |
+
total_success = succeeded + fallback_count
|
| 279 |
+
await on_progress(
|
| 280 |
+
f"{total_success}/{total} clips ready ({succeeded} video, {fallback_count} Ken Burns)",
|
| 281 |
+
100,
|
| 282 |
+
)
|
| 283 |
+
|
| 284 |
+
return storyboard
|
| 285 |
+
|
| 286 |
+
|
| 287 |
+
def _map_camera_to_ken_burns(camera: str) -> str:
|
| 288 |
+
"""Map professional camera notation to Ken Burns preset names."""
|
| 289 |
+
mapping = {
|
| 290 |
+
"FIX": "static",
|
| 291 |
+
"PAN_LEFT": "pan_left",
|
| 292 |
+
"PAN_RIGHT": "pan_right",
|
| 293 |
+
"TILT_UP": "pan_up",
|
| 294 |
+
"TILT_DOWN": "pan_down",
|
| 295 |
+
"T.U.": "zoom_in_slow",
|
| 296 |
+
"T.B.": "zoom_out",
|
| 297 |
+
"DOLLY_LEFT": "pan_left",
|
| 298 |
+
"DOLLY_RIGHT": "pan_right",
|
| 299 |
+
"FOLLOW": "pan_right",
|
| 300 |
+
"CRANE_UP": "pan_up",
|
| 301 |
+
"CRANE_DOWN": "pan_down",
|
| 302 |
+
"SHAKE": "shake",
|
| 303 |
+
}
|
| 304 |
+
return mapping.get(camera, camera) # Pass through if already a Ken Burns name
|
app/pipeline/stage_7_assembly.py
ADDED
|
@@ -0,0 +1,278 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 7: Video Assembly — concat clips, scene-aware transitions, subtitles, BGM."""
|
| 2 |
+
import logging
|
| 3 |
+
import os
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
from app.services.ffmpeg import concat_clips, add_subtitles, mix_background_music, get_duration
|
| 6 |
+
|
| 7 |
+
logger = logging.getLogger(__name__)
|
| 8 |
+
|
| 9 |
+
# Map storyboard mood keywords to BGM filenames
|
| 10 |
+
BGM_MOOD_MAP = {
|
| 11 |
+
"tense": "tense",
|
| 12 |
+
"mysterious": "tense",
|
| 13 |
+
"action": "epic",
|
| 14 |
+
"dramatic": "epic",
|
| 15 |
+
"epic": "epic",
|
| 16 |
+
"calm": "calm",
|
| 17 |
+
"romantic": "calm",
|
| 18 |
+
"comedic": "calm",
|
| 19 |
+
"melancholic": "sad",
|
| 20 |
+
"sad": "sad",
|
| 21 |
+
}
|
| 22 |
+
BGM_DIR = Path("data/bgm")
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
# Overlap durations for xfade transitions (seconds eaten from timeline)
|
| 26 |
+
TRANSITION_OVERLAP = {
|
| 27 |
+
"cut": 0.0,
|
| 28 |
+
"fade": 1.0,
|
| 29 |
+
"fade_black": 1.0,
|
| 30 |
+
"crossfade": 0.5,
|
| 31 |
+
}
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def _generate_srt(
|
| 35 |
+
storyboard: dict,
|
| 36 |
+
clip_durations: list[float],
|
| 37 |
+
transitions: list[str],
|
| 38 |
+
) -> str:
|
| 39 |
+
"""Generate SRT subtitle file using actual clip durations from ffprobe.
|
| 40 |
+
|
| 41 |
+
Uses measured clip durations instead of storyboard estimates, and accounts
|
| 42 |
+
for transition overlaps (xfade) that compress the timeline.
|
| 43 |
+
"""
|
| 44 |
+
# Collect cuts that have valid clip_paths (same order as clip_durations)
|
| 45 |
+
cuts_with_clips = []
|
| 46 |
+
for scene in storyboard.get("scenes", []):
|
| 47 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 48 |
+
clip_path = cut.get("clip_path")
|
| 49 |
+
if clip_path and os.path.exists(clip_path):
|
| 50 |
+
cuts_with_clips.append(cut)
|
| 51 |
+
|
| 52 |
+
lines = []
|
| 53 |
+
srt_idx = 1
|
| 54 |
+
elapsed = 0.0
|
| 55 |
+
prev_duration = 0.0
|
| 56 |
+
|
| 57 |
+
for i, (cut, duration) in enumerate(zip(cuts_with_clips, clip_durations)):
|
| 58 |
+
# Advance elapsed time, accounting for transition overlap
|
| 59 |
+
if i > 0:
|
| 60 |
+
overlap = TRANSITION_OVERLAP.get(transitions[i], 0.0) if i < len(transitions) else 0.0
|
| 61 |
+
elapsed += prev_duration - overlap
|
| 62 |
+
|
| 63 |
+
dialogue = cut.get("dialogue")
|
| 64 |
+
if dialogue and dialogue.get("text"):
|
| 65 |
+
start = elapsed
|
| 66 |
+
end = elapsed + duration
|
| 67 |
+
lines.append(str(srt_idx))
|
| 68 |
+
lines.append(f"{_srt_time(start)} --> {_srt_time(end)}")
|
| 69 |
+
|
| 70 |
+
prefix = ""
|
| 71 |
+
if dialogue.get("is_inner_thought"):
|
| 72 |
+
prefix = "\u266a " # Musical note for inner thoughts
|
| 73 |
+
elif dialogue.get("is_narration"):
|
| 74 |
+
prefix = ""
|
| 75 |
+
elif dialogue.get("speaker") and dialogue["speaker"] != "Narrator":
|
| 76 |
+
prefix = f"{dialogue['speaker']}: "
|
| 77 |
+
|
| 78 |
+
text = dialogue["text"]
|
| 79 |
+
# Wrap long lines
|
| 80 |
+
if len(text) > 60:
|
| 81 |
+
mid = len(text) // 2
|
| 82 |
+
space_pos = text.rfind(" ", 0, mid + 10)
|
| 83 |
+
if space_pos > mid - 15:
|
| 84 |
+
text = text[:space_pos] + "\n" + text[space_pos + 1:]
|
| 85 |
+
|
| 86 |
+
lines.append(f"{prefix}{text}")
|
| 87 |
+
lines.append("")
|
| 88 |
+
srt_idx += 1
|
| 89 |
+
|
| 90 |
+
prev_duration = duration
|
| 91 |
+
|
| 92 |
+
return "\n".join(lines)
|
| 93 |
+
|
| 94 |
+
|
| 95 |
+
def _srt_time(seconds: float) -> str:
|
| 96 |
+
"""Convert seconds to SRT timestamp format."""
|
| 97 |
+
h = int(seconds // 3600)
|
| 98 |
+
m = int((seconds % 3600) // 60)
|
| 99 |
+
s = int(seconds % 60)
|
| 100 |
+
ms = int((seconds % 1) * 1000)
|
| 101 |
+
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
|
| 102 |
+
|
| 103 |
+
|
| 104 |
+
def _build_transition_list(storyboard: dict) -> list[str]:
|
| 105 |
+
"""Build a list of transitions for each clip boundary.
|
| 106 |
+
|
| 107 |
+
Scene-aware: use the scene's transition_in at scene boundaries,
|
| 108 |
+
'cut' within scenes for continuous feel.
|
| 109 |
+
"""
|
| 110 |
+
transitions = []
|
| 111 |
+
prev_scene_id = None
|
| 112 |
+
|
| 113 |
+
for scene in storyboard.get("scenes", []):
|
| 114 |
+
scene_id = scene.get("scene_id", "")
|
| 115 |
+
scene_transition = scene.get("transition_in", "cut")
|
| 116 |
+
cuts = scene.get("cuts", scene.get("shots", []))
|
| 117 |
+
|
| 118 |
+
for idx, cut in enumerate(cuts):
|
| 119 |
+
if not cut.get("clip_path") or not os.path.exists(cut.get("clip_path", "")):
|
| 120 |
+
continue
|
| 121 |
+
|
| 122 |
+
if idx == 0 and prev_scene_id is not None:
|
| 123 |
+
# Scene boundary — use the scene's transition_in
|
| 124 |
+
transitions.append(scene_transition)
|
| 125 |
+
else:
|
| 126 |
+
# Within scene — use cut's transition_out or default to "cut"
|
| 127 |
+
if idx > 0:
|
| 128 |
+
prev_cut = cuts[idx - 1]
|
| 129 |
+
trans = prev_cut.get("transition_out", "cut")
|
| 130 |
+
transitions.append(trans)
|
| 131 |
+
else:
|
| 132 |
+
transitions.append("cut") # First clip overall
|
| 133 |
+
|
| 134 |
+
prev_scene_id = scene_id
|
| 135 |
+
|
| 136 |
+
return transitions
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
async def run_assembly(
|
| 140 |
+
storyboard: dict,
|
| 141 |
+
project_dir: str,
|
| 142 |
+
episode_number: int = 1,
|
| 143 |
+
on_progress: callable = None,
|
| 144 |
+
output_dir: str | None = None,
|
| 145 |
+
) -> dict:
|
| 146 |
+
"""Assemble all animated clips into a final episode video.
|
| 147 |
+
|
| 148 |
+
Args:
|
| 149 |
+
project_dir: Directory containing clips (episode_dir in per-episode mode).
|
| 150 |
+
output_dir: If set, final episode goes to output_dir/episodes/ instead.
|
| 151 |
+
|
| 152 |
+
Returns {video_path, duration_seconds, subtitle_path}.
|
| 153 |
+
"""
|
| 154 |
+
if on_progress:
|
| 155 |
+
await on_progress("Assembling episode...", 0)
|
| 156 |
+
|
| 157 |
+
episode_dir = Path(output_dir or project_dir) / "episodes"
|
| 158 |
+
episode_dir.mkdir(parents=True, exist_ok=True)
|
| 159 |
+
|
| 160 |
+
# Collect clip paths in order
|
| 161 |
+
clip_paths = []
|
| 162 |
+
for scene in storyboard.get("scenes", []):
|
| 163 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 164 |
+
clip_path = cut.get("clip_path")
|
| 165 |
+
if clip_path and os.path.exists(clip_path):
|
| 166 |
+
clip_paths.append(clip_path)
|
| 167 |
+
|
| 168 |
+
if not clip_paths:
|
| 169 |
+
raise ValueError("No animated clips to assemble")
|
| 170 |
+
|
| 171 |
+
if on_progress:
|
| 172 |
+
await on_progress(f"Concatenating {len(clip_paths)} clips...", 20)
|
| 173 |
+
|
| 174 |
+
# Build scene-aware transitions
|
| 175 |
+
transitions = _build_transition_list(storyboard)
|
| 176 |
+
|
| 177 |
+
# Measure actual clip durations via ffprobe (ground truth)
|
| 178 |
+
clip_durations = []
|
| 179 |
+
for path in clip_paths:
|
| 180 |
+
dur = await get_duration(path)
|
| 181 |
+
clip_durations.append(dur if dur > 0 else 2.0)
|
| 182 |
+
logger.info("Clip durations (ffprobe): %s", [round(d, 2) for d in clip_durations])
|
| 183 |
+
|
| 184 |
+
# Generate subtitles using actual durations + transition overlaps
|
| 185 |
+
srt_content = _generate_srt(storyboard, clip_durations, transitions)
|
| 186 |
+
srt_path = str(episode_dir / f"episode_{episode_number}.srt")
|
| 187 |
+
with open(srt_path, "w", encoding="utf-8") as f:
|
| 188 |
+
f.write(srt_content)
|
| 189 |
+
|
| 190 |
+
# Concat all clips with scene-aware transitions
|
| 191 |
+
raw_output = str(episode_dir / f"episode_{episode_number}_raw.mp4")
|
| 192 |
+
await concat_clips(clip_paths, raw_output, transitions=transitions)
|
| 193 |
+
|
| 194 |
+
if on_progress:
|
| 195 |
+
await on_progress("Burning subtitles...", 50)
|
| 196 |
+
|
| 197 |
+
# Burn subtitles into video
|
| 198 |
+
subbed_output = str(episode_dir / f"episode_{episode_number}_subbed.mp4")
|
| 199 |
+
try:
|
| 200 |
+
await add_subtitles(raw_output, srt_path, subbed_output)
|
| 201 |
+
# Clean up raw
|
| 202 |
+
if os.path.exists(raw_output):
|
| 203 |
+
os.unlink(raw_output)
|
| 204 |
+
except Exception as e:
|
| 205 |
+
logger.warning("Subtitle burning failed, using raw video: %s", e)
|
| 206 |
+
subbed_output = raw_output # fallback — no crash
|
| 207 |
+
|
| 208 |
+
if on_progress:
|
| 209 |
+
await on_progress("Adding background music...", 70)
|
| 210 |
+
|
| 211 |
+
# Select BGM based on dominant mood
|
| 212 |
+
final_output = str(episode_dir / f"episode_{episode_number}.mp4")
|
| 213 |
+
bgm_path = _select_bgm(storyboard)
|
| 214 |
+
|
| 215 |
+
if bgm_path:
|
| 216 |
+
try:
|
| 217 |
+
await mix_background_music(subbed_output, bgm_path, final_output)
|
| 218 |
+
# Clean up subbed
|
| 219 |
+
if os.path.exists(subbed_output) and subbed_output != final_output:
|
| 220 |
+
os.unlink(subbed_output)
|
| 221 |
+
except Exception as e:
|
| 222 |
+
logger.warning("BGM mixing failed, using video without BGM: %s", e)
|
| 223 |
+
if subbed_output != final_output:
|
| 224 |
+
os.rename(subbed_output, final_output)
|
| 225 |
+
else:
|
| 226 |
+
if subbed_output != final_output:
|
| 227 |
+
os.rename(subbed_output, final_output)
|
| 228 |
+
|
| 229 |
+
# Get final duration
|
| 230 |
+
duration = await get_duration(final_output)
|
| 231 |
+
|
| 232 |
+
if on_progress:
|
| 233 |
+
mins = int(duration // 60)
|
| 234 |
+
secs = int(duration % 60)
|
| 235 |
+
await on_progress(f"Episode assembled \u2014 {mins}:{secs:02d}", 100)
|
| 236 |
+
|
| 237 |
+
return {
|
| 238 |
+
"video_path": final_output,
|
| 239 |
+
"subtitle_path": srt_path,
|
| 240 |
+
"duration_seconds": duration,
|
| 241 |
+
}
|
| 242 |
+
|
| 243 |
+
|
| 244 |
+
def _select_bgm(storyboard: dict) -> str | None:
|
| 245 |
+
"""Select a BGM file based on the dominant mood in the storyboard."""
|
| 246 |
+
if not BGM_DIR.exists():
|
| 247 |
+
return None
|
| 248 |
+
|
| 249 |
+
# Count mood occurrences
|
| 250 |
+
mood_counts: dict[str, int] = {}
|
| 251 |
+
for scene in storyboard.get("scenes", []):
|
| 252 |
+
mood = scene.get("mood", "").lower()
|
| 253 |
+
bgm_key = BGM_MOOD_MAP.get(mood)
|
| 254 |
+
if bgm_key:
|
| 255 |
+
mood_counts[bgm_key] = mood_counts.get(bgm_key, 0) + 1
|
| 256 |
+
|
| 257 |
+
if not mood_counts:
|
| 258 |
+
# Default to calm
|
| 259 |
+
mood_counts = {"calm": 1}
|
| 260 |
+
|
| 261 |
+
# Pick the most common mood
|
| 262 |
+
dominant = max(mood_counts, key=mood_counts.get)
|
| 263 |
+
|
| 264 |
+
# Look for matching BGM file
|
| 265 |
+
for ext in (".mp3", ".wav", ".ogg", ".m4a"):
|
| 266 |
+
candidate = BGM_DIR / f"{dominant}{ext}"
|
| 267 |
+
if candidate.exists():
|
| 268 |
+
return str(candidate)
|
| 269 |
+
|
| 270 |
+
# Fallback: try any available BGM file
|
| 271 |
+
try:
|
| 272 |
+
for f in BGM_DIR.iterdir():
|
| 273 |
+
if f.suffix in (".mp3", ".wav", ".ogg", ".m4a"):
|
| 274 |
+
return str(f)
|
| 275 |
+
except Exception:
|
| 276 |
+
pass
|
| 277 |
+
|
| 278 |
+
return None
|
app/pipeline/stage_8_critic.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Stage 8: Critic Agent — AI review of generated storyboard/images."""
|
| 2 |
+
import logging
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
from app.services.llm import call_gemini_json, call_openrouter_vision_json
|
| 5 |
+
|
| 6 |
+
logger = logging.getLogger(__name__)
|
| 7 |
+
|
| 8 |
+
CRITIC_PROMPT = """You are an anime quality assurance critic. Review this storyboard for an anime episode and score it.
|
| 9 |
+
|
| 10 |
+
STORYBOARD METADATA:
|
| 11 |
+
- Total shots: {total_shots}
|
| 12 |
+
- Total duration: {total_duration}s
|
| 13 |
+
- Scenes: {scene_count}
|
| 14 |
+
|
| 15 |
+
SCENE BREAKDOWN:
|
| 16 |
+
{scene_summary}
|
| 17 |
+
|
| 18 |
+
Score each dimension from 0.0 to 10.0 and provide specific feedback.
|
| 19 |
+
|
| 20 |
+
Return JSON:
|
| 21 |
+
{{
|
| 22 |
+
"visual_consistency_score": float, // Are character descriptions consistent across shots?
|
| 23 |
+
"pacing_score": float, // Is the pacing good? Not too fast or slow?
|
| 24 |
+
"audio_sync_score": float, // Do dialogue durations match the content?
|
| 25 |
+
"overall_score": float, // Overall quality
|
| 26 |
+
"feedback": {{
|
| 27 |
+
"strengths": ["list of good aspects"],
|
| 28 |
+
"issues": [
|
| 29 |
+
{{"severity": "low/medium/high", "description": "what's wrong", "shot_id": "affected shot or null", "suggestion": "how to fix"}}
|
| 30 |
+
]
|
| 31 |
+
}}
|
| 32 |
+
}}"""
|
| 33 |
+
|
| 34 |
+
IMAGE_CRITIC_PROMPT = """You are an anime visual quality critic. You are reviewing {image_count} frames from an AI-generated anime episode.
|
| 35 |
+
|
| 36 |
+
Evaluate these images and return a JSON score card:
|
| 37 |
+
|
| 38 |
+
{{
|
| 39 |
+
"art_style_consistency": {{
|
| 40 |
+
"score": float (0-10),
|
| 41 |
+
"notes": "Are all images in a consistent anime art style?"
|
| 42 |
+
}},
|
| 43 |
+
"character_consistency": {{
|
| 44 |
+
"score": float (0-10),
|
| 45 |
+
"notes": "Do characters look the same across different shots?"
|
| 46 |
+
}},
|
| 47 |
+
"quality": {{
|
| 48 |
+
"score": float (0-10),
|
| 49 |
+
"notes": "Are there artifacts, weird anatomy, broken faces, extra fingers?"
|
| 50 |
+
}},
|
| 51 |
+
"anime_adherence": {{
|
| 52 |
+
"score": float (0-10),
|
| 53 |
+
"notes": "Does it look like anime vs photorealistic or other styles?"
|
| 54 |
+
}},
|
| 55 |
+
"overall_visual_score": float (0-10),
|
| 56 |
+
"visual_issues": [
|
| 57 |
+
{{"description": "what's wrong", "severity": "low/medium/high"}}
|
| 58 |
+
]
|
| 59 |
+
}}"""
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def _build_scene_summary(storyboard: dict) -> str:
|
| 63 |
+
"""Build a text summary of the storyboard for the critic."""
|
| 64 |
+
lines = []
|
| 65 |
+
for scene in storyboard.get("scenes", []):
|
| 66 |
+
scene_id = scene.get("scene_id", "?")
|
| 67 |
+
lines.append(f"\nScene {scene_id}: {scene.get('setting', 'unknown')} ({scene.get('mood', 'neutral')})")
|
| 68 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 69 |
+
cut_id = cut.get("cut_id") or cut.get("shot_id", "?")
|
| 70 |
+
st = cut.get("shot_type", "?")
|
| 71 |
+
dur = cut.get("duration_sec", 0)
|
| 72 |
+
chars = ", ".join(cut.get("characters_present", []))
|
| 73 |
+
dialogue = cut.get("dialogue") or {}
|
| 74 |
+
text_preview = (dialogue.get("text", "")[:50] + "...") if dialogue.get("text") else "no dialogue"
|
| 75 |
+
has_image = "yes" if cut.get("image_path") else "NO"
|
| 76 |
+
beat = cut.get("emotional_beat", "")
|
| 77 |
+
beat_str = f" [{beat}]" if beat else ""
|
| 78 |
+
lines.append(f" {cut_id}: {st}, {dur:.1f}s, chars=[{chars}], img={has_image}{beat_str}, \"{text_preview}\"")
|
| 79 |
+
return "\n".join(lines)
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def _collect_image_paths(storyboard: dict, project_dir: str) -> list[str]:
|
| 83 |
+
"""Collect all existing image paths from storyboard cuts."""
|
| 84 |
+
paths = []
|
| 85 |
+
for scene in storyboard.get("scenes", []):
|
| 86 |
+
for cut in scene.get("cuts", scene.get("shots", [])):
|
| 87 |
+
img_path = cut.get("image_path")
|
| 88 |
+
if img_path:
|
| 89 |
+
p = Path(img_path)
|
| 90 |
+
if p.exists():
|
| 91 |
+
paths.append(str(p.resolve()))
|
| 92 |
+
else:
|
| 93 |
+
# Try relative to project_dir
|
| 94 |
+
full = Path(project_dir) / img_path
|
| 95 |
+
if full.exists():
|
| 96 |
+
paths.append(str(full.resolve()))
|
| 97 |
+
return paths
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def _sample_images(paths: list[str], max_count: int = 6) -> list[str]:
|
| 101 |
+
"""Evenly sample up to max_count images from the list."""
|
| 102 |
+
if len(paths) <= max_count:
|
| 103 |
+
return paths
|
| 104 |
+
step = len(paths) / max_count
|
| 105 |
+
return [paths[int(i * step)] for i in range(max_count)]
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
async def _critique_images(storyboard: dict, project_dir: str) -> dict | None:
|
| 109 |
+
"""Run visual critique on generated images via OpenRouter vision."""
|
| 110 |
+
all_images = _collect_image_paths(storyboard, project_dir)
|
| 111 |
+
if not all_images:
|
| 112 |
+
logger.info("No images found for visual critique")
|
| 113 |
+
return None
|
| 114 |
+
|
| 115 |
+
sampled = _sample_images(all_images)
|
| 116 |
+
logger.info("Visual critique: %d/%d images sampled", len(sampled), len(all_images))
|
| 117 |
+
|
| 118 |
+
prompt = IMAGE_CRITIC_PROMPT.format(image_count=len(sampled))
|
| 119 |
+
result = await call_openrouter_vision_json(
|
| 120 |
+
prompt,
|
| 121 |
+
sampled,
|
| 122 |
+
system_prompt="You are a professional anime art director reviewing AI-generated frames.",
|
| 123 |
+
)
|
| 124 |
+
|
| 125 |
+
if not isinstance(result, dict):
|
| 126 |
+
return None
|
| 127 |
+
|
| 128 |
+
return result
|
| 129 |
+
|
| 130 |
+
|
| 131 |
+
async def run_critic(
|
| 132 |
+
storyboard: dict,
|
| 133 |
+
project_dir: str,
|
| 134 |
+
on_progress: callable = None,
|
| 135 |
+
) -> dict:
|
| 136 |
+
"""Run AI critic on the storyboard. Returns critic scores and feedback."""
|
| 137 |
+
if on_progress:
|
| 138 |
+
await on_progress("AI critic reviewing the episode...", 0)
|
| 139 |
+
|
| 140 |
+
metadata = storyboard.get("metadata", {})
|
| 141 |
+
scene_summary = _build_scene_summary(storyboard)
|
| 142 |
+
|
| 143 |
+
prompt = CRITIC_PROMPT.format(
|
| 144 |
+
total_shots=metadata.get("total_shots", 0),
|
| 145 |
+
total_duration=metadata.get("total_duration_estimate", 0),
|
| 146 |
+
scene_count=metadata.get("scene_count", 0),
|
| 147 |
+
scene_summary=scene_summary[:5000],
|
| 148 |
+
)
|
| 149 |
+
|
| 150 |
+
if on_progress:
|
| 151 |
+
await on_progress("Checking storyboard quality...", 20)
|
| 152 |
+
|
| 153 |
+
result = await call_gemini_json(prompt)
|
| 154 |
+
|
| 155 |
+
# Type-check the result
|
| 156 |
+
if not isinstance(result, dict):
|
| 157 |
+
result = {"overall_score": 5.0, "feedback": {"strengths": [], "issues": []}}
|
| 158 |
+
|
| 159 |
+
score = result.get("overall_score", 0)
|
| 160 |
+
if isinstance(score, (int, float)):
|
| 161 |
+
score = max(0, min(10, float(score)))
|
| 162 |
+
else:
|
| 163 |
+
score = 5.0
|
| 164 |
+
result["overall_score"] = score
|
| 165 |
+
|
| 166 |
+
# Image critique via OpenRouter vision (non-blocking on failure)
|
| 167 |
+
if on_progress:
|
| 168 |
+
await on_progress("Analyzing generated images...", 50)
|
| 169 |
+
|
| 170 |
+
try:
|
| 171 |
+
image_critique = await _critique_images(storyboard, project_dir)
|
| 172 |
+
if image_critique:
|
| 173 |
+
result["image_critique"] = image_critique
|
| 174 |
+
vis_score = image_critique.get("overall_visual_score")
|
| 175 |
+
if isinstance(vis_score, (int, float)):
|
| 176 |
+
vis_score = max(0, min(10, float(vis_score)))
|
| 177 |
+
result["visual_quality_score"] = vis_score
|
| 178 |
+
# Blend into overall: 60% storyboard, 40% visual
|
| 179 |
+
result["overall_score"] = round(score * 0.6 + vis_score * 0.4, 1)
|
| 180 |
+
score = result["overall_score"]
|
| 181 |
+
logger.info("Image critique complete: visual_score=%.1f", result.get("visual_quality_score", 0))
|
| 182 |
+
except Exception as e:
|
| 183 |
+
logger.warning("Image critique failed (non-fatal): %s", e)
|
| 184 |
+
result["image_critique_error"] = str(e)
|
| 185 |
+
|
| 186 |
+
if on_progress:
|
| 187 |
+
await on_progress(f"Quality score: {score}/10", 100)
|
| 188 |
+
|
| 189 |
+
return result
|
app/schemas/__init__.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from app.schemas.auth import (
|
| 2 |
+
UserCreate, UserLogin, UserResponse, TokenResponse, RefreshRequest,
|
| 3 |
+
)
|
| 4 |
+
from app.schemas.project import (
|
| 5 |
+
ProjectCreate, ProjectUpdate, ProjectResponse, ProjectListResponse,
|
| 6 |
+
)
|
| 7 |
+
from app.schemas.chapter import ChapterCreate, ChapterResponse
|
| 8 |
+
from app.schemas.character import CharacterCreate, CharacterUpdate, CharacterResponse
|
| 9 |
+
from app.schemas.episode import EpisodeCreate, EpisodeResponse
|
| 10 |
+
from app.schemas.generation import GenerationJobResponse, GenerationStart
|
| 11 |
+
|
| 12 |
+
__all__ = [
|
| 13 |
+
"UserCreate", "UserLogin", "UserResponse", "TokenResponse", "RefreshRequest",
|
| 14 |
+
"ProjectCreate", "ProjectUpdate", "ProjectResponse", "ProjectListResponse",
|
| 15 |
+
"ChapterCreate", "ChapterResponse",
|
| 16 |
+
"CharacterCreate", "CharacterUpdate", "CharacterResponse",
|
| 17 |
+
"EpisodeCreate", "EpisodeResponse",
|
| 18 |
+
"GenerationJobResponse", "GenerationStart",
|
| 19 |
+
]
|
app/schemas/auth.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel, EmailStr
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class UserCreate(BaseModel):
|
| 6 |
+
email: EmailStr
|
| 7 |
+
username: str
|
| 8 |
+
password: str
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class UserLogin(BaseModel):
|
| 12 |
+
email: EmailStr
|
| 13 |
+
password: str
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
class UserResponse(BaseModel):
|
| 17 |
+
id: int
|
| 18 |
+
email: str
|
| 19 |
+
username: str
|
| 20 |
+
role: str
|
| 21 |
+
created_at: datetime
|
| 22 |
+
|
| 23 |
+
model_config = {"from_attributes": True}
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class TokenResponse(BaseModel):
|
| 27 |
+
access_token: str
|
| 28 |
+
token_type: str = "bearer"
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
class RefreshRequest(BaseModel):
|
| 32 |
+
refresh_token: str
|
app/schemas/chapter.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class ChapterCreate(BaseModel):
|
| 6 |
+
chapter_number: int
|
| 7 |
+
title: str | None = None
|
| 8 |
+
source_text: str
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class ChapterResponse(BaseModel):
|
| 12 |
+
id: int
|
| 13 |
+
project_id: int
|
| 14 |
+
chapter_number: int
|
| 15 |
+
title: str | None
|
| 16 |
+
word_count: int
|
| 17 |
+
status: str
|
| 18 |
+
created_at: datetime
|
| 19 |
+
|
| 20 |
+
model_config = {"from_attributes": True}
|
app/schemas/character.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class CharacterCreate(BaseModel):
|
| 6 |
+
name: str
|
| 7 |
+
aliases_json: list[str] | None = None
|
| 8 |
+
role: str = "supporting"
|
| 9 |
+
gender: str | None = None
|
| 10 |
+
age_category: str | None = None
|
| 11 |
+
visual_prompt: str | None = None
|
| 12 |
+
voice_config_json: dict | None = None
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class CharacterUpdate(BaseModel):
|
| 16 |
+
name: str | None = None
|
| 17 |
+
aliases_json: list[str] | None = None
|
| 18 |
+
role: str | None = None
|
| 19 |
+
gender: str | None = None
|
| 20 |
+
age_category: str | None = None
|
| 21 |
+
visual_prompt: str | None = None
|
| 22 |
+
voice_config_json: dict | None = None
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class CharacterResponse(BaseModel):
|
| 26 |
+
id: int
|
| 27 |
+
project_id: int
|
| 28 |
+
name: str
|
| 29 |
+
aliases_json: list | None
|
| 30 |
+
role: str
|
| 31 |
+
gender: str | None
|
| 32 |
+
age_category: str | None
|
| 33 |
+
visual_prompt: str | None
|
| 34 |
+
reference_image_path: str | None
|
| 35 |
+
voice_config_json: dict | None
|
| 36 |
+
created_at: datetime
|
| 37 |
+
|
| 38 |
+
model_config = {"from_attributes": True}
|
app/schemas/episode.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class EpisodeCreate(BaseModel):
|
| 6 |
+
episode_number: int
|
| 7 |
+
title: str | None = None
|
| 8 |
+
chapter_ids_json: list[int] | None = None
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class EpisodeResponse(BaseModel):
|
| 12 |
+
id: int
|
| 13 |
+
project_id: int
|
| 14 |
+
episode_number: int
|
| 15 |
+
title: str | None
|
| 16 |
+
chapter_ids_json: list | None
|
| 17 |
+
duration_seconds: float | None
|
| 18 |
+
video_path: str | None
|
| 19 |
+
status: str
|
| 20 |
+
created_at: datetime
|
| 21 |
+
|
| 22 |
+
model_config = {"from_attributes": True}
|
app/schemas/generation.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class GenerationStart(BaseModel):
|
| 6 |
+
episode_id: int | None = None
|
| 7 |
+
chapter_ids: list[int] | None = None
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
class GenerationJobResponse(BaseModel):
|
| 11 |
+
id: int
|
| 12 |
+
project_id: int
|
| 13 |
+
episode_id: int | None
|
| 14 |
+
status: str
|
| 15 |
+
current_stage: str | None
|
| 16 |
+
progress_pct: float
|
| 17 |
+
progress_detail: str | None
|
| 18 |
+
error_message: str | None
|
| 19 |
+
started_at: datetime | None
|
| 20 |
+
completed_at: datetime | None
|
| 21 |
+
created_at: datetime
|
| 22 |
+
|
| 23 |
+
model_config = {"from_attributes": True}
|
app/schemas/project.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from pydantic import BaseModel
|
| 2 |
+
from datetime import datetime
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
class ProjectCreate(BaseModel):
|
| 6 |
+
title: str
|
| 7 |
+
novel_url: str | None = None
|
| 8 |
+
cover_image_url: str | None = None
|
| 9 |
+
synopsis: str | None = None
|
| 10 |
+
settings_json: dict | None = None
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
class ProjectUpdate(BaseModel):
|
| 14 |
+
title: str | None = None
|
| 15 |
+
novel_url: str | None = None
|
| 16 |
+
cover_image_url: str | None = None
|
| 17 |
+
synopsis: str | None = None
|
| 18 |
+
settings_json: dict | None = None
|
| 19 |
+
status: str | None = None
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class ProjectResponse(BaseModel):
|
| 23 |
+
id: int
|
| 24 |
+
user_id: int
|
| 25 |
+
title: str
|
| 26 |
+
novel_url: str | None
|
| 27 |
+
cover_image_url: str | None
|
| 28 |
+
synopsis: str | None
|
| 29 |
+
settings_json: dict | None
|
| 30 |
+
status: str
|
| 31 |
+
created_at: datetime
|
| 32 |
+
updated_at: datetime | None
|
| 33 |
+
|
| 34 |
+
model_config = {"from_attributes": True}
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
class ProjectListResponse(BaseModel):
|
| 38 |
+
projects: list[ProjectResponse]
|
| 39 |
+
total: int
|
app/services/__init__.py
ADDED
|
File without changes
|
app/services/auth.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from datetime import datetime, timedelta, timezone
|
| 2 |
+
import bcrypt
|
| 3 |
+
from jose import jwt, JWTError
|
| 4 |
+
from fastapi import Depends, HTTPException, status
|
| 5 |
+
from fastapi.security import OAuth2PasswordBearer
|
| 6 |
+
from sqlalchemy import select
|
| 7 |
+
from sqlalchemy.ext.asyncio import AsyncSession
|
| 8 |
+
from app.config import get_settings
|
| 9 |
+
from app.database import get_db
|
| 10 |
+
from app.models.user import User
|
| 11 |
+
|
| 12 |
+
settings = get_settings()
|
| 13 |
+
oauth2_scheme = OAuth2PasswordBearer(tokenUrl="/api/auth/login")
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def hash_password(password: str) -> str:
|
| 17 |
+
return bcrypt.hashpw(password.encode(), bcrypt.gensalt()).decode()
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def verify_password(plain: str, hashed: str) -> bool:
|
| 21 |
+
return bcrypt.checkpw(plain.encode(), hashed.encode())
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def create_token(data: dict, expires_delta: timedelta) -> str:
|
| 25 |
+
to_encode = data.copy()
|
| 26 |
+
to_encode["exp"] = datetime.now(timezone.utc) + expires_delta
|
| 27 |
+
return jwt.encode(to_encode, settings.jwt_secret_key, algorithm=settings.jwt_algorithm)
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def create_access_token(user_id: int) -> str:
|
| 31 |
+
return create_token(
|
| 32 |
+
{"sub": str(user_id), "type": "access"},
|
| 33 |
+
timedelta(minutes=settings.access_token_expire_minutes),
|
| 34 |
+
)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def create_refresh_token(user_id: int) -> str:
|
| 38 |
+
return create_token(
|
| 39 |
+
{"sub": str(user_id), "type": "refresh"},
|
| 40 |
+
timedelta(days=settings.refresh_token_expire_days),
|
| 41 |
+
)
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def decode_token(token: str) -> dict:
|
| 45 |
+
try:
|
| 46 |
+
return jwt.decode(token, settings.jwt_secret_key, algorithms=[settings.jwt_algorithm])
|
| 47 |
+
except JWTError:
|
| 48 |
+
raise HTTPException(
|
| 49 |
+
status_code=status.HTTP_401_UNAUTHORIZED,
|
| 50 |
+
detail="Invalid or expired token",
|
| 51 |
+
)
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
async def get_current_user(
|
| 55 |
+
token: str = Depends(oauth2_scheme),
|
| 56 |
+
db: AsyncSession = Depends(get_db),
|
| 57 |
+
) -> User:
|
| 58 |
+
payload = decode_token(token)
|
| 59 |
+
if payload.get("type") != "access":
|
| 60 |
+
raise HTTPException(status_code=401, detail="Invalid token type")
|
| 61 |
+
user_id = int(payload["sub"])
|
| 62 |
+
result = await db.execute(select(User).where(User.id == user_id))
|
| 63 |
+
user = result.scalar_one_or_none()
|
| 64 |
+
if not user:
|
| 65 |
+
raise HTTPException(status_code=401, detail="User not found")
|
| 66 |
+
return user
|
app/services/fandom.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Fandom/MediaWiki API client for character enrichment."""
|
| 2 |
+
import re
|
| 3 |
+
import httpx
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
async def search_fandom_wiki(
|
| 7 |
+
novel_title: str,
|
| 8 |
+
character_name: str,
|
| 9 |
+
) -> dict | None:
|
| 10 |
+
"""Search Fandom wikis for character appearance data.
|
| 11 |
+
Returns {description, image_url} or None."""
|
| 12 |
+
|
| 13 |
+
# Generate wiki subdomain from novel title
|
| 14 |
+
slug = re.sub(r'[^a-zA-Z0-9]', '', novel_title.lower().replace(' ', ''))
|
| 15 |
+
wiki_url = f"https://{slug}.fandom.com"
|
| 16 |
+
|
| 17 |
+
try:
|
| 18 |
+
async with httpx.AsyncClient(timeout=15.0) as client:
|
| 19 |
+
# Search for character page
|
| 20 |
+
resp = await client.get(
|
| 21 |
+
f"{wiki_url}/api.php",
|
| 22 |
+
params={
|
| 23 |
+
"action": "query",
|
| 24 |
+
"list": "search",
|
| 25 |
+
"srsearch": character_name,
|
| 26 |
+
"srnamespace": "0",
|
| 27 |
+
"srlimit": "3",
|
| 28 |
+
"format": "json",
|
| 29 |
+
},
|
| 30 |
+
)
|
| 31 |
+
if resp.status_code != 200:
|
| 32 |
+
return None
|
| 33 |
+
|
| 34 |
+
data = resp.json()
|
| 35 |
+
results = data.get("query", {}).get("search", [])
|
| 36 |
+
if not results:
|
| 37 |
+
return None
|
| 38 |
+
|
| 39 |
+
page_title = results[0]["title"]
|
| 40 |
+
|
| 41 |
+
# Get page content (parse for appearance section)
|
| 42 |
+
resp2 = await client.get(
|
| 43 |
+
f"{wiki_url}/api.php",
|
| 44 |
+
params={
|
| 45 |
+
"action": "parse",
|
| 46 |
+
"page": page_title,
|
| 47 |
+
"prop": "wikitext|images",
|
| 48 |
+
"format": "json",
|
| 49 |
+
},
|
| 50 |
+
)
|
| 51 |
+
if resp2.status_code != 200:
|
| 52 |
+
return None
|
| 53 |
+
|
| 54 |
+
parse_data = resp2.json().get("parse", {})
|
| 55 |
+
wikitext = parse_data.get("wikitext", {}).get("*", "")
|
| 56 |
+
images = parse_data.get("images", [])
|
| 57 |
+
|
| 58 |
+
# Extract appearance section
|
| 59 |
+
appearance = _extract_section(wikitext, ["Appearance", "Physical Description", "Description"])
|
| 60 |
+
|
| 61 |
+
# Get first image URL
|
| 62 |
+
image_url = None
|
| 63 |
+
if images:
|
| 64 |
+
img_name = images[0]
|
| 65 |
+
img_resp = await client.get(
|
| 66 |
+
f"{wiki_url}/api.php",
|
| 67 |
+
params={
|
| 68 |
+
"action": "query",
|
| 69 |
+
"titles": f"File:{img_name}",
|
| 70 |
+
"prop": "imageinfo",
|
| 71 |
+
"iiprop": "url",
|
| 72 |
+
"format": "json",
|
| 73 |
+
},
|
| 74 |
+
)
|
| 75 |
+
if img_resp.status_code == 200:
|
| 76 |
+
pages = img_resp.json().get("query", {}).get("pages", {})
|
| 77 |
+
for page in pages.values():
|
| 78 |
+
ii = page.get("imageinfo", [])
|
| 79 |
+
if ii:
|
| 80 |
+
image_url = ii[0].get("url")
|
| 81 |
+
|
| 82 |
+
return {
|
| 83 |
+
"description": _clean_wikitext(appearance) if appearance else None,
|
| 84 |
+
"image_url": image_url,
|
| 85 |
+
}
|
| 86 |
+
except Exception:
|
| 87 |
+
return None
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def _extract_section(wikitext: str, headings: list[str]) -> str | None:
|
| 91 |
+
"""Extract content under a wiki section heading."""
|
| 92 |
+
for heading in headings:
|
| 93 |
+
pattern = rf'==+\s*{re.escape(heading)}\s*==+(.*?)(?===|\Z)'
|
| 94 |
+
match = re.search(pattern, wikitext, re.DOTALL | re.IGNORECASE)
|
| 95 |
+
if match:
|
| 96 |
+
return match.group(1).strip()
|
| 97 |
+
return None
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def _clean_wikitext(text: str) -> str:
|
| 101 |
+
"""Remove wiki markup, leaving plain text."""
|
| 102 |
+
text = re.sub(r'\[\[(?:[^|\]]*\|)?([^\]]*)\]\]', r'\1', text) # [[link|text]] → text
|
| 103 |
+
text = re.sub(r'\{\{[^}]*\}\}', '', text) # {{templates}}
|
| 104 |
+
text = re.sub(r"'''?", '', text) # bold/italic
|
| 105 |
+
text = re.sub(r'<[^>]+>', '', text) # HTML tags
|
| 106 |
+
text = re.sub(r'\n{3,}', '\n\n', text)
|
| 107 |
+
return text.strip()
|