Initial commit

This commit is contained in:
justin committed 2026-04-13 18:03:33 -07:00
commit 73d16c9b69
26 files changed
+1660

No files matched your search

+112
View File
@@ -0,0 +1,112 @@
<p align="center">
<img src="screenshot.jpg" alt="epub-to-manga screenshot" />
</p>
# epub-to-manga
Convert EPUB novels into manga-style EPUBs using a local LLM for scene parsing and Stable Diffusion for image generation.
## How it works
1. Reads your EPUB and splits the text into scenes
2. Uses an Ollama LLM to parse each scene — extracting characters, dialogue, mood, and setting
3. Generates a manga-style illustration for each page via Stable Diffusion
4. Adds speech bubbles with character dialogue
5. Packages everything into a new EPUB
## Requirements
- Python 3.10+
- [Ollama](https://ollama.com) running locally
- One of the following Stable Diffusion backends:
- [AUTOMATIC1111 stable-diffusion-webui](https://github.com/AUTOMATIC1111/stable-diffusion-webui)
- [ComfyUI](https://github.com/comfyanonymous/ComfyUI) *(recommended for Apple Silicon)*
- [InvokeAI](https://github.com/invoke-ai/InvokeAI)
## Installation
```bash
pip install -r requirements.txt
```
Pull an Ollama model:
```bash
ollama pull llama3
```
## Usage
```bash
python3 main.py book.epub
python3 main.py book.epub output.epub
python3 main.py book.epub --layout tiny --model mistral
python3 main.py book.epub --sd-url http://192.168.1.10:7860 --steps 30 --size 512x768
python3 main.py book.epub --dry-run
```
If no Stable Diffusion backend is running, the tool will detect any local installations and offer to launch or install one for you.
## Options
| Flag | Default | Description |
|---|---|---|
| `--layout` | `normal` | `normal` = 3 scenes/page, `tiny` = 1 scene/page |
| `--model` | `llama3` | Ollama model for scene parsing |
| `--ollama-url` | `http://localhost:11434` | Ollama API URL |
| `--sd-url` | `http://127.0.0.1:7860` | Stable Diffusion API URL |
| `--sd-path` | — | Path to SD install; auto-starts if not running |
| `--steps` | `20` | Diffusion steps per image (higher = better quality) |
| `--style` | `lineart` | `lineart` = clean outlines, `manga` = screentone shading |
| `--size` | `768x1024` | Output image dimensions |
| `--workers` | `2` | Parallel image generation workers |
| `--hint` | — | Character description, e.g. `--hint "Rocky:alien who resembles a rock spider"` |
| `--jpeg-quality` | `85` | EPUB image quality (1–95, lower = smaller file) |
| `--dry-run [N]` | — | Preview first N prompts without generating images |
| `--verbose` / `-v` | — | Enable debug logging |
## Configuration
Default URLs and model can be changed in `config.py`:
```python
OLLAMA_URL = "http://localhost:11434/api/generate"
OLLAMA_MODEL = "llama3"
SD_API_URL = "http://127.0.0.1:7860/sdapi/v1/txt2img"
```
The tool saves its SD backend path to `~/.epub-to-manga` so you don't need `--sd-path` on subsequent runs.
## Resuming
Scene parsing and image generation both support resuming. Parsed scenes are cached to `manga-<bookname>.cache.json`, and generated page images are saved to `<output>_pages/`. Re-running the same command will pick up where it left off.
## Project structure
```
epub-to-manga/
├── main.py # Entry point
├── config.py # Default URLs and model
├── requirements.txt
├── core/
│ ├── epub_reader.py # EPUB text extraction
│ └── scene_splitter.py # Text → scenes
├── ai/
│ ├── llm_client.py # Ollama API client
│ ├── scene_parser.py # Scene → structured JSON
│ └── character_memory.py # Character tracking across scenes
├── manga/
│ ├── prompt_builder.py # SD prompt construction
│ ├── panel_layout.py # Scene grouping into pages
│ └── speech_bubbles.py # Dialogue overlay rendering
├── image/
│ ├── backend.py # Backend router (A1111 / ComfyUI)
│ ├── a1111_api.py # AUTOMATIC1111 API
│ ├── comfy_api.py # ComfyUI API + workflow builder
│ ├── sd_launcher.py # SD auto-start and install
│ └── device.py # Device detection
├── export/
│ └── epub_builder.py # Output EPUB assembly
└── utils/
└── naming.py # Output filename helpers
```
BIN
View File
Binary file not shown.
+37
View File
@@ -0,0 +1,37 @@
CHARACTER_DB = {}
def load(data):
global CHARACTER_DB
CHARACTER_DB = data if isinstance(data, dict) else {}
def dump():
return dict(CHARACTER_DB)
def update(name, description=""):
if name not in CHARACTER_DB:
CHARACTER_DB[name] = {"appearances": 0, "description": description}
else:
if description and not CHARACTER_DB[name].get("description"):
CHARACTER_DB[name]["description"] = description
CHARACTER_DB[name]["appearances"] += 1
def add_hint(name, description):
if name not in CHARACTER_DB:
CHARACTER_DB[name] = {"appearances": 0, "description": description}
else:
CHARACTER_DB[name]["description"] = description
def get_all():
return list(CHARACTER_DB.keys())
def build_consistency_tokens(characters):
parts = []
for c in characters:
if not c or not c.strip():
continue
desc = CHARACTER_DB.get(c, {}).get("description", "")
if desc:
parts.append(f"{c} ({desc})")
else:
parts.append(c)
return ", ".join(parts)
+36
View File
@@ -0,0 +1,36 @@
import requests
import logging
import time
import json
from config import OLLAMA_URL, OLLAMA_MODEL
log = logging.getLogger(__name__)
def call_llm(prompt, timeout=90):
deadline = time.time() + timeout
try:
with requests.post(
OLLAMA_URL,
json={"model": OLLAMA_MODEL, "prompt": prompt, "stream": True, "format": "json"},
stream=True,
timeout=30,
) as r:
r.raise_for_status()
chunks = []
for line in r.iter_lines():
if time.time() > deadline:
log.warning("LLM call exceeded wall-clock timeout of %ds", timeout)
break
if not line:
continue
try:
data = json.loads(line)
chunks.append(data.get("response", ""))
if data.get("done"):
break
except json.JSONDecodeError:
continue
return "".join(chunks)
except requests.RequestException as e:
log.error("LLM call failed: %s", e)
return ""
+88
View File
@@ -0,0 +1,88 @@
import json
import logging
from ai.llm_client import call_llm
log = logging.getLogger(__name__)
PROMPT_TEMPLATE = (
'Analyze the scene below and return a JSON object with exactly these keys:\n'
'"characters": array of proper name strings only, no pronouns\n'
'"dialogue": array of objects with "speaker" (proper name or "?") and "line" (spoken words only) for every quoted line\n'
'"visual_scene": string, one sentence describing the physical action to illustrate\n'
'"mood": string, exactly one of: neutral happy sad tense angry mysterious romantic action\n'
'"setting": string, brief location name\n\n'
'Scene:\n{scene}'
)
JUNK_SPEAKERS = {
"i", "me", "he", "she", "they", "we", "you", "it",
"him", "her", "them", "his", "hers", "their",
"narrator", "voice", "unknown", "someone", "anyone",
}
def _empty(scene):
return {
"characters": [],
"dialogue": [],
"visual_scene": scene[:120].strip(),
"mood": "neutral",
"setting": "",
}
def _clean_dialogue(raw):
if not isinstance(raw, list):
return []
cleaned = []
for entry in raw:
if not isinstance(entry, dict):
continue
line = entry.get("line") or entry.get("text") or entry.get("speech") or ""
speaker = entry.get("speaker") or entry.get("name") or "?"
line = str(line).strip().strip('"')
speaker = str(speaker).strip()
if not line:
continue
if len(speaker.split()) > 3 or speaker.endswith(('.', ',')):
speaker = "?"
if speaker.lower() in JUNK_SPEAKERS:
speaker = "?"
cleaned.append({"speaker": speaker, "line": line})
return cleaned
def _extract(result, scene):
if not isinstance(result, dict):
log.warning("LLM returned non-dict JSON: %s", str(result)[:200])
return _empty(scene)
out = _empty(scene)
out["characters"] = result.get("characters") or []
out["dialogue"] = _clean_dialogue(result.get("dialogue", []))
out["visual_scene"] = result.get("visual_scene") or scene[:120].strip()
out["mood"] = result.get("mood") or "neutral"
out["setting"] = result.get("setting") or ""
for key in ("scene", "response", "output", "result"):
if key in result and isinstance(result[key], dict):
log.warning("LLM wrapped response under key '%s', unwrapping", key)
return _extract(result[key], scene)
return out
def parse_scene(scene, timeout=90):
prompt = PROMPT_TEMPLATE.format(scene=scene)
raw = call_llm(prompt, timeout=timeout)
if not raw:
log.warning("Empty LLM response")
return _empty(scene)
try:
result = json.loads(raw)
return _extract(result, scene)
except json.JSONDecodeError as e:
log.warning("JSON decode failed: %s — raw: %s", e, raw[:300])
return _empty(scene)
def parse_in_batches(scenes, batch_size=1, progress_cb=None, timeout=90):
results = []
for i, scene in enumerate(scenes):
result = parse_scene(scene, timeout=timeout)
results.append(result)
if progress_cb:
progress_cb(i + 1, len(scenes))
return results
+3
View File
@@ -0,0 +1,3 @@
OLLAMA_URL = "http://localhost:11434/api/generate"
OLLAMA_MODEL = "llama3"
SD_API_URL = "http://127.0.0.1:7860/sdapi/v1/txt2img"
BIN
View File
Binary file not shown.
+25
View File
@@ -0,0 +1,25 @@
from ebooklib import epub
from ebooklib import ITEM_DOCUMENT
from bs4 import BeautifulSoup
CHAPTER_TAGS = {"h1", "h2", "h3", "h4"}
def read_epub(path):
book = epub.read_epub(path)
chapters = []
for item in book.get_items():
if item.get_type() == ITEM_DOCUMENT:
soup = BeautifulSoup(item.get_content(), "html.parser")
chapters.append(soup.get_text())
return chapters
def read_epub_flat(path):
return "\n\n".join(read_epub(path))
def read_epub_metadata(path):
book = epub.read_epub(path)
title = book.get_metadata('DC', 'title')
author = book.get_metadata('DC', 'creator')
title_str = title[0][0] if title else None
author_str = author[0][0] if author else None
return title_str, author_str
+62
View File
@@ -0,0 +1,62 @@
import re
JUNK_PATTERNS = [
r'all rights reserved',
r'copyright\s*©',
r'isbn[\s\-]',
r'published by',
r'first published',
r'printed in',
r'no part of this',
r'table of contents',
r'this is a work of fiction',
r'any resemblance to',
]
JUNK_RE = re.compile('|'.join(JUNK_PATTERNS), re.IGNORECASE)
def _is_junk(text):
if len(text.split()) < 30:
return True
if JUNK_RE.search(text):
return True
return False
def split_scenes(chapters, min_sentences=3, max_sentences=6):
scenes = []
for chapter_text in chapters:
paragraphs = [p.strip() for p in re.split(r'\n{2,}', chapter_text) if p.strip()]
buf = []
buf_sentences = 0
for para in paragraphs:
sentences = re.split(r'(?<=[.!?])\s+', para.strip())
sentences = [s for s in sentences if s]
for sentence in sentences:
buf.append(sentence)
buf_sentences += 1
if buf_sentences >= max_sentences:
candidate = " ".join(buf)
if not _is_junk(candidate):
scenes.append(candidate)
buf = []
buf_sentences = 0
if buf_sentences >= min_sentences:
candidate = " ".join(buf)
if not _is_junk(candidate):
scenes.append(candidate)
buf = []
buf_sentences = 0
if buf:
candidate = " ".join(buf)
if scenes and _is_junk(candidate):
pass
elif scenes:
scenes[-1] = scenes[-1] + " " + candidate
elif not _is_junk(candidate):
scenes.append(candidate)
return [s for s in scenes if len(s.strip()) > 20]
BIN
View File
Binary file not shown.
+50
View File
@@ -0,0 +1,50 @@
from ebooklib import epub
from PIL import Image
import io
import os
def _to_jpeg(img_path, quality=85):
img = Image.open(img_path).convert("L").convert("RGB")
buf = io.BytesIO()
img.save(buf, format="JPEG", quality=quality, optimize=True)
return buf.getvalue()
def build_epub(images, output, title="Manga Book", author="Manga", jpeg_quality=85):
book = epub.EpubBook()
book.set_title(title)
book.set_language("en")
book.add_author(author)
chapters = []
for i, img_path in enumerate(images):
if not os.path.exists(img_path):
continue
img_data = _to_jpeg(img_path, quality=jpeg_quality)
img_name = f"images/page_{i}.jpg"
epub_img = epub.EpubItem(
uid=f"img_{i}",
file_name=img_name,
media_type="image/jpeg",
content=img_data,
)
book.add_item(epub_img)
c = epub.EpubHtml(title=f"Page {i+1}", file_name=f"p{i}.xhtml", lang="en")
c.content = (
f'<html><body style="margin:0;padding:0;background:#000;">'
f'<img src="{img_name}" style="width:100%;height:auto;display:block;"/>'
f'</body></html>'
)
book.add_item(c)
chapters.append(c)
book.toc = tuple(chapters)
book.spine = ["nav"] + chapters
book.add_item(epub.EpubNcx())
book.add_item(epub.EpubNav())
epub.write_epub(output, book)
return output
BIN
View File
Binary file not shown.
+22
View File
@@ -0,0 +1,22 @@
import requests
import base64
def generate_image(base_url, positive, negative, steps=20, width=768, height=1024, cfg=4.5):
r = requests.post(f"{base_url}/sdapi/v1/txt2img", json={
"prompt": positive,
"negative_prompt": negative,
"steps": steps,
"width": width,
"height": height,
"cfg_scale": cfg,
})
r.raise_for_status()
return base64.b64decode(r.json()["images"][0])
def is_ready(base_url, timeout=3):
try:
requests.get(f"{base_url}/sdapi/v1/options", timeout=timeout)
return True
except Exception:
return False
+70
View File
@@ -0,0 +1,70 @@
import os
import sys
from image.sd_launcher import read_config, write_config
def get_backend_type(sd_path):
if not sd_path:
return None
if os.path.exists(os.path.join(sd_path, "webui.sh")):
return "a1111"
if os.path.exists(os.path.join(sd_path, "comfy_extras")) or "ComfyUI" in sd_path:
return "comfyui"
if "InvokeAI" in sd_path:
return "invokeai"
return "a1111"
def get_base_url(backend_type):
if backend_type == "comfyui":
return "http://127.0.0.1:8188"
return "http://127.0.0.1:7860"
def is_ready(backend_type, base_url):
if backend_type == "comfyui":
from image.comfy_api import is_ready as comfy_ready
return comfy_ready()
from image.a1111_api import is_ready as a1111_ready
return a1111_ready(base_url)
def ensure_model(backend_type, sd_path, style="manga"):
if backend_type == "comfyui":
from image import comfy_api
model = comfy_api.ensure_model_for_style(sd_path, style=style)
write_config("comfy-model", model)
return model
return None
def setup(sd_path):
backend_type = get_backend_type(sd_path)
write_config("sd-backend", backend_type)
if backend_type == "comfyui":
from image import comfy_api
comfy_api.install_dependencies(sd_path)
return backend_type
def _force_bw(img_bytes):
from PIL import Image
import io
img = Image.open(io.BytesIO(img_bytes)).convert("L").convert("RGB")
buf = io.BytesIO()
img.save(buf, format="PNG")
return buf.getvalue()
def generate_image(positive, negative, steps=20, width=768, height=1024, cfg=4.5, force_bw=True, style="manga"):
config = read_config()
sd_path = config.get("sd-path", "")
backend_type = config.get("sd-backend") or get_backend_type(sd_path)
base_url = get_base_url(backend_type)
if backend_type == "comfyui":
from image import comfy_api
model = comfy_api.ensure_model_for_style(sd_path, style=style)
write_config("comfy-model", model)
if not model:
print("\nError: no model found. Re-run to trigger model download.")
sys.exit(1)
result = comfy_api.generate_image(base_url, positive, negative, model, steps, width, height, cfg)
return _force_bw(result) if force_bw else result
from image import a1111_api
result = a1111_api.generate_image(base_url, positive, negative, steps, width, height, cfg)
return _force_bw(result) if force_bw else result
+188
View File
@@ -0,0 +1,188 @@
import requests
import json
import uuid
import time
import os
import subprocess
import sys
COMFY_PORT = 8188
POLL_TIMEOUT = 300
MODELS = {
"lineart": {
"name": "Deliberate_v2.safetensors",
"url": "https://huggingface.co/XpucT/Deliberate/resolve/main/Deliberate_v2.safetensors",
},
"manga": {
"name": "Deliberate_v2.safetensors",
"url": "https://huggingface.co/XpucT/Deliberate/resolve/main/Deliberate_v2.safetensors",
},
}
DEFAULT_MODEL_URL = MODELS["manga"]["url"]
DEFAULT_MODEL_NAME = MODELS["manga"]["name"]
def base_url(host="127.0.0.1"):
return f"http://{host}:{COMFY_PORT}"
def is_ready(host="127.0.0.1", timeout=3):
try:
requests.get(f"{base_url(host)}/system_stats", timeout=timeout)
return True
except Exception:
return False
def find_model(comfy_path, name=None):
model_dir = os.path.join(comfy_path, "models", "checkpoints")
if not os.path.isdir(model_dir):
return None
if name:
return name if os.path.exists(os.path.join(model_dir, name)) else None
for f in os.listdir(model_dir):
if f.endswith(".safetensors") or f.endswith(".ckpt"):
return f
return None
def download_model(comfy_path, style="manga"):
model_info = MODELS.get(style, MODELS["manga"])
model_name = model_info["name"]
model_url = model_info["url"]
style_label = "Anything V5 (anime/manga lineart)" if style == "lineart" else "Deliberate v2 (general purpose)"
model_dir = os.path.join(comfy_path, "models", "checkpoints")
os.makedirs(model_dir, exist_ok=True)
dest = os.path.join(model_dir, model_name)
if os.path.exists(dest):
return model_name
print(f"\n Model for --style {style} not found: {model_name}")
print(f" Recommended model: {style_label}")
print(f" Source: {model_url}")
confirm = input("\n Download it now? (~2GB) [Y/n] ").strip().lower()
if confirm in ("n", "no"):
print(" Aborted.")
sys.exit(0)
print(f"\n Downloading {model_name}...")
bar_width = 35
with requests.get(model_url, stream=True) as r:
r.raise_for_status()
total = int(r.headers.get("content-length", 0))
downloaded = 0
with open(dest, "wb") as f:
for chunk in r.iter_content(chunk_size=1024 * 1024):
f.write(chunk)
downloaded += len(chunk)
if total:
pct = downloaded / total
filled = int(bar_width * pct)
bar = "\u2588" * filled + "\u2591" * (bar_width - filled)
mb_done = downloaded / 1024 / 1024
mb_total = total / 1024 / 1024
print(f"\r [{bar}] {mb_done:.0f}/{mb_total:.0f} MB ", end="", flush=True)
print(f"\r Download complete: {dest} ")
return model_name
def ensure_model_for_style(comfy_path, style="manga"):
model_info = MODELS.get(style, MODELS["manga"])
model_name = model_info["name"]
existing = find_model(comfy_path, name=model_name)
if existing:
return existing
return download_model(comfy_path, style=style)
def install_dependencies(comfy_path):
req_file = os.path.join(comfy_path, "requirements.txt")
stamp = os.path.join(comfy_path, ".deps_installed")
if not os.path.exists(req_file):
return
if os.path.exists(stamp):
req_mtime = os.path.getmtime(req_file)
stamp_mtime = os.path.getmtime(stamp)
if stamp_mtime >= req_mtime:
return
print(" Installing ComfyUI dependencies...")
subprocess.run(
[sys.executable, "-m", "pip", "install", "-r", req_file, "--quiet"],
check=True,
)
open(stamp, "w").close()
def build_workflow(positive, negative, model_name, steps=20, width=768, height=1024, cfg=4.5):
seed = int(time.time()) % 2**32
return {
"3": {
"class_type": "KSampler",
"inputs": {
"seed": seed,
"steps": steps,
"cfg": cfg,
"sampler_name": "euler_ancestral",
"scheduler": "karras",
"denoise": 1.0,
"model": ["4", 0],
"positive": ["6", 0],
"negative": ["7", 0],
"latent_image": ["5", 0],
},
},
"4": {
"class_type": "CheckpointLoaderSimple",
"inputs": {"ckpt_name": model_name},
},
"5": {
"class_type": "EmptyLatentImage",
"inputs": {"width": width, "height": height, "batch_size": 1},
},
"6": {
"class_type": "CLIPTextEncode",
"inputs": {"text": positive, "clip": ["4", 1]},
},
"7": {
"class_type": "CLIPTextEncode",
"inputs": {"text": negative, "clip": ["4", 1]},
},
"8": {
"class_type": "VAEDecode",
"inputs": {"samples": ["3", 0], "vae": ["4", 2]},
},
"9": {
"class_type": "SaveImage",
"inputs": {"filename_prefix": "manga", "images": ["8", 0]},
},
}
def generate_image(base_url_str, positive, negative, model_name, steps=20, width=768, height=1024, cfg=4.5):
client_id = str(uuid.uuid4())
workflow = build_workflow(positive, negative, model_name, steps, width, height, cfg)
r = requests.post(f"{base_url_str}/prompt", json={"prompt": workflow, "client_id": client_id})
r.raise_for_status()
prompt_id = r.json()["prompt_id"]
deadline = time.time() + POLL_TIMEOUT
while time.time() < deadline:
time.sleep(1)
hist = requests.get(f"{base_url_str}/history/{prompt_id}").json()
if prompt_id in hist:
outputs = hist[prompt_id]["outputs"]
for node_id, node_output in outputs.items():
if "images" in node_output:
img_info = node_output["images"][0]
img_r = requests.get(
f"{base_url_str}/view",
params={
"filename": img_info["filename"],
"subfolder": img_info.get("subfolder", ""),
"type": img_info["type"],
},
)
img_r.raise_for_status()
return img_r.content
break
raise RuntimeError(f"ComfyUI: no images returned for prompt_id {prompt_id} within {POLL_TIMEOUT}s")
+14
View File
@@ -0,0 +1,14 @@
import torch
import platform
def detect_device():
if torch.cuda.is_available():
return "cuda"
if platform.system() == "Darwin":
try:
if torch.backends.mps.is_available():
return "mps"
except:
pass
return "cpu"
+337
View File
@@ -0,0 +1,337 @@
import os
import subprocess
import time
import threading
import sys
import requests
_sd_process = None
CONFIG_FILE = os.path.expanduser("~/.epub-to-manga")
BACKENDS = [
{
"name": "AUTOMATIC1111 Stable Diffusion WebUI",
"repo": "https://github.com/AUTOMATIC1111/stable-diffusion-webui",
"dir": "stable-diffusion-webui",
"type": "a1111",
"launch": ["bash", "webui.sh", "--api", "--nowebui"],
"ready_url": "http://127.0.0.1:7860/sdapi/v1/options",
},
{
"name": "ComfyUI (lighter, faster on Apple Silicon)",
"repo": "https://github.com/comfyanonymous/ComfyUI",
"dir": "ComfyUI",
"type": "comfyui",
"launch": [sys.executable, "main.py", "--listen"],
"ready_url": "http://127.0.0.1:8188/system_stats",
},
{
"name": "InvokeAI",
"repo": "https://github.com/invoke-ai/InvokeAI",
"dir": "InvokeAI",
"type": "invokeai",
"launch": ["invokeai-web"],
"ready_url": "http://127.0.0.1:9090/api/v1/app/version",
},
]
KNOWN_ERRORS = [
{
"markers": ["No module named 'pkg_resources'", "Couldn't install clip"],
"message": "AUTOMATIC1111 is not compatible with Python {python_version}. A1111 requires Python 3.10.",
"solutions": [
{
"label": "Install Python 3.10 via pyenv and relaunch A1111",
"action": "fix_a1111_python",
},
{
"label": "Switch to ComfyUI (better Apple Silicon support)",
"action": "switch_backend",
"backend_index": 1,
},
],
},
{
"markers": ["CUDA out of memory", "out of memory"],
"message": "GPU ran out of memory.",
"solutions": [
{
"label": "Re-run with a smaller image size (--size 256x384)",
"action": "suggest_flag",
"flag": "--size 256x384",
},
],
},
]
def read_config():
config = {}
if os.path.exists(CONFIG_FILE):
with open(CONFIG_FILE) as f:
for line in f:
line = line.strip()
if "=" in line and not line.startswith("#"):
key, _, val = line.partition("=")
config[key.strip()] = val.strip()
return config
def write_config(key, value):
config = read_config()
config[key] = value
with open(CONFIG_FILE, "w") as f:
for k, v in config.items():
f.write(f"{k}={v}\n")
def get_backend_for_path(sd_path):
if not sd_path:
return BACKENDS[0]
for b in BACKENDS:
if b["dir"] in sd_path or os.path.exists(os.path.join(sd_path, b["dir"])):
return b
if b["type"] == "a1111" and os.path.exists(os.path.join(sd_path, "webui.sh")):
return b
if b["type"] == "comfyui" and os.path.exists(os.path.join(sd_path, "comfy_extras")):
return b
return BACKENDS[0]
def is_running(ready_url, timeout=3):
try:
requests.get(ready_url, timeout=timeout)
return True
except Exception:
return False
def find_installed():
home = os.path.expanduser("~")
search_dirs = [home, os.path.join(home, "Downloads"), os.path.join(home, "Documents"), os.getcwd()]
for backend in BACKENDS:
for base in search_dirs:
candidate = os.path.join(base, backend["dir"])
if os.path.isdir(candidate):
return candidate, backend
return None, None
def get_python_version():
return f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}"
def handle_known_error(error_def, sd_path, backend):
python_version = get_python_version()
msg = error_def["message"].format(python_version=python_version)
solutions = error_def["solutions"]
print(f"\n\n ⚠️ {msg}\n")
print(" Solutions:\n")
for i, s in enumerate(solutions, 1):
print(f" {i}) {s['label']}")
print()
while True:
choice = input(" Choose a solution (or q to quit): ").strip().lower()
if choice == "q":
sys.exit(0)
if choice.isdigit() and 1 <= int(choice) <= len(solutions):
solution = solutions[int(choice) - 1]
break
print(f" Enter a number between 1 and {len(solutions)}")
action = solution["action"]
if action == "fix_a1111_python":
print("\n Installing pyenv and Python 3.10.14...")
subprocess.run(["brew", "install", "pyenv"], check=False)
subprocess.run(["pyenv", "install", "3.10.14"], check=False)
pyenv_python = os.path.expanduser("~/.pyenv/versions/3.10.14/bin/python3")
if not os.path.exists(pyenv_python):
print("\n Error: pyenv install failed. See https://github.com/pyenv/pyenv")
sys.exit(1)
env = os.environ.copy()
env["PYTHON"] = pyenv_python
print(f"\n Relaunching A1111 with Python 3.10.14...")
global _sd_process
_sd_process = subprocess.Popen(
["bash", "webui.sh", "--api", "--nowebui"],
cwd=sd_path,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
env=env,
)
stream_and_wait(backend, sd_path)
elif action == "switch_backend":
new_backend = BACKENDS[solution["backend_index"]]
dest = os.path.join(os.path.expanduser("~"), new_backend["dir"])
if os.path.isdir(dest):
print(f"\n Found existing {new_backend['name']} at {dest} — using it.")
else:
confirm = input(f"\n Install {new_backend['name']} to {dest}? [Y/n] ").strip().lower()
if confirm in ("n", "no"):
print(" Aborted.")
sys.exit(0)
print(f"\n Cloning {new_backend['repo']}...")
result = subprocess.run(["git", "clone", "--recursive", new_backend["repo"], dest], check=False)
if result.returncode != 0:
print("\n Error: git clone failed.")
sys.exit(1)
print(f"\n Installed to {dest}")
write_config("sd-path", dest)
write_config("sd-backend", new_backend["type"])
print(f" Re-run your original command — switching will take effect automatically.")
sys.exit(0)
elif action == "suggest_flag":
print(f"\n Re-run with: {solution['flag']}")
sys.exit(0)
def stream_logs(proc, ready_event, error_detected, sd_path, backend):
interesting = [
"Loading", "Downloading", "Installing", "Running", "Creating",
"Model loaded", "Starting", "Applying", "torch", "CUDA", "MPS",
"checkpoint", "venv", "pip", "Startup", "listen",
]
for raw in proc.stdout:
if ready_event.is_set():
break
line = raw.decode("utf-8", errors="replace").rstrip()
for err_def in KNOWN_ERRORS:
if any(m in line for m in err_def["markers"]):
ready_event.set()
error_detected["def"] = err_def
return
if any(kw in line for kw in interesting):
print(f"\r > {line[:100]:<100}")
print(" ", end="", flush=True)
def stream_and_wait(backend, sd_path):
global _sd_process
ready_event = threading.Event()
error_detected = {}
log_thread = threading.Thread(
target=stream_logs,
args=(_sd_process, ready_event, error_detected, sd_path, backend),
daemon=True,
)
log_thread.start()
start = time.time()
while not ready_event.is_set():
elapsed = int(time.time() - start)
mins, secs = divmod(elapsed, 60)
elapsed_str = f"{mins}m {secs:02d}s" if mins else f"{secs}s"
print(f"\r Waiting for API... {elapsed_str} elapsed ", end="", flush=True)
if is_running(backend["ready_url"]):
ready_event.set()
elapsed = int(time.time() - start)
mins, secs = divmod(elapsed, 60)
elapsed_str = f"{mins}m {secs:02d}s" if mins else f"{secs}s"
print(f"\r Ready in {elapsed_str}! \n")
return
if _sd_process.poll() is not None and not ready_event.is_set():
ready_event.set()
break
time.sleep(2)
log_thread.join(timeout=2)
if "def" in error_detected:
handle_known_error(error_detected["def"], sd_path, backend)
elif not is_running(backend["ready_url"]):
print("\n\n Error: SD process exited unexpectedly.")
sys.exit(1)
def launch(sd_path, backend):
global _sd_process
print(f"\nStarting {backend['name']} from {sd_path}...")
_sd_process = subprocess.Popen(
backend["launch"],
cwd=sd_path,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
)
stream_and_wait(backend, sd_path)
def prompt_install():
print("\nNo Stable Diffusion backend found on your system.")
print("\nAvailable backends to install:\n")
for i, b in enumerate(BACKENDS, 1):
print(f" {i}) {b['name']}")
print(f" {b['repo']}")
print()
while True:
choice = input("Enter number to install (or q to quit): ").strip().lower()
if choice == "q":
sys.exit(0)
if choice.isdigit() and 1 <= int(choice) <= len(BACKENDS):
backend = BACKENDS[int(choice) - 1]
break
print(f" Please enter a number between 1 and {len(BACKENDS)}")
dest = os.path.join(os.path.expanduser("~"), backend["dir"])
confirm = input(f"\nInstall {backend['name']} to {dest}? [Y/n] ").strip().lower()
if confirm in ("n", "no"):
print("Aborted.")
sys.exit(0)
if os.path.isdir(dest):
print(f"\nFound existing {backend['name']} at {dest} — using it.")
else:
print(f"\nCloning {backend['repo']}...")
result = subprocess.run(["git", "clone", "--recursive", backend["repo"], dest], check=False)
if result.returncode != 0:
print("\nError: git clone failed. Is git installed?")
sys.exit(1)
print(f"\nInstalled to {dest}")
write_config("sd-path", dest)
write_config("sd-backend", backend["type"])
print(f"Saved to {CONFIG_FILE}")
print(f"\nRe-run your original command — no --sd-path needed.")
sys.exit(0)
def ensure_running(sd_url=None, sd_path=None, style="manga"):
config = read_config()
if not sd_path:
sd_path = config.get("sd-path")
backend = get_backend_for_path(sd_path)
if is_running(backend["ready_url"]):
return
if sd_path and os.path.isdir(sd_path):
from image import backend as backend_mod
backend_mod.setup(sd_path)
backend_mod.ensure_model(backend["type"], sd_path, style=style)
launch(sd_path, backend)
return
installed_path, found_backend = find_installed()
if installed_path:
print(f"\nFound {found_backend['name']} at {installed_path}")
write_config("sd-path", installed_path)
write_config("sd-backend", found_backend["type"])
print(f"Saved to {CONFIG_FILE}")
from image import backend as backend_mod
backend_mod.setup(installed_path)
backend_mod.ensure_model(found_backend["type"], installed_path, style=style)
launch(installed_path, found_backend)
return
prompt_install()
def shutdown():
global _sd_process
if _sd_process:
_sd_process.terminate()
_sd_process = None
+333
View File
@@ -0,0 +1,333 @@
import sys
import argparse
import logging
import time
import os
import json
import requests
import concurrent.futures
from core.epub_reader import read_epub, read_epub_metadata
from core.scene_splitter import split_scenes
from manga.panel_layout import group_into_pages
from manga.prompt_builder import build_page_prompt, get_cfg
from manga.speech_bubbles import add_speech_bubbles
from ai.scene_parser import parse_scene
from ai.character_memory import update as update_character, load as load_character_db, dump as dump_character_db, add_hint
from image.backend import generate_image
from image.sd_launcher import ensure_running, shutdown
import atexit
from export.epub_builder import build_epub
from utils.naming import make_output_name
import config
logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(name)s: %(message)s")
log = logging.getLogger(__name__)
def fmt_time(seconds):
seconds = int(seconds)
if seconds < 60:
return f"{seconds}s"
m, s = divmod(seconds, 60)
if m < 60:
return f"{m}m {s:02d}s"
h, m = divmod(m, 60)
return f"{h}h {m:02d}m"
def progress_bar(current, total, label="", eta_str="", width=35):
pct = current / total if total else 0
filled = int(width * pct)
bar = "█" * filled + "░" * (width - filled)
eta = f" ETA {eta_str}" if eta_str else ""
print(f"\r [{bar}] {current}/{total} {label}{eta} ", end="", flush=True)
def parse_args():
parser = argparse.ArgumentParser(
prog="main.py",
description="Convert an EPUB novel into a manga-style EPUB using LLM scene parsing and Stable Diffusion.",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
examples:
python3 main.py book.epub
python3 main.py book.epub output.epub
python3 main.py book.epub --layout tiny --model mistral
python3 main.py book.epub --sd-url http://192.168.1.10:7860 --steps 30 --size 512x768
python3 main.py book.epub --dry-run
layout modes:
normal 3 scenes per page (default)
tiny 1 scene per page (more pages, more detail per image)
requirements:
Ollama running at OLLAMA_URL (default: http://localhost:11434)
Stable Diffusion WebUI running at SD_API_URL (default: http://127.0.0.1:7860)
Both URLs can be overridden via flags or by editing config.py
"""
)
parser.add_argument("input", help="Path to input .epub file")
parser.add_argument("output", nargs="?", help="Path for output .epub (default: manga-<input>.epub)")
parser.add_argument("--layout", choices=["normal", "tiny"], default="normal",
help="Page layout mode: normal=3 scenes/page, tiny=1 scene/page (default: normal)")
parser.add_argument("--model", default=config.OLLAMA_MODEL, metavar="MODEL",
help=f"Ollama model to use for scene parsing (default: {config.OLLAMA_MODEL})")
parser.add_argument("--ollama-url", default=config.OLLAMA_URL, metavar="URL",
help=f"Ollama API base URL (default: {config.OLLAMA_URL})")
parser.add_argument("--sd-url", default=config.SD_API_URL, metavar="URL",
help=f"Stable Diffusion WebUI API URL (default: {config.SD_API_URL})")
parser.add_argument("--sd-path", default=None, metavar="PATH",
help="Path to stable-diffusion-webui folder; if SD is not running, auto-starts it")
parser.add_argument("--steps", type=int, default=20, metavar="N",
help="Diffusion steps per image (default: 20, higher=better quality but slower)")
parser.add_argument("--style", choices=["lineart", "manga"], default="lineart",
help="Image style: lineart=simple clean outlines (default), manga=detailed screentone")
parser.add_argument("--size", default="768x1024", metavar="WxH",
help="Image dimensions in pixels (default: 768x1024)")
parser.add_argument("--workers", type=int, default=2, metavar="N",
help="Parallel image generation workers (default: 2)")
parser.add_argument("--hint", action="append", metavar="NAME:DESC",
help="Character description hint e.g. 'Rocky:alien who resembles a rock spider'. Can be used multiple times.")
parser.add_argument("--jpeg-quality", type=int, default=85, metavar="N",
help="JPEG quality for epub images 1-95 (default: 85, lower=smaller file)")
parser.add_argument("--dry-run", nargs="?", const=3, type=int, metavar="N",
help="Show N prompts without generating images (default: 3)")
parser.add_argument("--verbose", "-v", action="store_true",
help="Show debug logging")
return parser.parse_args()
def main():
args = parse_args()
if args.verbose:
logging.getLogger().setLevel(logging.DEBUG)
config.OLLAMA_MODEL = args.model
config.OLLAMA_URL = args.ollama_url
from image.sd_launcher import write_config as _wc
_wc("llm-model", args.model)
config.SD_API_URL = args.sd_url
atexit.register(shutdown)
ensure_running(args.sd_url, args.sd_path, style=args.style)
try:
width, height = (int(x) for x in args.size.split("x"))
except ValueError:
print(f"Error: --size must be in WxH format, e.g. 768x1024")
sys.exit(1)
if not os.path.isfile(args.input):
print(f"Error: input file not found: {args.input}")
sys.exit(1)
output = args.output or make_output_name(args.input)
output_dir = os.path.splitext(output)[0] + "_pages"
os.makedirs(output_dir, exist_ok=True)
total_start = time.time()
print(f"\nepub-to-manga")
print(f" input: {args.input}")
print(f" output: {output}")
print(f" layout: {args.layout} | model: {args.model} | steps: {args.steps} | size: {width}x{height} | style: {args.style} | workers: {args.workers}")
cache_file = make_output_name(args.input).replace(".epub", ".cache.json")
print(f"\nReading & splitting epub...")
source_title, source_author = read_epub_metadata(args.input)
chapters = read_epub(args.input)
scenes = split_scenes(chapters)
print(f" {len(scenes)} scenes found across {len(chapters)} chapters")
dry_run_n = args.dry_run if args.dry_run is not None else None
needed = dry_run_n if dry_run_n is not None else len(scenes)
cached_scenes = []
cached_chars = {}
if os.path.exists(cache_file):
with open(cache_file) as f:
cache_data = json.load(f)
cached_scenes = cache_data.get("parsed_scenes", [])
cached_chars = cache_data.get("character_db", {})
load_character_db(cached_chars)
if args.hint:
for hint in args.hint:
if ":" in hint:
name, _, desc = hint.partition(":")
add_hint(name.strip(), desc.strip())
print(f" Character hint: {name.strip()} = {desc.strip()}")
if len(cached_scenes) >= needed:
print(f" Loaded {len(cached_scenes)} scenes from cache ({cache_file})")
parsed_scenes = cached_scenes
else:
if cached_scenes:
print(f" Resuming parse — {len(cached_scenes)}/{needed} scenes cached")
remaining_scenes = scenes[len(cached_scenes):needed]
print(f"\nParsing scenes with {args.model} ({needed} total)")
progress_bar(len(cached_scenes), needed)
run_start = time.time()
new_parsed = []
scene_times = []
def on_scene_done(batch_idx, batch_len):
pass
for idx, scene in enumerate(remaining_scenes):
t0 = time.time()
result = parse_scene(scene, timeout=90)
elapsed_scene = time.time() - t0
new_parsed.append(result)
scene_times.append(elapsed_scene)
if len(scene_times) > 10:
scene_times.pop(0)
for char in result.get("characters", []):
if char and char.strip():
update_character(char.strip(), "")
combined = cached_scenes + new_parsed
with open(cache_file, "w") as f:
json.dump({"parsed_scenes": combined, "character_db": dump_character_db()}, f)
done = len(cached_scenes) + len(new_parsed)
avg = sum(scene_times) / len(scene_times)
remaining_count = needed - done
eta = fmt_time(avg * remaining_count) if remaining_count > 0 else fmt_time(time.time() - run_start)
progress_bar(done, needed, eta_str=eta)
print()
parsed_scenes = cached_scenes + new_parsed
print(f" Saved to {cache_file}")
for ps in parsed_scenes:
for char in ps.get("characters", []):
if char and char.strip():
update_character(char.strip(), "")
pages = group_into_pages(parsed_scenes, mode=args.layout)
print(f" {len(pages)} pages ({args.layout} layout)")
if args.dry_run is not None:
n = args.dry_run
print(f"\nDry run — first {n} prompts:")
for i, page in enumerate(pages[:n]):
positive, negative = build_page_prompt(page, style=args.style)
print(f"\n--- Page {i+1} ---\nPositive: {positive}\nNegative: {negative}")
return
from image.sd_launcher import read_config as _rc
_sd_cfg = _rc()
_sd_path = _sd_cfg.get("sd-path", "")
_backend = _sd_cfg.get("sd-backend", "comfyui")
_port = "8188" if _backend == "comfyui" else "7860"
if _backend == "comfyui":
from image import comfy_api
_model = comfy_api.ensure_model_for_style(_sd_path, style=args.style)
from image.sd_launcher import write_config as _wc2
_wc2("comfy-model", _model)
total = len(pages)
already_done = [
os.path.join(output_dir, f"page_{i}.png")
for i in range(total)
if os.path.exists(os.path.join(output_dir, f"page_{i}.png"))
]
resumed = len(already_done)
print(f"\nGenerating images... (http://127.0.0.1:{_port})")
print(f" {total} pages | ~{args.steps} steps each | {width}x{height} | {args.workers} workers")
if resumed:
print(f" Resuming — {resumed}/{total} pages already done, skipping...")
images = [None] * total
for i in range(total):
img_path = os.path.join(output_dir, f"page_{i}.png")
if os.path.exists(img_path):
images[i] = img_path
progress_bar(resumed, total)
img_start = time.time()
times = []
completed = resumed
lock = __import__("threading").Lock()
def gen_page(args_tuple):
i, page = args_tuple
img_path = os.path.join(output_dir, f"page_{i}.png")
if os.path.exists(img_path):
return i, img_path, None
positive, negative = build_page_prompt(page, style=args.style)
t0 = time.time()
try:
img_bytes = generate_image(
positive, negative,
steps=args.steps, width=width, height=height,
cfg=get_cfg(args.style),
force_bw=(args.style == "lineart"),
style=args.style
)
elapsed = time.time() - t0
with open(img_path, "wb") as f:
f.write(img_bytes)
dialogue = []
if isinstance(page, list):
for scene in page:
if isinstance(scene, dict):
dialogue.extend(scene.get("dialogue", []))
if dialogue:
add_speech_bubbles(img_path, dialogue)
return i, img_path, elapsed
except (requests.RequestException, KeyError, IndexError, RuntimeError) as e:
return i, None, str(e)
pending = [(i, page) for i, page in enumerate(pages) if images[i] is None]
with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as executor:
futures = {executor.submit(gen_page, item): item[0] for item in pending}
for future in concurrent.futures.as_completed(futures):
i, img_path, result = future.result()
with lock:
if img_path:
images[i] = img_path
if isinstance(result, float):
times.append(result)
if len(times) > 8:
times.pop(0)
else:
print(f"\n [!] Page {i} failed: {result}")
completed += 1
avg = sum(times) / len(times) if times else 0
remaining_count = total - completed
eta = fmt_time(avg * remaining_count / args.workers) if avg and remaining_count else ""
progress_bar(completed, total, eta_str=eta)
print()
final_images = [img for img in images if img]
if not final_images:
print(f"\nError: no images generated. Is the SD API running at {args.sd_url}?")
sys.exit(1)
print(f"\nBuilding epub...")
result = build_epub(final_images, output, title=source_title or "Manga Book", author="Manga", jpeg_quality=args.jpeg_quality)
total_elapsed = time.time() - total_start
print(f" Done in {fmt_time(total_elapsed)}: {result}\n")
if __name__ == "__main__":
try:
main()
except KeyboardInterrupt:
print("\n\nStopped. Progress saved - re-run to resume.")
BIN
View File
Binary file not shown.
+13
View File
@@ -0,0 +1,13 @@
def group_into_pages(scenes, mode="normal"):
if mode == "tiny":
return [[s] for s in scenes]
pages = []
temp = []
for s in scenes:
temp.append(s)
if len(temp) == 3:
pages.append(temp)
temp = []
if temp:
pages.append(temp)
return pages
+137
View File
@@ -0,0 +1,137 @@
import re
from ai.character_memory import build_consistency_tokens
STYLES = {
"lineart": {
"positive": (
"anime illustration, manga style, black and white, monochrome, "
"clean ink linework, expressive characters, dynamic composition, "
"professional manga art, single scene, full image"
),
"negative": (
"color, colorful, coloured, vibrant, saturated, "
"multiple panels, panel borders, panel grid, split panels, "
"collage, triptych, diptych, "
"realistic, photorealistic, photograph, 3d render, "
"ugly, blurry, watermark, text, signature, lowres, "
"bad anatomy, deformed, extra limbs"
),
"cfg": 7.0,
},
"manga": {
"positive": (
"single full-page manga illustration, full-bleed scene, "
"black and white ink, screentone shading, "
"expressive faces, detailed backgrounds, "
"professional manga art, one continuous scene"
),
"negative": (
"multiple panels, panel borders, panel grid, comic layout, split panels, "
"panel dividers, gutters, multi-panel page, page layout, comic book grid, "
"collage, triptych, diptych, "
"lowres, bad anatomy, blurry, watermark, text, ugly, deformed, "
"color, coloured, western comic style"
),
"cfg": 7.5,
},
}
MOOD_MAP = {
"tense": "tense dramatic scene, characters look worried or scared",
"sad": "sad melancholy scene, characters look downcast",
"happy": "happy cheerful scene, characters smiling",
"angry": "angry confrontational scene, characters look furious",
"mysterious": "mysterious eerie scene, shadowy atmosphere",
"neutral": "calm everyday scene",
"romantic": "romantic gentle scene, soft atmosphere",
"action": "action dynamic scene, characters in motion",
"excited": "excited energetic scene, characters enthusiastic",
"fearful": "fearful tense scene, characters look afraid",
}
JUNK_NAMES = {
"he", "she", "they", "him", "her", "them", "his", "hers", "their",
"i", "me", "we", "us", "it", "you", "location", "unknown", "none",
"character", "person", "man", "woman", "boy", "girl",
"the man", "the woman", "the boy", "the girl", "the person",
"a man", "a woman", "old man", "young man", "young woman",
"his secretary", "his secretarys", "her secretary",
"narrator", "voice", "someone", "anyone", "everyone",
}
PLACEHOLDER_SETTINGS = {"location", "unknown", "none", "'location'", '"location"', ""}
POSSESSIVE_RE = re.compile(r"'s$|s'$", re.IGNORECASE)
def clean_text(s):
s = s.replace("\\u0022", "").replace("\\u0027", "")
s = s.replace('\\"', "").replace("\\'", "")
s = re.sub(r'[\"\'`]', "", s)
s = re.sub(r"\s+", " ", s).strip()
return s
def is_valid_name(name):
n = name.strip().lower()
if not n or len(n) < 2:
return False
if n in JUNK_NAMES:
return False
if POSSESSIVE_RE.search(n):
return False
return True
def build_page_prompt(parsed_scenes, style="lineart"):
style_def = STYLES.get(style, STYLES["lineart"])
if not parsed_scenes:
return style_def["positive"], style_def["negative"]
all_characters = []
visual_parts = []
settings = []
moods = []
for scene in parsed_scenes:
if isinstance(scene, str):
visual_parts.append(scene)
continue
for char in scene.get("characters", []):
char = clean_text(char)
if is_valid_name(char):
all_characters.append(char)
visual = clean_text(scene.get("visual_scene", ""))
if visual:
visual_parts.append(visual)
setting = clean_text(scene.get("setting", ""))
if setting.lower() not in PLACEHOLDER_SETTINGS:
settings.append(setting)
mood = scene.get("mood", "neutral").strip().lower()
if mood:
moods.append(mood)
unique_chars = list(dict.fromkeys(all_characters))
char_tokens = build_consistency_tokens(unique_chars)
dominant_mood = moods[0] if moods else "neutral"
mood_desc = MOOD_MAP.get(dominant_mood, MOOD_MAP["neutral"])
setting_str = settings[0][:60] if settings else ""
scene_str = visual_parts[0][:80] if visual_parts else ""
style_prefix = (
"(anime style:1.4), (manga illustration:1.3), "
"(black and white:1.3), (monochrome:1.2), "
) if style == "lineart" else ""
parts = [style_prefix + style_def["positive"], mood_desc]
if setting_str:
parts.append(setting_str)
if char_tokens:
parts.append(char_tokens)
if scene_str:
parts.append(scene_str)
return ", ".join(parts), style_def["negative"]
def get_cfg(style="lineart"):
return STYLES.get(style, STYLES["lineart"])["cfg"]
+115
View File
@@ -0,0 +1,115 @@
import textwrap
from PIL import Image, ImageDraw, ImageFont
import os
MAX_BUBBLES = 3
MAX_LINE_WIDTH = 22
BUBBLE_PADDING = 14
FONT_SIZE = 22
TAIL_SIZE = 14
def get_font(size=FONT_SIZE):
candidates = [
"/System/Library/Fonts/Helvetica.ttc",
"/System/Library/Fonts/Arial.ttf",
"/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf",
"/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf",
]
for path in candidates:
if os.path.exists(path):
try:
return ImageFont.truetype(path, size)
except Exception:
pass
return ImageFont.load_default()
def wrap_text(text, max_width=MAX_LINE_WIDTH):
return textwrap.fill(text, max_width)
def draw_bubble(draw, x, y, w, h, tail_side="bottom"):
r = 16
draw.rounded_rectangle([x, y, x+w, y+h], radius=r, fill="white", outline="black", width=3)
if tail_side == "bottom":
tx = x + w // 2
ty = y + h
draw.polygon([
(tx - TAIL_SIZE, ty - 4),
(tx + TAIL_SIZE, ty - 4),
(tx, ty + TAIL_SIZE),
], fill="white")
draw.line([(tx - TAIL_SIZE, ty - 4), (tx, ty + TAIL_SIZE)], fill="black", width=3)
draw.line([(tx + TAIL_SIZE, ty - 4), (tx, ty + TAIL_SIZE)], fill="black", width=3)
elif tail_side == "top":
tx = x + w // 2
ty = y
draw.polygon([
(tx - TAIL_SIZE, ty + 4),
(tx + TAIL_SIZE, ty + 4),
(tx, ty - TAIL_SIZE),
], fill="white")
draw.line([(tx - TAIL_SIZE, ty + 4), (tx, ty - TAIL_SIZE)], fill="black", width=3)
draw.line([(tx + TAIL_SIZE, ty + 4), (tx, ty - TAIL_SIZE)], fill="black", width=3)
def add_speech_bubbles(img_path, dialogue):
if not dialogue:
return
entries = [d for d in dialogue if d.get("line", "").strip()][:MAX_BUBBLES]
if not entries:
return
img = Image.open(img_path).convert("RGB")
draw = ImageDraw.Draw(img)
font = get_font(FONT_SIZE)
small_font = get_font(FONT_SIZE - 6)
iw, ih = img.size
margin = 18
zones = [
ih // 8,
ih * 5 // 8,
ih // 4,
]
for i, entry in enumerate(entries):
speaker = entry.get("speaker", "").strip()
line = entry.get("line", "").strip()
if not line:
continue
wrapped = wrap_text(line)
bbox = draw.textbbox((0, 0), wrapped, font=font)
text_w = bbox[2] - bbox[0]
text_h = bbox[3] - bbox[1]
if speaker and speaker != "?":
spk_bbox = draw.textbbox((0, 0), speaker, font=small_font)
spk_h = spk_bbox[3] - spk_bbox[1] + 4
else:
spk_h = 0
bw = min(text_w + BUBBLE_PADDING * 2, iw - margin * 2)
bh = text_h + spk_h + BUBBLE_PADDING * 2
x_offset = margin if i % 2 == 0 else iw - bw - margin
by = zones[i % len(zones)]
by = max(margin, min(by, ih - bh - margin))
tail = "bottom" if by < ih // 2 else "top"
draw_bubble(draw, x_offset, by, bw, bh, tail_side=tail)
if speaker and speaker != "?":
draw.text((x_offset + BUBBLE_PADDING, by + BUBBLE_PADDING), speaker, font=small_font, fill="#444444")
draw.text(
(x_offset + BUBBLE_PADDING, by + BUBBLE_PADDING + spk_h),
wrapped,
font=font,
fill="black",
)
img.save(img_path)
+10
View File
@@ -0,0 +1,10 @@
ebooklib
beautifulsoup4
requests
torch
diffusers
transformers
accelerate
pillow
tqdm
BIN
View File
Binary file not shown.

After

Width:  |  Height:  |  Size: 2.4 MiB

BIN
View File
Binary file not shown.
+8
View File
@@ -0,0 +1,8 @@
import os
def make_output_name(path, override=None):
if override:
return override
base = os.path.splitext(os.path.basename(path))[0]
return f"manga-{base}.epub"