commit 73d16c9b690795bee7106685d4aa1134d952af69 Author: Justin Oros Date: Mon Apr 13 18:03:33 2026 -0700 Initial commit diff --git a/README.md b/README.md new file mode 100644 index 0000000..dadece2 --- /dev/null +++ b/README.md @@ -0,0 +1,112 @@ +

+ epub-to-manga screenshot +

+ +# epub-to-manga + +Convert EPUB novels into manga-style EPUBs using a local LLM for scene parsing and Stable Diffusion for image generation. + +## How it works + +1. Reads your EPUB and splits the text into scenes +2. Uses an Ollama LLM to parse each scene — extracting characters, dialogue, mood, and setting +3. Generates a manga-style illustration for each page via Stable Diffusion +4. Adds speech bubbles with character dialogue +5. Packages everything into a new EPUB + +## Requirements + +- Python 3.10+ +- [Ollama](https://ollama.com) running locally +- One of the following Stable Diffusion backends: + - [AUTOMATIC1111 stable-diffusion-webui](https://github.com/AUTOMATIC1111/stable-diffusion-webui) + - [ComfyUI](https://github.com/comfyanonymous/ComfyUI) *(recommended for Apple Silicon)* + - [InvokeAI](https://github.com/invoke-ai/InvokeAI) + +## Installation + +```bash +pip install -r requirements.txt +``` + +Pull an Ollama model: + +```bash +ollama pull llama3 +``` + +## Usage + +```bash +python3 main.py book.epub +python3 main.py book.epub output.epub +python3 main.py book.epub --layout tiny --model mistral +python3 main.py book.epub --sd-url http://192.168.1.10:7860 --steps 30 --size 512x768 +python3 main.py book.epub --dry-run +``` + +If no Stable Diffusion backend is running, the tool will detect any local installations and offer to launch or install one for you. + +## Options + +| Flag | Default | Description | +|---|---|---| +| `--layout` | `normal` | `normal` = 3 scenes/page, `tiny` = 1 scene/page | +| `--model` | `llama3` | Ollama model for scene parsing | +| `--ollama-url` | `http://localhost:11434` | Ollama API URL | +| `--sd-url` | `http://127.0.0.1:7860` | Stable Diffusion API URL | +| `--sd-path` | — | Path to SD install; auto-starts if not running | +| `--steps` | `20` | Diffusion steps per image (higher = better quality) | +| `--style` | `lineart` | `lineart` = clean outlines, `manga` = screentone shading | +| `--size` | `768x1024` | Output image dimensions | +| `--workers` | `2` | Parallel image generation workers | +| `--hint` | — | Character description, e.g. `--hint "Rocky:alien who resembles a rock spider"` | +| `--jpeg-quality` | `85` | EPUB image quality (1–95, lower = smaller file) | +| `--dry-run [N]` | — | Preview first N prompts without generating images | +| `--verbose` / `-v` | — | Enable debug logging | + +## Configuration + +Default URLs and model can be changed in `config.py`: + +```python +OLLAMA_URL = "http://localhost:11434/api/generate" +OLLAMA_MODEL = "llama3" +SD_API_URL = "http://127.0.0.1:7860/sdapi/v1/txt2img" +``` + +The tool saves its SD backend path to `~/.epub-to-manga` so you don't need `--sd-path` on subsequent runs. + +## Resuming + +Scene parsing and image generation both support resuming. Parsed scenes are cached to `manga-.cache.json`, and generated page images are saved to `_pages/`. Re-running the same command will pick up where it left off. + +## Project structure + +``` +epub-to-manga/ +├── main.py # Entry point +├── config.py # Default URLs and model +├── requirements.txt +├── core/ +│ ├── epub_reader.py # EPUB text extraction +│ └── scene_splitter.py # Text → scenes +├── ai/ +│ ├── llm_client.py # Ollama API client +│ ├── scene_parser.py # Scene → structured JSON +│ └── character_memory.py # Character tracking across scenes +├── manga/ +│ ├── prompt_builder.py # SD prompt construction +│ ├── panel_layout.py # Scene grouping into pages +│ └── speech_bubbles.py # Dialogue overlay rendering +├── image/ +│ ├── backend.py # Backend router (A1111 / ComfyUI) +│ ├── a1111_api.py # AUTOMATIC1111 API +│ ├── comfy_api.py # ComfyUI API + workflow builder +│ ├── sd_launcher.py # SD auto-start and install +│ └── device.py # Device detection +├── export/ +│ └── epub_builder.py # Output EPUB assembly +└── utils/ + └── naming.py # Output filename helpers +``` diff --git a/ai/.DS_Store b/ai/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/ai/.DS_Store differ diff --git a/ai/character_memory.py b/ai/character_memory.py new file mode 100644 index 0000000..1d946a7 --- /dev/null +++ b/ai/character_memory.py @@ -0,0 +1,37 @@ +CHARACTER_DB = {} + +def load(data): + global CHARACTER_DB + CHARACTER_DB = data if isinstance(data, dict) else {} + +def dump(): + return dict(CHARACTER_DB) + +def update(name, description=""): + if name not in CHARACTER_DB: + CHARACTER_DB[name] = {"appearances": 0, "description": description} + else: + if description and not CHARACTER_DB[name].get("description"): + CHARACTER_DB[name]["description"] = description + CHARACTER_DB[name]["appearances"] += 1 + +def add_hint(name, description): + if name not in CHARACTER_DB: + CHARACTER_DB[name] = {"appearances": 0, "description": description} + else: + CHARACTER_DB[name]["description"] = description + +def get_all(): + return list(CHARACTER_DB.keys()) + +def build_consistency_tokens(characters): + parts = [] + for c in characters: + if not c or not c.strip(): + continue + desc = CHARACTER_DB.get(c, {}).get("description", "") + if desc: + parts.append(f"{c} ({desc})") + else: + parts.append(c) + return ", ".join(parts) diff --git a/ai/llm_client.py b/ai/llm_client.py new file mode 100644 index 0000000..218f2ef --- /dev/null +++ b/ai/llm_client.py @@ -0,0 +1,36 @@ +import requests +import logging +import time +import json +from config import OLLAMA_URL, OLLAMA_MODEL + +log = logging.getLogger(__name__) + +def call_llm(prompt, timeout=90): + deadline = time.time() + timeout + try: + with requests.post( + OLLAMA_URL, + json={"model": OLLAMA_MODEL, "prompt": prompt, "stream": True, "format": "json"}, + stream=True, + timeout=30, + ) as r: + r.raise_for_status() + chunks = [] + for line in r.iter_lines(): + if time.time() > deadline: + log.warning("LLM call exceeded wall-clock timeout of %ds", timeout) + break + if not line: + continue + try: + data = json.loads(line) + chunks.append(data.get("response", "")) + if data.get("done"): + break + except json.JSONDecodeError: + continue + return "".join(chunks) + except requests.RequestException as e: + log.error("LLM call failed: %s", e) + return "" diff --git a/ai/scene_parser.py b/ai/scene_parser.py new file mode 100644 index 0000000..3c41405 --- /dev/null +++ b/ai/scene_parser.py @@ -0,0 +1,88 @@ +import json +import logging +from ai.llm_client import call_llm + +log = logging.getLogger(__name__) + +PROMPT_TEMPLATE = ( + 'Analyze the scene below and return a JSON object with exactly these keys:\n' + '"characters": array of proper name strings only, no pronouns\n' + '"dialogue": array of objects with "speaker" (proper name or "?") and "line" (spoken words only) for every quoted line\n' + '"visual_scene": string, one sentence describing the physical action to illustrate\n' + '"mood": string, exactly one of: neutral happy sad tense angry mysterious romantic action\n' + '"setting": string, brief location name\n\n' + 'Scene:\n{scene}' +) + +JUNK_SPEAKERS = { + "i", "me", "he", "she", "they", "we", "you", "it", + "him", "her", "them", "his", "hers", "their", + "narrator", "voice", "unknown", "someone", "anyone", +} + +def _empty(scene): + return { + "characters": [], + "dialogue": [], + "visual_scene": scene[:120].strip(), + "mood": "neutral", + "setting": "", + } + +def _clean_dialogue(raw): + if not isinstance(raw, list): + return [] + cleaned = [] + for entry in raw: + if not isinstance(entry, dict): + continue + line = entry.get("line") or entry.get("text") or entry.get("speech") or "" + speaker = entry.get("speaker") or entry.get("name") or "?" + line = str(line).strip().strip('"') + speaker = str(speaker).strip() + if not line: + continue + if len(speaker.split()) > 3 or speaker.endswith(('.', ',')): + speaker = "?" + if speaker.lower() in JUNK_SPEAKERS: + speaker = "?" + cleaned.append({"speaker": speaker, "line": line}) + return cleaned + +def _extract(result, scene): + if not isinstance(result, dict): + log.warning("LLM returned non-dict JSON: %s", str(result)[:200]) + return _empty(scene) + out = _empty(scene) + out["characters"] = result.get("characters") or [] + out["dialogue"] = _clean_dialogue(result.get("dialogue", [])) + out["visual_scene"] = result.get("visual_scene") or scene[:120].strip() + out["mood"] = result.get("mood") or "neutral" + out["setting"] = result.get("setting") or "" + for key in ("scene", "response", "output", "result"): + if key in result and isinstance(result[key], dict): + log.warning("LLM wrapped response under key '%s', unwrapping", key) + return _extract(result[key], scene) + return out + +def parse_scene(scene, timeout=90): + prompt = PROMPT_TEMPLATE.format(scene=scene) + raw = call_llm(prompt, timeout=timeout) + if not raw: + log.warning("Empty LLM response") + return _empty(scene) + try: + result = json.loads(raw) + return _extract(result, scene) + except json.JSONDecodeError as e: + log.warning("JSON decode failed: %s — raw: %s", e, raw[:300]) + return _empty(scene) + +def parse_in_batches(scenes, batch_size=1, progress_cb=None, timeout=90): + results = [] + for i, scene in enumerate(scenes): + result = parse_scene(scene, timeout=timeout) + results.append(result) + if progress_cb: + progress_cb(i + 1, len(scenes)) + return results diff --git a/config.py b/config.py new file mode 100644 index 0000000..3c018c4 --- /dev/null +++ b/config.py @@ -0,0 +1,3 @@ +OLLAMA_URL = "http://localhost:11434/api/generate" +OLLAMA_MODEL = "llama3" +SD_API_URL = "http://127.0.0.1:7860/sdapi/v1/txt2img" diff --git a/core/.DS_Store b/core/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/core/.DS_Store differ diff --git a/core/epub_reader.py b/core/epub_reader.py new file mode 100644 index 0000000..bc38873 --- /dev/null +++ b/core/epub_reader.py @@ -0,0 +1,25 @@ +from ebooklib import epub +from ebooklib import ITEM_DOCUMENT +from bs4 import BeautifulSoup + +CHAPTER_TAGS = {"h1", "h2", "h3", "h4"} + +def read_epub(path): + book = epub.read_epub(path) + chapters = [] + for item in book.get_items(): + if item.get_type() == ITEM_DOCUMENT: + soup = BeautifulSoup(item.get_content(), "html.parser") + chapters.append(soup.get_text()) + return chapters + +def read_epub_flat(path): + return "\n\n".join(read_epub(path)) + +def read_epub_metadata(path): + book = epub.read_epub(path) + title = book.get_metadata('DC', 'title') + author = book.get_metadata('DC', 'creator') + title_str = title[0][0] if title else None + author_str = author[0][0] if author else None + return title_str, author_str diff --git a/core/scene_splitter.py b/core/scene_splitter.py new file mode 100644 index 0000000..cff6e14 --- /dev/null +++ b/core/scene_splitter.py @@ -0,0 +1,62 @@ +import re + +JUNK_PATTERNS = [ + r'all rights reserved', + r'copyright\s*©', + r'isbn[\s\-]', + r'published by', + r'first published', + r'printed in', + r'no part of this', + r'table of contents', + r'this is a work of fiction', + r'any resemblance to', +] +JUNK_RE = re.compile('|'.join(JUNK_PATTERNS), re.IGNORECASE) + +def _is_junk(text): + if len(text.split()) < 30: + return True + if JUNK_RE.search(text): + return True + return False + +def split_scenes(chapters, min_sentences=3, max_sentences=6): + scenes = [] + for chapter_text in chapters: + paragraphs = [p.strip() for p in re.split(r'\n{2,}', chapter_text) if p.strip()] + buf = [] + buf_sentences = 0 + + for para in paragraphs: + sentences = re.split(r'(?<=[.!?])\s+', para.strip()) + sentences = [s for s in sentences if s] + + for sentence in sentences: + buf.append(sentence) + buf_sentences += 1 + + if buf_sentences >= max_sentences: + candidate = " ".join(buf) + if not _is_junk(candidate): + scenes.append(candidate) + buf = [] + buf_sentences = 0 + + if buf_sentences >= min_sentences: + candidate = " ".join(buf) + if not _is_junk(candidate): + scenes.append(candidate) + buf = [] + buf_sentences = 0 + + if buf: + candidate = " ".join(buf) + if scenes and _is_junk(candidate): + pass + elif scenes: + scenes[-1] = scenes[-1] + " " + candidate + elif not _is_junk(candidate): + scenes.append(candidate) + + return [s for s in scenes if len(s.strip()) > 20] diff --git a/export/.DS_Store b/export/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/export/.DS_Store differ diff --git a/export/epub_builder.py b/export/epub_builder.py new file mode 100644 index 0000000..e91aa63 --- /dev/null +++ b/export/epub_builder.py @@ -0,0 +1,50 @@ +from ebooklib import epub +from PIL import Image +import io +import os + +def _to_jpeg(img_path, quality=85): + img = Image.open(img_path).convert("L").convert("RGB") + buf = io.BytesIO() + img.save(buf, format="JPEG", quality=quality, optimize=True) + return buf.getvalue() + +def build_epub(images, output, title="Manga Book", author="Manga", jpeg_quality=85): + book = epub.EpubBook() + book.set_title(title) + book.set_language("en") + book.add_author(author) + + chapters = [] + + for i, img_path in enumerate(images): + if not os.path.exists(img_path): + continue + + img_data = _to_jpeg(img_path, quality=jpeg_quality) + img_name = f"images/page_{i}.jpg" + + epub_img = epub.EpubItem( + uid=f"img_{i}", + file_name=img_name, + media_type="image/jpeg", + content=img_data, + ) + book.add_item(epub_img) + + c = epub.EpubHtml(title=f"Page {i+1}", file_name=f"p{i}.xhtml", lang="en") + c.content = ( + f'' + f'' + f'' + ) + book.add_item(c) + chapters.append(c) + + book.toc = tuple(chapters) + book.spine = ["nav"] + chapters + book.add_item(epub.EpubNcx()) + book.add_item(epub.EpubNav()) + + epub.write_epub(output, book) + return output diff --git a/image/.DS_Store b/image/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/image/.DS_Store differ diff --git a/image/a1111_api.py b/image/a1111_api.py new file mode 100644 index 0000000..e632dbe --- /dev/null +++ b/image/a1111_api.py @@ -0,0 +1,22 @@ + +import requests +import base64 + +def generate_image(base_url, positive, negative, steps=20, width=768, height=1024, cfg=4.5): + r = requests.post(f"{base_url}/sdapi/v1/txt2img", json={ + "prompt": positive, + "negative_prompt": negative, + "steps": steps, + "width": width, + "height": height, + "cfg_scale": cfg, + }) + r.raise_for_status() + return base64.b64decode(r.json()["images"][0]) + +def is_ready(base_url, timeout=3): + try: + requests.get(f"{base_url}/sdapi/v1/options", timeout=timeout) + return True + except Exception: + return False diff --git a/image/backend.py b/image/backend.py new file mode 100644 index 0000000..e109899 --- /dev/null +++ b/image/backend.py @@ -0,0 +1,70 @@ +import os +import sys +from image.sd_launcher import read_config, write_config + +def get_backend_type(sd_path): + if not sd_path: + return None + if os.path.exists(os.path.join(sd_path, "webui.sh")): + return "a1111" + if os.path.exists(os.path.join(sd_path, "comfy_extras")) or "ComfyUI" in sd_path: + return "comfyui" + if "InvokeAI" in sd_path: + return "invokeai" + return "a1111" + +def get_base_url(backend_type): + if backend_type == "comfyui": + return "http://127.0.0.1:8188" + return "http://127.0.0.1:7860" + +def is_ready(backend_type, base_url): + if backend_type == "comfyui": + from image.comfy_api import is_ready as comfy_ready + return comfy_ready() + from image.a1111_api import is_ready as a1111_ready + return a1111_ready(base_url) + +def ensure_model(backend_type, sd_path, style="manga"): + if backend_type == "comfyui": + from image import comfy_api + model = comfy_api.ensure_model_for_style(sd_path, style=style) + write_config("comfy-model", model) + return model + return None + +def setup(sd_path): + backend_type = get_backend_type(sd_path) + write_config("sd-backend", backend_type) + if backend_type == "comfyui": + from image import comfy_api + comfy_api.install_dependencies(sd_path) + return backend_type + +def _force_bw(img_bytes): + from PIL import Image + import io + img = Image.open(io.BytesIO(img_bytes)).convert("L").convert("RGB") + buf = io.BytesIO() + img.save(buf, format="PNG") + return buf.getvalue() + +def generate_image(positive, negative, steps=20, width=768, height=1024, cfg=4.5, force_bw=True, style="manga"): + config = read_config() + sd_path = config.get("sd-path", "") + backend_type = config.get("sd-backend") or get_backend_type(sd_path) + base_url = get_base_url(backend_type) + + if backend_type == "comfyui": + from image import comfy_api + model = comfy_api.ensure_model_for_style(sd_path, style=style) + write_config("comfy-model", model) + if not model: + print("\nError: no model found. Re-run to trigger model download.") + sys.exit(1) + result = comfy_api.generate_image(base_url, positive, negative, model, steps, width, height, cfg) + return _force_bw(result) if force_bw else result + + from image import a1111_api + result = a1111_api.generate_image(base_url, positive, negative, steps, width, height, cfg) + return _force_bw(result) if force_bw else result diff --git a/image/comfy_api.py b/image/comfy_api.py new file mode 100644 index 0000000..5e394c3 --- /dev/null +++ b/image/comfy_api.py @@ -0,0 +1,188 @@ +import requests +import json +import uuid +import time +import os +import subprocess +import sys + +COMFY_PORT = 8188 +POLL_TIMEOUT = 300 + +MODELS = { + "lineart": { + "name": "Deliberate_v2.safetensors", + "url": "https://huggingface.co/XpucT/Deliberate/resolve/main/Deliberate_v2.safetensors", + }, + "manga": { + "name": "Deliberate_v2.safetensors", + "url": "https://huggingface.co/XpucT/Deliberate/resolve/main/Deliberate_v2.safetensors", + }, +} + +DEFAULT_MODEL_URL = MODELS["manga"]["url"] +DEFAULT_MODEL_NAME = MODELS["manga"]["name"] + +def base_url(host="127.0.0.1"): + return f"http://{host}:{COMFY_PORT}" + +def is_ready(host="127.0.0.1", timeout=3): + try: + requests.get(f"{base_url(host)}/system_stats", timeout=timeout) + return True + except Exception: + return False + +def find_model(comfy_path, name=None): + model_dir = os.path.join(comfy_path, "models", "checkpoints") + if not os.path.isdir(model_dir): + return None + if name: + return name if os.path.exists(os.path.join(model_dir, name)) else None + for f in os.listdir(model_dir): + if f.endswith(".safetensors") or f.endswith(".ckpt"): + return f + return None + +def download_model(comfy_path, style="manga"): + model_info = MODELS.get(style, MODELS["manga"]) + model_name = model_info["name"] + model_url = model_info["url"] + style_label = "Anything V5 (anime/manga lineart)" if style == "lineart" else "Deliberate v2 (general purpose)" + + model_dir = os.path.join(comfy_path, "models", "checkpoints") + os.makedirs(model_dir, exist_ok=True) + dest = os.path.join(model_dir, model_name) + + if os.path.exists(dest): + return model_name + + print(f"\n Model for --style {style} not found: {model_name}") + print(f" Recommended model: {style_label}") + print(f" Source: {model_url}") + confirm = input("\n Download it now? (~2GB) [Y/n] ").strip().lower() + if confirm in ("n", "no"): + print(" Aborted.") + sys.exit(0) + + print(f"\n Downloading {model_name}...") + + bar_width = 35 + with requests.get(model_url, stream=True) as r: + r.raise_for_status() + total = int(r.headers.get("content-length", 0)) + downloaded = 0 + with open(dest, "wb") as f: + for chunk in r.iter_content(chunk_size=1024 * 1024): + f.write(chunk) + downloaded += len(chunk) + if total: + pct = downloaded / total + filled = int(bar_width * pct) + bar = "\u2588" * filled + "\u2591" * (bar_width - filled) + mb_done = downloaded / 1024 / 1024 + mb_total = total / 1024 / 1024 + print(f"\r [{bar}] {mb_done:.0f}/{mb_total:.0f} MB ", end="", flush=True) + + print(f"\r Download complete: {dest} ") + return model_name + +def ensure_model_for_style(comfy_path, style="manga"): + model_info = MODELS.get(style, MODELS["manga"]) + model_name = model_info["name"] + existing = find_model(comfy_path, name=model_name) + if existing: + return existing + return download_model(comfy_path, style=style) + +def install_dependencies(comfy_path): + req_file = os.path.join(comfy_path, "requirements.txt") + stamp = os.path.join(comfy_path, ".deps_installed") + if not os.path.exists(req_file): + return + if os.path.exists(stamp): + req_mtime = os.path.getmtime(req_file) + stamp_mtime = os.path.getmtime(stamp) + if stamp_mtime >= req_mtime: + return + print(" Installing ComfyUI dependencies...") + subprocess.run( + [sys.executable, "-m", "pip", "install", "-r", req_file, "--quiet"], + check=True, + ) + open(stamp, "w").close() + +def build_workflow(positive, negative, model_name, steps=20, width=768, height=1024, cfg=4.5): + seed = int(time.time()) % 2**32 + return { + "3": { + "class_type": "KSampler", + "inputs": { + "seed": seed, + "steps": steps, + "cfg": cfg, + "sampler_name": "euler_ancestral", + "scheduler": "karras", + "denoise": 1.0, + "model": ["4", 0], + "positive": ["6", 0], + "negative": ["7", 0], + "latent_image": ["5", 0], + }, + }, + "4": { + "class_type": "CheckpointLoaderSimple", + "inputs": {"ckpt_name": model_name}, + }, + "5": { + "class_type": "EmptyLatentImage", + "inputs": {"width": width, "height": height, "batch_size": 1}, + }, + "6": { + "class_type": "CLIPTextEncode", + "inputs": {"text": positive, "clip": ["4", 1]}, + }, + "7": { + "class_type": "CLIPTextEncode", + "inputs": {"text": negative, "clip": ["4", 1]}, + }, + "8": { + "class_type": "VAEDecode", + "inputs": {"samples": ["3", 0], "vae": ["4", 2]}, + }, + "9": { + "class_type": "SaveImage", + "inputs": {"filename_prefix": "manga", "images": ["8", 0]}, + }, + } + +def generate_image(base_url_str, positive, negative, model_name, steps=20, width=768, height=1024, cfg=4.5): + client_id = str(uuid.uuid4()) + workflow = build_workflow(positive, negative, model_name, steps, width, height, cfg) + + r = requests.post(f"{base_url_str}/prompt", json={"prompt": workflow, "client_id": client_id}) + r.raise_for_status() + prompt_id = r.json()["prompt_id"] + + deadline = time.time() + POLL_TIMEOUT + while time.time() < deadline: + time.sleep(1) + hist = requests.get(f"{base_url_str}/history/{prompt_id}").json() + if prompt_id in hist: + outputs = hist[prompt_id]["outputs"] + for node_id, node_output in outputs.items(): + if "images" in node_output: + img_info = node_output["images"][0] + img_r = requests.get( + f"{base_url_str}/view", + params={ + "filename": img_info["filename"], + "subfolder": img_info.get("subfolder", ""), + "type": img_info["type"], + }, + ) + img_r.raise_for_status() + return img_r.content + break + + raise RuntimeError(f"ComfyUI: no images returned for prompt_id {prompt_id} within {POLL_TIMEOUT}s") diff --git a/image/device.py b/image/device.py new file mode 100644 index 0000000..159252d --- /dev/null +++ b/image/device.py @@ -0,0 +1,14 @@ + +import torch +import platform + +def detect_device(): + if torch.cuda.is_available(): + return "cuda" + if platform.system() == "Darwin": + try: + if torch.backends.mps.is_available(): + return "mps" + except: + pass + return "cpu" diff --git a/image/sd_launcher.py b/image/sd_launcher.py new file mode 100644 index 0000000..9bd0833 --- /dev/null +++ b/image/sd_launcher.py @@ -0,0 +1,337 @@ + +import os +import subprocess +import time +import threading +import sys +import requests + +_sd_process = None + +CONFIG_FILE = os.path.expanduser("~/.epub-to-manga") + +BACKENDS = [ + { + "name": "AUTOMATIC1111 Stable Diffusion WebUI", + "repo": "https://github.com/AUTOMATIC1111/stable-diffusion-webui", + "dir": "stable-diffusion-webui", + "type": "a1111", + "launch": ["bash", "webui.sh", "--api", "--nowebui"], + "ready_url": "http://127.0.0.1:7860/sdapi/v1/options", + }, + { + "name": "ComfyUI (lighter, faster on Apple Silicon)", + "repo": "https://github.com/comfyanonymous/ComfyUI", + "dir": "ComfyUI", + "type": "comfyui", + "launch": [sys.executable, "main.py", "--listen"], + "ready_url": "http://127.0.0.1:8188/system_stats", + }, + { + "name": "InvokeAI", + "repo": "https://github.com/invoke-ai/InvokeAI", + "dir": "InvokeAI", + "type": "invokeai", + "launch": ["invokeai-web"], + "ready_url": "http://127.0.0.1:9090/api/v1/app/version", + }, +] + +KNOWN_ERRORS = [ + { + "markers": ["No module named 'pkg_resources'", "Couldn't install clip"], + "message": "AUTOMATIC1111 is not compatible with Python {python_version}. A1111 requires Python 3.10.", + "solutions": [ + { + "label": "Install Python 3.10 via pyenv and relaunch A1111", + "action": "fix_a1111_python", + }, + { + "label": "Switch to ComfyUI (better Apple Silicon support)", + "action": "switch_backend", + "backend_index": 1, + }, + ], + }, + { + "markers": ["CUDA out of memory", "out of memory"], + "message": "GPU ran out of memory.", + "solutions": [ + { + "label": "Re-run with a smaller image size (--size 256x384)", + "action": "suggest_flag", + "flag": "--size 256x384", + }, + ], + }, +] + +def read_config(): + config = {} + if os.path.exists(CONFIG_FILE): + with open(CONFIG_FILE) as f: + for line in f: + line = line.strip() + if "=" in line and not line.startswith("#"): + key, _, val = line.partition("=") + config[key.strip()] = val.strip() + return config + +def write_config(key, value): + config = read_config() + config[key] = value + with open(CONFIG_FILE, "w") as f: + for k, v in config.items(): + f.write(f"{k}={v}\n") + +def get_backend_for_path(sd_path): + if not sd_path: + return BACKENDS[0] + for b in BACKENDS: + if b["dir"] in sd_path or os.path.exists(os.path.join(sd_path, b["dir"])): + return b + if b["type"] == "a1111" and os.path.exists(os.path.join(sd_path, "webui.sh")): + return b + if b["type"] == "comfyui" and os.path.exists(os.path.join(sd_path, "comfy_extras")): + return b + return BACKENDS[0] + +def is_running(ready_url, timeout=3): + try: + requests.get(ready_url, timeout=timeout) + return True + except Exception: + return False + +def find_installed(): + home = os.path.expanduser("~") + search_dirs = [home, os.path.join(home, "Downloads"), os.path.join(home, "Documents"), os.getcwd()] + for backend in BACKENDS: + for base in search_dirs: + candidate = os.path.join(base, backend["dir"]) + if os.path.isdir(candidate): + return candidate, backend + return None, None + +def get_python_version(): + return f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}" + +def handle_known_error(error_def, sd_path, backend): + python_version = get_python_version() + msg = error_def["message"].format(python_version=python_version) + solutions = error_def["solutions"] + + print(f"\n\n ⚠️ {msg}\n") + print(" Solutions:\n") + for i, s in enumerate(solutions, 1): + print(f" {i}) {s['label']}") + print() + + while True: + choice = input(" Choose a solution (or q to quit): ").strip().lower() + if choice == "q": + sys.exit(0) + if choice.isdigit() and 1 <= int(choice) <= len(solutions): + solution = solutions[int(choice) - 1] + break + print(f" Enter a number between 1 and {len(solutions)}") + + action = solution["action"] + + if action == "fix_a1111_python": + print("\n Installing pyenv and Python 3.10.14...") + subprocess.run(["brew", "install", "pyenv"], check=False) + subprocess.run(["pyenv", "install", "3.10.14"], check=False) + pyenv_python = os.path.expanduser("~/.pyenv/versions/3.10.14/bin/python3") + if not os.path.exists(pyenv_python): + print("\n Error: pyenv install failed. See https://github.com/pyenv/pyenv") + sys.exit(1) + env = os.environ.copy() + env["PYTHON"] = pyenv_python + print(f"\n Relaunching A1111 with Python 3.10.14...") + global _sd_process + _sd_process = subprocess.Popen( + ["bash", "webui.sh", "--api", "--nowebui"], + cwd=sd_path, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + env=env, + ) + stream_and_wait(backend, sd_path) + + elif action == "switch_backend": + new_backend = BACKENDS[solution["backend_index"]] + dest = os.path.join(os.path.expanduser("~"), new_backend["dir"]) + if os.path.isdir(dest): + print(f"\n Found existing {new_backend['name']} at {dest} — using it.") + else: + confirm = input(f"\n Install {new_backend['name']} to {dest}? [Y/n] ").strip().lower() + if confirm in ("n", "no"): + print(" Aborted.") + sys.exit(0) + print(f"\n Cloning {new_backend['repo']}...") + result = subprocess.run(["git", "clone", "--recursive", new_backend["repo"], dest], check=False) + if result.returncode != 0: + print("\n Error: git clone failed.") + sys.exit(1) + print(f"\n Installed to {dest}") + write_config("sd-path", dest) + write_config("sd-backend", new_backend["type"]) + print(f" Re-run your original command — switching will take effect automatically.") + sys.exit(0) + + elif action == "suggest_flag": + print(f"\n Re-run with: {solution['flag']}") + sys.exit(0) + +def stream_logs(proc, ready_event, error_detected, sd_path, backend): + interesting = [ + "Loading", "Downloading", "Installing", "Running", "Creating", + "Model loaded", "Starting", "Applying", "torch", "CUDA", "MPS", + "checkpoint", "venv", "pip", "Startup", "listen", + ] + + for raw in proc.stdout: + if ready_event.is_set(): + break + line = raw.decode("utf-8", errors="replace").rstrip() + + for err_def in KNOWN_ERRORS: + if any(m in line for m in err_def["markers"]): + ready_event.set() + error_detected["def"] = err_def + return + + if any(kw in line for kw in interesting): + print(f"\r > {line[:100]:<100}") + print(" ", end="", flush=True) + +def stream_and_wait(backend, sd_path): + global _sd_process + ready_event = threading.Event() + error_detected = {} + + log_thread = threading.Thread( + target=stream_logs, + args=(_sd_process, ready_event, error_detected, sd_path, backend), + daemon=True, + ) + log_thread.start() + + start = time.time() + while not ready_event.is_set(): + elapsed = int(time.time() - start) + mins, secs = divmod(elapsed, 60) + elapsed_str = f"{mins}m {secs:02d}s" if mins else f"{secs}s" + print(f"\r Waiting for API... {elapsed_str} elapsed ", end="", flush=True) + + if is_running(backend["ready_url"]): + ready_event.set() + elapsed = int(time.time() - start) + mins, secs = divmod(elapsed, 60) + elapsed_str = f"{mins}m {secs:02d}s" if mins else f"{secs}s" + print(f"\r Ready in {elapsed_str}! \n") + return + + if _sd_process.poll() is not None and not ready_event.is_set(): + ready_event.set() + break + + time.sleep(2) + + log_thread.join(timeout=2) + + if "def" in error_detected: + handle_known_error(error_detected["def"], sd_path, backend) + elif not is_running(backend["ready_url"]): + print("\n\n Error: SD process exited unexpectedly.") + sys.exit(1) + +def launch(sd_path, backend): + global _sd_process + print(f"\nStarting {backend['name']} from {sd_path}...") + + _sd_process = subprocess.Popen( + backend["launch"], + cwd=sd_path, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + ) + stream_and_wait(backend, sd_path) + +def prompt_install(): + print("\nNo Stable Diffusion backend found on your system.") + print("\nAvailable backends to install:\n") + for i, b in enumerate(BACKENDS, 1): + print(f" {i}) {b['name']}") + print(f" {b['repo']}") + print() + + while True: + choice = input("Enter number to install (or q to quit): ").strip().lower() + if choice == "q": + sys.exit(0) + if choice.isdigit() and 1 <= int(choice) <= len(BACKENDS): + backend = BACKENDS[int(choice) - 1] + break + print(f" Please enter a number between 1 and {len(BACKENDS)}") + + dest = os.path.join(os.path.expanduser("~"), backend["dir"]) + confirm = input(f"\nInstall {backend['name']} to {dest}? [Y/n] ").strip().lower() + if confirm in ("n", "no"): + print("Aborted.") + sys.exit(0) + + if os.path.isdir(dest): + print(f"\nFound existing {backend['name']} at {dest} — using it.") + else: + print(f"\nCloning {backend['repo']}...") + result = subprocess.run(["git", "clone", "--recursive", backend["repo"], dest], check=False) + if result.returncode != 0: + print("\nError: git clone failed. Is git installed?") + sys.exit(1) + print(f"\nInstalled to {dest}") + + write_config("sd-path", dest) + write_config("sd-backend", backend["type"]) + print(f"Saved to {CONFIG_FILE}") + print(f"\nRe-run your original command — no --sd-path needed.") + sys.exit(0) + +def ensure_running(sd_url=None, sd_path=None, style="manga"): + config = read_config() + + if not sd_path: + sd_path = config.get("sd-path") + + backend = get_backend_for_path(sd_path) + + if is_running(backend["ready_url"]): + return + + if sd_path and os.path.isdir(sd_path): + from image import backend as backend_mod + backend_mod.setup(sd_path) + backend_mod.ensure_model(backend["type"], sd_path, style=style) + launch(sd_path, backend) + return + + installed_path, found_backend = find_installed() + if installed_path: + print(f"\nFound {found_backend['name']} at {installed_path}") + write_config("sd-path", installed_path) + write_config("sd-backend", found_backend["type"]) + print(f"Saved to {CONFIG_FILE}") + from image import backend as backend_mod + backend_mod.setup(installed_path) + backend_mod.ensure_model(found_backend["type"], installed_path, style=style) + launch(installed_path, found_backend) + return + + prompt_install() + +def shutdown(): + global _sd_process + if _sd_process: + _sd_process.terminate() + _sd_process = None diff --git a/main.py b/main.py new file mode 100644 index 0000000..82f166a --- /dev/null +++ b/main.py @@ -0,0 +1,333 @@ +import sys +import argparse +import logging +import time +import os +import json +import requests +import concurrent.futures +from core.epub_reader import read_epub, read_epub_metadata +from core.scene_splitter import split_scenes +from manga.panel_layout import group_into_pages +from manga.prompt_builder import build_page_prompt, get_cfg +from manga.speech_bubbles import add_speech_bubbles +from ai.scene_parser import parse_scene +from ai.character_memory import update as update_character, load as load_character_db, dump as dump_character_db, add_hint +from image.backend import generate_image +from image.sd_launcher import ensure_running, shutdown +import atexit +from export.epub_builder import build_epub +from utils.naming import make_output_name +import config + +logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(name)s: %(message)s") +log = logging.getLogger(__name__) + +def fmt_time(seconds): + seconds = int(seconds) + if seconds < 60: + return f"{seconds}s" + m, s = divmod(seconds, 60) + if m < 60: + return f"{m}m {s:02d}s" + h, m = divmod(m, 60) + return f"{h}h {m:02d}m" + +def progress_bar(current, total, label="", eta_str="", width=35): + pct = current / total if total else 0 + filled = int(width * pct) + bar = "█" * filled + "░" * (width - filled) + eta = f" ETA {eta_str}" if eta_str else "" + print(f"\r [{bar}] {current}/{total} {label}{eta} ", end="", flush=True) + +def parse_args(): + parser = argparse.ArgumentParser( + prog="main.py", + description="Convert an EPUB novel into a manga-style EPUB using LLM scene parsing and Stable Diffusion.", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=""" +examples: + python3 main.py book.epub + python3 main.py book.epub output.epub + python3 main.py book.epub --layout tiny --model mistral + python3 main.py book.epub --sd-url http://192.168.1.10:7860 --steps 30 --size 512x768 + python3 main.py book.epub --dry-run + +layout modes: + normal 3 scenes per page (default) + tiny 1 scene per page (more pages, more detail per image) + +requirements: + Ollama running at OLLAMA_URL (default: http://localhost:11434) + Stable Diffusion WebUI running at SD_API_URL (default: http://127.0.0.1:7860) + Both URLs can be overridden via flags or by editing config.py + """ + ) + + parser.add_argument("input", help="Path to input .epub file") + parser.add_argument("output", nargs="?", help="Path for output .epub (default: manga-.epub)") + parser.add_argument("--layout", choices=["normal", "tiny"], default="normal", + help="Page layout mode: normal=3 scenes/page, tiny=1 scene/page (default: normal)") + parser.add_argument("--model", default=config.OLLAMA_MODEL, metavar="MODEL", + help=f"Ollama model to use for scene parsing (default: {config.OLLAMA_MODEL})") + parser.add_argument("--ollama-url", default=config.OLLAMA_URL, metavar="URL", + help=f"Ollama API base URL (default: {config.OLLAMA_URL})") + parser.add_argument("--sd-url", default=config.SD_API_URL, metavar="URL", + help=f"Stable Diffusion WebUI API URL (default: {config.SD_API_URL})") + parser.add_argument("--sd-path", default=None, metavar="PATH", + help="Path to stable-diffusion-webui folder; if SD is not running, auto-starts it") + parser.add_argument("--steps", type=int, default=20, metavar="N", + help="Diffusion steps per image (default: 20, higher=better quality but slower)") + parser.add_argument("--style", choices=["lineart", "manga"], default="lineart", + help="Image style: lineart=simple clean outlines (default), manga=detailed screentone") + parser.add_argument("--size", default="768x1024", metavar="WxH", + help="Image dimensions in pixels (default: 768x1024)") + parser.add_argument("--workers", type=int, default=2, metavar="N", + help="Parallel image generation workers (default: 2)") + parser.add_argument("--hint", action="append", metavar="NAME:DESC", + help="Character description hint e.g. 'Rocky:alien who resembles a rock spider'. Can be used multiple times.") + parser.add_argument("--jpeg-quality", type=int, default=85, metavar="N", + help="JPEG quality for epub images 1-95 (default: 85, lower=smaller file)") + parser.add_argument("--dry-run", nargs="?", const=3, type=int, metavar="N", + help="Show N prompts without generating images (default: 3)") + parser.add_argument("--verbose", "-v", action="store_true", + help="Show debug logging") + + return parser.parse_args() + +def main(): + args = parse_args() + + if args.verbose: + logging.getLogger().setLevel(logging.DEBUG) + + config.OLLAMA_MODEL = args.model + config.OLLAMA_URL = args.ollama_url + from image.sd_launcher import write_config as _wc + _wc("llm-model", args.model) + config.SD_API_URL = args.sd_url + + atexit.register(shutdown) + ensure_running(args.sd_url, args.sd_path, style=args.style) + + try: + width, height = (int(x) for x in args.size.split("x")) + except ValueError: + print(f"Error: --size must be in WxH format, e.g. 768x1024") + sys.exit(1) + + if not os.path.isfile(args.input): + print(f"Error: input file not found: {args.input}") + sys.exit(1) + + output = args.output or make_output_name(args.input) + output_dir = os.path.splitext(output)[0] + "_pages" + os.makedirs(output_dir, exist_ok=True) + + total_start = time.time() + + print(f"\nepub-to-manga") + print(f" input: {args.input}") + print(f" output: {output}") + print(f" layout: {args.layout} | model: {args.model} | steps: {args.steps} | size: {width}x{height} | style: {args.style} | workers: {args.workers}") + + cache_file = make_output_name(args.input).replace(".epub", ".cache.json") + + print(f"\nReading & splitting epub...") + source_title, source_author = read_epub_metadata(args.input) + chapters = read_epub(args.input) + scenes = split_scenes(chapters) + print(f" {len(scenes)} scenes found across {len(chapters)} chapters") + + dry_run_n = args.dry_run if args.dry_run is not None else None + needed = dry_run_n if dry_run_n is not None else len(scenes) + + cached_scenes = [] + cached_chars = {} + if os.path.exists(cache_file): + with open(cache_file) as f: + cache_data = json.load(f) + cached_scenes = cache_data.get("parsed_scenes", []) + cached_chars = cache_data.get("character_db", {}) + + load_character_db(cached_chars) + + if args.hint: + for hint in args.hint: + if ":" in hint: + name, _, desc = hint.partition(":") + add_hint(name.strip(), desc.strip()) + print(f" Character hint: {name.strip()} = {desc.strip()}") + + if len(cached_scenes) >= needed: + print(f" Loaded {len(cached_scenes)} scenes from cache ({cache_file})") + parsed_scenes = cached_scenes + else: + if cached_scenes: + print(f" Resuming parse — {len(cached_scenes)}/{needed} scenes cached") + + remaining_scenes = scenes[len(cached_scenes):needed] + + print(f"\nParsing scenes with {args.model} ({needed} total)") + progress_bar(len(cached_scenes), needed) + run_start = time.time() + + new_parsed = [] + scene_times = [] + + def on_scene_done(batch_idx, batch_len): + pass + + for idx, scene in enumerate(remaining_scenes): + t0 = time.time() + result = parse_scene(scene, timeout=90) + elapsed_scene = time.time() - t0 + new_parsed.append(result) + scene_times.append(elapsed_scene) + if len(scene_times) > 10: + scene_times.pop(0) + + for char in result.get("characters", []): + if char and char.strip(): + update_character(char.strip(), "") + + combined = cached_scenes + new_parsed + with open(cache_file, "w") as f: + json.dump({"parsed_scenes": combined, "character_db": dump_character_db()}, f) + + done = len(cached_scenes) + len(new_parsed) + avg = sum(scene_times) / len(scene_times) + remaining_count = needed - done + eta = fmt_time(avg * remaining_count) if remaining_count > 0 else fmt_time(time.time() - run_start) + progress_bar(done, needed, eta_str=eta) + + print() + parsed_scenes = cached_scenes + new_parsed + print(f" Saved to {cache_file}") + + for ps in parsed_scenes: + for char in ps.get("characters", []): + if char and char.strip(): + update_character(char.strip(), "") + + pages = group_into_pages(parsed_scenes, mode=args.layout) + print(f" {len(pages)} pages ({args.layout} layout)") + + if args.dry_run is not None: + n = args.dry_run + print(f"\nDry run — first {n} prompts:") + for i, page in enumerate(pages[:n]): + positive, negative = build_page_prompt(page, style=args.style) + print(f"\n--- Page {i+1} ---\nPositive: {positive}\nNegative: {negative}") + return + + + from image.sd_launcher import read_config as _rc + _sd_cfg = _rc() + _sd_path = _sd_cfg.get("sd-path", "") + _backend = _sd_cfg.get("sd-backend", "comfyui") + _port = "8188" if _backend == "comfyui" else "7860" + if _backend == "comfyui": + from image import comfy_api + _model = comfy_api.ensure_model_for_style(_sd_path, style=args.style) + from image.sd_launcher import write_config as _wc2 + _wc2("comfy-model", _model) + + total = len(pages) + already_done = [ + os.path.join(output_dir, f"page_{i}.png") + for i in range(total) + if os.path.exists(os.path.join(output_dir, f"page_{i}.png")) + ] + resumed = len(already_done) + + print(f"\nGenerating images... (http://127.0.0.1:{_port})") + print(f" {total} pages | ~{args.steps} steps each | {width}x{height} | {args.workers} workers") + if resumed: + print(f" Resuming — {resumed}/{total} pages already done, skipping...") + + images = [None] * total + for i in range(total): + img_path = os.path.join(output_dir, f"page_{i}.png") + if os.path.exists(img_path): + images[i] = img_path + + progress_bar(resumed, total) + img_start = time.time() + times = [] + completed = resumed + lock = __import__("threading").Lock() + + def gen_page(args_tuple): + i, page = args_tuple + img_path = os.path.join(output_dir, f"page_{i}.png") + if os.path.exists(img_path): + return i, img_path, None + + positive, negative = build_page_prompt(page, style=args.style) + t0 = time.time() + try: + img_bytes = generate_image( + positive, negative, + steps=args.steps, width=width, height=height, + cfg=get_cfg(args.style), + force_bw=(args.style == "lineart"), + style=args.style + ) + elapsed = time.time() - t0 + + with open(img_path, "wb") as f: + f.write(img_bytes) + + dialogue = [] + if isinstance(page, list): + for scene in page: + if isinstance(scene, dict): + dialogue.extend(scene.get("dialogue", [])) + if dialogue: + add_speech_bubbles(img_path, dialogue) + + return i, img_path, elapsed + except (requests.RequestException, KeyError, IndexError, RuntimeError) as e: + return i, None, str(e) + + pending = [(i, page) for i, page in enumerate(pages) if images[i] is None] + + with concurrent.futures.ThreadPoolExecutor(max_workers=args.workers) as executor: + futures = {executor.submit(gen_page, item): item[0] for item in pending} + for future in concurrent.futures.as_completed(futures): + i, img_path, result = future.result() + with lock: + if img_path: + images[i] = img_path + if isinstance(result, float): + times.append(result) + if len(times) > 8: + times.pop(0) + else: + print(f"\n [!] Page {i} failed: {result}") + completed += 1 + avg = sum(times) / len(times) if times else 0 + remaining_count = total - completed + eta = fmt_time(avg * remaining_count / args.workers) if avg and remaining_count else "" + progress_bar(completed, total, eta_str=eta) + + print() + + final_images = [img for img in images if img] + + if not final_images: + print(f"\nError: no images generated. Is the SD API running at {args.sd_url}?") + sys.exit(1) + + print(f"\nBuilding epub...") + result = build_epub(final_images, output, title=source_title or "Manga Book", author="Manga", jpeg_quality=args.jpeg_quality) + + total_elapsed = time.time() - total_start + print(f" Done in {fmt_time(total_elapsed)}: {result}\n") + +if __name__ == "__main__": + try: + main() + except KeyboardInterrupt: + print("\n\nStopped. Progress saved - re-run to resume.") diff --git a/manga/.DS_Store b/manga/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/manga/.DS_Store differ diff --git a/manga/panel_layout.py b/manga/panel_layout.py new file mode 100644 index 0000000..3fe4319 --- /dev/null +++ b/manga/panel_layout.py @@ -0,0 +1,13 @@ +def group_into_pages(scenes, mode="normal"): + if mode == "tiny": + return [[s] for s in scenes] + pages = [] + temp = [] + for s in scenes: + temp.append(s) + if len(temp) == 3: + pages.append(temp) + temp = [] + if temp: + pages.append(temp) + return pages diff --git a/manga/prompt_builder.py b/manga/prompt_builder.py new file mode 100644 index 0000000..c17c91c --- /dev/null +++ b/manga/prompt_builder.py @@ -0,0 +1,137 @@ +import re +from ai.character_memory import build_consistency_tokens + +STYLES = { + "lineart": { + "positive": ( + "anime illustration, manga style, black and white, monochrome, " + "clean ink linework, expressive characters, dynamic composition, " + "professional manga art, single scene, full image" + ), + "negative": ( + "color, colorful, coloured, vibrant, saturated, " + "multiple panels, panel borders, panel grid, split panels, " + "collage, triptych, diptych, " + "realistic, photorealistic, photograph, 3d render, " + "ugly, blurry, watermark, text, signature, lowres, " + "bad anatomy, deformed, extra limbs" + ), + "cfg": 7.0, + }, + "manga": { + "positive": ( + "single full-page manga illustration, full-bleed scene, " + "black and white ink, screentone shading, " + "expressive faces, detailed backgrounds, " + "professional manga art, one continuous scene" + ), + "negative": ( + "multiple panels, panel borders, panel grid, comic layout, split panels, " + "panel dividers, gutters, multi-panel page, page layout, comic book grid, " + "collage, triptych, diptych, " + "lowres, bad anatomy, blurry, watermark, text, ugly, deformed, " + "color, coloured, western comic style" + ), + "cfg": 7.5, + }, +} + +MOOD_MAP = { + "tense": "tense dramatic scene, characters look worried or scared", + "sad": "sad melancholy scene, characters look downcast", + "happy": "happy cheerful scene, characters smiling", + "angry": "angry confrontational scene, characters look furious", + "mysterious": "mysterious eerie scene, shadowy atmosphere", + "neutral": "calm everyday scene", + "romantic": "romantic gentle scene, soft atmosphere", + "action": "action dynamic scene, characters in motion", + "excited": "excited energetic scene, characters enthusiastic", + "fearful": "fearful tense scene, characters look afraid", +} + +JUNK_NAMES = { + "he", "she", "they", "him", "her", "them", "his", "hers", "their", + "i", "me", "we", "us", "it", "you", "location", "unknown", "none", + "character", "person", "man", "woman", "boy", "girl", + "the man", "the woman", "the boy", "the girl", "the person", + "a man", "a woman", "old man", "young man", "young woman", + "his secretary", "his secretarys", "her secretary", + "narrator", "voice", "someone", "anyone", "everyone", +} + +PLACEHOLDER_SETTINGS = {"location", "unknown", "none", "'location'", '"location"', ""} + +POSSESSIVE_RE = re.compile(r"'s$|s'$", re.IGNORECASE) + +def clean_text(s): + s = s.replace("\\u0022", "").replace("\\u0027", "") + s = s.replace('\\"', "").replace("\\'", "") + s = re.sub(r'[\"\'`]', "", s) + s = re.sub(r"\s+", " ", s).strip() + return s + +def is_valid_name(name): + n = name.strip().lower() + if not n or len(n) < 2: + return False + if n in JUNK_NAMES: + return False + if POSSESSIVE_RE.search(n): + return False + return True + +def build_page_prompt(parsed_scenes, style="lineart"): + style_def = STYLES.get(style, STYLES["lineart"]) + + if not parsed_scenes: + return style_def["positive"], style_def["negative"] + + all_characters = [] + visual_parts = [] + settings = [] + moods = [] + + for scene in parsed_scenes: + if isinstance(scene, str): + visual_parts.append(scene) + continue + for char in scene.get("characters", []): + char = clean_text(char) + if is_valid_name(char): + all_characters.append(char) + visual = clean_text(scene.get("visual_scene", "")) + if visual: + visual_parts.append(visual) + setting = clean_text(scene.get("setting", "")) + if setting.lower() not in PLACEHOLDER_SETTINGS: + settings.append(setting) + mood = scene.get("mood", "neutral").strip().lower() + if mood: + moods.append(mood) + + unique_chars = list(dict.fromkeys(all_characters)) + char_tokens = build_consistency_tokens(unique_chars) + + dominant_mood = moods[0] if moods else "neutral" + mood_desc = MOOD_MAP.get(dominant_mood, MOOD_MAP["neutral"]) + + setting_str = settings[0][:60] if settings else "" + scene_str = visual_parts[0][:80] if visual_parts else "" + + style_prefix = ( + "(anime style:1.4), (manga illustration:1.3), " + "(black and white:1.3), (monochrome:1.2), " + ) if style == "lineart" else "" + + parts = [style_prefix + style_def["positive"], mood_desc] + if setting_str: + parts.append(setting_str) + if char_tokens: + parts.append(char_tokens) + if scene_str: + parts.append(scene_str) + + return ", ".join(parts), style_def["negative"] + +def get_cfg(style="lineart"): + return STYLES.get(style, STYLES["lineart"])["cfg"] diff --git a/manga/speech_bubbles.py b/manga/speech_bubbles.py new file mode 100644 index 0000000..e744b10 --- /dev/null +++ b/manga/speech_bubbles.py @@ -0,0 +1,115 @@ +import textwrap +from PIL import Image, ImageDraw, ImageFont +import os + +MAX_BUBBLES = 3 +MAX_LINE_WIDTH = 22 +BUBBLE_PADDING = 14 +FONT_SIZE = 22 +TAIL_SIZE = 14 + +def get_font(size=FONT_SIZE): + candidates = [ + "/System/Library/Fonts/Helvetica.ttc", + "/System/Library/Fonts/Arial.ttf", + "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf", + "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf", + ] + for path in candidates: + if os.path.exists(path): + try: + return ImageFont.truetype(path, size) + except Exception: + pass + return ImageFont.load_default() + +def wrap_text(text, max_width=MAX_LINE_WIDTH): + return textwrap.fill(text, max_width) + +def draw_bubble(draw, x, y, w, h, tail_side="bottom"): + r = 16 + draw.rounded_rectangle([x, y, x+w, y+h], radius=r, fill="white", outline="black", width=3) + + if tail_side == "bottom": + tx = x + w // 2 + ty = y + h + draw.polygon([ + (tx - TAIL_SIZE, ty - 4), + (tx + TAIL_SIZE, ty - 4), + (tx, ty + TAIL_SIZE), + ], fill="white") + draw.line([(tx - TAIL_SIZE, ty - 4), (tx, ty + TAIL_SIZE)], fill="black", width=3) + draw.line([(tx + TAIL_SIZE, ty - 4), (tx, ty + TAIL_SIZE)], fill="black", width=3) + elif tail_side == "top": + tx = x + w // 2 + ty = y + draw.polygon([ + (tx - TAIL_SIZE, ty + 4), + (tx + TAIL_SIZE, ty + 4), + (tx, ty - TAIL_SIZE), + ], fill="white") + draw.line([(tx - TAIL_SIZE, ty + 4), (tx, ty - TAIL_SIZE)], fill="black", width=3) + draw.line([(tx + TAIL_SIZE, ty + 4), (tx, ty - TAIL_SIZE)], fill="black", width=3) + +def add_speech_bubbles(img_path, dialogue): + if not dialogue: + return + + entries = [d for d in dialogue if d.get("line", "").strip()][:MAX_BUBBLES] + if not entries: + return + + img = Image.open(img_path).convert("RGB") + draw = ImageDraw.Draw(img) + font = get_font(FONT_SIZE) + small_font = get_font(FONT_SIZE - 6) + + iw, ih = img.size + margin = 18 + + zones = [ + ih // 8, + ih * 5 // 8, + ih // 4, + ] + + for i, entry in enumerate(entries): + speaker = entry.get("speaker", "").strip() + line = entry.get("line", "").strip() + if not line: + continue + + wrapped = wrap_text(line) + + bbox = draw.textbbox((0, 0), wrapped, font=font) + text_w = bbox[2] - bbox[0] + text_h = bbox[3] - bbox[1] + + if speaker and speaker != "?": + spk_bbox = draw.textbbox((0, 0), speaker, font=small_font) + spk_h = spk_bbox[3] - spk_bbox[1] + 4 + else: + spk_h = 0 + + bw = min(text_w + BUBBLE_PADDING * 2, iw - margin * 2) + bh = text_h + spk_h + BUBBLE_PADDING * 2 + + x_offset = margin if i % 2 == 0 else iw - bw - margin + by = zones[i % len(zones)] + + by = max(margin, min(by, ih - bh - margin)) + + tail = "bottom" if by < ih // 2 else "top" + draw_bubble(draw, x_offset, by, bw, bh, tail_side=tail) + + if speaker and speaker != "?": + draw.text((x_offset + BUBBLE_PADDING, by + BUBBLE_PADDING), speaker, font=small_font, fill="#444444") + + draw.text( + (x_offset + BUBBLE_PADDING, by + BUBBLE_PADDING + spk_h), + wrapped, + font=font, + fill="black", + ) + + img.save(img_path) diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..e488054 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,10 @@ + +ebooklib +beautifulsoup4 +requests +torch +diffusers +transformers +accelerate +pillow +tqdm diff --git a/screenshot.jpg b/screenshot.jpg new file mode 100644 index 0000000..f2457dd Binary files /dev/null and b/screenshot.jpg differ diff --git a/utils/.DS_Store b/utils/.DS_Store new file mode 100644 index 0000000..5008ddf Binary files /dev/null and b/utils/.DS_Store differ diff --git a/utils/naming.py b/utils/naming.py new file mode 100644 index 0000000..9d225a9 --- /dev/null +++ b/utils/naming.py @@ -0,0 +1,8 @@ + +import os + +def make_output_name(path, override=None): + if override: + return override + base = os.path.splitext(os.path.basename(path))[0] + return f"manga-{base}.epub"