Initial commit
This commit is contained in:
commit
73d16c9b69
26 files changed
+1660
No files matched your search
Vendored
BIN
Binary file not shown.
@@ -0,0 +1,37 @@
|
||||
CHARACTER_DB = {}
|
||||
|
||||
def load(data):
|
||||
global CHARACTER_DB
|
||||
CHARACTER_DB = data if isinstance(data, dict) else {}
|
||||
|
||||
def dump():
|
||||
return dict(CHARACTER_DB)
|
||||
|
||||
def update(name, description=""):
|
||||
if name not in CHARACTER_DB:
|
||||
CHARACTER_DB[name] = {"appearances": 0, "description": description}
|
||||
else:
|
||||
if description and not CHARACTER_DB[name].get("description"):
|
||||
CHARACTER_DB[name]["description"] = description
|
||||
CHARACTER_DB[name]["appearances"] += 1
|
||||
|
||||
def add_hint(name, description):
|
||||
if name not in CHARACTER_DB:
|
||||
CHARACTER_DB[name] = {"appearances": 0, "description": description}
|
||||
else:
|
||||
CHARACTER_DB[name]["description"] = description
|
||||
|
||||
def get_all():
|
||||
return list(CHARACTER_DB.keys())
|
||||
|
||||
def build_consistency_tokens(characters):
|
||||
parts = []
|
||||
for c in characters:
|
||||
if not c or not c.strip():
|
||||
continue
|
||||
desc = CHARACTER_DB.get(c, {}).get("description", "")
|
||||
if desc:
|
||||
parts.append(f"{c} ({desc})")
|
||||
else:
|
||||
parts.append(c)
|
||||
return ", ".join(parts)
|
||||
@@ -0,0 +1,36 @@
|
||||
import requests
|
||||
import logging
|
||||
import time
|
||||
import json
|
||||
from config import OLLAMA_URL, OLLAMA_MODEL
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
def call_llm(prompt, timeout=90):
|
||||
deadline = time.time() + timeout
|
||||
try:
|
||||
with requests.post(
|
||||
OLLAMA_URL,
|
||||
json={"model": OLLAMA_MODEL, "prompt": prompt, "stream": True, "format": "json"},
|
||||
stream=True,
|
||||
timeout=30,
|
||||
) as r:
|
||||
r.raise_for_status()
|
||||
chunks = []
|
||||
for line in r.iter_lines():
|
||||
if time.time() > deadline:
|
||||
log.warning("LLM call exceeded wall-clock timeout of %ds", timeout)
|
||||
break
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
data = json.loads(line)
|
||||
chunks.append(data.get("response", ""))
|
||||
if data.get("done"):
|
||||
break
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return "".join(chunks)
|
||||
except requests.RequestException as e:
|
||||
log.error("LLM call failed: %s", e)
|
||||
return ""
|
||||
@@ -0,0 +1,88 @@
|
||||
import json
|
||||
import logging
|
||||
from ai.llm_client import call_llm
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
PROMPT_TEMPLATE = (
|
||||
'Analyze the scene below and return a JSON object with exactly these keys:\n'
|
||||
'"characters": array of proper name strings only, no pronouns\n'
|
||||
'"dialogue": array of objects with "speaker" (proper name or "?") and "line" (spoken words only) for every quoted line\n'
|
||||
'"visual_scene": string, one sentence describing the physical action to illustrate\n'
|
||||
'"mood": string, exactly one of: neutral happy sad tense angry mysterious romantic action\n'
|
||||
'"setting": string, brief location name\n\n'
|
||||
'Scene:\n{scene}'
|
||||
)
|
||||
|
||||
JUNK_SPEAKERS = {
|
||||
"i", "me", "he", "she", "they", "we", "you", "it",
|
||||
"him", "her", "them", "his", "hers", "their",
|
||||
"narrator", "voice", "unknown", "someone", "anyone",
|
||||
}
|
||||
|
||||
def _empty(scene):
|
||||
return {
|
||||
"characters": [],
|
||||
"dialogue": [],
|
||||
"visual_scene": scene[:120].strip(),
|
||||
"mood": "neutral",
|
||||
"setting": "",
|
||||
}
|
||||
|
||||
def _clean_dialogue(raw):
|
||||
if not isinstance(raw, list):
|
||||
return []
|
||||
cleaned = []
|
||||
for entry in raw:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
line = entry.get("line") or entry.get("text") or entry.get("speech") or ""
|
||||
speaker = entry.get("speaker") or entry.get("name") or "?"
|
||||
line = str(line).strip().strip('"')
|
||||
speaker = str(speaker).strip()
|
||||
if not line:
|
||||
continue
|
||||
if len(speaker.split()) > 3 or speaker.endswith(('.', ',')):
|
||||
speaker = "?"
|
||||
if speaker.lower() in JUNK_SPEAKERS:
|
||||
speaker = "?"
|
||||
cleaned.append({"speaker": speaker, "line": line})
|
||||
return cleaned
|
||||
|
||||
def _extract(result, scene):
|
||||
if not isinstance(result, dict):
|
||||
log.warning("LLM returned non-dict JSON: %s", str(result)[:200])
|
||||
return _empty(scene)
|
||||
out = _empty(scene)
|
||||
out["characters"] = result.get("characters") or []
|
||||
out["dialogue"] = _clean_dialogue(result.get("dialogue", []))
|
||||
out["visual_scene"] = result.get("visual_scene") or scene[:120].strip()
|
||||
out["mood"] = result.get("mood") or "neutral"
|
||||
out["setting"] = result.get("setting") or ""
|
||||
for key in ("scene", "response", "output", "result"):
|
||||
if key in result and isinstance(result[key], dict):
|
||||
log.warning("LLM wrapped response under key '%s', unwrapping", key)
|
||||
return _extract(result[key], scene)
|
||||
return out
|
||||
|
||||
def parse_scene(scene, timeout=90):
|
||||
prompt = PROMPT_TEMPLATE.format(scene=scene)
|
||||
raw = call_llm(prompt, timeout=timeout)
|
||||
if not raw:
|
||||
log.warning("Empty LLM response")
|
||||
return _empty(scene)
|
||||
try:
|
||||
result = json.loads(raw)
|
||||
return _extract(result, scene)
|
||||
except json.JSONDecodeError as e:
|
||||
log.warning("JSON decode failed: %s — raw: %s", e, raw[:300])
|
||||
return _empty(scene)
|
||||
|
||||
def parse_in_batches(scenes, batch_size=1, progress_cb=None, timeout=90):
|
||||
results = []
|
||||
for i, scene in enumerate(scenes):
|
||||
result = parse_scene(scene, timeout=timeout)
|
||||
results.append(result)
|
||||
if progress_cb:
|
||||
progress_cb(i + 1, len(scenes))
|
||||
return results
|
||||
Reference in new issue
Block a user