Add Ollama output limits and timeout, source language filter, and kids theme filter
This commit is contained in:
1 parent
bedf9d937b
commit
9205e4614b
2 files changed
+60
-13
No files matched your search
@@ -43,6 +43,7 @@ SCOPES = [
|
|||||||
]
|
]
|
||||||
MASK = "********"
|
MASK = "********"
|
||||||
MIN_FREE_GB = 2
|
MIN_FREE_GB = 2
|
||||||
|
LATIN_LANGS = {"en", "es", "fr", "de", "it", "pt", "nl", "sv", "no", "da", "fi", "pl", "cs", "ro", "hu", "tr", "id", "ms", "vi", "tl"}
|
||||||
|
|
||||||
GLOBAL_DEFAULTS = {
|
GLOBAL_DEFAULTS = {
|
||||||
"active_profile": "default",
|
"active_profile": "default",
|
||||||
@@ -52,6 +53,7 @@ GLOBAL_DEFAULTS = {
|
|||||||
"ollama_url": "http://localhost:11434",
|
"ollama_url": "http://localhost:11434",
|
||||||
"ollama_model": "llama3.1:8b",
|
"ollama_model": "llama3.1:8b",
|
||||||
"vision_model": "llama3.2-vision",
|
"vision_model": "llama3.2-vision",
|
||||||
|
"ollama_timeout": 300,
|
||||||
"gemini_api_key": "",
|
"gemini_api_key": "",
|
||||||
"veo_model": "veo-3.0-fast-generate-001",
|
"veo_model": "veo-3.0-fast-generate-001",
|
||||||
"aspect_ratio": "9:16",
|
"aspect_ratio": "9:16",
|
||||||
@@ -71,6 +73,7 @@ PROFILE_DEFAULTS = {
|
|||||||
"trending_days": 7,
|
"trending_days": 7,
|
||||||
"category_id": "",
|
"category_id": "",
|
||||||
"region_code": "US",
|
"region_code": "US",
|
||||||
|
"source_language": "en",
|
||||||
"trending_count": 25,
|
"trending_count": 25,
|
||||||
"videos_per_day": 0,
|
"videos_per_day": 0,
|
||||||
"minutes_between_videos": 0,
|
"minutes_between_videos": 0,
|
||||||
@@ -98,7 +101,8 @@ Return JSON only: {"safe": true or false, "reason": "short explanation"}"""
|
|||||||
|
|
||||||
KIDS_BANNED = re.compile(
|
KIDS_BANNED = re.compile(
|
||||||
r"\b(kill\w*|blood\w*|bleed\w*|guns?|knife|knives|swords?|weapons?|bombs?|murder\w*|dead|die|dies|dying|death|"
|
r"\b(kill\w*|blood\w*|bleed\w*|guns?|knife|knives|swords?|weapons?|bombs?|murder\w*|dead|die|dies|dying|death|"
|
||||||
r"scary|terrif\w*|horror|creepy|nightmare\w*|sexy|kiss\w*|beer|wine|drunk|cigar\w*|smok\w*|drugs?|vape\w*|hate)\b",
|
r"scary|spooky|terrif\w*|horror|creepy|nightmare\w*|ghost\w*|haunt\w*|zombie\w*|demon\w*|devil\w*|bhoot|"
|
||||||
|
r"sexy|kiss\w*|beer|wine|drunk|cigar\w*|smok\w*|drugs?|vape\w*|hate)\b",
|
||||||
re.I,
|
re.I,
|
||||||
)
|
)
|
||||||
KIDS_CONTACT = re.compile(
|
KIDS_CONTACT = re.compile(
|
||||||
@@ -380,16 +384,41 @@ def map_videos(items):
|
|||||||
"channel": i["snippet"].get("channelTitle", ""),
|
"channel": i["snippet"].get("channelTitle", ""),
|
||||||
"description": i["snippet"].get("description", "")[:1500],
|
"description": i["snippet"].get("description", "")[:1500],
|
||||||
"tags": i["snippet"].get("tags", [])[:20],
|
"tags": i["snippet"].get("tags", [])[:20],
|
||||||
|
"language": (i["snippet"].get("defaultAudioLanguage") or i["snippet"].get("defaultLanguage") or "").lower(),
|
||||||
"views": int(i.get("statistics", {}).get("viewCount", 0)),
|
"views": int(i.get("statistics", {}).get("viewCount", 0)),
|
||||||
}
|
}
|
||||||
for i in items
|
for i in items
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def mostly_latin(text):
|
||||||
|
letters = [c for c in text if c.isalpha()]
|
||||||
|
if not letters:
|
||||||
|
return True
|
||||||
|
latin = sum(1 for c in letters if ord(c) < 0x250)
|
||||||
|
return latin / len(letters) >= 0.6
|
||||||
|
|
||||||
|
|
||||||
|
def filter_language(videos, lang):
|
||||||
|
if not lang:
|
||||||
|
return videos
|
||||||
|
kept = []
|
||||||
|
for v in videos:
|
||||||
|
if v["language"] and not v["language"].startswith(lang):
|
||||||
|
continue
|
||||||
|
if lang in LATIN_LANGS and not mostly_latin(v["title"]):
|
||||||
|
continue
|
||||||
|
kept.append(v)
|
||||||
|
if len(kept) < len(videos):
|
||||||
|
logger.info("Skipped %d source videos not in language '%s'", len(videos) - len(kept), lang)
|
||||||
|
return kept
|
||||||
|
|
||||||
|
|
||||||
def fetch_trending(s):
|
def fetch_trending(s):
|
||||||
yt = youtube_client(s["profile_id"])
|
yt = youtube_client(s["profile_id"])
|
||||||
count = max(1, min(int(s["trending_count"]), 50))
|
count = max(1, min(int(s["trending_count"]), 50))
|
||||||
region = s["region_code"] or "US"
|
region = s["region_code"] or "US"
|
||||||
|
lang = s["source_language"].strip().lower()
|
||||||
query = s["trending_query"].strip()
|
query = s["trending_query"].strip()
|
||||||
if query:
|
if query:
|
||||||
after = (datetime.now(timezone.utc) - timedelta(days=max(1, int(s["trending_days"])))).strftime("%Y-%m-%dT%H:%M:%SZ")
|
after = (datetime.now(timezone.utc) - timedelta(days=max(1, int(s["trending_days"])))).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||||
@@ -403,11 +432,13 @@ def fetch_trending(s):
|
|||||||
"regionCode": region,
|
"regionCode": region,
|
||||||
"maxResults": count,
|
"maxResults": count,
|
||||||
}
|
}
|
||||||
|
if lang:
|
||||||
|
params["relevanceLanguage"] = lang
|
||||||
if s["category_id"]:
|
if s["category_id"]:
|
||||||
params["videoCategoryId"] = s["category_id"]
|
params["videoCategoryId"] = s["category_id"]
|
||||||
ids = [i["id"]["videoId"] for i in yt.search().list(**params).execute().get("items", []) if i.get("id", {}).get("videoId")]
|
ids = [i["id"]["videoId"] for i in yt.search().list(**params).execute().get("items", []) if i.get("id", {}).get("videoId")]
|
||||||
items = yt.videos().list(part="snippet,statistics", id=",".join(ids)).execute().get("items", []) if ids else []
|
items = yt.videos().list(part="snippet,statistics", id=",".join(ids)).execute().get("items", []) if ids else []
|
||||||
logger.info("Search '%s' returned %d videos from the last %s days (region %s)", query, len(items), s["trending_days"], region)
|
logger.info("Search '%s' returned %d videos from the last %s days (region %s, language %s)", query, len(items), s["trending_days"], region, lang or "any")
|
||||||
else:
|
else:
|
||||||
params = {
|
params = {
|
||||||
"part": "snippet,statistics",
|
"part": "snippet,statistics",
|
||||||
@@ -419,23 +450,31 @@ def fetch_trending(s):
|
|||||||
params["videoCategoryId"] = s["category_id"]
|
params["videoCategoryId"] = s["category_id"]
|
||||||
items = yt.videos().list(**params).execute().get("items", [])
|
items = yt.videos().list(**params).execute().get("items", [])
|
||||||
logger.info("Fetched %d trending chart videos (region %s, category %s)", len(items), region, s["category_id"] or "all")
|
logger.info("Fetched %d trending chart videos (region %s, category %s)", len(items), region, s["category_id"] or "all")
|
||||||
return map_videos(items)
|
return filter_language(map_videos(items), lang)
|
||||||
|
|
||||||
|
|
||||||
def ollama_json(s, prompt, temperature=0.9, model=None, images=None):
|
def ollama_json(s, prompt, temperature=0.9, model=None, images=None, num_predict=2048):
|
||||||
url = f"{s['ollama_url'].rstrip('/')}/api/generate"
|
url = f"{s['ollama_url'].rstrip('/')}/api/generate"
|
||||||
model = model or s["ollama_model"]
|
model = model or s["ollama_model"]
|
||||||
|
timeout = max(30, int(s["ollama_timeout"]))
|
||||||
payload = {
|
payload = {
|
||||||
"model": model,
|
"model": model,
|
||||||
"prompt": prompt,
|
"prompt": prompt,
|
||||||
"format": "json",
|
"format": "json",
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"options": {"temperature": temperature},
|
"keep_alive": "30m",
|
||||||
|
"options": {"temperature": temperature, "num_predict": num_predict, "num_ctx": 8192},
|
||||||
}
|
}
|
||||||
if images:
|
if images:
|
||||||
payload["images"] = images
|
payload["images"] = images
|
||||||
started = datetime.now()
|
started = datetime.now()
|
||||||
r = requests.post(url, json=payload, timeout=900)
|
logger.debug("Waiting on Ollama %s (timeout %ds)", model, timeout)
|
||||||
|
try:
|
||||||
|
r = requests.post(url, json=payload, timeout=(10, timeout))
|
||||||
|
except requests.exceptions.ReadTimeout:
|
||||||
|
raise RuntimeError(f"Ollama model '{model}' did not respond within {timeout}s. Check 'ollama ps' and free RAM, or use a smaller model")
|
||||||
|
except requests.exceptions.ConnectionError:
|
||||||
|
raise RuntimeError(f"Cannot reach Ollama at {s['ollama_url']}. Is it running?")
|
||||||
if r.status_code == 404:
|
if r.status_code == 404:
|
||||||
raise RuntimeError(f"Ollama model '{model}' not found. Run: ollama pull {model}")
|
raise RuntimeError(f"Ollama model '{model}' not found. Run: ollama pull {model}")
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
@@ -466,7 +505,7 @@ Description: {src['description']}
|
|||||||
Tags: {', '.join(src['tags'])}
|
Tags: {', '.join(src['tags'])}
|
||||||
|
|
||||||
{niche_block}{kids_block}{feedback_block}
|
{niche_block}{kids_block}{feedback_block}
|
||||||
Identify the underlying topic and why it appeals to viewers. Then write a completely ORIGINAL {n * clip}-second YouTube Short.
|
Identify the underlying topic and why it appeals to viewers. Then write a completely ORIGINAL {n * clip}-second YouTube Short in English.
|
||||||
Do not reuse the source's title, script, characters, jokes, branding, or channel identity.
|
Do not reuse the source's title, script, characters, jokes, branding, or channel identity.
|
||||||
Do not depict real people, celebrities, brands, logos, or copyrighted characters. Invent new characters.
|
Do not depict real people, celebrities, brands, logos, or copyrighted characters. Invent new characters.
|
||||||
|
|
||||||
@@ -537,7 +576,7 @@ Return JSON only: {{"tags": ["15 to 25 tags, most specific first"]}}"""
|
|||||||
for attempt in range(1, 3):
|
for attempt in range(1, 3):
|
||||||
check()
|
check()
|
||||||
try:
|
try:
|
||||||
data = ollama_json(s, prompt)
|
data = ollama_json(s, prompt, num_predict=512)
|
||||||
raw = data.get("tags", [])
|
raw = data.get("tags", [])
|
||||||
if isinstance(raw, str):
|
if isinstance(raw, str):
|
||||||
raw = raw.split(",")
|
raw = raw.split(",")
|
||||||
@@ -583,7 +622,7 @@ Scenes:
|
|||||||
|
|
||||||
Return JSON only: {{"pass": true or false, "issues": ["each specific problem"]}}"""
|
Return JSON only: {{"pass": true or false, "issues": ["each specific problem"]}}"""
|
||||||
try:
|
try:
|
||||||
data = ollama_json(s, prompt, temperature=0.1)
|
data = ollama_json(s, prompt, temperature=0.1, num_predict=1024)
|
||||||
except Cancelled:
|
except Cancelled:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -639,7 +678,7 @@ def kids_vision_check(s, clip_path):
|
|||||||
raise RuntimeError(f"Could not extract frame at {t}s for vision check")
|
raise RuntimeError(f"Could not extract frame at {t}s for vision check")
|
||||||
images.append(base64.b64encode(frame.read_bytes()).decode())
|
images.append(base64.b64encode(frame.read_bytes()).decode())
|
||||||
frame.unlink(missing_ok=True)
|
frame.unlink(missing_ok=True)
|
||||||
data = ollama_json(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images)
|
data = ollama_json(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images, num_predict=256)
|
||||||
if not truthy(data.get("safe")):
|
if not truthy(data.get("safe")):
|
||||||
raise ValueError(f"Vision check rejected {clip_path.name}: {data.get('reason', 'no reason given')}")
|
raise ValueError(f"Vision check rejected {clip_path.name}: {data.get('reason', 'no reason given')}")
|
||||||
logger.info("Vision check passed for %s", clip_path.name)
|
logger.info("Vision check passed for %s", clip_path.name)
|
||||||
@@ -750,8 +789,13 @@ def run_job(s):
|
|||||||
trending = fetch_trending(s)
|
trending = fetch_trending(s)
|
||||||
state = load_state()
|
state = load_state()
|
||||||
candidates = [v for v in sorted(trending, key=lambda v: -v["views"]) if v["id"] not in state["processed"]]
|
candidates = [v for v in sorted(trending, key=lambda v: -v["views"]) if v["id"] not in state["processed"]]
|
||||||
|
if kids:
|
||||||
|
safe = [v for v in candidates if not KIDS_BANNED.search(v["title"])]
|
||||||
|
if len(safe) < len(candidates):
|
||||||
|
logger.info("Skipped %d source videos with themes unsuitable for kids", len(candidates) - len(safe))
|
||||||
|
candidates = safe
|
||||||
if not candidates:
|
if not candidates:
|
||||||
logger.warning("No unprocessed trending videos, checking again in 30 minutes")
|
logger.warning("No usable trending videos, checking again in 30 minutes")
|
||||||
set_stage("waiting for new trending videos")
|
set_stage("waiting for new trending videos")
|
||||||
stop_event.wait(1800)
|
stop_event.wait(1800)
|
||||||
return
|
return
|
||||||
@@ -947,6 +991,7 @@ async def post_settings(request: Request):
|
|||||||
if k in PROFILE_DEFAULTS:
|
if k in PROFILE_DEFAULTS:
|
||||||
p[k] = coerce(PROFILE_DEFAULTS, k, v)
|
p[k] = coerce(PROFILE_DEFAULTS, k, v)
|
||||||
g["storage_dir"] = g["storage_dir"].strip().strip('"') or "videos"
|
g["storage_dir"] = g["storage_dir"].strip().strip('"') or "videos"
|
||||||
|
p["source_language"] = p["source_language"].lower()[:5]
|
||||||
root = validate_storage(g["storage_dir"])
|
root = validate_storage(g["storage_dir"])
|
||||||
except (TypeError, ValueError) as e:
|
except (TypeError, ValueError) as e:
|
||||||
return JSONResponse({"detail": str(e)}, status_code=400)
|
return JSONResponse({"detail": str(e)}, status_code=400)
|
||||||
|
|||||||
+4
-2
@@ -111,17 +111,19 @@ const FIELDS=[
|
|||||||
["trending_days","Search window (days)","number","Only used with a search query","p"],
|
["trending_days","Search window (days)","number","Only used with a search query","p"],
|
||||||
["category_id","Source category","select:=All|"+CATS,"Filters source videos","p"],
|
["category_id","Source category","select:=All|"+CATS,"Filters source videos","p"],
|
||||||
["region_code","Region","text","Two-letter country code, e.g. US","p"],
|
["region_code","Region","text","Two-letter country code, e.g. US","p"],
|
||||||
|
["source_language","Source language","text","Two-letter code, e.g. en. Skips source videos in other languages. Blank for any","p"],
|
||||||
["trending_count","Videos to scan","number","Max 50","p"],
|
["trending_count","Videos to scan","number","Max 50","p"],
|
||||||
["videos_per_day","Videos per day","number","0 runs continuously","p"],
|
["videos_per_day","Videos per day","number","0 runs continuously","p"],
|
||||||
["minutes_between_videos","Minutes between videos","number","0 for no gap","p"],
|
["minutes_between_videos","Minutes between videos","number","0 for no gap","p"],
|
||||||
["","Storage (shared by all profiles)","header","",""],
|
["","Storage (shared by all profiles)","header","",""],
|
||||||
["storage_dir","Storage folder","wide","Full path on a large disk, e.g. D:\\ShortsVideos. Jobs are built in output\\ and moved to published\\ after upload","g"],
|
["storage_dir","Storage folder","wide","Full path on a large disk, e.g. /Volumes/BigDisk/ShortsVideos. Jobs are built in output/ and moved to published/ after upload","g"],
|
||||||
["delete_after_publish","Delete videos after publishing","checkbox","Removes the job folder once the upload succeeds instead of moving it to published","g"],
|
["delete_after_publish","Delete videos after publishing","checkbox","Removes the job folder once the upload succeeds instead of moving it to published","g"],
|
||||||
["keep_clips","Keep individual clips","checkbox","Keeps the 8 second clips alongside final.mp4","g"],
|
["keep_clips","Keep individual clips","checkbox","Keeps the 8 second clips alongside final.mp4","g"],
|
||||||
["","Generation (shared by all profiles)","header","",""],
|
["","Generation (shared by all profiles)","header","",""],
|
||||||
["ollama_url","Ollama URL","text","","g"],
|
["ollama_url","Ollama URL","text","","g"],
|
||||||
["ollama_model","Ollama model","text","Any pulled model, e.g. llama3.1:8b","g"],
|
["ollama_model","Ollama model","text","Any pulled model, e.g. llama3.1:8b, or llama3.2:3b on 8 GB Macs","g"],
|
||||||
["vision_model","Ollama vision model","text","Required for made-for-kids profiles. Run: ollama pull llama3.2-vision","g"],
|
["vision_model","Ollama vision model","text","Required for made-for-kids profiles. Run: ollama pull llama3.2-vision","g"],
|
||||||
|
["ollama_timeout","Ollama timeout (seconds)","number","Max wait per Ollama request before retrying","g"],
|
||||||
["gemini_api_key","Gemini API key (Veo)","password","Billing must be enabled","g"],
|
["gemini_api_key","Gemini API key (Veo)","password","Billing must be enabled","g"],
|
||||||
["veo_model","Veo model","text","Update if Google renames or retires the model","g"],
|
["veo_model","Veo model","text","Update if Google renames or retires the model","g"],
|
||||||
["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"],
|
["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"],
|
||||||
|
|||||||
Reference in new issue
Block a user