Batch vision checks after generation, reject inconclusive answers, recommend llava

This commit is contained in:
Justin Oros
2026-09-20 22:24:40 -07:00
parent d9fc18fcca
commit 47433dbfa8
2 changed files with 64 additions and 41 deletions
+63 -40
View File
@@ -48,7 +48,7 @@ REVIEW_MODES = ("advisory", "block", "off")
ENGINES = ("local", "veo") ENGINES = ("local", "veo")
TTS_ENGINES = ("auto", "kokoro", "piper", "say", "espeak") TTS_ENGINES = ("auto", "kokoro", "piper", "say", "espeak")
SUGGESTED_TEXT_MODELS = ["llama3.1:8b", "llama3.2:3b", "qwen2.5:7b", "mistral:7b"] SUGGESTED_TEXT_MODELS = ["llama3.1:8b", "llama3.2:3b", "qwen2.5:7b", "mistral:7b"]
SUGGESTED_VISION_MODELS = ["moondream", "llava:7b", "llama3.2-vision", "qwen2.5vl:7b"] SUGGESTED_VISION_MODELS = ["llava:7b", "qwen2.5vl:7b", "llama3.2-vision", "moondream"]
SMALL_VISION_MODELS = ("moondream", "llava-phi3", "bakllava") SMALL_VISION_MODELS = ("moondream", "llava-phi3", "bakllava")
GLOBAL_DEFAULTS = { GLOBAL_DEFAULTS = {
@@ -61,7 +61,7 @@ GLOBAL_DEFAULTS = {
"video_engine": "local", "video_engine": "local",
"ollama_url": "http://localhost:11434", "ollama_url": "http://localhost:11434",
"ollama_model": "llama3.1:8b", "ollama_model": "llama3.1:8b",
"vision_model": "moondream", "vision_model": "llava:7b",
"ollama_timeout": 300, "ollama_timeout": 300,
"sd_model": "stabilityai/sdxl-turbo", "sd_model": "stabilityai/sdxl-turbo",
"sd_steps": 4, "sd_steps": 4,
@@ -1129,6 +1129,9 @@ def unload_pipeline():
logger.debug("Released image model from memory") logger.debug("Released image model from memory")
INCONCLUSIVE = ("don't know", "do not know", "not sure", "cannot", "can't", "unable", "no image", "what you see", "short explanation")
def kids_vision_check(s, clip_path): def kids_vision_check(s, clip_path):
model = s["vision_model"].strip() model = s["vision_model"].strip()
if not model: if not model:
@@ -1142,30 +1145,35 @@ def kids_vision_check(s, clip_path):
raise RuntimeError("Could not extract a frame for the vision check") raise RuntimeError("Could not extract a frame for the vision check")
images = [base64.b64encode(frame.read_bytes()).decode()] images = [base64.b64encode(frame.read_bytes()).decode()]
frame.unlink(missing_ok=True) frame.unlink(missing_ok=True)
if not model.split(":")[0] in SMALL_VISION_MODELS: verdict, detail = None, ""
unload_pipeline()
safe, reason = True, ""
try: try:
try: try:
data = ollama_call(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images, num_predict=256) data = ollama_call(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images, num_predict=300)
desc = str(data.get("description", "")) desc = str(data.get("description", "")).strip()
reason = str(data.get("reason", "")) reason = str(data.get("reason", "")).strip()
echoed = "it is unsafe if" in reason.lower() or reason.strip().lower() in ("why", "short explanation") low = f"{desc} {reason}".lower()
if echoed or desc.strip().lower() in ("what you see", "description"): echoed = "it is unsafe if" in low
raise ValueError("model echoed the prompt") if desc and not echoed and not any(x in low for x in INCONCLUSIVE) and "safe" in data:
safe = truthy(data.get("safe")) verdict = truthy(data.get("safe"))
detail = f"{desc} | {reason}"
except (requests.exceptions.HTTPError, ValueError): except (requests.exceptions.HTTPError, ValueError):
text = ollama_call(s, VISION_TEXT_PROMPT, temperature=0.1, model=model, images=images, num_predict=128, as_json=False) pass
if verdict is None:
text = ollama_call(s, VISION_TEXT_PROMPT, temperature=0.1, model=model, images=images, num_predict=150, as_json=False).strip()
low = text.lower() low = text.lower()
safe = "unsafe" not in low if low.startswith("unsafe") or re.search(r"\bunsafe\b", low):
reason = text.strip()[:200] verdict, detail = False, text
elif re.match(r"^\W*safe\b", low) and not any(x in low for x in INCONCLUSIVE):
verdict, detail = True, text
except Cancelled: except Cancelled:
raise raise
except Exception as e: except Exception as e:
raise VisionUnavailable(f"vision model '{model}' could not run: {e}") from e raise VisionUnavailable(f"vision model '{model}' could not run: {e}") from e
if not safe: if verdict is None:
raise ValueError(f"Vision check rejected {clip_path.name}: {reason or 'no reason given'}") raise VisionUnavailable(f"vision model '{model}' gave no clear answer, try llava:7b")
logger.info("Vision check passed for %s", clip_path.name) if not verdict:
raise ValueError(f"Vision check rejected {clip_path.name}: {detail[:200] or 'no reason given'}")
logger.info("Vision check passed for %s: %s", clip_path.name, detail[:160])
def load_pipeline(s): def load_pipeline(s):
@@ -1497,16 +1505,18 @@ def run_job(s):
extras = [] extras = []
vision_log = [] vision_log = []
n = len(script["scenes"]) n = len(script["scenes"])
for i, scene in enumerate(script["scenes"], 1):
check() def build_scene(i, scene, path):
set_stage(f"generating scene {i}/{n}", step=3 + (i - 1) * 2)
path = job_dir / f"clip_{i:02d}.mp4"
for attempt in range(1, 4): for attempt in range(1, 4):
check()
try: try:
if engine == "local": if engine == "local":
extra = make_clip_local(s, scene, script, path, kids) extra = make_clip_local(s, scene, script, path, kids)
else: else:
extra = make_clip_veo(s, scene, script, path, kids) extra = make_clip_veo(s, scene, script, path, kids)
if extra and extra not in extras:
extras.append(extra)
return
except Cancelled: except Cancelled:
raise raise
except Exception as e: except Exception as e:
@@ -1515,32 +1525,45 @@ def run_job(s):
if attempt == 3: if attempt == 3:
raise raise
stop_event.wait(10) stop_event.wait(10)
continue
if kids: for i, scene in enumerate(script["scenes"], 1):
set_stage(f"vision check scene {i}/{n}", step=4 + (i - 1) * 2) set_stage(f"generating scene {i}/{n}", step=2 + i)
path = job_dir / f"clip_{i:02d}.mp4"
build_scene(i, scene, path)
clips.append(path)
if kids:
pending = list(range(1, n + 1))
for rnd in range(1, 4):
unload_pipeline()
rejected = []
for i in pending:
check()
set_stage(f"vision check scene {i}/{n}", step=2 + n + i)
try: try:
kids_vision_check(s, path) kids_vision_check(s, clips[i - 1])
vision_log.append({"clip": i, "attempt": attempt, "result": "passed"}) vision_log.append({"clip": i, "round": rnd, "result": "passed"})
except Cancelled: except Cancelled:
raise raise
except VisionUnavailable as e: except VisionUnavailable as e:
if s["kids_manual_review"]: if s["kids_manual_review"]:
logger.warning("Scene %d: %s. Continuing, check this video manually before publishing", i, e) logger.warning("Scene %d: %s. Continuing, check this video manually before publishing", i, e)
vision_log.append({"clip": i, "attempt": attempt, "result": f"skipped: {e}"[:300]}) vision_log.append({"clip": i, "round": rnd, "result": f"skipped: {e}"[:300]})
else: else:
raise RuntimeError(f"{e}. Turn on manual review or pick a working vision model") raise RuntimeError(f"{e}. Frames were not verified, so nothing was uploaded")
except Exception as e: except Exception as e:
logger.warning("Scene %d attempt %d rejected: %s", i, attempt, e) logger.warning("Scene %d rejected (round %d): %s", i, rnd, e)
vision_log.append({"clip": i, "attempt": attempt, "result": str(e)[:300]}) vision_log.append({"clip": i, "round": rnd, "result": str(e)[:300]})
path.unlink(missing_ok=True) rejected.append(i)
if attempt == 3: if not rejected:
raise break
stop_event.wait(5) if rnd == 3:
continue raise RuntimeError(f"Scenes {rejected} still failed the vision check after 3 rounds")
if extra: for i in rejected:
extras.append(extra) set_stage(f"regenerating scene {i}/{n}", step=2 + i)
break clips[i - 1].unlink(missing_ok=True)
clips.append(path) build_scene(i, script["scenes"][i - 1], clips[i - 1])
pending = rejected
if kids: if kids:
compliance["vision_checks"] = vision_log compliance["vision_checks"] = vision_log
set_stage("stitching video", step=3 + n * 2) set_stage("stitching video", step=3 + n * 2)
+1 -1
View File
@@ -153,7 +153,7 @@ const FIELDS=[
["","Text and shared options","header","",""], ["","Text and shared options","header","",""],
["ollama_url","Ollama URL","text","","g"], ["ollama_url","Ollama URL","text","","g"],
["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"], ["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"],
["vision_model","Ollama vision model","model:vision","Checks generated scenes on made-for-kids profiles. moondream suits 16 GB machines","g"], ["vision_model","Ollama vision model","model:vision","Checks every scene on made-for-kids profiles. llava:7b is recommended, moondream is too weak to judge anatomy","g"],
["ollama_timeout","Ollama timeout (seconds)","select:120|300|600|900","Max wait per Ollama request before retrying","g"], ["ollama_timeout","Ollama timeout (seconds)","select:120|300|600|900","Max wait per Ollama request before retrying","g"],
["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"], ["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"],
["target_seconds","Target length (seconds)","select:30|45|60|90|120|180","Split into scenes of the length below","g"], ["target_seconds","Target length (seconds)","select:30|45|60|90|120|180","Split into scenes of the length below","g"],