Batch vision checks after generation, reject inconclusive answers, recommend llava
This commit is contained in:
@@ -48,7 +48,7 @@ REVIEW_MODES = ("advisory", "block", "off")
|
|||||||
ENGINES = ("local", "veo")
|
ENGINES = ("local", "veo")
|
||||||
TTS_ENGINES = ("auto", "kokoro", "piper", "say", "espeak")
|
TTS_ENGINES = ("auto", "kokoro", "piper", "say", "espeak")
|
||||||
SUGGESTED_TEXT_MODELS = ["llama3.1:8b", "llama3.2:3b", "qwen2.5:7b", "mistral:7b"]
|
SUGGESTED_TEXT_MODELS = ["llama3.1:8b", "llama3.2:3b", "qwen2.5:7b", "mistral:7b"]
|
||||||
SUGGESTED_VISION_MODELS = ["moondream", "llava:7b", "llama3.2-vision", "qwen2.5vl:7b"]
|
SUGGESTED_VISION_MODELS = ["llava:7b", "qwen2.5vl:7b", "llama3.2-vision", "moondream"]
|
||||||
SMALL_VISION_MODELS = ("moondream", "llava-phi3", "bakllava")
|
SMALL_VISION_MODELS = ("moondream", "llava-phi3", "bakllava")
|
||||||
|
|
||||||
GLOBAL_DEFAULTS = {
|
GLOBAL_DEFAULTS = {
|
||||||
@@ -61,7 +61,7 @@ GLOBAL_DEFAULTS = {
|
|||||||
"video_engine": "local",
|
"video_engine": "local",
|
||||||
"ollama_url": "http://localhost:11434",
|
"ollama_url": "http://localhost:11434",
|
||||||
"ollama_model": "llama3.1:8b",
|
"ollama_model": "llama3.1:8b",
|
||||||
"vision_model": "moondream",
|
"vision_model": "llava:7b",
|
||||||
"ollama_timeout": 300,
|
"ollama_timeout": 300,
|
||||||
"sd_model": "stabilityai/sdxl-turbo",
|
"sd_model": "stabilityai/sdxl-turbo",
|
||||||
"sd_steps": 4,
|
"sd_steps": 4,
|
||||||
@@ -1129,6 +1129,9 @@ def unload_pipeline():
|
|||||||
logger.debug("Released image model from memory")
|
logger.debug("Released image model from memory")
|
||||||
|
|
||||||
|
|
||||||
|
INCONCLUSIVE = ("don't know", "do not know", "not sure", "cannot", "can't", "unable", "no image", "what you see", "short explanation")
|
||||||
|
|
||||||
|
|
||||||
def kids_vision_check(s, clip_path):
|
def kids_vision_check(s, clip_path):
|
||||||
model = s["vision_model"].strip()
|
model = s["vision_model"].strip()
|
||||||
if not model:
|
if not model:
|
||||||
@@ -1142,30 +1145,35 @@ def kids_vision_check(s, clip_path):
|
|||||||
raise RuntimeError("Could not extract a frame for the vision check")
|
raise RuntimeError("Could not extract a frame for the vision check")
|
||||||
images = [base64.b64encode(frame.read_bytes()).decode()]
|
images = [base64.b64encode(frame.read_bytes()).decode()]
|
||||||
frame.unlink(missing_ok=True)
|
frame.unlink(missing_ok=True)
|
||||||
if not model.split(":")[0] in SMALL_VISION_MODELS:
|
verdict, detail = None, ""
|
||||||
unload_pipeline()
|
|
||||||
safe, reason = True, ""
|
|
||||||
try:
|
try:
|
||||||
try:
|
try:
|
||||||
data = ollama_call(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images, num_predict=256)
|
data = ollama_call(s, KIDS_VISION_PROMPT, temperature=0.1, model=model, images=images, num_predict=300)
|
||||||
desc = str(data.get("description", ""))
|
desc = str(data.get("description", "")).strip()
|
||||||
reason = str(data.get("reason", ""))
|
reason = str(data.get("reason", "")).strip()
|
||||||
echoed = "it is unsafe if" in reason.lower() or reason.strip().lower() in ("why", "short explanation")
|
low = f"{desc} {reason}".lower()
|
||||||
if echoed or desc.strip().lower() in ("what you see", "description"):
|
echoed = "it is unsafe if" in low
|
||||||
raise ValueError("model echoed the prompt")
|
if desc and not echoed and not any(x in low for x in INCONCLUSIVE) and "safe" in data:
|
||||||
safe = truthy(data.get("safe"))
|
verdict = truthy(data.get("safe"))
|
||||||
|
detail = f"{desc} | {reason}"
|
||||||
except (requests.exceptions.HTTPError, ValueError):
|
except (requests.exceptions.HTTPError, ValueError):
|
||||||
text = ollama_call(s, VISION_TEXT_PROMPT, temperature=0.1, model=model, images=images, num_predict=128, as_json=False)
|
pass
|
||||||
|
if verdict is None:
|
||||||
|
text = ollama_call(s, VISION_TEXT_PROMPT, temperature=0.1, model=model, images=images, num_predict=150, as_json=False).strip()
|
||||||
low = text.lower()
|
low = text.lower()
|
||||||
safe = "unsafe" not in low
|
if low.startswith("unsafe") or re.search(r"\bunsafe\b", low):
|
||||||
reason = text.strip()[:200]
|
verdict, detail = False, text
|
||||||
|
elif re.match(r"^\W*safe\b", low) and not any(x in low for x in INCONCLUSIVE):
|
||||||
|
verdict, detail = True, text
|
||||||
except Cancelled:
|
except Cancelled:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise VisionUnavailable(f"vision model '{model}' could not run: {e}") from e
|
raise VisionUnavailable(f"vision model '{model}' could not run: {e}") from e
|
||||||
if not safe:
|
if verdict is None:
|
||||||
raise ValueError(f"Vision check rejected {clip_path.name}: {reason or 'no reason given'}")
|
raise VisionUnavailable(f"vision model '{model}' gave no clear answer, try llava:7b")
|
||||||
logger.info("Vision check passed for %s", clip_path.name)
|
if not verdict:
|
||||||
|
raise ValueError(f"Vision check rejected {clip_path.name}: {detail[:200] or 'no reason given'}")
|
||||||
|
logger.info("Vision check passed for %s: %s", clip_path.name, detail[:160])
|
||||||
|
|
||||||
|
|
||||||
def load_pipeline(s):
|
def load_pipeline(s):
|
||||||
@@ -1497,16 +1505,18 @@ def run_job(s):
|
|||||||
extras = []
|
extras = []
|
||||||
vision_log = []
|
vision_log = []
|
||||||
n = len(script["scenes"])
|
n = len(script["scenes"])
|
||||||
for i, scene in enumerate(script["scenes"], 1):
|
|
||||||
check()
|
def build_scene(i, scene, path):
|
||||||
set_stage(f"generating scene {i}/{n}", step=3 + (i - 1) * 2)
|
|
||||||
path = job_dir / f"clip_{i:02d}.mp4"
|
|
||||||
for attempt in range(1, 4):
|
for attempt in range(1, 4):
|
||||||
|
check()
|
||||||
try:
|
try:
|
||||||
if engine == "local":
|
if engine == "local":
|
||||||
extra = make_clip_local(s, scene, script, path, kids)
|
extra = make_clip_local(s, scene, script, path, kids)
|
||||||
else:
|
else:
|
||||||
extra = make_clip_veo(s, scene, script, path, kids)
|
extra = make_clip_veo(s, scene, script, path, kids)
|
||||||
|
if extra and extra not in extras:
|
||||||
|
extras.append(extra)
|
||||||
|
return
|
||||||
except Cancelled:
|
except Cancelled:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -1515,32 +1525,45 @@ def run_job(s):
|
|||||||
if attempt == 3:
|
if attempt == 3:
|
||||||
raise
|
raise
|
||||||
stop_event.wait(10)
|
stop_event.wait(10)
|
||||||
continue
|
|
||||||
if kids:
|
for i, scene in enumerate(script["scenes"], 1):
|
||||||
set_stage(f"vision check scene {i}/{n}", step=4 + (i - 1) * 2)
|
set_stage(f"generating scene {i}/{n}", step=2 + i)
|
||||||
|
path = job_dir / f"clip_{i:02d}.mp4"
|
||||||
|
build_scene(i, scene, path)
|
||||||
|
clips.append(path)
|
||||||
|
|
||||||
|
if kids:
|
||||||
|
pending = list(range(1, n + 1))
|
||||||
|
for rnd in range(1, 4):
|
||||||
|
unload_pipeline()
|
||||||
|
rejected = []
|
||||||
|
for i in pending:
|
||||||
|
check()
|
||||||
|
set_stage(f"vision check scene {i}/{n}", step=2 + n + i)
|
||||||
try:
|
try:
|
||||||
kids_vision_check(s, path)
|
kids_vision_check(s, clips[i - 1])
|
||||||
vision_log.append({"clip": i, "attempt": attempt, "result": "passed"})
|
vision_log.append({"clip": i, "round": rnd, "result": "passed"})
|
||||||
except Cancelled:
|
except Cancelled:
|
||||||
raise
|
raise
|
||||||
except VisionUnavailable as e:
|
except VisionUnavailable as e:
|
||||||
if s["kids_manual_review"]:
|
if s["kids_manual_review"]:
|
||||||
logger.warning("Scene %d: %s. Continuing, check this video manually before publishing", i, e)
|
logger.warning("Scene %d: %s. Continuing, check this video manually before publishing", i, e)
|
||||||
vision_log.append({"clip": i, "attempt": attempt, "result": f"skipped: {e}"[:300]})
|
vision_log.append({"clip": i, "round": rnd, "result": f"skipped: {e}"[:300]})
|
||||||
else:
|
else:
|
||||||
raise RuntimeError(f"{e}. Turn on manual review or pick a working vision model")
|
raise RuntimeError(f"{e}. Frames were not verified, so nothing was uploaded")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning("Scene %d attempt %d rejected: %s", i, attempt, e)
|
logger.warning("Scene %d rejected (round %d): %s", i, rnd, e)
|
||||||
vision_log.append({"clip": i, "attempt": attempt, "result": str(e)[:300]})
|
vision_log.append({"clip": i, "round": rnd, "result": str(e)[:300]})
|
||||||
path.unlink(missing_ok=True)
|
rejected.append(i)
|
||||||
if attempt == 3:
|
if not rejected:
|
||||||
raise
|
break
|
||||||
stop_event.wait(5)
|
if rnd == 3:
|
||||||
continue
|
raise RuntimeError(f"Scenes {rejected} still failed the vision check after 3 rounds")
|
||||||
if extra:
|
for i in rejected:
|
||||||
extras.append(extra)
|
set_stage(f"regenerating scene {i}/{n}", step=2 + i)
|
||||||
break
|
clips[i - 1].unlink(missing_ok=True)
|
||||||
clips.append(path)
|
build_scene(i, script["scenes"][i - 1], clips[i - 1])
|
||||||
|
pending = rejected
|
||||||
if kids:
|
if kids:
|
||||||
compliance["vision_checks"] = vision_log
|
compliance["vision_checks"] = vision_log
|
||||||
set_stage("stitching video", step=3 + n * 2)
|
set_stage("stitching video", step=3 + n * 2)
|
||||||
|
|||||||
+1
-1
@@ -153,7 +153,7 @@ const FIELDS=[
|
|||||||
["","Text and shared options","header","",""],
|
["","Text and shared options","header","",""],
|
||||||
["ollama_url","Ollama URL","text","","g"],
|
["ollama_url","Ollama URL","text","","g"],
|
||||||
["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"],
|
["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"],
|
||||||
["vision_model","Ollama vision model","model:vision","Checks generated scenes on made-for-kids profiles. moondream suits 16 GB machines","g"],
|
["vision_model","Ollama vision model","model:vision","Checks every scene on made-for-kids profiles. llava:7b is recommended, moondream is too weak to judge anatomy","g"],
|
||||||
["ollama_timeout","Ollama timeout (seconds)","select:120|300|600|900","Max wait per Ollama request before retrying","g"],
|
["ollama_timeout","Ollama timeout (seconds)","select:120|300|600|900","Max wait per Ollama request before retrying","g"],
|
||||||
["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"],
|
["aspect_ratio","Aspect ratio","select:9:16|16:9","","g"],
|
||||||
["target_seconds","Target length (seconds)","select:30|45|60|90|120|180","Split into scenes of the length below","g"],
|
["target_seconds","Target length (seconds)","select:30|45|60|90|120|180","Split into scenes of the length below","g"],
|
||||||
|
|||||||
Reference in New Issue
Block a user