Add low memory mode, model unloading between stages, and a CPU thread cap

This commit is contained in:
Justin Oros
2026-09-26 21:17:39 -07:00
parent fcd9d77a4d
commit df5bcbb185
2 changed files with 38 additions and 1 deletions
+33
View File
@@ -82,6 +82,9 @@ GLOBAL_DEFAULTS = {
"target_seconds": 60, "target_seconds": 60,
"clip_seconds": 8, "clip_seconds": 8,
"scene_timing": "narration", "scene_timing": "narration",
"low_memory": True,
"free_models_between_stages": True,
"cpu_threads": 0,
"error_cooldown_minutes": 5, "error_cooldown_minutes": 5,
"work_hours_enabled": False, "work_hours_enabled": False,
"work_days": "mon,tue,wed,thu,fri,sat,sun", "work_days": "mon,tue,wed,thu,fri,sat,sun",
@@ -956,6 +959,21 @@ def ollama_pull(s, model):
return True return True
def unload_ollama(s, model):
model = (model or "").strip()
if not model or not s.get("free_models_between_stages", True):
return
try:
requests.post(
f"{s['ollama_url'].rstrip('/')}/api/generate",
json={"model": model, "prompt": "", "keep_alive": 0, "stream": False},
timeout=20,
)
logger.debug("Asked Ollama to release %s", model)
except Exception as e:
logger.debug("Could not release %s: %s", model, e)
def ensure_local_deps(s): def ensure_local_deps(s):
if not ensure_module(s, "torch", "torch") or not ensure_module(s, "diffusers", "diffusers accelerate safetensors transformers".split()[0]): if not ensure_module(s, "torch", "torch") or not ensure_module(s, "diffusers", "diffusers accelerate safetensors transformers".split()[0]):
raise RuntimeError("Could not install the image packages. Run ./install-local.sh") raise RuntimeError("Could not install the image packages. Run ./install-local.sh")
@@ -1470,10 +1488,21 @@ def load_pipeline(s):
device, dtype = "cuda", torch.float16 device, dtype = "cuda", torch.float16
else: else:
device, dtype = "cpu", torch.float32 device, dtype = "cpu", torch.float32
threads = int(s.get("cpu_threads", 0) or 0)
if threads > 0:
torch.set_num_threads(threads)
logger.info("Limiting image generation to %d CPU threads", threads)
logger.info("Loading image model %s on %s (first run downloads several GB)", model, device) logger.info("Loading image model %s on %s (first run downloads several GB)", model, device)
pipe = AutoPipelineForText2Image.from_pretrained(model, torch_dtype=dtype, variant="fp16" if dtype == torch.float16 else None) pipe = AutoPipelineForText2Image.from_pretrained(model, torch_dtype=dtype, variant="fp16" if dtype == torch.float16 else None)
pipe = pipe.to(device) pipe = pipe.to(device)
pipe.set_progress_bar_config(disable=True) pipe.set_progress_bar_config(disable=True)
if s.get("low_memory", True):
for enable in ("enable_attention_slicing", "enable_vae_slicing", "enable_vae_tiling"):
try:
getattr(pipe, enable)()
except Exception:
pass
logger.info("Low memory mode is on for image generation")
pipeline["id"] = model pipeline["id"] = model
pipeline["obj"] = pipe pipeline["obj"] = pipe
logger.info("Image model ready") logger.info("Image model ready")
@@ -1800,6 +1829,8 @@ def run_job(s):
logger.info("Loaded %d existing channel titles to avoid repeats", len(s["_channel_titles"])) logger.info("Loaded %d existing channel titles to avoid repeats", len(s["_channel_titles"]))
script, tags, compliance = produce_script(s, src) script, tags, compliance = produce_script(s, src)
remember_story(s["profile_id"], script) remember_story(s["profile_id"], script)
if engine == "local":
unload_ollama(s, s["ollama_model"])
compliance["engine"] = engine compliance["engine"] = engine
update_history(job_id, title=script["title"], notes="; ".join(compliance.get("reviewer_notes", []))[:400]) update_history(job_id, title=script["title"], notes="; ".join(compliance.get("reviewer_notes", []))[:400])
write_json(job_dir / "script.json", {"profile": s["name"], "source": src, "script": script, "tags": tags}) write_json(job_dir / "script.json", {"profile": s["name"], "source": src, "script": script, "tags": tags})
@@ -1909,6 +1940,8 @@ def run_job(s):
build_scene(i, script["scenes"][i - 1], clips[i - 1]) build_scene(i, script["scenes"][i - 1], clips[i - 1])
pending = failed pending = failed
if kids:
unload_ollama(s, s["vision_model"])
anatomy_note = f"Possible anatomy glitches kept in scenes {', '.join(str(i) for i in anatomy_flags)}" if anatomy_flags else "" anatomy_note = f"Possible anatomy glitches kept in scenes {', '.join(str(i) for i in anatomy_flags)}" if anatomy_flags else ""
if kids: if kids:
compliance["vision_checks"] = vision_log compliance["vision_checks"] = vision_log
+5 -1
View File
@@ -319,6 +319,10 @@ const FIELDS=[
["piper_model","Piper voice file","wide","Blank uses the first voice found in the voices folder under Models","g"], ["piper_model","Piper voice file","wide","Blank uses the first voice found in the voices folder under Models","g"],
["gemini_api_key","Gemini API key (Veo)","password","Only needed for the Veo engine. Billing must be enabled","g"], ["gemini_api_key","Gemini API key (Veo)","password","Only needed for the Veo engine. Billing must be enabled","g"],
["veo_model","Veo model","text","e.g. veo-3.1-fast-generate-preview","g"], ["veo_model","Veo model","text","e.g. veo-3.1-fast-generate-preview","g"],
["","Performance","header","Shared by all profiles. Lower settings here if the Mac becomes slow or unstable",""],
["low_memory","Low memory mode","checkbox","Generates images in smaller pieces. A little slower, much less memory. Leave on for Macs with 16 GB or less","g"],
["free_models_between_stages","Free models between stages","checkbox","Unloads the writing model before images are drawn and the vision model after checks","g"],
["cpu_threads","CPU threads","select:0=Automatic|2|4|6|8|10","Caps how many cores image generation uses so the rest of the Mac stays responsive","g"],
["","Writing and timing","header","Shared by all profiles",""], ["","Writing and timing","header","Shared by all profiles",""],
["ollama_url","Ollama URL","text","","g"], ["ollama_url","Ollama URL","text","","g"],
["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"], ["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"],
@@ -332,7 +336,7 @@ const FIELDS=[
]; ];
const DAYS=[["mon","Mon"],["tue","Tue"],["wed","Wed"],["thu","Thu"],["fri","Fri"],["sat","Sat"],["sun","Sun"]]; const DAYS=[["mon","Mon"],["tue","Tue"],["wed","Wed"],["thu","Thu"],["fri","Fri"],["sat","Sat"],["sun","Sun"]];
const TIMES=[];for(let h=0;h<24;h++)for(const m of [0,30]){const v=String(h).padStart(2,"0")+":"+String(m).padStart(2,"0");const hr=h%12||12;TIMES.push([v,hr+":"+String(m).padStart(2,"0")+(h<12?" AM":" PM")])} const TIMES=[];for(let h=0;h<24;h++)for(const m of [0,30]){const v=String(h).padStart(2,"0")+":"+String(m).padStart(2,"0");const hr=h%12||12;TIMES.push([v,hr+":"+String(m).padStart(2,"0")+(h<12?" AM":" PM")])}
const NUMERIC=new Set(["sd_steps","sd_guidance","sd_width","sd_height","trending_days","trending_count","videos_per_day","minutes_between_videos","target_seconds","clip_seconds","ollama_timeout","error_cooldown_minutes","tts_rate"]); const NUMERIC=new Set(["sd_steps","sd_guidance","sd_width","sd_height","trending_days","trending_count","videos_per_day","minutes_between_videos","target_seconds","clip_seconds","ollama_timeout","error_cooldown_minutes","tts_rate","cpu_threads"]);
const $=s=>document.querySelector(s); const $=s=>document.querySelector(s);
const esc=s=>String(s??"").replace(/[&<>"]/g,c=>({"&":"&amp;","<":"&lt;",">":"&gt;",'"':"&quot;"}[c])); const esc=s=>String(s??"").replace(/[&<>"]/g,c=>({"&":"&amp;","<":"&lt;",">":"&gt;",'"':"&quot;"}[c]));
async function api(path,opts={}){const r=await fetch(path,{headers:{"Content-Type":"application/json"},...opts});const j=await r.json().catch(()=>({}));if(!r.ok)throw new Error(j.detail||r.statusText);return j} async function api(path,opts={}){const r=await fetch(path,{headers:{"Content-Type":"application/json"},...opts});const j=await r.json().catch(()=>({}));if(!r.ok)throw new Error(j.detail||r.statusText);return j}