From df5bcbb185f7d199815a34354d5cb1e2c51fc6c5 Mon Sep 17 00:00:00 2001 From: Justin Oros Date: Sat, 26 Sep 2026 21:17:39 -0700 Subject: [PATCH] Add low memory mode, model unloading between stages, and a CPU thread cap --- app.py | 33 +++++++++++++++++++++++++++++++++ static/index.html | 6 +++++- 2 files changed, 38 insertions(+), 1 deletion(-) diff --git a/app.py b/app.py index 3cdf6cb..769bcc3 100644 --- a/app.py +++ b/app.py @@ -82,6 +82,9 @@ GLOBAL_DEFAULTS = { "target_seconds": 60, "clip_seconds": 8, "scene_timing": "narration", + "low_memory": True, + "free_models_between_stages": True, + "cpu_threads": 0, "error_cooldown_minutes": 5, "work_hours_enabled": False, "work_days": "mon,tue,wed,thu,fri,sat,sun", @@ -956,6 +959,21 @@ def ollama_pull(s, model): return True +def unload_ollama(s, model): + model = (model or "").strip() + if not model or not s.get("free_models_between_stages", True): + return + try: + requests.post( + f"{s['ollama_url'].rstrip('/')}/api/generate", + json={"model": model, "prompt": "", "keep_alive": 0, "stream": False}, + timeout=20, + ) + logger.debug("Asked Ollama to release %s", model) + except Exception as e: + logger.debug("Could not release %s: %s", model, e) + + def ensure_local_deps(s): if not ensure_module(s, "torch", "torch") or not ensure_module(s, "diffusers", "diffusers accelerate safetensors transformers".split()[0]): raise RuntimeError("Could not install the image packages. Run ./install-local.sh") @@ -1470,10 +1488,21 @@ def load_pipeline(s): device, dtype = "cuda", torch.float16 else: device, dtype = "cpu", torch.float32 + threads = int(s.get("cpu_threads", 0) or 0) + if threads > 0: + torch.set_num_threads(threads) + logger.info("Limiting image generation to %d CPU threads", threads) logger.info("Loading image model %s on %s (first run downloads several GB)", model, device) pipe = AutoPipelineForText2Image.from_pretrained(model, torch_dtype=dtype, variant="fp16" if dtype == torch.float16 else None) pipe = pipe.to(device) pipe.set_progress_bar_config(disable=True) + if s.get("low_memory", True): + for enable in ("enable_attention_slicing", "enable_vae_slicing", "enable_vae_tiling"): + try: + getattr(pipe, enable)() + except Exception: + pass + logger.info("Low memory mode is on for image generation") pipeline["id"] = model pipeline["obj"] = pipe logger.info("Image model ready") @@ -1800,6 +1829,8 @@ def run_job(s): logger.info("Loaded %d existing channel titles to avoid repeats", len(s["_channel_titles"])) script, tags, compliance = produce_script(s, src) remember_story(s["profile_id"], script) + if engine == "local": + unload_ollama(s, s["ollama_model"]) compliance["engine"] = engine update_history(job_id, title=script["title"], notes="; ".join(compliance.get("reviewer_notes", []))[:400]) write_json(job_dir / "script.json", {"profile": s["name"], "source": src, "script": script, "tags": tags}) @@ -1909,6 +1940,8 @@ def run_job(s): build_scene(i, script["scenes"][i - 1], clips[i - 1]) pending = failed + if kids: + unload_ollama(s, s["vision_model"]) anatomy_note = f"Possible anatomy glitches kept in scenes {', '.join(str(i) for i in anatomy_flags)}" if anatomy_flags else "" if kids: compliance["vision_checks"] = vision_log diff --git a/static/index.html b/static/index.html index ffb5ae5..1951279 100644 --- a/static/index.html +++ b/static/index.html @@ -319,6 +319,10 @@ const FIELDS=[ ["piper_model","Piper voice file","wide","Blank uses the first voice found in the voices folder under Models","g"], ["gemini_api_key","Gemini API key (Veo)","password","Only needed for the Veo engine. Billing must be enabled","g"], ["veo_model","Veo model","text","e.g. veo-3.1-fast-generate-preview","g"], +["","Performance","header","Shared by all profiles. Lower settings here if the Mac becomes slow or unstable",""], +["low_memory","Low memory mode","checkbox","Generates images in smaller pieces. A little slower, much less memory. Leave on for Macs with 16 GB or less","g"], +["free_models_between_stages","Free models between stages","checkbox","Unloads the writing model before images are drawn and the vision model after checks","g"], +["cpu_threads","CPU threads","select:0=Automatic|2|4|6|8|10","Caps how many cores image generation uses so the rest of the Mac stays responsive","g"], ["","Writing and timing","header","Shared by all profiles",""], ["ollama_url","Ollama URL","text","","g"], ["ollama_model","Ollama model","model:text","Writes scripts and tags. Models not installed yet are downloaded automatically","g"], @@ -332,7 +336,7 @@ const FIELDS=[ ]; const DAYS=[["mon","Mon"],["tue","Tue"],["wed","Wed"],["thu","Thu"],["fri","Fri"],["sat","Sat"],["sun","Sun"]]; const TIMES=[];for(let h=0;h<24;h++)for(const m of [0,30]){const v=String(h).padStart(2,"0")+":"+String(m).padStart(2,"0");const hr=h%12||12;TIMES.push([v,hr+":"+String(m).padStart(2,"0")+(h<12?" AM":" PM")])} -const NUMERIC=new Set(["sd_steps","sd_guidance","sd_width","sd_height","trending_days","trending_count","videos_per_day","minutes_between_videos","target_seconds","clip_seconds","ollama_timeout","error_cooldown_minutes","tts_rate"]); +const NUMERIC=new Set(["sd_steps","sd_guidance","sd_width","sd_height","trending_days","trending_count","videos_per_day","minutes_between_videos","target_seconds","clip_seconds","ollama_timeout","error_cooldown_minutes","tts_rate","cpu_threads"]); const $=s=>document.querySelector(s); const esc=s=>String(s??"").replace(/[&<>"]/g,c=>({"&":"&","<":"<",">":">",'"':"""}[c])); async function api(path,opts={}){const r=await fetch(path,{headers:{"Content-Type":"application/json"},...opts});const j=await r.json().catch(()=>({}));if(!r.ok)throw new Error(j.detail||r.statusText);return j}