diff --git a/app.py b/app.py index 2d2bc6e..f16331a 100644 --- a/app.py +++ b/app.py @@ -46,7 +46,7 @@ MIN_FREE_GB = 2 LATIN_LANGS = {"en", "es", "fr", "de", "it", "pt", "nl", "sv", "no", "da", "fi", "pl", "cs", "ro", "hu", "tr", "id", "ms", "vi", "tl"} REVIEW_MODES = ("advisory", "block", "off") ENGINES = ("local", "veo") -TTS_ENGINES = ("auto", "say", "espeak", "piper") +TTS_ENGINES = ("auto", "kokoro", "piper", "say", "espeak") SUGGESTED_TEXT_MODELS = ["llama3.1:8b", "llama3.2:3b", "qwen2.5:7b", "mistral:7b"] SUGGESTED_VISION_MODELS = ["moondream", "llava:7b", "llama3.2-vision", "qwen2.5vl:7b"] SMALL_VISION_MODELS = ("moondream", "llava-phi3", "bakllava") @@ -69,9 +69,10 @@ GLOBAL_DEFAULTS = { "sd_width": 768, "sd_height": 1344, "tts_engine": "auto", - "tts_voice": "Samantha", + "tts_voice": "af_heart", "tts_rate": 170, "piper_model": "", + "auto_install": True, "gemini_api_key": "", "veo_model": "veo-3.1-fast-generate-preview", "aspect_ratio": "9:16", @@ -182,6 +183,7 @@ state_lock = threading.RLock() settings_lock = threading.RLock() pipe_lock = threading.Lock() pipeline = {"id": None, "obj": None} +kokoro = {"obj": None} stop_event = threading.Event() runner = None pending_flow = None @@ -280,6 +282,40 @@ def free_gb(path): return None +def venv_python(): + cand = BASE / ".venv" / "bin" / "python" + return str(cand) if cand.exists() else sys.executable + + +def pip_install(*args): + cmd = [venv_python(), "-m", "pip", "install", *args] + logger.info("Installing %s (this can take a few minutes)", " ".join(args)) + r = subprocess.run(cmd, capture_output=True, text=True) + if r.returncode != 0: + logger.warning("Install failed: %s", (r.stderr or r.stdout).strip()[-500:]) + return False + logger.info("Installed %s", " ".join(args)) + return True + + +def have_module(name): + try: + __import__(name) + return True + except Exception: + return False + + +def ensure_module(s, module, package): + if have_module(module): + return True + if not s.get("auto_install", True): + logger.warning("%s is missing and auto install is off", package) + return False + set_stage(f"installing {package}") + return pip_install(package) and have_module(module) + + def voices_dir(s): return resolve_sub(s, "models_dir", "models") / "voices" @@ -295,6 +331,37 @@ def find_piper_voice(s): return "" +def ensure_piper_voice(s): + voice = find_piper_voice(s) + if voice: + return voice + if not s.get("auto_install", True): + return "" + vdir = voices_dir(s) + vdir.mkdir(parents=True, exist_ok=True) + base = "https://huggingface.co/rhasspy/piper-voices/resolve/main/en/en_US/lessac/medium/en_US-lessac-medium" + set_stage("downloading narration voice") + try: + for suffix in (".onnx", ".onnx.json"): + dest = vdir / f"en_US-lessac-medium{suffix}" + if dest.exists(): + continue + logger.info("Downloading narration voice%s", suffix) + with requests.get(base + suffix, stream=True, timeout=(10, 600)) as r: + r.raise_for_status() + with open(dest, "wb") as fh: + for chunk in r.iter_content(chunk_size=1 << 20): + check() + fh.write(chunk) + logger.info("Narration voice ready") + except Cancelled: + raise + except Exception as e: + logger.warning("Could not download the Piper voice: %s", e) + return "" + return find_piper_voice(s) + + def piper_binary(): for cand in (BASE / ".venv" / "bin" / "piper", Path("/opt/homebrew/bin/piper")): if cand.exists(): @@ -302,6 +369,24 @@ def piper_binary(): return shutil.which("piper") or "" +def kokoro_speak(s, text, out_wav): + import numpy as np + import soundfile as sf + from kokoro import KPipeline + with pipe_lock: + if kokoro["obj"] is None: + logger.info("Loading narration model (first run downloads about 350 MB)") + kokoro["obj"] = KPipeline(lang_code="a") + logger.info("Narration model ready") + voice = s["tts_voice"].strip() or "af_heart" + if not re.fullmatch(r"[a-z]{2}_[a-z_]+", voice): + voice = "af_heart" + chunks = [audio for _, _, audio in kokoro["obj"](text, voice=voice, speed=0.95)] + if not chunks: + raise RuntimeError("Narration model produced no audio") + sf.write(str(out_wav), np.concatenate(chunks), 24000) + + def apply_paths(s): models = writable(resolve_sub(s, "models_dir", "models")) temp = writable(resolve_sub(s, "temp_dir", "tmp")) @@ -599,6 +684,24 @@ def ollama_pull(s, model): return True +def ensure_local_deps(s): + if not ensure_module(s, "torch", "torch") or not ensure_module(s, "diffusers", "diffusers accelerate safetensors transformers".split()[0]): + raise RuntimeError("Could not install the image packages. Run ./install-local.sh") + for module, package in (("accelerate", "accelerate"), ("safetensors", "safetensors"), ("transformers", "transformers")): + ensure_module(s, module, package) + want = s["tts_engine"] if s["tts_engine"] in TTS_ENGINES else "auto" + if want in ("auto", "kokoro") and not have_module("kokoro"): + if ensure_module(s, "kokoro", "kokoro") : + ensure_module(s, "soundfile", "soundfile") + elif want == "kokoro": + logger.warning("Kokoro could not be installed, narration will fall back") + if want == "piper": + if not piper_binary(): + ensure_module(s, "piper", "piper-tts") + ensure_piper_voice(s) + logger.info("Narration engine: %s", pick_tts(s)) + + def ensure_models(s, engine): needed = [s["ollama_model"].strip()] if s["made_for_kids"] and s["vision_model"].strip(): @@ -945,7 +1048,7 @@ def load_pipeline(s): import torch from diffusers import AutoPipelineForText2Image except ImportError: - raise RuntimeError("Local engine needs extra packages. Run ./install-local.sh") + raise RuntimeError("Image packages are missing. Turn on auto install in Settings or run ./install-local.sh") if torch.backends.mps.is_available(): device, dtype = "mps", torch.float16 elif torch.cuda.is_available(): @@ -979,36 +1082,22 @@ def generate_image(s, prompt, path): logger.info("Generated %s in %.1fs", path.name, (datetime.now() - started).total_seconds()) -def tts_speak(s, text, out_wav): +def pick_tts(s): engine = s["tts_engine"] if s["tts_engine"] in TTS_ENGINES else "auto" - voice = find_piper_voice(s) - binary = piper_binary() - if engine == "auto": - if voice and binary: - engine = "piper" - elif sys.platform == "darwin": - engine = "say" - else: - engine = "espeak" - text = text.strip() or "..." - if engine == "piper": - if not voice: - raise RuntimeError("Piper needs a voice file. Run ./install-local.sh or set the path in Settings") - if not binary: - raise RuntimeError("Piper is not installed. Run ./install-local.sh") - r = subprocess.run([binary, "--model", voice, "--output_file", str(out_wav)], input=text, capture_output=True, text=True) - if r.returncode != 0: - raise RuntimeError(f"piper failed: {r.stderr.strip()[:200]}") - elif engine == "say": + if engine != "auto": + return engine + if have_module("kokoro"): + return "kokoro" + if find_piper_voice(s) and piper_binary(): + return "piper" + return "say" if sys.platform == "darwin" else "espeak" + + +def system_speak(s, text, out_wav): + if sys.platform == "darwin": aiff = out_wav.with_suffix(".aiff") - cmd = ["say", "-r", str(int(s["tts_rate"])), "-o", str(aiff)] - if s["tts_voice"].strip(): - cmd += ["-v", s["tts_voice"].strip()] - cmd.append(text) + cmd = ["say", "-r", str(int(s["tts_rate"])), "-o", str(aiff), text] r = subprocess.run(cmd, capture_output=True, text=True) - if r.returncode != 0 and s["tts_voice"].strip(): - logger.warning("Voice '%s' is not installed, using the system default", s["tts_voice"].strip()) - r = subprocess.run(["say", "-r", str(int(s["tts_rate"])), "-o", str(aiff), text], capture_output=True, text=True) if r.returncode != 0: raise RuntimeError(f"say failed: {r.stderr.strip()[:200]}") conv = subprocess.run(["ffmpeg", "-y", "-i", str(aiff), "-ar", "44100", "-ac", "2", str(out_wav)], capture_output=True, text=True) @@ -1016,10 +1105,35 @@ def tts_speak(s, text, out_wav): if conv.returncode != 0: raise RuntimeError("Could not convert narration audio") else: - v = s["tts_voice"].strip() or "en-us" - r = subprocess.run(["espeak-ng", "-v", v, "-s", str(int(s["tts_rate"])), "-w", str(out_wav), text], capture_output=True, text=True) + r = subprocess.run(["espeak-ng", "-v", "en-us", "-s", str(int(s["tts_rate"])), "-w", str(out_wav), text], capture_output=True, text=True) if r.returncode != 0: raise RuntimeError(f"espeak-ng failed: {r.stderr.strip()[:200]}") + + +def tts_speak(s, text, out_wav): + text = text.strip() or "..." + engine = pick_tts(s) + try: + if engine == "kokoro": + kokoro_speak(s, text, out_wav) + elif engine == "piper": + voice = ensure_piper_voice(s) + binary = piper_binary() + if not voice or not binary: + raise RuntimeError("Piper is not available") + r = subprocess.run([binary, "--model", voice, "--output_file", str(out_wav)], input=text, capture_output=True, text=True) + if r.returncode != 0: + raise RuntimeError(f"piper failed: {r.stderr.strip()[:200]}") + else: + system_speak(s, text, out_wav) + except Cancelled: + raise + except Exception as e: + if engine in ("kokoro", "piper"): + logger.warning("%s narration failed (%s), using the system voice for this scene", engine, e) + system_speak(s, text, out_wav) + else: + raise if not out_wav.exists() or out_wav.stat().st_size < 1000: raise RuntimeError("Narration audio was empty") @@ -1184,7 +1298,9 @@ def run_job(s): prepare_storage(s) ensure_models(s, engine) if engine == "local": - logger.info("Narration: %s | Image: %s at %sx%s", "piper" if find_piper_voice(s) and piper_binary() else s["tts_engine"], s["sd_model"], s["sd_width"], s["sd_height"]) + ensure_local_deps(s) + if engine == "local": + logger.info("Image model: %s at %sx%s", s["sd_model"], s["sd_width"], s["sd_height"]) logger.info("Profile: %s | Engine: %s | Niche: %s | Made for kids: %s | Upload category: %s", s["name"], engine, s["channel_niche"] or "none", kids, s["upload_category_id"]) if kids: logger.info( diff --git a/static/index.html b/static/index.html index 25cfb7a..a7a6dc6 100644 --- a/static/index.html +++ b/static/index.html @@ -126,6 +126,7 @@ const FIELDS=[ ["temp_dir","Temp folder","wide","Scratch space used while rendering. Blank puts it in the storage folder","g"], ["delete_after_publish","Delete videos after publishing","checkbox","Removes the job folder once the upload succeeds instead of moving it to published","g"], ["keep_clips","Keep individual clips","checkbox","Keeps the scene clips, images, and narration alongside final.mp4","g"], +["auto_install","Install missing pieces automatically","checkbox","Downloads models and installs packages when a run needs them","g"], ["","Video engine (shared by all profiles)","header","",""], ["video_engine","Video engine","select:local=Local (free: AI images, narration, motion)|veo=Veo (paid Google API)","Local runs entirely on this machine at no cost. Veo makes true AI video but needs billing","g"], ["sd_model","Image model","text","Hugging Face model id, e.g. stabilityai/sdxl-turbo or stabilityai/stable-diffusion-xl-base-1.0","g"], @@ -133,9 +134,9 @@ const FIELDS=[ ["sd_guidance","Image guidance","number","0 for turbo models, 7 for standard models","g"], ["sd_width","Image width","number","768 avoids duplicated subjects. Larger is slower","g"], ["sd_height","Image height","number","1344 is the correct vertical ratio for SDXL","g"], -["tts_engine","Narration engine","select:auto=Auto|piper=Piper (natural)|say=macOS say|espeak=espeak-ng","Auto uses Piper when a voice file is present, otherwise the system voice","g"], -["tts_voice","Narration voice","text","macOS: Samantha, Alex, Daniel. espeak: en-us","g"], -["tts_rate","Narration speed","number","macOS words per minute (170). espeak uses a similar scale","g"], +["tts_engine","Narration engine","select:auto=Auto (best available)|kokoro=Kokoro (most natural)|piper=Piper|say=macOS say|espeak=espeak-ng","Missing engines are installed automatically on the next run","g"], +["tts_voice","Narration voice","text","Kokoro: af_heart, af_bella, am_michael, bf_emma. Ignored by the system voice","g"], +["tts_rate","Narration speed","number","Only used by the system voice. Words per minute","g"], ["piper_model","Piper voice file","wide","Blank uses the first voice found in the voices folder under Models","g"], ["gemini_api_key","Gemini API key (Veo)","password","Only needed for the Veo engine. Billing must be enabled","g"], ["veo_model","Veo model","text","e.g. veo-3.1-fast-generate-preview","g"],