skip unchanged folders in cleanup and lower service worker process priority

This commit is contained in:
Justin Oro
2026-08-27 00:09:25 -07:00
parent 981f4bf40e
commit f6cbaa1a6f
2 changed files with 82 additions and 10 deletions
+3 -1
View File
@@ -170,7 +170,9 @@ python plex-library-tool.py --service start # start using the curr
python plex-library-tool.py --service stop # stop the service python plex-library-tool.py --service stop # stop the service
``` ```
Each cycle runs the equivalent of `-rcy` (rename + cleanup, no prompts) on every configured share. It runs as a detached background process, so it keeps running after you close the terminal. Progress is written to `logs/service.log` instead of the screen. Settings (shares + interval) are stored in `service.json` next to the script; you can add more shares later by running `--service <path>` again while it's running, no need to stop it first. Each cycle runs the equivalent of `-rcy` (rename + cleanup, no prompts) on every configured share. It runs as a detached background process, so it keeps running after you close the terminal, at a lowered OS priority so it yields to things like Plex during playback. Progress is written to `logs/service.log` instead of the screen. Settings (shares + interval) are stored in `service.json` next to the script; you can add more shares later by running `--service <path>` again while it's running, no need to stop it first.
Both rename and cleanup skip folders that haven't changed since the last pass, so most cycles do little more than a quick unchanged-folder check rather than a full rescan.
The service reuses your saved TMDb API key and language from `.env`, so run the script normally at least once first if you haven't already. The service reuses your saved TMDb API key and language from `.env`, so run the script normally at least once first if you haven't already.
+79 -9
View File
@@ -2445,6 +2445,19 @@ def save_scan_cache(cache):
CACHE_FILE.write_text(json.dumps(cache, indent=2)) CACHE_FILE.write_text(json.dumps(cache, indent=2))
def compute_cleanup_rules_signature(delete_folder_names, delete_file_patterns, keep_languages, delete_languages):
hasher = hashlib.sha256()
for name in sorted(delete_folder_names):
hasher.update(f"f:{name.lower()}\n".encode("utf-8"))
for pattern in sorted(delete_file_patterns):
hasher.update(f"p:{pattern.lower()}\n".encode("utf-8"))
for lang in sorted(keep_languages or []):
hasher.update(f"k:{lang}\n".encode("utf-8"))
for lang in sorted(delete_languages or []):
hasher.update(f"d:{lang}\n".encode("utf-8"))
return hasher.hexdigest()
def parse_simple_yaml_list(path, key): def parse_simple_yaml_list(path, key):
if not path.exists(): if not path.exists():
return [] return []
@@ -2636,16 +2649,13 @@ def detect_srt_language(path):
scores = {lang: sum(1 for w in words if w in stopwords) for lang, stopwords in SUBTITLE_STOPWORDS.items()} scores = {lang: sum(1 for w in words if w in stopwords) for lang, stopwords in SUBTITLE_STOPWORDS.items()}
best_lang = max(scores, key=scores.get) best_lang = max(scores, key=scores.get)
best_score = scores[best_lang] best_score = scores[best_lang]
en_score = scores.get("en", 0) second_score = max((s for lang, s in scores.items() if lang != best_lang), default=0)
vprint(f" {path.name}: word count={total}, scores={scores}") vprint(f" {path.name}: word count={total}, scores={scores}")
if best_lang == "en": if best_score >= 5 and best_score >= second_score * 1.5 and best_score >= second_score + 3:
vprint(f" {path.name}: classified as en") vprint(f" {path.name}: classified as {best_lang} (second_score={second_score}, {best_lang}_score={best_score})")
return "en"
if best_score >= 5 and best_score >= en_score * 1.5 and best_score >= en_score + 3:
vprint(f" {path.name}: classified as {best_lang} (en_score={en_score}, {best_lang}_score={best_score})")
return best_lang return best_lang
vprint(f" {path.name}: inconclusive (en_score={en_score}, best={best_lang}:{best_score}) -> unknown") vprint(f" {path.name}: inconclusive (best={best_lang}:{best_score}, second={second_score}) -> unknown")
return "unknown" return "unknown"
@@ -2693,10 +2703,12 @@ def matches_file_name(name, patterns, path=None):
return True return True
def find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_languages=None, delete_languages=None): def find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_languages=None, delete_languages=None, unchanged_folder_names=None):
targets = [] targets = []
keep_languages = keep_languages or set() keep_languages = keep_languages or set()
delete_languages = delete_languages or set() delete_languages = delete_languages or set()
unchanged_folder_names = unchanged_folder_names or set()
share_root = Path(share)
def walk(folder): def walk(folder):
vprint(f"Scanning folder: {folder}") vprint(f"Scanning folder: {folder}")
@@ -2711,6 +2723,9 @@ def find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_
vprint(f" Match (folder name): {entry}") vprint(f" Match (folder name): {entry}")
targets.append(("folder", entry)) targets.append(("folder", entry))
continue continue
if entry.parent == share_root and entry.name in unchanged_folder_names:
vprint(f" Unchanged since last cleanup, skipping: {entry}")
continue
vprint(f" No match, descending into: {entry}") vprint(f" No match, descending into: {entry}")
walk(entry) walk(entry)
elif entry.is_file(): elif entry.is_file():
@@ -2803,11 +2818,41 @@ def run_cleanup(args, log):
print(f"Edit {DELETE_FILE.name} to enable cleanup.") print(f"Edit {DELETE_FILE.name} to enable cleanup.")
return return
rules_signature = compute_cleanup_rules_signature(
delete_folder_names, delete_file_patterns, keep_languages, delete_languages
)
cache = load_scan_cache()
share_key = str(Path(share).resolve())
cleanup_cache_key = f"cleanup::{share_key}"
cleanup_baseline = cache.get(cleanup_cache_key)
if not isinstance(cleanup_baseline, dict):
cleanup_baseline = {}
share_path = Path(share)
try:
top_level_folders = [
p for p in share_path.iterdir() if p.is_dir() and not p.name.startswith(".")
]
except OSError:
top_level_folders = []
unchanged_folder_names = set()
if not getattr(args, "force", False):
for folder in top_level_folders:
sig = compute_folder_signature(folder)
combined = f"{sig}|{rules_signature}"
if cleanup_baseline.get(folder.name) == combined:
unchanged_folder_names.add(folder.name)
if unchanged_folder_names:
vprint(f"Skipping content scan for {len(unchanged_folder_names)} unchanged folder(s)")
print() print()
print(f"Scanning: {share}") print(f"Scanning: {share}")
print() print()
targets = find_cleanup_targets(share, delete_folder_names, delete_file_patterns, keep_languages, delete_languages) targets = find_cleanup_targets(
share, delete_folder_names, delete_file_patterns, keep_languages, delete_languages, unchanged_folder_names
)
vprint(f"Scan complete. {len(targets)} item(s) matched.") vprint(f"Scan complete. {len(targets)} item(s) matched.")
test_mode = args.test is not None test_mode = args.test is not None
@@ -2888,6 +2933,18 @@ def run_cleanup(args, log):
folders_deleted = moved_folders + empty_removed folders_deleted = moved_folders + empty_removed
folders_deleted_skipped = skipped_folders + empty_skipped folders_deleted_skipped = skipped_folders + empty_skipped
new_cleanup_baseline = {}
for folder in top_level_folders:
if folder.name in unchanged_folder_names:
new_cleanup_baseline[folder.name] = cleanup_baseline.get(folder.name)
continue
if not folder.exists():
continue
sig = compute_folder_signature(folder)
new_cleanup_baseline[folder.name] = f"{sig}|{rules_signature}"
cache[cleanup_cache_key] = new_cleanup_baseline
save_scan_cache(cache)
if not targets and not folders_deleted and not moved_files and not folders_deleted_skipped and not skipped_files: if not targets and not folders_deleted and not moved_files and not folders_deleted_skipped and not skipped_files:
print("Nothing to clean up.") print("Nothing to clean up.")
return return
@@ -4028,6 +4085,18 @@ def run_service_pass(path_str, media_type):
run_cleanup(args, log) run_cleanup(args, log)
def lower_process_priority():
try:
if platform.system() == "Windows":
BELOW_NORMAL_PRIORITY_CLASS = 0x00004000
handle = ctypes.windll.kernel32.GetCurrentProcess()
ctypes.windll.kernel32.SetPriorityClass(handle, BELOW_NORMAL_PRIORITY_CLASS)
else:
os.nice(10)
except (AttributeError, OSError):
pass
def run_service_worker(): def run_service_worker():
global VERBOSE global VERBOSE
VERBOSE = False VERBOSE = False
@@ -4035,6 +4104,7 @@ def run_service_worker():
def ts(): def ts():
return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
lower_process_priority()
print(f"[{ts()}] Service worker started (PID {os.getpid()}).") print(f"[{ts()}] Service worker started (PID {os.getpid()}).")
sys.stdout.flush() sys.stdout.flush()