From b76ba153651c10b361928d87c2de598c88555e54 Mon Sep 17 00:00:00 2001 From: JustinOros Date: Thu, 24 Jul 2025 10:36:24 -0700 Subject: [PATCH] Added support for email notifications via SNMP. --- diff-web.py | 113 ++++++++++++++++++++++++++-------------------------- 1 file changed, 57 insertions(+), 56 deletions(-) diff --git a/diff-web.py b/diff-web.py index b7b280b..c483b39 100644 --- a/diff-web.py +++ b/diff-web.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 -# Description: Monitor a website for changes. -# Usage: python3 diff-web.py +# Description: Monitor a website for changes. +# Usage: python3 diff-web.py https://example.com --email user@example.com # Author: Justin Oros # Source: https://github.com/JustinOros @@ -9,50 +9,36 @@ import hashlib import os import time import argparse -from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode +from urllib.parse import urlparse, urlunparse, parse_qs, urlencode from datetime import datetime -from bs4 import BeautifulSoup, Comment +from bs4 import BeautifulSoup +import smtplib +from email.mime.text import MIMEText -# List of known tracking parameters to strip -TRACKING_PARAMS = { - 'utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', - 'fbclid', 'gclid', 'mc_cid', 'mc_eid', 'ref', 'ref_src' -} +SMTP_SERVER = "smtp.gmail.com" +SMTP_PORT = 587 +SMTP_USERNAME = os.getenv("EMAIL_USER") +SMTP_PASSWORD = os.getenv("EMAIL_PASS") +EMAIL_FROM = SMTP_USERNAME def normalize_url(url): - if not urlparse(url).scheme: - url = 'https://' + url - return strip_tracking_params(url) - -def strip_tracking_params(url): parsed = urlparse(url) - clean_query = [(k, v) for k, v in parse_qsl(parsed.query) if k not in TRACKING_PARAMS] - new_query = urlencode(clean_query) + if not parsed.scheme: + url = 'https://' + url + parsed = urlparse(url) + + query = parse_qs(parsed.query) + stripped_query = {k: v for k, v in query.items() if not k.startswith(('utm_', 'fbclid', 'gclid'))} + new_query = urlencode(stripped_query, doseq=True) return urlunparse(parsed._replace(query=new_query)) def fetch_content(url): response = requests.get(url, timeout=10) response.raise_for_status() - return response.text - -def clean_content(html): - soup = BeautifulSoup(html, 'html.parser') - - # Remove dynamic or non-content elements - for tag in soup(['script', 'style', 'noscript', 'meta']): + soup = BeautifulSoup(response.text, 'html.parser') + for tag in soup(["script", "style", "noscript", "iframe", "svg", "canvas"]): tag.decompose() - - # Remove HTML comments - for comment in soup.find_all(string=lambda text: isinstance(text, Comment)): - comment.extract() - - # Remove tracking params from all anchor links - for a in soup.find_all('a', href=True): - a['href'] = strip_tracking_params(a['href']) - - # Get visible text, normalize whitespace - text = soup.get_text(separator=' ', strip=True) - return ' '.join(text.split()) + return soup.get_text(separator=' ', strip=True) def get_hash(content): return hashlib.sha256(content.encode('utf-8')).hexdigest() @@ -67,16 +53,32 @@ def get_cache_filename(url): domain = urlparse(url).netloc return f".{domain}.cache" -def timestamp(): - return datetime.now().strftime("%Y-%m-%d %H:%M:%S") - def log_message(message, log_file, quiet=False): if not quiet: print(message) with open(log_file, 'a') as f: f.write(message + '\n') -def monitor_website(url, interval, log_file_override=None, max_checks=None, quiet=False): +def timestamp(): + return datetime.now().strftime('%Y-%m-%d %H:%M:%S') + +def send_email(subject, body, recipients): + if not SMTP_USERNAME or not SMTP_PASSWORD: + print("Email credentials not configured in environment.") + return + msg = MIMEText(body) + msg['Subject'] = subject + msg['From'] = EMAIL_FROM + msg['To'] = ", ".join(recipients) + try: + with smtplib.SMTP(SMTP_SERVER, SMTP_PORT) as server: + server.starttls() + server.login(SMTP_USERNAME, SMTP_PASSWORD) + server.sendmail(EMAIL_FROM, recipients, msg.as_string()) + except Exception as e: + print(f"Failed to send email: {e}") + +def monitor_website(url, interval, log_file_override=None, quiet=False, recipients=None): url = normalize_url(url) domain = urlparse(url).netloc log_file = get_log_filename(url, log_file_override) @@ -84,13 +86,10 @@ def monitor_website(url, interval, log_file_override=None, max_checks=None, quie log_message(f"[{timestamp()}] Monitoring {domain}", log_file, quiet) - checks_done = 0 - while True: try: - raw_html = fetch_content(url) - cleaned_text = clean_content(raw_html) - current_hash = get_hash(cleaned_text) + current_content = fetch_content(url) + current_hash = get_hash(current_content) except Exception as e: log_message(f"[{timestamp()}] [ERROR] Failed to fetch {url}: {e}", log_file, quiet) time.sleep(interval) @@ -101,20 +100,18 @@ def monitor_website(url, interval, log_file_override=None, max_checks=None, quie with open(cache_file, 'r') as f: old_hash = f.read().strip() is_changed = current_hash != old_hash - if is_changed: log_message(f"[{timestamp()}] Change detected on {url}", log_file, quiet) + if recipients: + subject = f"Website Change Detected: {domain}" + body = f"A change was detected on {url} at {timestamp()}." + send_email(subject, body, recipients) else: log_message(f"[{timestamp()}] First-time check for {url}: storing baseline.", log_file, quiet) with open(cache_file, 'w') as f: f.write(current_hash) - checks_done += 1 - if max_checks is not None and checks_done >= max_checks: - log_message(f"[{timestamp()}] Reached max checks ({max_checks}). Stopping.", log_file, quiet) - break - time.sleep(interval) if __name__ == "__main__": @@ -124,15 +121,19 @@ if __name__ == "__main__": help="Time interval between checks in seconds (default: 60).") parser.add_argument("-l", "--log", type=str, help="Optional log file name (default: domain.tld.log)") - parser.add_argument("-c", "--count", type=int, - help="Optional number of times to check before stopping.") parser.add_argument("-q", "--quiet", action="store_true", - help="Quiet mode: suppress console output") + help="Suppress console output.") + parser.add_argument("-e", "--email", nargs='+', + help="Email address(es) to notify on changes.") args = parser.parse_args() + log_file = get_log_filename(args.url, args.log) try: - monitor_website(args.url, args.time, args.log, args.count, args.quiet) + monitor_website(args.url, args.time, args.log, args.quiet, args.email) except KeyboardInterrupt: - log_file = get_log_filename(args.url, args.log) - log_message(f"[{timestamp()}] Monitoring stopped by user (KeyboardInterrupt).", log_file, args.quiet) + message = f"[{timestamp()}] Monitoring halted by user (^C)." + if not args.quiet: + print(message) + with open(log_file, 'a') as f: + f.write(message + '\n')