#!/usr/bin/env python3
"""
m3u8 Clipboard Watcher
======================
Monitors clipboard for .m3u8 links, deduplicates, and downloads
the highest quality video via yt-dlp (with ffmpeg fallback).
Converts output to .mp4 if needed.
Dependencies:
pip install pyperclip yt-dlp
System requirements:
- ffmpeg must be installed and on PATH (yt-dlp uses it for remuxing)
Usage:
python m3u8_watcher.py
python m3u8_watcher.py --output-dir ./downloads
python m3u8_watcher.py --interval 1.5
"""
import argparse
import os
import re
import signal
import subprocess
import sys
import time
import threading
from pathlib import Path
from datetime import datetime
try:
import pyperclip
except ImportError:
print("Missing dependency: pip install pyperclip")
sys.exit(1)
# ---------------------------------------------------------------------------
# Config
# ---------------------------------------------------------------------------
M3U8_PATTERN = re.compile(r'https?://[^\s<>"\']+\.(?:m3u8|ts|mpd)\b', re.IGNORECASE)
DEFAULT_INTERVAL = 1.0 # seconds between clipboard checks
MAX_RETRIES = 3
STATE_FILE = "m3u8_history.json"
# ---------------------------------------------------------------------------
# Globals
# ---------------------------------------------------------------------------
running = True
download_queue = []
queue_lock = threading.Lock()
seen_links = set()
seen_lock = threading.Lock()
def signal_handler(sig, frame):
global running
print("\n[!] Interrupt received, finishing current download then exiting...")
running = False
signal.signal(signal.SIGINT, signal_handler)
# ---------------------------------------------------------------------------
# Utilities
# ---------------------------------------------------------------------------
def timestamp():
return datetime.now().strftime("%H:%M:%S")
def load_history(output_dir: Path):
"""Load previously downloaded links so reruns don't re-download."""
global seen_links
state_file = output_dir / STATE_FILE
if state_file.exists():
try:
import json
with open(state_file, "r", encoding="utf-8") as f:
data = json.load(f)
seen_links.update(data.get("links", []))
print(f"[*] Loaded {len(seen_links)} previously seen links from history.")
except Exception:
pass
def save_history(output_dir: Path):
"""Persist the seen-links set to disk."""
state_file = output_dir / STATE_FILE
try:
import json
with open(state_file, "w", encoding="utf-8") as f:
json.dump({"links": sorted(seen_links)}, f, indent=2)
except Exception:
pass
def check_dependencies():
"""Verify yt-dlp and ffmpeg are available."""
missing = []
# yt-dlp
try:
subprocess.run(
["yt-dlp", "--version"],
capture_output=True, timeout=10
)
except FileNotFoundError:
missing.append("yt-dlp (pip install yt-dlp)")
except Exception:
pass
# ffmpeg
try:
subprocess.run(
["ffmpeg", "-version"],
capture_output=True, timeout=10
)
except FileNotFoundError:
missing.append("ffmpeg. Search on google how to install ffmpeg")
if missing:
print("[!] Missing required tools:")
for m in missing:
print(f" - {m}")
sys.exit(1)
# ---------------------------------------------------------------------------
# Clipboard monitoring
# ---------------------------------------------------------------------------
def poll_clipboard(output_dir: Path, interval: float):
"""Poll clipboard for new m3u8 links and enqueue them."""
last_clip = ""
print(f"[*] Clipboard watcher active (poll every {interval}s)")
print("[*] Copy an m3u8 / .ts / .mpd link to start a download.\n")
while running:
try:
current = pyperclip.paste().strip()
except Exception:
time.sleep(interval)
continue
if current and current != last_clip:
last_clip = current
matches = M3U8_PATTERN.findall(current)
for url in matches:
with seen_lock:
if url in seen_links:
print(f"[{timestamp()}] [SKIP] Duplicate: {url[:80]}...")
continue
seen_links.add(url)
with queue_lock:
download_queue.append(url)
print(f"[{timestamp()}] [QUEUED] {url[:80]}...")
# Persist after each new link so reruns remember
save_history(output_dir)
time.sleep(interval)
# ---------------------------------------------------------------------------
# Downloader
# ---------------------------------------------------------------------------
def download_m3u8(url: str, output_dir: Path, attempt: int = 1) -> bool:
"""
Download an m3u8 stream using yt-dlp with best quality, remux to mp4.
yt-dlp args explained:
-f "bv*+ba/b" -> best video+audio merged, or best single stream
--remux-video mp4 -> remux container to mp4 if needed
--merge-output-format mp4 -> force mp4 output
-o template -> output file naming
--no-part -> no .part files (cleaner)
--no-overwrites -> skip if file exists
--retries 5 -> handle network hiccups
"""
safe_name = re.sub(r'[^\w\-.]', '_', url.split("?")[0].split("/")[-1])
# Strip extension from name so yt-dlp appends the right one
safe_name = re.sub(r'\.(m3u8|ts|mp4|mkv)$', '', safe_name, flags=re.IGNORECASE)
if not safe_name or len(safe_name) < 3:
safe_name = f"stream_{int(time.time())}"
output_template = str(output_dir / f"{safe_name}.%(ext)s")
cmd = [
"yt-dlp",
"-f", "bv*+ba/b",
"--remux-video", "mp4",
"--merge-output-format", "mp4",
"-o", output_template,
"--no-part",
"--retries", "5",
"--retry-sleep", "3",
"--fragment-retries", "5",
"--concurrent-fragments", "4",
"--embed-metadata",
"--embed-thumbnail",
url
]
print(f"\n[{timestamp()}] [DOWNLOAD] {url[:80]}...")
print(f" -> Output: {safe_name}.mp4")
try:
result = subprocess.run(
cmd,
capture_output=True,
text=True,
timeout=3600, # 1 hour max per stream
)
if result.returncode == 0:
print(f"[{timestamp()}] [DONE] {safe_name}.mp4")
return True
else:
stderr = result.stderr.strip()
print(f"[{timestamp()}] [WARN] yt-dlp exit {result.returncode}")
# Print last few lines of stderr for debugging
lines = stderr.splitlines()
for line in lines[-5:]:
print(f" {line}")
# Fallback: try direct ffmpeg download
if attempt <= MAX_RETRIES:
print(f"[{timestamp()}] [RETRY] Attempting ffmpeg fallback...")
return download_with_ffmpeg(url, output_dir, safe_name)
return False
except subprocess.TimeoutExpired:
print(f"[{timestamp()}] [ERROR] Download timed out after 1 hour.")
return False
except Exception as e:
print(f"[{timestamp()}] [ERROR] {e}")
return False
def download_with_ffmpeg(url: str, output_dir: Path, name: str) -> bool:
"""
Fallback: download directly with ffmpeg.
ffmpeg can read m3u8 natively and remux to mp4.
"""
output_path = output_dir / f"{name}.mp4"
cmd = [
"ffmpeg",
"-i", url,
"-c", "copy", # no re-encoding (fast)
"-bsf:a", "aac_adtstoasc", # fix AAC for mp4 container
"-movflags", "+faststart", # web-friendly mp4
"-y", # overwrite
str(output_path)
]
print(f" [ffmpeg] Downloading with direct ffmpeg...")
try:
result = subprocess.run(
cmd,
capture_output=True,
text=True,
timeout=3600,
)
if result.returncode == 0 and output_path.exists():
print(f"[{timestamp()}] [DONE via ffmpeg] {name}.mp4")
return True
else:
print(f"[{timestamp()}] [FAIL] ffmpeg also failed.")
return False
except Exception as e:
print(f"[{timestamp()}] [ERROR] ffmpeg: {e}")
return False
# ---------------------------------------------------------------------------
# Download worker
# ---------------------------------------------------------------------------
def download_worker(output_dir: Path):
"""Process the download queue sequentially."""
while running:
url = None
with queue_lock:
if download_queue:
url = download_queue.pop(0)
if url is None:
time.sleep(0.5)
continue
success = download_m3u8(url, output_dir)
if not success:
print(f"[{timestamp()}] [FAILED] Could not download: {url[:80]}...")
# Re-add to seen so we don't retry endlessly,
# but mark it so user knows
with seen_lock:
seen_links.discard(url)
save_history(output_dir)
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main():
parser = argparse.ArgumentParser(
description="Watch clipboard for m3u8 links and download them."
)
parser.add_argument(
"--output-dir", "-o",
type=str,
default="./m3u8_downloads",
help="Directory to save downloads (default: ./m3u8_downloads)"
)
parser.add_argument(
"--interval", "-i",
type=float,
default=DEFAULT_INTERVAL,
help=f"Clipboard poll interval in seconds (default: {DEFAULT_INTERVAL})"
)
parser.add_argument(
"--list", "-l",
action="store_true",
help="Show seen links and exit"
)
parser.add_argument(
"--clear-history",
action="store_true",
help="Clear download history and exit"
)
args = parser.parse_args()
output_dir = Path(args.output_dir).resolve()
output_dir.mkdir(parents=True, exist_ok=True)
if args.clear_history:
state_file = output_dir / STATE_FILE
if state_file.exists():
state_file.unlink()
print("[*] History cleared.")
else:
print("[*] No history to clear.")
return
# Load previous history
load_history(output_dir)
if args.list:
if seen_links:
print(f"Previously seen links ({len(seen_links)}):")
for link in sorted(seen_links):
print(f" {link}")
else:
print("No links in history.")
return
# Check deps
check_dependencies()
print("=" * 60)
print(" m3u8 Clipboard Watcher")
print(f" Output: {output_dir}")
print(f" Previously seen: {len(seen_links)} links")
print(" Press Ctrl+C to stop")
print("=" * 60)
# Start download worker in background
worker = threading.Thread(
target=download_worker,
args=(output_dir,),
daemon=True
)
worker.start()
# Start clipboard poller on main thread
try:
poll_clipboard(output_dir, args.interval)
except KeyboardInterrupt:
pass
finally:
# Wait for queue to drain (up to 30s)
print("\n[*] Waiting for active downloads to finish...")
deadline = time.time() + 30
while time.time() < deadline:
with queue_lock:
if not download_queue and not any(
t.is_alive() for t in [worker]
):
break
time.sleep(0.5)
save_history(output_dir)
print("[*] Goodbye!")
if __name__ == "__main__":
main()