geocrop-platform./apps/musicseerr/tasks.py

463 lines
19 KiB
Python

import os
import shutil
import subprocess
import logging
import re
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from mutagen.mp3 import EasyMP3
from mutagen.id3 import ID3
# Configure Logger
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger("musicseerr-tasks")
def scrape_tubidy_link(url: str, proxy_url: str) -> str:
"""
Scrapes a Tubidy page to find the direct MP3 download URL.
"""
proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
logger.info(f"Fetching Tubidy page: {url} via proxy: {proxy_url}")
response = requests.get(url, headers=headers, proxies=proxies, timeout=20)
response.raise_for_status()
soup = BeautifulSoup(response.text, "html.parser")
# 1. Search for direct mp3 links in audio tags
audio_tag = soup.find("audio")
if audio_tag:
source = audio_tag.find("source")
if source and source.get("src"):
return urljoin(url, source["src"])
if audio_tag.get("src"):
return urljoin(url, audio_tag["src"])
# 2. Search for <a> tags containing .mp3 in href
for a in soup.find_all("a", href=True):
href = a["href"]
if ".mp3" in href.lower():
return urljoin(url, href)
# 3. Follow potential download/mp3 buttons if nested
for a in soup.find_all("a", href=True):
text = a.text.lower()
if "mp3" in text or "download" in text:
sub_url = urljoin(url, a["href"])
logger.info(f"Following nested Tubidy download link: {sub_url}")
try:
sub_resp = requests.get(sub_url, headers=headers, proxies=proxies, timeout=15)
if sub_resp.status_code == 200:
sub_soup = BeautifulSoup(sub_resp.text, "html.parser")
for sub_a in sub_soup.find_all("a", href=True):
sub_href = sub_a["href"]
if ".mp3" in sub_href.lower():
return urljoin(sub_url, sub_href)
except Exception as e:
logger.warning(f"Failed to fetch nested link {sub_url}: {e}")
# 4. Fallback: search raw text for mp3 URLs
mp3_matches = re.findall(r'https?://[^\s"\'>]+\.mp3', response.text)
if mp3_matches:
return mp3_matches[0]
raise ValueError("Could not find any MP3 download link on the Tubidy page.")
def search_tubidy(query: str, proxy_url: str) -> list:
"""
Searches Tubidy for the query and returns a list of results with ID, title, and link.
"""
proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
endpoint = "https://mp3.tubidy.cool"
search_url = f"{endpoint}/search.php"
params = {
"q": query,
"si": 7,
"pn": 1
}
logger.info(f"Searching Tubidy for '{query}' via proxy: {proxy_url}")
try:
response = requests.get(search_url, headers=headers, params=params, proxies=proxies, timeout=15)
response.raise_for_status()
soup = BeautifulSoup(response.text, "html.parser")
results = []
for media_body in soup.find_all("div", class_="media-body"):
a_tag = media_body.find("a")
if a_tag:
href = a_tag.get("href")
if href:
title = a_tag.get("aria-label") or a_tag.text.strip()
# Extract ID from /watch/content_id
match = re.search(r'/watch/([^/]+)', href)
content_id = match.group(1) if match else None
link = href
if href.startswith("//"):
link = f"https:{href}"
elif href.startswith("/"):
link = f"{endpoint}{href}"
if content_id:
results.append({
"id": content_id,
"title": title,
"link": link
})
return results
except Exception as e:
logger.error(f"Failed to search Tubidy: {e}")
return []
def get_tubidy_download_link(content_id: str, proxy_url: str) -> str:
"""
Queries watch.php for the given content ID to find the direct MP3 download URL.
"""
proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
endpoint = "https://mp3.tubidy.cool"
watch_url = f"{endpoint}/watch.php"
params = {
"id": content_id,
"p": "mp4",
"lnk": 6,
"act": "down",
"t": "ssl"
}
logger.info(f"Fetching watch.php for content ID '{content_id}'")
try:
response = requests.get(watch_url, headers=headers, params=params, proxies=proxies, timeout=15)
response.raise_for_status()
soup = BeautifulSoup(response.text, "html.parser")
for li in soup.find_all("li", class_=lambda x: x and "list-group-item" in x and "big" in x):
a_tag = li.find("a")
if a_tag:
text = a_tag.text.lower()
href = a_tag.get("href")
if href and "download" in text:
return href
# Fallback to general scraping of the watch.php page if structure differs
for a_tag in soup.find_all("a", href=True):
text = a_tag.text.lower()
if "download" in text and "act=process" not in a_tag["href"]:
return a_tag["href"]
except Exception as e:
logger.error(f"Failed to fetch download link for {content_id}: {e}")
raise ValueError("Could not find download link on watch.php page.")
def download_file_stream(url: str, dest_path: str, proxy_url: str):
"""
Downloads a file in chunks using requests through the proxy.
"""
proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
logger.info(f"Streaming download from {url} to {dest_path}")
response = requests.get(url, headers=headers, proxies=proxies, stream=True, timeout=60)
response.raise_for_status()
with open(dest_path, "wb") as f:
for chunk in response.iter_content(chunk_size=8192):
if chunk:
f.write(chunk)
def tag_mp3(file_path: str, filename: str):
"""
Parses the filename to extract artist/title and injects ID3 tags using Mutagen.
"""
base_name = os.path.splitext(os.path.basename(file_path))[0]
artist = "Unknown Artist"
title = base_name
# Try parsing "Artist - Title" format
if " - " in base_name:
parts = base_name.split(" - ", 1)
artist = parts[0].strip()
title = parts[1].strip()
logger.info(f"Applying tags - Title: '{title}', Artist: '{artist}' to {file_path}")
# Initialize ID3 tags if they do not exist
try:
audio = EasyMP3(file_path)
except Exception:
id3 = ID3()
id3.save(file_path)
audio = EasyMP3(file_path)
audio["title"] = title
audio["artist"] = artist
audio["album"] = "Tubidy Ingestion"
audio.save()
def send_telegram_notification(chat_id: int, text: str):
"""
Sends a message back to Telegram.
"""
token = os.getenv("TELEGRAM_BOT_TOKEN")
if not token or not chat_id:
return
url = f"https://api.telegram.org/bot{token}/sendMessage"
try:
requests.post(url, json={"chat_id": chat_id, "text": text}, timeout=10)
except Exception as e:
logger.error(f"Failed to send Telegram notification: {e}")
def symlink_existing_tracks(temp_dir: str):
"""
Finds all MP3 files recursively in /remote-music/music and creates symlinks to them
in the temp_dir so spotDL and yt-dlp skip downloading duplicates.
"""
remote_base = "/remote-music/music"
if not os.path.isdir(remote_base):
return
logger.info(f"Scanning {remote_base} to symlink existing tracks in {temp_dir}...")
count = 0
for root, dirs, files in os.walk(remote_base):
for file in files:
if file.endswith(".mp3"):
src_path = os.path.join(root, file)
dest_path = os.path.join(temp_dir, file)
if not os.path.exists(dest_path):
try:
os.symlink(src_path, dest_path)
count += 1
except Exception as e:
logger.warning(f"Failed to create symlink for {file}: {e}")
logger.info(f"Created {count} symlinks for existing tracks.")
def run_download_task(task_id: str, query: str, proxy_url: str, jobs_dict: dict, chat_id: int = None):
"""
Core download manager task that runs in the background.
"""
jobs_dict[task_id]["status"] = "downloading"
temp_dir = f"/tmp/downloads/{task_id}"
os.makedirs(temp_dir, exist_ok=True)
symlink_existing_tracks(temp_dir)
# Resolve proxy hostname to IP address for spotDL compatibility (spotDL requires IP in proxy URL)
resolved_proxy_url = proxy_url
if proxy_url:
try:
from urllib.parse import urlparse
import socket
parsed = urlparse(proxy_url)
if parsed.hostname and not parsed.hostname.replace('.', '').isdigit():
ip = socket.gethostbyname(parsed.hostname)
netloc = ip
if parsed.port:
netloc = f"{ip}:{parsed.port}"
if parsed.username or parsed.password:
auth = ""
if parsed.username:
auth += parsed.username
if parsed.password:
auth += f":{parsed.password}"
netloc = f"{auth}@{netloc}"
resolved_proxy_url = parsed._replace(netloc=netloc).geturl()
logger.info(f"Resolved proxy hostname '{parsed.hostname}' to IP '{ip}'. New proxy URL: '{resolved_proxy_url}'")
except Exception as e:
logger.error(f"Failed to resolve proxy hostname: {e}")
try:
is_tubidy = "tubidy" in query.lower()
is_spotify = "spotify.com" in query.lower()
is_youtube = "youtube.com" in query.lower() or "youtu.be" in query.lower()
# Priority 1: spotDL (text search or Spotify URL)
if not is_tubidy and not is_youtube:
jobs_dict[task_id]["priority_used"] = "spotDL"
logger.info(f"[{task_id}] Attempting Priority 1: spotDL for '{query}'")
env = os.environ.copy()
if resolved_proxy_url:
env["http_proxy"] = resolved_proxy_url
env["https_proxy"] = resolved_proxy_url
cmd = ["spotdl", "download", query]
if resolved_proxy_url:
cmd += ["--proxy", resolved_proxy_url]
# Run spotDL in the temp directory
result = subprocess.run(cmd, cwd=temp_dir, env=env, capture_output=True, text=True)
if result.returncode == 0:
logger.info(f"[{task_id}] spotDL download succeeded.")
else:
logger.warning(f"[{task_id}] spotDL failed: {result.stderr or result.stdout}. Falling back to yt-dlp.")
# Clear state for P2 fallback
is_spotify = False
# Priority 2: yt-dlp (direct YouTube or fallback for failed spotDL)
if not is_tubidy and not is_spotify:
jobs_dict[task_id]["priority_used"] = "yt-dlp"
logger.info(f"[{task_id}] Attempting Priority 2: yt-dlp for '{query}'")
env = os.environ.copy()
if resolved_proxy_url:
env["http_proxy"] = resolved_proxy_url
env["https_proxy"] = resolved_proxy_url
# If not a link, treat as search query
search_query = query
if not query.startswith("http"):
search_query = f"ytsearch1:{query}"
cmd = [
"yt-dlp",
"-x",
"--audio-format", "mp3",
"--embed-metadata",
"--yes-playlist",
"--output", "%(title)s.%(ext)s"
]
if resolved_proxy_url:
cmd += ["--proxy", resolved_proxy_url]
cmd.append(search_query)
result = subprocess.run(cmd, cwd=temp_dir, env=env, capture_output=True, text=True)
if result.returncode == 0:
logger.info(f"[{task_id}] yt-dlp download succeeded.")
else:
logger.warning(f"[{task_id}] yt-dlp failed: {result.stderr or result.stdout}. Falling back to Tubidy Search.")
# Priority 3: Custom Tubidy Link Scraper (if URL is directly passed)
if is_tubidy:
jobs_dict[task_id]["priority_used"] = "Tubidy"
logger.info(f"[{task_id}] Attempting Priority 3: Tubidy Scraper for '{query}'")
# Scrape MP3 URL
mp3_url = scrape_tubidy_link(query, resolved_proxy_url)
logger.info(f"[{task_id}] Scraped Tubidy MP3 URL: {mp3_url}")
# Formulate filename
filename = query.split("/")[-1]
if not filename or ".html" in filename or "?" in filename:
filename = "tubidy_download"
filename = filename.replace(".html", "").replace(".php", "").strip()
if not filename.endswith(".mp3"):
filename += ".mp3"
dest_path = os.path.join(temp_dir, filename)
# Download file stream
download_file_stream(mp3_url, dest_path, resolved_proxy_url)
logger.info(f"[{task_id}] Tubidy download completed.")
# Tag metadata
tag_mp3(dest_path, filename)
logger.info(f"[{task_id}] Tubidy ID3 tagging complete.")
# Priority 4: Tubidy Search Fallback (if spotDL and yt-dlp failed and no files are present)
downloaded_files = os.listdir(temp_dir)
if not downloaded_files and not is_tubidy:
jobs_dict[task_id]["priority_used"] = "Tubidy-Search"
logger.info(f"[{task_id}] Attempting Fallback Priority 4: Tubidy Search Scraper for '{query}'")
search_results = search_tubidy(query, resolved_proxy_url)
if search_results:
first_result = search_results[0]
logger.info(f"[{task_id}] Found Tubidy match: '{first_result['title']}' (ID: {first_result['id']})")
try:
mp3_url = get_tubidy_download_link(first_result["id"], resolved_proxy_url)
logger.info(f"[{task_id}] Resolved Tubidy download URL: {mp3_url}")
filename = f"{first_result['title']}.mp3"
# Clean filename
filename = "".join([c for c in filename if c.isalnum() or c in " .-_()"]).strip()
if not filename.endswith(".mp3"):
filename += ".mp3"
if not filename or filename == ".mp3":
filename = "tubidy_fallback.mp3"
dest_path = os.path.join(temp_dir, filename)
# Download file stream
download_file_stream(mp3_url, dest_path, resolved_proxy_url)
logger.info(f"[{task_id}] Tubidy fallback download completed.")
# Tag metadata
tag_mp3(dest_path, filename)
logger.info(f"[{task_id}] Tubidy fallback ID3 tagging complete.")
except Exception as e:
logger.error(f"[{task_id}] Tubidy fallback download failed: {e}")
raise Exception(f"All download options (spotDL, yt-dlp, Tubidy) failed. Last error: {e}")
else:
logger.error(f"[{task_id}] No Tubidy search results found for query '{query}'")
raise Exception("All download options (spotDL, yt-dlp, Tubidy) failed. No Tubidy search results found.")
# Verify files and move to /remote-music/music
downloaded_files = os.listdir(temp_dir)
actual_files = [f for f in downloaded_files if not os.path.islink(os.path.join(temp_dir, f))]
if not actual_files:
if downloaded_files:
logger.info(f"[{task_id}] All tracks in this query were already downloaded (skipped duplicates).")
jobs_dict[task_id]["files"] = []
jobs_dict[task_id]["status"] = "completed"
if chat_id:
send_telegram_notification(chat_id, f"✅ Successfully ingested request: '{query}'\n\n(All tracks already existed - skipped duplicates)")
return
else:
raise Exception("Ingestion finished but no files were found in local temp folder.")
os.makedirs("/remote-music/music", exist_ok=True)
moved_files = []
for file_name in actual_files:
src_file = os.path.join(temp_dir, file_name)
dest_file = os.path.join("/remote-music/music", file_name)
logger.info(f"[{task_id}] Moving new file '{file_name}' to remote Seedbox directory")
shutil.move(src_file, dest_file)
moved_files.append(file_name)
jobs_dict[task_id]["files"] = moved_files
jobs_dict[task_id]["status"] = "completed"
logger.info(f"[{task_id}] Music Ingestion task fully completed.")
if chat_id:
files_str = "\n".join(moved_files)
send_telegram_notification(chat_id, f"✅ Successfully ingested request: '{query}'\n\nFiles imported to Navidrome:\n{files_str}")
except Exception as e:
logger.error(f"[{task_id}] Ingestion failed: {e}")
jobs_dict[task_id]["status"] = "failed"
jobs_dict[task_id]["error"] = str(e)
if chat_id:
send_telegram_notification(chat_id, f"❌ Failed to ingest request: '{query}'\nError: {e}")
finally:
# Clean up local emptyDir workspace directory for this task
if os.path.exists(temp_dir):
shutil.rmtree(temp_dir)