import os import shutil import subprocess import logging import re import requests from bs4 import BeautifulSoup from urllib.parse import urljoin from mutagen.mp3 import EasyMP3 from mutagen.id3 import ID3 # Configure Logger logging.basicConfig(level=logging.INFO) logger = logging.getLogger("musicseerr-tasks") def scrape_tubidy_link(url: str, proxy_url: str) -> str: """ Scrapes a Tubidy page to find the direct MP3 download URL. """ proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } logger.info(f"Fetching Tubidy page: {url} via proxy: {proxy_url}") response = requests.get(url, headers=headers, proxies=proxies, timeout=20) response.raise_for_status() soup = BeautifulSoup(response.text, "html.parser") # 1. Search for direct mp3 links in audio tags audio_tag = soup.find("audio") if audio_tag: source = audio_tag.find("source") if source and source.get("src"): return urljoin(url, source["src"]) if audio_tag.get("src"): return urljoin(url, audio_tag["src"]) # 2. Search for tags containing .mp3 in href for a in soup.find_all("a", href=True): href = a["href"] if ".mp3" in href.lower(): return urljoin(url, href) # 3. Follow potential download/mp3 buttons if nested for a in soup.find_all("a", href=True): text = a.text.lower() if "mp3" in text or "download" in text: sub_url = urljoin(url, a["href"]) logger.info(f"Following nested Tubidy download link: {sub_url}") try: sub_resp = requests.get(sub_url, headers=headers, proxies=proxies, timeout=15) if sub_resp.status_code == 200: sub_soup = BeautifulSoup(sub_resp.text, "html.parser") for sub_a in sub_soup.find_all("a", href=True): sub_href = sub_a["href"] if ".mp3" in sub_href.lower(): return urljoin(sub_url, sub_href) except Exception as e: logger.warning(f"Failed to fetch nested link {sub_url}: {e}") # 4. Fallback: search raw text for mp3 URLs mp3_matches = re.findall(r'https?://[^\s"\'>]+\.mp3', response.text) if mp3_matches: return mp3_matches[0] raise ValueError("Could not find any MP3 download link on the Tubidy page.") def search_tubidy(query: str, proxy_url: str) -> list: """ Searches Tubidy for the query and returns a list of results with ID, title, and link. """ proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } endpoint = "https://mp3.tubidy.cool" search_url = f"{endpoint}/search.php" params = { "q": query, "si": 7, "pn": 1 } logger.info(f"Searching Tubidy for '{query}' via proxy: {proxy_url}") try: response = requests.get(search_url, headers=headers, params=params, proxies=proxies, timeout=15) response.raise_for_status() soup = BeautifulSoup(response.text, "html.parser") results = [] for media_body in soup.find_all("div", class_="media-body"): a_tag = media_body.find("a") if a_tag: href = a_tag.get("href") if href: title = a_tag.get("aria-label") or a_tag.text.strip() # Extract ID from /watch/content_id match = re.search(r'/watch/([^/]+)', href) content_id = match.group(1) if match else None link = href if href.startswith("//"): link = f"https:{href}" elif href.startswith("/"): link = f"{endpoint}{href}" if content_id: results.append({ "id": content_id, "title": title, "link": link }) return results except Exception as e: logger.error(f"Failed to search Tubidy: {e}") return [] def get_tubidy_download_link(content_id: str, proxy_url: str) -> str: """ Queries watch.php for the given content ID to find the direct MP3 download URL. """ proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } endpoint = "https://mp3.tubidy.cool" watch_url = f"{endpoint}/watch.php" params = { "id": content_id, "p": "mp4", "lnk": 6, "act": "down", "t": "ssl" } logger.info(f"Fetching watch.php for content ID '{content_id}'") try: response = requests.get(watch_url, headers=headers, params=params, proxies=proxies, timeout=15) response.raise_for_status() soup = BeautifulSoup(response.text, "html.parser") for li in soup.find_all("li", class_=lambda x: x and "list-group-item" in x and "big" in x): a_tag = li.find("a") if a_tag: text = a_tag.text.lower() href = a_tag.get("href") if href and "download" in text: return href # Fallback to general scraping of the watch.php page if structure differs for a_tag in soup.find_all("a", href=True): text = a_tag.text.lower() if "download" in text and "act=process" not in a_tag["href"]: return a_tag["href"] except Exception as e: logger.error(f"Failed to fetch download link for {content_id}: {e}") raise ValueError("Could not find download link on watch.php page.") def download_file_stream(url: str, dest_path: str, proxy_url: str): """ Downloads a file in chunks using requests through the proxy. """ proxies = {"http": proxy_url, "https": proxy_url} if proxy_url else None headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } logger.info(f"Streaming download from {url} to {dest_path}") response = requests.get(url, headers=headers, proxies=proxies, stream=True, timeout=60) response.raise_for_status() with open(dest_path, "wb") as f: for chunk in response.iter_content(chunk_size=8192): if chunk: f.write(chunk) def tag_mp3(file_path: str, filename: str): """ Parses the filename to extract artist/title and injects ID3 tags using Mutagen. """ base_name = os.path.splitext(os.path.basename(file_path))[0] artist = "Unknown Artist" title = base_name # Try parsing "Artist - Title" format if " - " in base_name: parts = base_name.split(" - ", 1) artist = parts[0].strip() title = parts[1].strip() logger.info(f"Applying tags - Title: '{title}', Artist: '{artist}' to {file_path}") # Initialize ID3 tags if they do not exist try: audio = EasyMP3(file_path) except Exception: id3 = ID3() id3.save(file_path) audio = EasyMP3(file_path) audio["title"] = title audio["artist"] = artist audio["album"] = "Tubidy Ingestion" audio.save() def send_telegram_notification(chat_id: int, text: str): """ Sends a message back to Telegram. """ token = os.getenv("TELEGRAM_BOT_TOKEN") if not token or not chat_id: return url = f"https://api.telegram.org/bot{token}/sendMessage" try: requests.post(url, json={"chat_id": chat_id, "text": text}, timeout=10) except Exception as e: logger.error(f"Failed to send Telegram notification: {e}") def symlink_existing_tracks(temp_dir: str): """ Finds all audio files recursively in /remote-music/music, extracts artist and title, and creates symlinks named "Artist - Title.mp3" in temp_dir so spotDL skips them. """ remote_base = "/remote-music/music" if not os.path.isdir(remote_base): return logger.info(f"Scanning {remote_base} to symlink existing tracks in {temp_dir}...") count = 0 for root, dirs, files in os.walk(remote_base): for file in files: if file.lower().endswith((".mp3", ".flac", ".m4a", ".ogg", ".wav", ".opus")): # Try to clean/extract artist and title base_name = os.path.splitext(file)[0] # Strip leading numbers like "01. ", "01 - ", etc. cleaned = re.sub(r'^\d+[\s.-]+', '', base_name).strip() # We expect the file to have a " - " separator if " - " in cleaned: parts = cleaned.split(" - ", 1) artist = parts[0].strip() title = parts[1].strip() # Target filename that spotDL expects target_name = f"{artist} - {title}.mp3" else: # Fallback to direct name with .mp3 extension target_name = f"{cleaned}.mp3" # Replaces invalid chars target_name = "".join([c for c in target_name if c.isalnum() or c in " .-_()"]).strip() if not target_name.endswith(".mp3"): target_name += ".mp3" dest_path = os.path.join(temp_dir, target_name) if not os.path.exists(dest_path): try: src_path = os.path.join(root, file) os.symlink(src_path, dest_path) count += 1 except Exception as e: logger.warning(f"Failed to create symlink for {file} -> {target_name}: {e}") logger.info(f"Created {count} symlinks for existing tracks.") def run_download_task(task_id: str, query: str, proxy_url: str, jobs_dict: dict, chat_id: int = None): """ Core download manager task that runs in the background. """ jobs_dict[task_id]["status"] = "downloading" temp_dir = f"/tmp/downloads/{task_id}" os.makedirs(temp_dir, exist_ok=True) symlink_existing_tracks(temp_dir) # Resolve proxy hostname to IP address for spotDL compatibility (spotDL requires IP in proxy URL) resolved_proxy_url = proxy_url if proxy_url: try: from urllib.parse import urlparse import socket parsed = urlparse(proxy_url) if parsed.hostname and not parsed.hostname.replace('.', '').isdigit(): ip = socket.gethostbyname(parsed.hostname) netloc = ip if parsed.port: netloc = f"{ip}:{parsed.port}" if parsed.username or parsed.password: auth = "" if parsed.username: auth += parsed.username if parsed.password: auth += f":{parsed.password}" netloc = f"{auth}@{netloc}" resolved_proxy_url = parsed._replace(netloc=netloc).geturl() logger.info(f"Resolved proxy hostname '{parsed.hostname}' to IP '{ip}'. New proxy URL: '{resolved_proxy_url}'") except Exception as e: logger.error(f"Failed to resolve proxy hostname: {e}") try: is_tubidy = "tubidy" in query.lower() is_spotify = "spotify.com" in query.lower() is_youtube = "youtube.com" in query.lower() or "youtu.be" in query.lower() # Priority 1: spotDL (text search or Spotify URL) if not is_tubidy and not is_youtube: jobs_dict[task_id]["priority_used"] = "spotDL" logger.info(f"[{task_id}] Attempting Priority 1: spotDL for '{query}'") env = os.environ.copy() if resolved_proxy_url: env["http_proxy"] = resolved_proxy_url env["https_proxy"] = resolved_proxy_url cmd = ["spotdl", "download", query] if resolved_proxy_url: cmd += ["--proxy", resolved_proxy_url] # Run spotDL in the temp directory result = subprocess.run(cmd, cwd=temp_dir, env=env, capture_output=True, text=True) if result.returncode == 0: logger.info(f"[{task_id}] spotDL download succeeded.") else: logger.warning(f"[{task_id}] spotDL failed: {result.stderr or result.stdout}. Falling back to yt-dlp.") # Clear state for P2 fallback is_spotify = False # Priority 2: yt-dlp (direct YouTube or fallback for failed spotDL) if not is_tubidy and not is_spotify: jobs_dict[task_id]["priority_used"] = "yt-dlp" logger.info(f"[{task_id}] Attempting Priority 2: yt-dlp for '{query}'") env = os.environ.copy() if resolved_proxy_url: env["http_proxy"] = resolved_proxy_url env["https_proxy"] = resolved_proxy_url # If not a link, treat as search query search_query = query if not query.startswith("http"): search_query = f"ytsearch1:{query}" cmd = [ "yt-dlp", "-x", "--audio-format", "mp3", "--embed-metadata", "--yes-playlist", "--output", "%(title)s.%(ext)s" ] if resolved_proxy_url: cmd += ["--proxy", resolved_proxy_url] cmd.append(search_query) result = subprocess.run(cmd, cwd=temp_dir, env=env, capture_output=True, text=True) if result.returncode == 0: logger.info(f"[{task_id}] yt-dlp download succeeded.") else: logger.warning(f"[{task_id}] yt-dlp failed: {result.stderr or result.stdout}. Falling back to Tubidy Search.") # Priority 3: Custom Tubidy Link Scraper (if URL is directly passed) if is_tubidy: jobs_dict[task_id]["priority_used"] = "Tubidy" logger.info(f"[{task_id}] Attempting Priority 3: Tubidy Scraper for '{query}'") # Scrape MP3 URL mp3_url = scrape_tubidy_link(query, resolved_proxy_url) logger.info(f"[{task_id}] Scraped Tubidy MP3 URL: {mp3_url}") # Formulate filename filename = query.split("/")[-1] if not filename or ".html" in filename or "?" in filename: filename = "tubidy_download" filename = filename.replace(".html", "").replace(".php", "").strip() if not filename.endswith(".mp3"): filename += ".mp3" dest_path = os.path.join(temp_dir, filename) # Download file stream download_file_stream(mp3_url, dest_path, resolved_proxy_url) logger.info(f"[{task_id}] Tubidy download completed.") # Tag metadata tag_mp3(dest_path, filename) logger.info(f"[{task_id}] Tubidy ID3 tagging complete.") # Priority 4: Tubidy Search Fallback (if spotDL and yt-dlp failed and no files are present) downloaded_files = os.listdir(temp_dir) if not downloaded_files and not is_tubidy: jobs_dict[task_id]["priority_used"] = "Tubidy-Search" logger.info(f"[{task_id}] Attempting Fallback Priority 4: Tubidy Search Scraper for '{query}'") search_results = search_tubidy(query, resolved_proxy_url) if search_results: first_result = search_results[0] logger.info(f"[{task_id}] Found Tubidy match: '{first_result['title']}' (ID: {first_result['id']})") try: mp3_url = get_tubidy_download_link(first_result["id"], resolved_proxy_url) logger.info(f"[{task_id}] Resolved Tubidy download URL: {mp3_url}") filename = f"{first_result['title']}.mp3" # Clean filename filename = "".join([c for c in filename if c.isalnum() or c in " .-_()"]).strip() if not filename.endswith(".mp3"): filename += ".mp3" if not filename or filename == ".mp3": filename = "tubidy_fallback.mp3" dest_path = os.path.join(temp_dir, filename) # Download file stream download_file_stream(mp3_url, dest_path, resolved_proxy_url) logger.info(f"[{task_id}] Tubidy fallback download completed.") # Tag metadata tag_mp3(dest_path, filename) logger.info(f"[{task_id}] Tubidy fallback ID3 tagging complete.") except Exception as e: logger.error(f"[{task_id}] Tubidy fallback download failed: {e}") raise Exception(f"All download options (spotDL, yt-dlp, Tubidy) failed. Last error: {e}") else: logger.error(f"[{task_id}] No Tubidy search results found for query '{query}'") raise Exception("All download options (spotDL, yt-dlp, Tubidy) failed. No Tubidy search results found.") # Verify files and move to /remote-music/music downloaded_files = os.listdir(temp_dir) actual_files = [f for f in downloaded_files if not os.path.islink(os.path.join(temp_dir, f))] if not actual_files: if downloaded_files: logger.info(f"[{task_id}] All tracks in this query were already downloaded (skipped duplicates).") jobs_dict[task_id]["files"] = [] jobs_dict[task_id]["status"] = "completed" if chat_id: send_telegram_notification(chat_id, f"✅ Successfully ingested request: '{query}'\n\n(All tracks already existed - skipped duplicates)") return else: raise Exception("Ingestion finished but no files were found in local temp folder.") os.makedirs("/remote-music/music", exist_ok=True) moved_files = [] for file_name in actual_files: src_file = os.path.join(temp_dir, file_name) dest_file = os.path.join("/remote-music/music", file_name) logger.info(f"[{task_id}] Moving new file '{file_name}' to remote Seedbox directory") shutil.move(src_file, dest_file) moved_files.append(file_name) jobs_dict[task_id]["files"] = moved_files jobs_dict[task_id]["status"] = "completed" logger.info(f"[{task_id}] Music Ingestion task fully completed.") if chat_id: files_str = "\n".join(moved_files) send_telegram_notification(chat_id, f"✅ Successfully ingested request: '{query}'\n\nFiles imported to Navidrome:\n{files_str}") except Exception as e: logger.error(f"[{task_id}] Ingestion failed: {e}") jobs_dict[task_id]["status"] = "failed" jobs_dict[task_id]["error"] = str(e) if chat_id: send_telegram_notification(chat_id, f"❌ Failed to ingest request: '{query}'\nError: {e}") finally: # Clean up local emptyDir workspace directory for this task if os.path.exists(temp_dir): shutil.rmtree(temp_dir)