Files
biblioteca_conocimiento_lab…/videoclub/videoclub_buscador.py

376 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
videoclub_buscador.py — Buscador interactivo de peliculas y series.
Reemplaza el listado diario pasivo por un buscador on-demand:
/pelicula <nombre> → lista numerada de peliculas
/serie <nombre> → lista series con temporadas/capitulos numerados
/descargar <numero> → descarga a NAS via Download Station
/novedades → novedades del catalogo
Usa la API de Torrentio (Stremio addon) + RealDebrid para enlaces directos.
"""
import json
import re
import sys
import urllib.request
from pathlib import Path
BASE = Path(__file__).resolve().parent
sys.path.insert(0, str(BASE))
import config
CATALOGO_URL = ("https://torrentio.strem.fun/"
"{config}/catalog/other/torrentio-realdebrid.json")
# Patrones de clasificacion
YEAR_RE = re.compile(r"\((19\d\d|20\d\d)\)")
YEAR_BARE_RE = re.compile(r"\.(19\d\d|20\d\d)\.")
SERIE_EP_RE = re.compile(r"\bS(\d{1,2})E(\d{1,3})\b", re.IGNORECASE)
CAP_RE = re.compile(r"(?:Cap[\.\s]*(\d+))", re.IGNORECASE)
ADULTO_RE = re.compile(
r"(\.XXX\.|XXX\.|private\.|brazzers|onlytarts|spankbang|xhamster|"
r"porntrex|missav|sexandsubmission|evilangel|monstersofcock|itsanal|thelifeerotic)",
re.IGNORECASE)
SERIE_PALABRAS = {
"star trek", "stargate", "heidi", "euphoria", "from", "the pitt",
"supergirl", "monarch", "paradise", "desaparecida", "vanished",
"strange new worlds", "star trek strange", "the studio",
"one punch man", "conan", "detective conan"
}
EXTRA_SUCIO = [
"wolfmax4k.com", "pctfenix", "pctfenix.com", "yts", "yts.mx",
"rarbg", "verpeliculasonline", "viruse", "proyecto", "kikorip",
"camrip", "telesync", "hdtv", "bluray", "remux", "webrip", "bdrip",
"repack", "proper", "xtrem", "webdl", "web-dl", "web", "dl",
"x264", "x265", "h264", "h265", "aac", "1080p", "720p", "4k",
"2160p", "spanish", "castellano", "dual", "lat", "es", "en",
"audio", "sub", "subs"
]
CALIDAD_MALA = re.compile(
r"(CAMRip|TS\.|HDTS|Telesync|KinoRip|Screener|HDCAM|HC\b|r5|360p|240p)",
re.IGNORECASE)
CASTELLANO = re.compile(
r"(Esp|Castellano|Span|Spanish|dual|lat|\bES\b|audio)", re.IGNORECASE)
INGLES = re.compile(r"(Eng|English|Ingl[eé]s|\bEN\b)", re.IGNORECASE)
OTRO_IDIOMA = re.compile(
r"(Ita\b|MIRCrew|RO\s?EN|South Wind|Silver Zero|Francais|Deutsch|Portugu[eê]s)",
re.IGNORECASE)
def clean_filename(name):
"""Limpia un nombre de fichero y devuelve (titulo_limpio, year, tipo)."""
name = re.sub(r"\.(mkv|mp4|avi|mov|webm|m4v)$", "", name.strip(),
flags=re.IGNORECASE)
year = None
m = YEAR_RE.search(name)
if m:
year = m.group(1)
name = name.replace(m.group(0), " ")
else:
m = YEAR_BARE_RE.search(name)
if m:
year = m.group(1)
name = name.replace(m.group(0), " ")
es_serie = bool(SERIE_EP_RE.search(name) or CAP_RE.search(name)) or \
any(p in name.lower() for p in SERIE_PALABRAS)
es_adulto = bool(ADULTO_RE.search(name))
name = re.sub(r"\[[^\]]*\]", " ", name)
name = re.sub(r"[\(\)\[\]]", " ", name)
name = name.replace("_", " ").replace(".", " ")
name = re.sub(r"\s+", " ", name).strip()
tokens = [t for t in name.split(" ") if t.lower() not in EXTRA_SUCIO]
name = " ".join(tokens).strip(" -–")
name = re.sub(r"\s+", " ", name).strip()
tipo = "adulto" if es_adulto else ("serie" if es_serie else "pelicula")
return name, year, tipo
def parsear_episodio(name):
"""Extrae (temporada, capitulo) de un nombre, o (0, 0)."""
m = SERIE_EP_RE.search(name)
if m:
return int(m.group(1)), int(m.group(2))
m = CAP_RE.search(name)
if m:
return 1, int(m.group(1))
return 0, 0
def serie_base(name):
"""Nombre base de la serie (antes del marcador SxxEyy, Cap.X, o anio)."""
m = SERIE_EP_RE.search(name) or CAP_RE.search(name)
if m:
base = name[:m.start()]
else:
base = name
# Quitar el anio si quedo pegado al final del nombre base
ym = YEAR_RE.search(base)
if ym:
base = base[:ym.start()] + base[ym.end():]
else:
ym2 = YEAR_BARE_RE.search(base)
if ym2:
base = base[:ym2.start()] + base[ym2.end():]
return _limpiar_titulo_ep(base)
def _limpiar_titulo_ep(texto):
"""Limpia un titulo de episodio."""
texto = re.sub(r"\.(mkv|mp4|avi|mov|webm|m4v)$", "", texto,
flags=re.IGNORECASE)
texto = re.sub(r"\[[^\]]*\]", " ", texto)
texto = re.sub(r"\b\d+x\d+\b", " ", texto, flags=re.IGNORECASE)
texto = texto.replace("_", " ").replace(".", " ").replace("-", " ")
texto = re.sub(r"[\[\]\(\)]", " ", texto)
tokens = [t for t in texto.split(" ") if t.lower() not in EXTRA_SUCIO]
texto = " ".join(tokens).strip(" -–")
return re.sub(r"\s+", " ", texto).strip()
def fetch_catalogo():
"""Descarga el catalogo de Torrentio + enlaces directos RD."""
if not config.TORRENTIO_RD_TOKEN:
return []
url = CATALOGO_URL.format(config=config.TORRENTIO_CONFIG_WITH_TOKEN)
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
with urllib.request.urlopen(req, timeout=40) as r:
data = json.loads(r.read().decode("utf-8"))
enlaces = {}
try:
req2 = urllib.request.Request(
"https://api.real-debrid.com/rest/1.0/torrents",
headers={"Authorization": f"Bearer {config.TORRENTIO_RD_TOKEN}"})
with urllib.request.urlopen(req2, timeout=40) as r2:
for t in json.loads(r2.read().decode("utf-8")):
links = t.get("links") or []
if links:
enlaces[t["id"]] = links[0]
except Exception as e:
print(f"[videoclub] No se pudieron obtener enlaces RD: {e}")
items = []
for m in data.get("metas", []):
name = m.get("name", "")
clean, year, tipo = clean_filename(name)
rd_id = m.get("id", "").split(":", 1)[-1]
items.append({
"id": m.get("id"),
"name": name,
"clean": clean,
"year": year,
"tipo": tipo,
"enlace": enlaces.get(rd_id),
"buena_calidad": not bool(CALIDAD_MALA.search(name)),
"castellano": bool(CASTELLANO.search(name)),
})
return items
def separar_adultos(items):
"""Separa items adultos del resto."""
normales, adultos = [], []
for it in items:
(adultos if it["tipo"] == "adulto" else normales).append(it)
return normales, adultos
def buscar_peliculas(termino, items=None):
"""Busca peliculas por nombre. Devuelve lista ordenada de resultados."""
term = (termino or "").strip().lower()
if not term:
return []
if items is None:
items = fetch_catalogo()
normales, _ = separar_adultos(items)
resultados = []
for it in normales:
if it["tipo"] != "pelicula":
continue
if not it["buena_calidad"]:
continue
if not (it.get("castellano") or bool(INGLES.search(it.get("name", "")))):
continue
clean = (it.get("clean") or "").lower()
name = (it.get("name") or "").lower()
if term in clean or term in name:
resultados.append(it)
resultados.sort(key=lambda x: (int(x["year"] or 0), x["clean"] or ""),
reverse=True)
return resultados
def buscar_series(termino, items=None):
"""Busca series por nombre. Devuelve dict agrupado por serie base."""
term = (termino or "").strip().lower()
if not term:
return {}
if items is None:
items = fetch_catalogo()
normales, _ = separar_adultos(items)
grupos = {}
for it in normales:
if it["tipo"] != "serie":
continue
if not it["buena_calidad"]:
continue
if OTRO_IDIOMA.search(it.get("name", "")):
continue
base = serie_base(it["name"]) or it["clean"] or ""
if term in base.lower() or term in (it["name"] or "").lower():
grupos.setdefault(base, []).append(it)
for base in grupos:
grupos[base].sort(key=lambda x: parsear_episodio(x["name"]))
return grupos
def trocear_texto(lineas, max_len=3500):
"""Convierte lista de lineas en mensajes Telegram <= max_len sin partir lineas."""
mensajes, actual, long = [], [], 0
for linea in lineas:
if actual and long + len(linea) + 1 > max_len:
mensajes.append("\n".join(actual))
actual, long = [], 0
actual.append(linea)
long += len(linea) + 1
if actual:
mensajes.append("\n".join(actual))
return mensajes
def formatear_peliculas(resultados):
"""Formatea resultados de peliculas como lista de mensajes Telegram numerados."""
if not resultados:
return ["No se encontraron peliculas con ese termino."]
lineas = [f"🎬 <b>Resultados ({len(resultados)}):</b>", ""]
for i, it in enumerate(resultados, 1):
y = f" ({it['year']})" if it["year"] else ""
calidad = "4K" if "4k" in (it.get("name") or "").lower() or \
"2160p" in (it.get("name") or "").lower() else "HD"
idioma = "cast." if it.get("castellano") else "ing."
enlace = it.get("enlace") or ""
lineas.append(f"<b>{i}.</b> {it['clean']}{y} — {calidad}, {idioma}")
if enlace:
lineas.append(f" <code>descargar {enlace}</code>")
lineas.append("")
lineas.append("Responde con el <b>numero</b> para descargar.")
return trocear_texto(lineas)
def formatear_series(grupos):
"""Formatea resultados de series como lista de mensajes Telegram numerados."""
if not grupos:
return ["No se encontraron series con ese termino."]
lineas = []
for base in sorted(grupos, key=lambda x: x.lower()):
eps = grupos[base]
temp_max = max(parsear_episodio(e["name"])[0] for e in eps)
lineas.append(f"📺 <b>{base}</b> — {temp_max} temporadas")
lineas.append("")
temp_actual = 0
num_global = 1
for it in eps:
temp, cap = parsear_episodio(it["name"])
if temp != temp_actual:
temp_actual = temp
if temp > 0:
lineas.append(f"<b>Temporada {temp}:</b>")
resto = _limpiar_titulo_ep(it["name"][SERIE_EP_RE.search(it["name"]).end():]) \
if SERIE_EP_RE.search(it["name"]) else it["clean"]
enlace = it.get("enlace") or ""
if temp > 0:
etiqueta = f"S{temp:02d}E{cap:02d}"
else:
etiqueta = it["clean"]
lineas.append(f" <b>{num_global}.</b> {etiqueta} — {resto}")
if enlace:
lineas.append(f" <code>descargar {enlace}</code>")
num_global += 1
lineas.append("")
lineas.append("Responde con el <b>numero</b> de episodio para descargar.")
return trocear_texto(lineas)
def enlace_realdebrid(rd_id):
"""Obtiene el enlace directo de descarga de un torrent RD."""
import urllib.parse
rd_id = rd_id.split(":", 1)[-1]
tok = config.TORRENTIO_RD_TOKEN
url = f"https://api.real-debrid.com/rest/1.0/torrents/info/{urllib.parse.quote(rd_id)}"
req = urllib.request.Request(url, headers={"Authorization": f"Bearer {tok}"})
with urllib.request.urlopen(req, timeout=30) as r:
info = json.loads(r.read().decode("utf-8"))
links = info.get("links") or []
return links[0] if links else None
def generar_novedades(items=None, limite=20):
"""Devuelve listado de novedades recientes."""
if items is None:
items = fetch_catalogo()
normales, _ = separar_adultos(items)
buenos = [it for it in normales if it["buena_calidad"]]
buenos.sort(key=lambda x: (int(x["year"] or 0), x["clean"] or ""),
reverse=True)
return buenos[:limite]
def formatear_novedades(items):
"""Formatea novedades como lista de mensajes Telegram numerados."""
if not items:
return ["No hay novedades recientes."]
lineas = [f"🎬 <b>Ultimas novedades ({len(items)}):</b>", ""]
for i, it in enumerate(items, 1):
y = f" ({it['year']})" if it["year"] else ""
tipo = "📺" if it["tipo"] == "serie" else "🎬"
idioma = "cast." if it.get("castellano") else "ing."
enlace = it.get("enlace") or ""
lineas.append(f"{tipo} <b>{i}.</b> {it['clean']}{y} — {idioma}")
if enlace:
lineas.append(f" <code>descargar {enlace}</code>")
return trocear_texto(lineas)
if __name__ == "__main__":
import sys as _sys
import os as _os
try:
_os.environ.setdefault("PYTHONIOENCODING", "utf-8")
_sys.stdout.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
if len(_sys.argv) < 2:
print("Uso:")
print(" python videoclub_buscador.py pelicula <nombre>")
print(" python videoclub_buscador.py serie <nombre>")
print(" python videoclub_buscador.py novedades")
_sys.exit(1)
cmd = _sys.argv[1]
arg = " ".join(_sys.argv[2:]) if len(_sys.argv) > 2 else ""
if cmd == "pelicula":
res = buscar_peliculas(arg)
for m in formatear_peliculas(res):
print(m)
print("---")
elif cmd == "serie":
res = buscar_series(arg)
for m in formatear_series(res):
print(m)
print("---")
elif cmd == "novedades":
res = generar_novedades()
for m in formatear_novedades(res):
print(m)
print("---")
else:
print(f"Comando desconocido: {cmd}")