Files
biblioteca_conocimiento_lab…/videoclub/global_buscar.py

316 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
global_buscar.py — Busqueda GLOBAL de peliculas/series que NO estan en tu
biblioteca RealDebrid. Busca "miles de pelis" en internet.
Flujo (2 pasos, sin estado persistente):
1) Texto -> titulo: Cinemeta (Stremio, sin clave) devuelve los titulos
mas relevantes con su id IMDb (ttXXXXXXX).
2) Titulo -> torrente: Torrentio (Stremio) devuelve los torrents reales
scaneados de 1337x, TPB, TorrentGalaxy, Wolfmax4k, ...
cada uno con infoHash para montar el magnet.
Zona "miles de peliculas": Torrentio consulta decenas de proveedores en
tiempo real y devuelve cientos de stream por titulo, no solo tu biblioteca.
Bajar a NAS: ./nas_dl.py "<magnet>" (Download Station se encarga).
"""
import json
import re
import sys
import urllib.parse
import urllib.request
from pathlib import Path
BASE = Path(__file__).resolve().parent
sys.path.insert(0, str(BASE))
CINEMETA = "https://v3-cinemeta.strem.io"
TORRENTIO = "https://torrentio.strem.fun/language=spanish"
# Catalogo por genero (Cinemeta). "top" = mas descargadas, "imdbRating" = mejor valoradas.
CINEMETA_CAT = "https://cinemeta-catalogs.strem.io"
CATEGORIAS = ("top", "imdbRating")
GENEROS = (
"Action", "Adventure", "Animation", "Biography", "Comedy", "Crime",
"Documentary", "Drama", "Family", "Fantasy", "History", "Horror",
"Mystery", "Romance", "Sci-Fi", "Sport", "Thriller", "War", "Western",
)
HEADERS = {"User-Agent": "Mozilla/5.0"}
# Utilidades de clasificacion reutilizadas de videoclub_buscador
import videoclub_buscador as vcb
# Idiomas que no interesan en el catalogo global (ruso, tamil, hindi, etc.)
OTRO_IDIOMA_GLOBAL = re.compile(
r"(Tamil|Hindi|Telugu|Malayalam|Turkish|Tur[cç]e|Zulu|Arabic|"
r"Cyrillic|Russian|Greek|Korean|Chinese|Japanese|Thai)",
re.IGNORECASE)
RUSO = re.compile(r"[а-яА-ЯёЁ]")
# AUDIO EN CASTELLANO (doblada). Marcadores reales de doblaje ES o pista
# multiaudio que incluye ES. Se exige que coincida para mostrar el torrent.
CASTELLANO_AUDIO = re.compile(
r"(Esp[añ]ol|Castellano|Spanish|Latino|DualLat|\bdual\b|\bmulti[ _-]?audio\b|"
r"^Es[ ._-]|\bEs-Es\b|\bESP\b)",
re.IGNORECASE)
def _get_json(url, timeout=35):
req = urllib.request.Request(url, headers=HEADERS)
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read().decode("utf-8"))
def buscar_titulos(termino, limite=10):
"""Texto -> lista de {name, type, imdb_id, year}. Usa Cinemeta (sin clave)."""
term = (termino or "").strip()
if not term:
return []
resultado, lista = [], []
for _type in ("movie", "series"):
url = (f"{CINEMETA}/catalog/{_type}/top/"
f"search={urllib.parse.quote(term)}.json")
try:
metas = _get_json(url).get("metas", [])
except Exception:
metas = []
for m in metas[:limite]:
imdb = m.get("id", "")
if not re.match(r"^tt\d+$", imdb):
continue
name = m.get("name", "")
year = m.get("year") or None
resultado.append({
"name": name,
"type": _type,
"imdb_id": imdb,
"year": str(year) if year else None,
})
# Intercalamos movie/series y deduplicamos por imdb_id
vistos = set()
for r in resultado:
if r["imdb_id"] in vistos:
continue
vistos.add(r["imdb_id"])
lista.append(r)
return lista[:20]
def torrents_de(imdb_id, tipo="movie"):
"""IMDb id -> lista de torrents reales de Torrentio.
Cada item: {name, title, info_hash, file_id, filename, seeders, size}."""
if not re.match(r"^tt\d+$", imdb_id or ""):
return []
url = f"{TORRENTIO}/stream/{tipo}/{imdb_id}.json"
try:
data = _get_json(url)
except Exception:
return []
torrents = []
for s in data.get("streams", []):
if not s.get("infoHash"):
continue
filename = (s.get("behaviorHints", {}) or {}).get("filename") or ""
title = (s.get("title") or "").replace("\n", " ")
torrents.append({
"name": s.get("name", "").replace("\n", " "),
"title": title,
"info_hash": s["infoHash"],
"file_id": s.get("fileIdx", 0),
"filename": filename,
"seeders": _seeders(title),
"size": _size(title),
"calidad": _calidad(filename + " " + title),
})
return torrents
def generos_disponibles():
"""Devuelve la lista de generos con su numero (1-based) para el usuario."""
return list(enumerate(GENEROS, 1))
def catalogo_por_genero(tipo, genero, categoria="top", limite=30):
"""'las N mas descargadas/valoradas' de un genero y tipo.
tipo: movie|series. genero: nombre en ingles (Action, Horror...).
categoria: top (mas descargadas) | imdbRating (mejor valoradas).
Devuelve lista de {name, type, imdb_id, year}."""
if categoria not in CATEGORIAS:
categoria = "top"
if genero not in GENEROS:
return []
url = (f"{CINEMETA_CAT}/{categoria}/catalog/{tipo}/{categoria}/"
f"genre={urllib.parse.quote(genero)}&skip=0.json")
try:
metas = _get_json(url).get("metas", [])
except Exception:
return []
lista = []
for m in metas:
imdb = m.get("id", "")
if not re.match(r"^tt\d+$", imdb):
continue
name = m.get("name", "")
year = m.get("year") or m.get("releaseInfo") or None
lista.append({
"name": name,
"type": tipo,
"imdb_id": imdb,
"year": str(year) if year else None,
})
return lista[:limite]
def formatear_generos():
"""Numeros la lista de generos para elegir por numero en Telegram."""
lineas = ["🎭 <b>Elige un genero por numero:</b>", ""]
for n, g in generos_disponibles():
lineas.append(f"<b>{n}.</b> {g}")
return vcb.trocear_texto(lineas)
def formatear_catalogo(resultados, tipo, genero, categoria):
"""Numera los titulos de un catalogo por genero para elegir luego /torrents."""
if not resultados:
return [f"No hay resultados para <b>{genero}</b>."]
etiqueta = "🎬 Peliculas" if tipo == "movie" else "📺 Series"
orden = "mas descargadas" if categoria == "top" else "mejor valoradas"
lineas = [f"{etiqueta} · <b>{genero}</b> — {len(resultados)} {orden}:",
"Responde <b>/torrents &lt;numero&gt;</b> para ver torrents.", ""]
for i, r in enumerate(resultados, 1):
icono = "📺" if r["type"] == "series" else "🎬"
y = f" ({r['year']})" if r.get("year") else ""
lineas.append(f"<b>{i}.</b> {icono} {r['name']}{y}")
return vcb.trocear_texto(lineas)
def _seeders(title):
m = re.search(r"👤\s*(\d+)", title)
return int(m.group(1)) if m else 0
def _size(title):
m = re.search(r"💾\s*([\d.]+)\s*(TB|GB|MB)", title)
if not m:
return None
val = float(m.group(1))
if m.group(2) == "TB":
return val * 1024
if m.group(2) == "MB":
return val / 1024
return val
def _calidad(texto):
t = " " + texto.lower() + " "
if "2160p" in t or ("4k" in t and "1080p" not in t):
return "4K"
if "1080p" in t or "1080" in t:
return "1080p"
if "720p" in t:
return "720p"
if "4k" in t or "uhd" in t:
return "4K"
return "?"
def filtrar_torrents(torrents, limite=20):
"""Solo torrents con audio en CASTELLANO (doblada/dual). Descarta el resto."""
utiles = []
for t in torrents:
titulo = t["title"]
if vcb.CALIDAD_MALA.search(titulo):
continue
if vcb.OTRO_IDIOMA.search(titulo) or OTRO_IDIOMA_GLOBAL.search(titulo):
continue
if RUSO.search(titulo):
continue
# EXIGIR audio en castellano: sin marcador de doblaje ES, se descarta
if not CASTELLANO_AUDIO.search(titulo):
continue
utiles.append(t)
# Ordenar: mejor calidad, mas seeders
orden = {"4K": 0, "1080p": 1, "720p": 2, "?": 3}
utiles.sort(key=lambda t: (orden.get(t["calidad"], 3), -t["seeders"]))
return utiles[:limite]
def magnet_de(t):
"""Construye el magnet a partir del infoHash y nombre de fichero."""
dn = urllib.parse.quote((t.get("filename") or t.get("title") or "video"))
return f"magnet:?xt=urn:btih:{t['info_hash']}&dn={dn}"
def formatear_titulos(resultados):
"""Formatea resultados de Cinemeta numerados para Telegram."""
if not resultados:
return ["No se encontraron resultados para ese termino."]
lineas = [f"🎬 <b>Resultados globales ({len(resultados)}):</b>",
"Responde <b>/torrents &lt;numero&gt;</b> para ver torrents.", ""]
for i, r in enumerate(resultados, 1):
icono = "📺" if r["type"] == "series" else "🎬"
y = f" ({r['year']})" if r.get("year") else ""
lineas.append(f"<b>{i}.</b> {icono} {r['name']}{y}")
lineas.append(f" <code>{r['imdb_id']}</code>")
return vcb.trocear_texto(lineas)
def formatear_torrents(torrents, titulo):
"""Formatea los torrents de un titulo numerados con calidad/size/seeders."""
if not torrents:
return [f"No se encontraron torrents utiles para <b>{titulo}</b>."]
lineas = [
f"🧲 <b>{titulo}</b> — {len(torrents)} torrents.",
"Responde <b>/bajar &lt;numero&gt;</b> de estos para descargarlo al NAS.",
"",
]
for i, t in enumerate(torrents, 1):
size = f"{t['size']:.1f}GB" if t.get("size") else "?"
lineas.append(
f"<b>{i}.</b> {t['calidad']} | {size} | 👤 {t['seeders']} | {t['filename'] or t['name']}")
return vcb.trocear_texto(lineas)
if __name__ == "__main__":
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
if len(sys.argv) < 2:
print("Uso:")
print(" python global_buscar.py busca <texto>")
print(" python global_buscar.py torrents <imdb_id> [movie|series]")
print(" python global_buscar.py generos")
print(" python global_buscar.py catalogo <movie|series> <genero> [top|imdbRating] [limite]")
sys.exit(1)
cmd = sys.argv[1]
if cmd == "busca":
q = " ".join(sys.argv[2:])
for m in formatear_titulos(buscar_titulos(q)):
print(m)
print("---")
elif cmd == "torrents":
imdb = sys.argv[2]
tipo = sys.argv[3] if len(sys.argv) > 3 else "movie"
res = torrents_de(imdb, tipo)
res = filtrar_torrents(res)
for m in formatear_torrents(res, imdb):
print(m)
print("---")
elif cmd == "generos":
for m in formatear_generos():
print(m)
print("---")
elif cmd == "catalogo":
tipo = sys.argv[2]
genero = sys.argv[3]
cat = sys.argv[4] if len(sys.argv) > 4 else "top"
lim = int(sys.argv[5]) if len(sys.argv) > 5 else 30
res = catalogo_por_genero(tipo, genero, cat, lim)
for m in formatear_catalogo(res, tipo, genero, cat):
print(m)
print("---")