Files
biblioteca_conocimiento_lab…/videoclub/global_buscar.py

224 lines
7.4 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
global_buscar.py — Busqueda GLOBAL de peliculas/series que NO estan en tu
biblioteca RealDebrid. Busca "miles de pelis" en internet.
Flujo (2 pasos, sin estado persistente):
1) Texto -> titulo: Cinemeta (Stremio, sin clave) devuelve los titulos
mas relevantes con su id IMDb (ttXXXXXXX).
2) Titulo -> torrente: Torrentio (Stremio) devuelve los torrents reales
scaneados de 1337x, TPB, TorrentGalaxy, Wolfmax4k, ...
cada uno con infoHash para montar el magnet.
Zona "miles de peliculas": Torrentio consulta decenas de proveedores en
tiempo real y devuelve cientos de stream por titulo, no solo tu biblioteca.
Bajar a NAS: ./nas_dl.py "<magnet>" (Download Station se encarga).
"""
import json
import re
import sys
import urllib.parse
import urllib.request
from pathlib import Path
BASE = Path(__file__).resolve().parent
sys.path.insert(0, str(BASE))
CINEMETA = "https://v3-cinemeta.strem.io"
TORRENTIO = "https://torrentio.strem.fun/language=spanish"
HEADERS = {"User-Agent": "Mozilla/5.0"}
# Utilidades de clasificacion reutilizadas de videoclub_buscador
import videoclub_buscador as vcb
# Idiomas que no interesan en el catalogo global (ruso, tamil, hindi, etc.)
OTRO_IDIOMA_GLOBAL = re.compile(
r"(Tamil|Hindi|Telugu|Malayalam|Turkish|Tur[cç]e|Zulu|Arabic|"
r"Cyrillic|Russian|Greek|Korean|Chinese|Japanese|Thai)",
re.IGNORECASE)
RUSO = re.compile(r"[а-яА-ЯёЁ]")
def _get_json(url, timeout=35):
req = urllib.request.Request(url, headers=HEADERS)
with urllib.request.urlopen(req, timeout=timeout) as r:
return json.loads(r.read().decode("utf-8"))
def buscar_titulos(termino, limite=10):
"""Texto -> lista de {name, type, imdb_id, year}. Usa Cinemeta (sin clave)."""
term = (termino or "").strip()
if not term:
return []
resultado, lista = [], []
for _type in ("movie", "series"):
url = (f"{CINEMETA}/catalog/{_type}/top/"
f"search={urllib.parse.quote(term)}.json")
try:
metas = _get_json(url).get("metas", [])
except Exception:
metas = []
for m in metas[:limite]:
imdb = m.get("id", "")
if not re.match(r"^tt\d+$", imdb):
continue
name = m.get("name", "")
year = m.get("year") or None
resultado.append({
"name": name,
"type": _type,
"imdb_id": imdb,
"year": str(year) if year else None,
})
# Intercalamos movie/series y deduplicamos por imdb_id
vistos = set()
for r in resultado:
if r["imdb_id"] in vistos:
continue
vistos.add(r["imdb_id"])
lista.append(r)
return lista[:20]
def torrents_de(imdb_id, tipo="movie"):
"""IMDb id -> lista de torrents reales de Torrentio.
Cada item: {name, title, info_hash, file_id, filename, seeders, size}."""
if not re.match(r"^tt\d+$", imdb_id or ""):
return []
url = f"{TORRENTIO}/stream/{tipo}/{imdb_id}.json"
try:
data = _get_json(url)
except Exception:
return []
torrents = []
for s in data.get("streams", []):
if not s.get("infoHash"):
continue
filename = (s.get("behaviorHints", {}) or {}).get("filename") or ""
title = (s.get("title") or "").replace("\n", " ")
torrents.append({
"name": s.get("name", "").replace("\n", " "),
"title": title,
"info_hash": s["infoHash"],
"file_id": s.get("fileIdx", 0),
"filename": filename,
"seeders": _seeders(title),
"size": _size(title),
"calidad": _calidad(filename + " " + title),
})
return torrents
def _seeders(title):
m = re.search(r"👤\s*(\d+)", title)
return int(m.group(1)) if m else 0
def _size(title):
m = re.search(r"💾\s*([\d.]+)\s*(TB|GB|MB)", title)
if not m:
return None
val = float(m.group(1))
if m.group(2) == "TB":
return val * 1024
if m.group(2) == "MB":
return val / 1024
return val
def _calidad(texto):
t = " " + texto.lower() + " "
if "2160p" in t or ("4k" in t and "1080p" not in t):
return "4K"
if "1080p" in t or "1080" in t:
return "1080p"
if "720p" in t:
return "720p"
if "4k" in t or "uhd" in t:
return "4K"
return "?"
def filtrar_torrents(torrents, limite=20):
"""Descarta calidad mala y contenido claramente no en castellano/ingles."""
utiles = []
for t in torrents:
if vcb.CALIDAD_MALA.search(t["title"]):
continue
if vcb.OTRO_IDIOMA.search(t["title"]) or OTRO_IDIOMA_GLOBAL.search(t["title"]):
continue
if RUSO.search(t["title"]):
continue
if not (vcb.CASTELLANO.search(t["title"]) or vcb.INGLES.search(t["title"])):
continue
utiles.append(t)
# Ordenar: mejor calidad, mas seeders
orden = {"4K": 0, "1080p": 1, "720p": 2, "?": 3}
utiles.sort(key=lambda t: (orden.get(t["calidad"], 3), -t["seeders"]))
return utiles[:limite]
def magnet_de(t):
"""Construye el magnet a partir del infoHash y nombre de fichero."""
dn = urllib.parse.quote((t.get("filename") or t.get("title") or "video"))
return f"magnet:?xt=urn:btih:{t['info_hash']}&dn={dn}"
def formatear_titulos(resultados):
"""Formatea resultados de Cinemeta numerados para Telegram."""
if not resultados:
return ["No se encontraron resultados para ese termino."]
lineas = [f"🎬 <b>Resultados globales ({len(resultados)}):</b>",
"Responde <b>/bajar &lt;numero&gt;</b> para ver torrents.", ""]
for i, r in enumerate(resultados, 1):
icono = "📺" if r["type"] == "series" else "🎬"
y = f" ({r['year']})" if r.get("year") else ""
lineas.append(f"<b>{i}.</b> {icono} {r['name']}{y}")
lineas.append(f" <code>{r['imdb_id']}</code>")
return vcb.trocear_texto(lineas)
def formatear_torrents(torrents, titulo):
"""Formatea los torrents de un titulo numerados con calidad/size/seeders."""
if not torrents:
return [f"No se encontraron torrents utiles para <b>{titulo}</b>."]
lineas = [
f"🧲 <b>{titulo}</b> — {len(torrents)} torrents.",
"Responde <b>/bajar &lt;numero&gt;</b> de estos para descargarlo al NAS.",
"",
]
for i, t in enumerate(torrents, 1):
size = f"{t['size']:.1f}GB" if t.get("size") else "?"
lineas.append(
f"<b>{i}.</b> {t['calidad']} | {size} | 👤 {t['seeders']} | {t['filename'] or t['name']}")
return vcb.trocear_texto(lineas)
if __name__ == "__main__":
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
except Exception:
pass
if len(sys.argv) < 2:
print("Uso:")
print(" python global_buscar.py busca <texto>")
print(" python global_buscar.py torrents <imdb_id> [movie|series]")
sys.exit(1)
cmd = sys.argv[1]
if cmd == "busca":
q = " ".join(sys.argv[2:])
for m in formatear_titulos(buscar_titulos(q)):
print(m)
print("---")
elif cmd == "torrents":
imdb = sys.argv[2]
tipo = sys.argv[3] if len(sys.argv) > 3 else "movie"
res = torrents_de(imdb, tipo)
res = filtrar_torrents(res)
for m in formatear_torrents(res, imdb):
print(m)
print("---")