477 lines
18 KiB
Python
477 lines
18 KiB
Python
#!/usr/bin/env python3
|
|
# -*- coding: utf-8 -*-
|
|
"""
|
|
HLS / MP4 비디오 다운로더 (maccms 계열 스트리밍 사이트 대응)
|
|
==========================================================
|
|
- 페이지 URL 입력 시 영상 주소(m3u8/mp4) 자동 추출 (maccms player_aaaa, 정규식, iframe)
|
|
- HLS 완전 지원: 마스터 재생목록(최고 화질 자동 선택), AES-128 복호화, fMP4(#EXT-X-MAP) 초기화 세그먼트
|
|
- 세그먼트 병렬 다운로드 + 이어받기(중단 후 재실행 시 받은 분할 건너뜀)
|
|
- ffmpeg 자동 리먹스(TS/fMP4 -> MP4)
|
|
- Referer 등 커스텀 헤더 지원 (CDN 토큰 인증 대응)
|
|
|
|
의존성: requests (필수) / pycryptodome (AES 암호화 영상, 옵션) / tqdm (진행률, 옵션) / ffmpeg (리먹스, 권장)
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from urllib.parse import urljoin, urlparse
|
|
|
|
import requests
|
|
|
|
# Windows 콘솔 한글 깨짐 방지
|
|
try:
|
|
sys.stdout.reconfigure(encoding="utf-8")
|
|
sys.stderr.reconfigure(encoding="utf-8")
|
|
except Exception:
|
|
pass
|
|
|
|
DEFAULT_UA = (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
|
)
|
|
DEFAULT_HEADERS = {
|
|
"User-Agent": DEFAULT_UA,
|
|
"Accept": "*/*",
|
|
"Accept-Language": "ko-KR,ko;q=0.9,en;q=0.8",
|
|
}
|
|
|
|
# m3u8 속성 파싱: KEY="VAL",KEY2=VAL
|
|
_ATTR_RE = re.compile(r'([A-Z0-9\-]+)=("([^"]*)"|([^,]*))')
|
|
|
|
# 페이지에서 영상 주소 추출용 정규식
|
|
_PLAYER_AAAA_RE = re.compile(r"player_aaaa\s*=\s*(\{.*?\})\s*[;<]", re.S)
|
|
_PLAYER_DATA_RE = re.compile(r"player_data\s*=\s*(\{.*?\})\s*[;<]", re.S)
|
|
_M3U8_RE = re.compile(r"https?://[^\s\"'<>]*\.m3u8[^\s\"'<>]*", re.I)
|
|
_MP4_RE = re.compile(r"https?://[^\s\"'<>]*\.mp4[^\s\"'<>]*", re.I)
|
|
_IFRAME_RE = re.compile(r'<iframe[^>]+src=["\']([^"\']+)["\']', re.I)
|
|
|
|
|
|
def parse_attrs(s):
|
|
d = {}
|
|
for m in _ATTR_RE.finditer(s):
|
|
d[m.group(1)] = m.group(3) if m.group(3) is not None else m.group(4)
|
|
return d
|
|
|
|
|
|
def parse_m3u8(text, base_url):
|
|
"""m3u8 재생목록 파싱. 마스터(변환재생목록)와 미디어 재생목록을 모두 처리."""
|
|
result = {"variants": [], "segments": [], "init": None, "is_master": False}
|
|
pending_variant = None # 직전 #EXT-X-STREAM-INF 속성
|
|
cur_key = None # 현재 유효한 #EXT-X-KEY (세그먼트별로 참조 복사)
|
|
seq = 0
|
|
idx = 0
|
|
saw_seq = False
|
|
|
|
for raw in text.splitlines():
|
|
line = raw.strip()
|
|
if not line:
|
|
continue
|
|
|
|
if line.startswith("#EXT-X-STREAM-INF:"):
|
|
pending_variant = parse_attrs(line[len("#EXT-X-STREAM-INF:"):])
|
|
continue
|
|
|
|
if line.startswith("#EXT-X-MEDIA-SEQUENCE:"):
|
|
try:
|
|
seq = int(line.split(":", 1)[1].strip())
|
|
saw_seq = True
|
|
except ValueError:
|
|
pass
|
|
continue
|
|
|
|
if line.startswith("#EXT-X-KEY:"):
|
|
a = parse_attrs(line[len("#EXT-X-KEY:"):])
|
|
method = a.get("METHOD", "NONE").upper()
|
|
if method == "AES-128":
|
|
uri = a.get("URI")
|
|
if uri:
|
|
uri = urljoin(base_url, uri) # 재생목록 기준 절대경로화
|
|
cur_key = {"uri": uri, "iv": a.get("IV")}
|
|
else:
|
|
cur_key = None
|
|
continue
|
|
|
|
if line.startswith("#EXT-X-MAP:"):
|
|
a = parse_attrs(line[len("#EXT-X-MAP:"):])
|
|
if a.get("URI"):
|
|
result["init"] = urljoin(base_url, a["URI"]) # fMP4 초기화 세그먼트
|
|
continue
|
|
|
|
if line.startswith("#"):
|
|
continue
|
|
|
|
# URL 라인
|
|
abs_url = urljoin(base_url, line)
|
|
if pending_variant is not None:
|
|
bw = 0
|
|
try:
|
|
bw = int(pending_variant.get("BANDWIDTH", "0"))
|
|
except ValueError:
|
|
pass
|
|
result["variants"].append(
|
|
{
|
|
"url": abs_url,
|
|
"bandwidth": bw,
|
|
"res": pending_variant.get("RESOLUTION", ""),
|
|
"codecs": pending_variant.get("CODECS", ""),
|
|
}
|
|
)
|
|
pending_variant = None
|
|
result["is_master"] = True
|
|
else:
|
|
result["segments"].append(
|
|
{"url": abs_url, "index": idx, "seq": seq, "key": cur_key}
|
|
)
|
|
idx += 1
|
|
seq += 1
|
|
|
|
return result
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# 진행률 표시 (tqdm이 있으면 사용, 없으면 단순 폴백)
|
|
# --------------------------------------------------------------------------- #
|
|
class _SimpleProgress:
|
|
def __init__(self, total=None, desc="", unit=""):
|
|
self.total = total
|
|
self.n = 0
|
|
self.desc = desc
|
|
self.unit = unit
|
|
self._last_pct = -1
|
|
|
|
def update(self, n=1):
|
|
self.n += n
|
|
if self.total:
|
|
pct = int(self.n * 100 / self.total)
|
|
if pct != self._last_pct and pct % 5 == 0:
|
|
self._last_pct = pct
|
|
if self.unit == "B":
|
|
sys.stdout.write(
|
|
f"\r{self.desc}: {pct}% "
|
|
f"({self.n/1048576:.1f}/{self.total/1048576:.1f} MB)"
|
|
)
|
|
else:
|
|
sys.stdout.write(f"\r{self.desc}: {pct}% ({self.n}/{self.total})")
|
|
sys.stdout.flush()
|
|
|
|
def close(self):
|
|
sys.stdout.write("\n")
|
|
sys.stdout.flush()
|
|
|
|
|
|
def make_progress(total=None, **kw):
|
|
try:
|
|
from tqdm import tqdm
|
|
|
|
return tqdm(total=total, **kw)
|
|
except Exception:
|
|
return _SimpleProgress(total=total, desc=kw.get("desc", ""), unit=kw.get("unit", ""))
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
# 다운로더 본체
|
|
# --------------------------------------------------------------------------- #
|
|
class Downloader:
|
|
def __init__(self, output, headers=None, concurrency=8, retries=6, timeout=30):
|
|
self.session = requests.Session()
|
|
self.session.headers.update(DEFAULT_HEADERS)
|
|
for k, v in (headers or {}).items():
|
|
self.session.headers[k] = v
|
|
self.concurrency = max(1, concurrency)
|
|
self.retries = retries
|
|
self.timeout = timeout
|
|
self.output = output
|
|
self.tmp = output + ".parts"
|
|
os.makedirs(self.tmp, exist_ok=True)
|
|
self._key_cache = {}
|
|
|
|
# ---- 공통 HTTP ---- #
|
|
def _get_text(self, url):
|
|
r = self.session.get(url, timeout=self.timeout)
|
|
r.raise_for_status()
|
|
return r.text
|
|
|
|
def _download_file(self, url, path):
|
|
tmp = path + ".tmp"
|
|
with self.session.get(url, timeout=self.timeout, stream=True) as r:
|
|
r.raise_for_status()
|
|
with open(tmp, "wb") as f:
|
|
for chunk in r.iter_content(65536):
|
|
if chunk:
|
|
f.write(chunk)
|
|
os.replace(tmp, path)
|
|
|
|
# ---- 진입: URL 종류 판별 ---- #
|
|
def run(self, url):
|
|
path = urlparse(url).path.lower()
|
|
if ".m3u8" in path:
|
|
return self.run_hls(url)
|
|
if path.endswith(".mp4"):
|
|
return self.run_direct(url)
|
|
return self.run_page(url)
|
|
|
|
# ---- 페이지에서 영상 주소 추출 ---- #
|
|
def run_page(self, page_url):
|
|
print(f"페이지 분석 중: {page_url}")
|
|
self.session.headers.setdefault("Referer", page_url)
|
|
html = self._get_text(page_url)
|
|
|
|
cands = [] # (kind, url)
|
|
|
|
def scan(text):
|
|
for rx, kind in ((_PLAYER_AAAA_RE, "player_aaaa"), (_PLAYER_DATA_RE, "player_data")):
|
|
for m in rx.finditer(text):
|
|
try:
|
|
obj = json.loads(m.group(1))
|
|
except ValueError:
|
|
continue
|
|
for field in ("url", "parse", "parse2"):
|
|
u = obj.get(field)
|
|
if isinstance(u, str) and u.startswith("http"):
|
|
cands.append((f"{kind}.{field}", u))
|
|
for u in _M3U8_RE.findall(text):
|
|
cands.append(("m3u8", u))
|
|
for u in _MP4_RE.findall(text):
|
|
cands.append(("mp4", u))
|
|
|
|
scan(html)
|
|
|
|
# iframe 내부 플레이어 페이지도 1단계까지 탐색
|
|
for src in _IFRAME_RE.findall(html):
|
|
abs_src = urljoin(page_url, src)
|
|
if "http" not in abs_src:
|
|
continue
|
|
try:
|
|
scan(self._get_text(abs_src))
|
|
except Exception:
|
|
pass
|
|
|
|
# 중복 제거(순서 유지) 후 m3u8 우선
|
|
seen, uniq = set(), []
|
|
for kind, u in cands:
|
|
if u in seen:
|
|
continue
|
|
seen.add(u)
|
|
uniq.append((kind, u))
|
|
|
|
if not uniq:
|
|
raise RuntimeError(
|
|
"페이지에서 영상 주소를 찾지 못했습니다. README의 "
|
|
"'m3u8 주소 직접 구하기' 방법으로 주소를 얻어 .m3u8 주소를 직접 넘겨주세요."
|
|
)
|
|
|
|
print("발견된 영상 주소 후보:")
|
|
for i, (kind, u) in enumerate(uniq):
|
|
print(f" [{i}] ({kind}) {u}")
|
|
|
|
chosen = next((u for _, u in uniq if ".m3u8" in u.lower()), None)
|
|
if chosen is None:
|
|
chosen = next((u for _, u in uniq if ".mp4" in u.lower()), uniq[0][1])
|
|
print(f"\n선택: {chosen}\n")
|
|
|
|
if ".m3u8" in chosen.lower():
|
|
return self.run_hls(chosen)
|
|
return self.run_direct(chosen)
|
|
|
|
# ---- HLS 다운로드 ---- #
|
|
def run_hls(self, m3u8_url):
|
|
print(f"HLS 재생목록 로드: {m3u8_url}")
|
|
parsed = parse_m3u8(self._get_text(m3u8_url), m3u8_url)
|
|
|
|
# 마스터 재생목록이면 최고 화질 변환 선택 후 미디어 재생목록 로드
|
|
while parsed["is_master"]:
|
|
best = max(parsed["variants"], key=lambda v: v["bandwidth"])
|
|
label = best["res"] or f"{best['bandwidth']}bps"
|
|
print(f"마스터 재생목록 -> 최고 화질 선택: {label}")
|
|
parsed = parse_m3u8(self._get_text(best["url"]), best["url"])
|
|
|
|
segs = parsed["segments"]
|
|
if not segs:
|
|
raise RuntimeError("재생목록에 다운로드할 세그먼트가 없습니다.")
|
|
|
|
encrypted = any(s["key"] for s in segs)
|
|
print(f"세그먼트 수: {len(segs)} | 암호화: {'AES-128' if encrypted else '없음'}")
|
|
|
|
# fMP4 초기화 세그먼트
|
|
init_path = None
|
|
if parsed["init"]:
|
|
init_path = os.path.join(self.tmp, "init.fmp4")
|
|
print("fMP4 초기화 세그먼트 다운로드 ...")
|
|
self._download_file(parsed["init"], init_path)
|
|
|
|
self._download_segments(segs)
|
|
merged = self._merge(segs, init_path)
|
|
final = self._remux(merged, init_path is not None)
|
|
return final
|
|
|
|
def _seg_path(self, seg):
|
|
return os.path.join(self.tmp, f"{seg['index']:06d}.seg")
|
|
|
|
def _download_segments(self, segs):
|
|
todo = [s for s in segs if not (os.path.exists(self._seg_path(s)) and os.path.getsize(self._seg_path(s)) > 0)]
|
|
skipped = len(segs) - len(todo)
|
|
print(f"다운로드: {len(todo)}개 (이미 받은 {skipped}개 건너뜀), 동시 {self.concurrency}연결")
|
|
|
|
prog = make_progress(total=len(segs), initial=skipped, desc="세그먼트", unit="seg")
|
|
failures = []
|
|
|
|
def work(seg):
|
|
for attempt in range(1, self.retries + 1):
|
|
try:
|
|
self._download_file(seg["url"], self._seg_path(seg))
|
|
return None
|
|
except Exception as e:
|
|
if attempt == self.retries:
|
|
return (seg, str(e))
|
|
time.sleep(min(2 ** attempt, 10))
|
|
return None
|
|
|
|
with ThreadPoolExecutor(max_workers=self.concurrency) as ex:
|
|
futures = [ex.submit(work, s) for s in todo]
|
|
for fut in as_completed(futures):
|
|
err = fut.result()
|
|
if err:
|
|
failures.append(err)
|
|
prog.update(1)
|
|
prog.close()
|
|
|
|
if failures:
|
|
sample = "; ".join(f"{s['index']}({m})" for s, m in failures[:3])
|
|
raise RuntimeError(f"{len(failures)}개 세그먼트 실패(재시도 후): {sample} ...")
|
|
|
|
@staticmethod
|
|
def _aes():
|
|
try:
|
|
from Crypto.Cipher import AES
|
|
return AES
|
|
except ImportError:
|
|
raise RuntimeError("암호화 영상입니다. 'pip install pycryptodome' 설치 후 재시도하세요.")
|
|
|
|
def _decrypt(self, data, seg):
|
|
key = seg["key"]
|
|
if not key:
|
|
return data
|
|
if key["uri"] not in self._key_cache:
|
|
r = self.session.get(key["uri"], timeout=self.timeout)
|
|
r.raise_for_status()
|
|
self._key_cache[key["uri"]] = r.content # 보통 16바이트
|
|
keybytes = self._key_cache[key["uri"]]
|
|
if key["iv"]:
|
|
ivb = bytes.fromhex(key["iv"].replace("0x", "").replace(" ", ""))
|
|
else:
|
|
ivb = seg["seq"].to_bytes(16, "big") # IV 미지정 시 시퀀스 번호
|
|
cipher = self._aes().new(keybytes, self._aes().MODE_CBC, ivb)
|
|
return cipher.decrypt(data)
|
|
|
|
def _merge(self, segs, init_path):
|
|
is_fmp4 = init_path is not None
|
|
ext = ".mp4" if is_fmp4 else ".ts"
|
|
merged = os.path.abspath(self.output + ".merged" + ext)
|
|
print("세그먼트 병합 중 ...")
|
|
with open(merged, "wb") as out:
|
|
if init_path and os.path.exists(init_path):
|
|
with open(init_path, "rb") as f:
|
|
out.write(f.read())
|
|
for seg in sorted(segs, key=lambda x: x["index"]):
|
|
with open(self._seg_path(seg), "rb") as f:
|
|
out.write(self._decrypt(f.read(), seg))
|
|
return merged
|
|
|
|
def _remux(self, merged, is_fmp4):
|
|
base, ext = os.path.splitext(self.output)
|
|
if ext.lower() in (".ts", ".m3u8", ".m4s", ".fmp4"):
|
|
ext = ""
|
|
ffmpeg = shutil.which("ffmpeg")
|
|
if ffmpeg:
|
|
final = (base or "video") + ".mp4"
|
|
if os.path.abspath(final) == os.path.abspath(merged):
|
|
final = (base or "video") + ".out.mp4"
|
|
print(f"ffmpeg 리먹스(-c copy) -> {os.path.basename(final)}")
|
|
ret = subprocess.run(
|
|
[ffmpeg, "-y", "-hide_banner", "-loglevel", "error", "-i", merged, "-c", "copy", final]
|
|
)
|
|
if ret.returncode == 0 and os.path.exists(final) and os.path.getsize(final) > 0:
|
|
os.remove(merged)
|
|
return final
|
|
print("경고: ffmpeg 리먹스 실패, 병합본을 그대로 유지합니다.", file=sys.stderr)
|
|
return merged
|
|
# ffmpeg 없음: 확장자만 맞춰 이동
|
|
final = (base or "video") + (".mp4" if is_fmp4 else ".ts")
|
|
os.replace(merged, final)
|
|
print("안내: ffmpeg가 없어 리먹스를 생략했습니다. MP4 변환을 권장합니다.", file=sys.stderr)
|
|
return final
|
|
|
|
# ---- 직접 MP4 다운로드 ---- #
|
|
def run_direct(self, url):
|
|
base, ext = os.path.splitext(self.output)
|
|
final = self.output if ext.lower() in (".mp4", ".mkv", ".ts") else (base or "video") + ".mp4"
|
|
print(f"직접 다운로드: {url}")
|
|
with self.session.get(url, timeout=self.timeout, stream=True) as r:
|
|
r.raise_for_status()
|
|
total = int(r.headers.get("Content-Length", 0)) or None
|
|
prog = make_progress(total=total, desc="다운로드", unit="B", unit_scale=True)
|
|
tmp = final + ".tmp"
|
|
with open(tmp, "wb") as f:
|
|
for chunk in r.iter_content(65536):
|
|
if chunk:
|
|
f.write(chunk)
|
|
prog.update(len(chunk))
|
|
prog.close()
|
|
os.replace(tmp, final)
|
|
return final
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser(
|
|
description="HLS/MP4 비디오 다운로더 (maccms 계열 사이트 대응)",
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog=(
|
|
"예제:\n"
|
|
' python downloader.py "https://site/vod/play/id/123/sid/1/nid/1.html" -o movie\n'
|
|
' python downloader.py "https://cdn.example.com/v/index.m3u8" -o movie --referer "https://site/"\n'
|
|
' python downloader.py "https://cdn.example.com/v/index.m3u8" --referer "https://site/" -H "Origin: https://site"\n'
|
|
),
|
|
)
|
|
ap.add_argument("url", help="재생페이지 URL 또는 .m3u8 / .mp4 직접 주소")
|
|
ap.add_argument("-o", "--output", default="video", help="출력 파일명(확장자 생략 가능, 기본 video)")
|
|
ap.add_argument("--referer", help="Referer 헤더 (CDN 토큰 인증에 필요한 경우)")
|
|
ap.add_argument("-H", "--header", action="append", default=[], metavar='"Key: Value"', help="추가 헤더 (여러개 가능)")
|
|
ap.add_argument("-c", "--concurrency", type=int, default=8, help="동시 다운로드 연결 수 (기본 8)")
|
|
ap.add_argument("--user-agent", default=None, help="User-Agent 재정의")
|
|
ap.add_argument("--keep-parts", action="store_true", help="병합 후 세그먼트 임시파일 유지(디버그용)")
|
|
args = ap.parse_args()
|
|
|
|
headers = {}
|
|
for h in args.header:
|
|
if ":" in h:
|
|
k, v = h.split(":", 1)
|
|
headers[k.strip()] = v.strip()
|
|
if args.referer:
|
|
headers["Referer"] = args.referer
|
|
if args.user_agent:
|
|
headers["User-Agent"] = args.user_agent
|
|
|
|
dl = Downloader(args.output, headers, concurrency=args.concurrency)
|
|
try:
|
|
final = dl.run(args.url)
|
|
except KeyboardInterrupt:
|
|
print("\n중단됨. 받은 세그먼트는 유지되어 재실행 시 이어받기됩니다.")
|
|
sys.exit(130)
|
|
except Exception as e:
|
|
print(f"\n오류: {e}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
print(f"\n완료: {final}")
|
|
if not args.keep_parts:
|
|
shutil.rmtree(dl.tmp, ignore_errors=True)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|