Plugin cleanup and tweaks

This commit is contained in:
2023-02-20 19:18:45 -06:00
parent 372e4ff3dc
commit 3ad9e1c7bb
1138 changed files with 48878 additions and 40445 deletions

View File

@@ -1,13 +1,12 @@
# coding: utf-8
from __future__ import unicode_literals
from .common import InfoExtractor
from ..utils import (
ExtractorError,
get_first,
int_or_none,
traverse_obj,
try_get,
unified_strdate,
unified_timestamp
unified_timestamp,
)
from ..compat import compat_str
@@ -17,45 +16,49 @@ class OpenRecBaseIE(InfoExtractor):
return self._parse_json(
self._search_regex(r'(?m)window\.pageStore\s*=\s*(\{.+?\});$', webpage, 'window.pageStore'), video_id)
def _extract_movie(self, webpage, video_id, name, is_live):
window_stores = self._extract_pagestore(webpage, video_id)
movie_store = traverse_obj(
window_stores,
('v8', 'state', 'movie'),
('v8', 'movie'),
expected_type=dict)
if not movie_store:
raise ExtractorError(f'Failed to extract {name} info')
title = movie_store.get('title')
description = movie_store.get('introduction')
thumbnail = movie_store.get('thumbnailUrl')
uploader = traverse_obj(movie_store, ('channel', 'user', 'name'), expected_type=compat_str)
uploader_id = traverse_obj(movie_store, ('channel', 'user', 'id'), expected_type=compat_str)
timestamp = int_or_none(traverse_obj(movie_store, ('publishedAt', 'time')), scale=1000)
m3u8_playlists = movie_store.get('media') or {}
formats = []
for name, m3u8_url in m3u8_playlists.items():
def _expand_media(self, video_id, media):
for name, m3u8_url in (media or {}).items():
if not m3u8_url:
continue
formats.extend(self._extract_m3u8_formats(
m3u8_url, video_id, ext='mp4', entry_protocol='m3u8',
m3u8_id='hls-%s' % name, live=True))
yield from self._extract_m3u8_formats(
m3u8_url, video_id, ext='mp4', m3u8_id=name)
self._sort_formats(formats)
def _extract_movie(self, webpage, video_id, name, is_live):
window_stores = self._extract_pagestore(webpage, video_id)
movie_stores = [
# extract all three important data (most of data are duplicated each other, but slightly different!)
traverse_obj(window_stores, ('v8', 'state', 'movie'), expected_type=dict),
traverse_obj(window_stores, ('v8', 'movie'), expected_type=dict),
traverse_obj(window_stores, 'movieStore', expected_type=dict),
]
if not any(movie_stores):
raise ExtractorError(f'Failed to extract {name} info')
formats = list(self._expand_media(video_id, get_first(movie_stores, 'media')))
if not formats:
# archived livestreams or subscriber-only videos
cookies = self._get_cookies('https://www.openrec.tv/')
detail = self._download_json(
f'https://apiv5.openrec.tv/api/v5/movies/{video_id}/detail', video_id,
headers={
'Origin': 'https://www.openrec.tv',
'Referer': 'https://www.openrec.tv/',
'access-token': try_get(cookies, lambda x: x.get('access_token').value),
'uuid': try_get(cookies, lambda x: x.get('uuid').value),
})
new_media = traverse_obj(detail, ('data', 'items', ..., 'media'), get_all=False)
formats = list(self._expand_media(video_id, new_media))
is_live = False
return {
'id': video_id,
'title': title,
'description': description,
'thumbnail': thumbnail,
'title': get_first(movie_stores, 'title'),
'description': get_first(movie_stores, 'introduction'),
'thumbnail': get_first(movie_stores, 'thumbnailUrl'),
'formats': formats,
'uploader': uploader,
'uploader_id': uploader_id,
'timestamp': timestamp,
'uploader': get_first(movie_stores, ('channel', 'user', 'name')),
'uploader_id': get_first(movie_stores, ('channel', 'user', 'id')),
'timestamp': int_or_none(get_first(movie_stores, ['publishedAt', 'time']), scale=1000) or unified_timestamp(get_first(movie_stores, 'publishedAt')),
'is_live': is_live,
}
@@ -73,7 +76,7 @@ class OpenRecIE(OpenRecBaseIE):
def _real_extract(self, url):
video_id = self._match_id(url)
webpage = self._download_webpage('https://www.openrec.tv/live/%s' % video_id, video_id)
webpage = self._download_webpage(f'https://www.openrec.tv/live/{video_id}', video_id)
return self._extract_movie(webpage, video_id, 'live', True)
@@ -97,7 +100,7 @@ class OpenRecCaptureIE(OpenRecBaseIE):
def _real_extract(self, url):
video_id = self._match_id(url)
webpage = self._download_webpage('https://www.openrec.tv/capture/%s' % video_id, video_id)
webpage = self._download_webpage(f'https://www.openrec.tv/capture/{video_id}', video_id)
window_stores = self._extract_pagestore(webpage, video_id)
movie_store = window_stores.get('movie')
@@ -105,29 +108,19 @@ class OpenRecCaptureIE(OpenRecBaseIE):
capture_data = window_stores.get('capture')
if not capture_data:
raise ExtractorError('Cannot extract title')
title = capture_data.get('title')
thumbnail = capture_data.get('thumbnailUrl')
upload_date = unified_strdate(capture_data.get('createdAt'))
uploader = traverse_obj(movie_store, ('channel', 'name'), expected_type=compat_str)
uploader_id = traverse_obj(movie_store, ('channel', 'id'), expected_type=compat_str)
timestamp = traverse_obj(movie_store, 'createdAt', expected_type=compat_str)
timestamp = unified_timestamp(timestamp)
formats = self._extract_m3u8_formats(
capture_data.get('source'), video_id, ext='mp4')
self._sort_formats(formats)
return {
'id': video_id,
'title': title,
'thumbnail': thumbnail,
'title': capture_data.get('title'),
'thumbnail': capture_data.get('thumbnailUrl'),
'formats': formats,
'timestamp': timestamp,
'uploader': uploader,
'uploader_id': uploader_id,
'upload_date': upload_date,
'timestamp': unified_timestamp(traverse_obj(movie_store, 'createdAt', expected_type=compat_str)),
'uploader': traverse_obj(movie_store, ('channel', 'name'), expected_type=compat_str),
'uploader_id': traverse_obj(movie_store, ('channel', 'id'), expected_type=compat_str),
'upload_date': unified_strdate(capture_data.get('createdAt')),
}
@@ -149,6 +142,6 @@ class OpenRecMovieIE(OpenRecBaseIE):
def _real_extract(self, url):
video_id = self._match_id(url)
webpage = self._download_webpage('https://www.openrec.tv/movie/%s' % video_id, video_id)
webpage = self._download_webpage(f'https://www.openrec.tv/movie/{video_id}', video_id)
return self._extract_movie(webpage, video_id, 'movie', False)