restructured manifest and plugins loading; updated plugins

2025-12-29 22:50:05 -06:00
parent c74f97aca7
commit 21120cd61e
324 changed files with 18088 additions and 15974 deletions
--- a/plugins/youtube_download/yt_dlp/extractor/newspicks.py
+++ b/plugins/youtube_download/yt_dlp/extractor/newspicks.py
@@ -1,53 +1,77 @@
-import re
-
 from .common import InfoExtractor
-from ..utils import ExtractorError
+from ..utils import (
+    clean_html,
+    parse_iso8601,
+    parse_qs,
+    url_or_none,
+)
+from ..utils.traversal import require, traverse_obj


 class NewsPicksIE(InfoExtractor):
-    _VALID_URL = r'https?://newspicks\.com/movie-series/(?P<channel_id>\d+)\?movieId=(?P<id>\d+)'
-
+    _VALID_URL = r'https?://newspicks\.com/movie-series/(?P<id>[^?/#]+)'
    _TESTS = [{
-        'url': 'https://newspicks.com/movie-series/11?movieId=1813',
+        'url': 'https://newspicks.com/movie-series/11/?movieId=1813',
        'info_dict': {
            'id': '1813',
-            'title': '日本の課題を破壊せよ【ゲスト：成田悠輔】',
-            'description': 'md5:09397aad46d6ded6487ff13f138acadf',
-            'channel': 'HORIE ONE',
-            'channel_id': '11',
-            'release_date': '20220117',
-            'thumbnail': r're:https://.+jpg',
            'ext': 'mp4',
+            'title': '日本の課題を破壊せよ【ゲスト：成田悠輔】',
+            'cast': 'count:4',
+            'description': 'md5:09397aad46d6ded6487ff13f138acadf',
+            'release_date': '20220117',
+            'release_timestamp': 1642424400,
+            'series': 'HORIE ONE',
+            'series_id': '11',
+            'thumbnail': r're:https?://resources\.newspicks\.com/.+\.(?:jpe?g|png)',
+            'timestamp': 1642424420,
+            'upload_date': '20220117',
+        },
+    }, {
+        'url': 'https://newspicks.com/movie-series/158/?movieId=3932',
+        'info_dict': {
+            'id': '3932',
+            'ext': 'mp4',
+            'title': '【検証】専門家は、KADOKAWAをどう見るか',
+            'cast': 'count:3',
+            'description': 'md5:2c2d4bf77484a4333ec995d676f9a91d',
+            'release_date': '20240622',
+            'release_timestamp': 1719088080,
+            'series': 'NPレポート',
+            'series_id': '158',
+            'thumbnail': r're:https?://resources\.newspicks\.com/.+\.(?:jpe?g|png)',
+            'timestamp': 1719086400,
+            'upload_date': '20240622',
        },
    }]

    def _real_extract(self, url):
-        video_id, channel_id = self._match_valid_url(url).group('id', 'channel_id')
+        series_id = self._match_id(url)
+        video_id = traverse_obj(parse_qs(url), ('movieId', -1, {str}, {require('movie ID')}))
        webpage = self._download_webpage(url, video_id)
-        entries = self._parse_html5_media_entries(
-            url, webpage.replace('movie-for-pc', 'movie'), video_id, 'hls')
-        if not entries:
-            raise ExtractorError('No HTML5 media elements found')
-        info = entries[0]

-        title = self._html_search_meta('og:title', webpage, fatal=False)
-        description = self._html_search_meta(
-            ('og:description', 'twitter:title'), webpage, fatal=False)
-        channel = self._html_search_regex(
-            r'value="11".+?<div\s+class="title">(.+?)</div', webpage, 'channel name', fatal=False)
-        if not title or not channel:
-            title, channel = re.split(r'\s*|\s*', self._html_extract_title(webpage))
+        fragment = self._search_nextjs_data(webpage, video_id)['props']['pageProps']['fragment']
+        movie = fragment['movie']

-        release_date = self._search_regex(
-            r'<span\s+class="on-air-date">\s*(\d+)年(\d+)月(\d+)日\s*</span>',
-            webpage, 'release date', fatal=False, group=(1, 2, 3))
+        if traverse_obj(movie, ('viewable', {str})) == 'PARTIAL_FREE' and not traverse_obj(movie, ('canWatch', {bool})):
+            self.report_warning(
+                'Full video is for Premium members. Without cookies, '
+                f'only the preview is downloaded. {self._login_hint()}', video_id)

-        info.update({
+        m3u8_url = traverse_obj(movie, ('movieUrl', {url_or_none}, {require('m3u8 URL')}))
+        formats, subtitles = self._extract_m3u8_formats_and_subtitles(m3u8_url, video_id, 'mp4')
+
+        return {
            'id': video_id,
-            'title': title,
-            'description': description,
-            'channel': channel,
-            'channel_id': channel_id,
-            'release_date': ('%04d%02d%02d' % tuple(map(int, release_date))) if release_date else None,
-        })
-        return info
+            'formats': formats,
+            'series': traverse_obj(fragment, ('series', 'title', {str})),
+            'series_id': series_id,
+            'subtitles': subtitles,
+            **traverse_obj(movie, {
+                'title': ('title', {str}),
+                'cast': ('relatedUsers', ..., 'displayName', {str}, filter, all, filter),
+                'description': ('explanation', {clean_html}),
+                'release_timestamp': ('onAirStartDate', {parse_iso8601}),
+                'thumbnail': (('image', 'coverImageUrl'), {url_or_none}, any),
+                'timestamp': ('published', {parse_iso8601}),
+            }),
+        }