Merge fe354fa548 into e3b42d8b1b

[ie/facebook] Fix DASH formats extraction (#9734 )
Closes #9720 Authored by: bashonly
2024-04-21 04:46:31 +05:30 · 2024-04-20 10:23:12 +00:00 · 2024-04-18 23:18:56 +00:00 · 2024-04-18 23:11:12 +00:00 · 2024-03-20 18:51:52 +02:00 · 2024-03-20 18:51:42 +02:00
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@ -254,7 +254,7 @@ jobs:
          # We need to fuse our own universal2 wheels for curl_cffi
          python3 -m pip install -U --user delocate
          mkdir curl_cffi_whls curl_cffi_universal2
-          python3 devscripts/install_deps.py --print -o --include curl_cffi > requirements.txt
+          python3 devscripts/install_deps.py --print -o --include curl-cffi > requirements.txt
          for platform in "macosx_11_0_arm64" "macosx_11_0_x86_64"; do
            python3 -m pip download \
              --only-binary=:all: \
@ -362,7 +362,7 @@ jobs:
      - name: Install Requirements
        run: | # Custom pyinstaller built with https://github.com/yt-dlp/pyinstaller-builds
          python devscripts/install_deps.py -o --include build
-          python devscripts/install_deps.py --include py2exe --include curl_cffi
+          python devscripts/install_deps.py --include py2exe --include curl-cffi
          python -m pip install -U "https://yt-dlp.github.io/Pyinstaller-Builds/x86_64/pyinstaller-5.8.0-py3-none-any.whl"

      - name: Prepare
--- a/README.md
+++ b/README.md
@ -202,7 +202,7 @@ While all the other dependencies are optional, `ffmpeg` and `ffprobe` are highly
 The following provide support for impersonating browser requests. This may be required for some sites that employ TLS fingerprinting. 

 * [**curl_cffi**](https://github.com/yifeikong/curl_cffi) (recommended) - Python binding for [curl-impersonate](https://github.com/lwthiker/curl-impersonate). Provides impersonation targets for Chrome, Edge and Safari. Licensed under [MIT](https://github.com/yifeikong/curl_cffi/blob/main/LICENSE)
-  * Can be installed with the `curl_cffi` group, e.g. `pip install yt-dlp[default,curl_cffi]`
+  * Can be installed with the `curl-cffi` group, e.g. `pip install yt-dlp[default,curl-cffi]`
  * Currently only included in `yt-dlp.exe` and `yt-dlp_macos` builds


--- a/pyproject.toml
+++ b/pyproject.toml
@ -53,7 +53,7 @@ dependencies = [

 [project.optional-dependencies]
 default = []
-curl_cffi = ["curl-cffi==0.5.10; implementation_name=='cpython'"]
+curl-cffi = ["curl-cffi==0.5.10; implementation_name=='cpython'"]
 secretstorage = [
    "cffi",
    "secretstorage",
--- a/yt_dlp/extractor/facebook.py
+++ b/yt_dlp/extractor/facebook.py
@ -560,7 +560,7 @@ class FacebookIE(InfoExtractor):
                    js_data, lambda x: x['jsmods']['instances'], list) or [])

        def extract_dash_manifest(video, formats):
-            dash_manifest = video.get('dash_manifest')
+            dash_manifest = traverse_obj(video, 'dash_manifest', 'playlist', expected_type=str)
            if dash_manifest:
                formats.extend(self._parse_mpd_formats(
                    compat_etree_fromstring(urllib.parse.unquote_plus(dash_manifest)),
--- a/yt_dlp/extractor/patreon.py
+++ b/yt_dlp/extractor/patreon.py
@ -1,8 +1,8 @@
 import itertools
+import urllib.parse

 from .common import InfoExtractor
 from .vimeo import VimeoIE
-from ..compat import compat_urllib_parse_unquote
 from ..networking.exceptions import HTTPError
 from ..utils import (
    KNOWN_EXTENSIONS,
@ -14,7 +14,6 @@ from ..utils import (
    parse_iso8601,
    str_or_none,
    traverse_obj,
-    try_get,
    url_or_none,
    urljoin,
 )
@ -199,6 +198,27 @@ class PatreonIE(PatreonBaseIE):
            'channel_id': '2147162',
            'uploader_url': 'https://www.patreon.com/yaboyroshi',
        },
+    }, {
+        # NSFW vimeo embed URL
+        'url': 'https://www.patreon.com/posts/4k-spiderman-4k-96414599',
+        'info_dict': {
+            'id': '902250943',
+            'ext': 'mp4',
+            'title': '❤️(4K) Spiderman Girl Yeonhwa’s Gift ❤️(4K) 스파이더맨걸 연화의 선물',
+            'description': '❤️(4K) Spiderman Girl Yeonhwa’s Gift \n❤️(4K) 스파이더맨걸 연화의 선물',
+            'uploader': 'Npickyeonhwa',
+            'uploader_id': '90574422',
+            'uploader_url': 'https://www.patreon.com/Yeonhwa726',
+            'channel_id': '10237902',
+            'channel_url': 'https://www.patreon.com/Yeonhwa726',
+            'duration': 70,
+            'timestamp': 1705150153,
+            'upload_date': '20240113',
+            'comment_count': int,
+            'like_count': int,
+            'thumbnail': r're:^https?://.+',
+        },
+        'params': {'skip_download': 'm3u8'},
    }]

    def _real_extract(self, url):
@ -268,16 +288,19 @@ class PatreonIE(PatreonBaseIE):
                })

        # handle Vimeo embeds
-        if try_get(attributes, lambda x: x['embed']['provider']) == 'Vimeo':
-            embed_html = try_get(attributes, lambda x: x['embed']['html'])
-            v_url = url_or_none(compat_urllib_parse_unquote(
-                self._search_regex(r'(https(?:%3A%2F%2F|://)player\.vimeo\.com.+app_id(?:=|%3D)+\d+)', embed_html, 'vimeo url', fatal=False)))
-            if v_url:
-                v_url = VimeoIE._smuggle_referrer(v_url, 'https://patreon.com')
-                if self._request_webpage(v_url, video_id, 'Checking Vimeo embed URL', fatal=False, errnote=False):
-                    return self.url_result(v_url, VimeoIE, url_transparent=True, **info)
+        if traverse_obj(attributes, ('embed', 'provider')) == 'Vimeo':
+            v_url = urllib.parse.unquote(self._html_search_regex(
+                r'(https(?:%3A%2F%2F|://)player\.vimeo\.com.+app_id(?:=|%3D)+\d+)',
+                traverse_obj(attributes, ('embed', 'html', {str})), 'vimeo url', fatal=False) or '')
+            if url_or_none(v_url) and self._request_webpage(
+                    v_url, video_id, 'Checking Vimeo embed URL',
+                    headers={'Referer': 'https://patreon.com/'},
+                    fatal=False, errnote=False):
+                return self.url_result(
+                    VimeoIE._smuggle_referrer(v_url, 'https://patreon.com/'),
+                    VimeoIE, url_transparent=True, **info)

-        embed_url = try_get(attributes, lambda x: x['embed']['url'])
+        embed_url = traverse_obj(attributes, ('embed', 'url', {url_or_none}))
        if embed_url and self._request_webpage(embed_url, video_id, 'Checking embed URL', fatal=False, errnote=False):
            return self.url_result(embed_url, **info)

--- a/yt_dlp/extractor/senategov.py
+++ b/yt_dlp/extractor/senategov.py
@ -12,37 +12,37 @@ from ..utils import (
 )

 _COMMITTEES = {
-    'ag': ('76440', 'http://ag-f.akamaihd.net'),
-    'aging': ('76442', 'http://aging-f.akamaihd.net'),
-    'approps': ('76441', 'http://approps-f.akamaihd.net'),
-    'arch': ('', 'http://ussenate-f.akamaihd.net'),
-    'armed': ('76445', 'http://armed-f.akamaihd.net'),
-    'banking': ('76446', 'http://banking-f.akamaihd.net'),
-    'budget': ('76447', 'http://budget-f.akamaihd.net'),
-    'cecc': ('76486', 'http://srs-f.akamaihd.net'),
-    'commerce': ('80177', 'http://commerce1-f.akamaihd.net'),
-    'csce': ('75229', 'http://srs-f.akamaihd.net'),
-    'dpc': ('76590', 'http://dpc-f.akamaihd.net'),
-    'energy': ('76448', 'http://energy-f.akamaihd.net'),
-    'epw': ('76478', 'http://epw-f.akamaihd.net'),
-    'ethics': ('76449', 'http://ethics-f.akamaihd.net'),
-    'finance': ('76450', 'http://finance-f.akamaihd.net'),
-    'foreign': ('76451', 'http://foreign-f.akamaihd.net'),
-    'govtaff': ('76453', 'http://govtaff-f.akamaihd.net'),
-    'help': ('76452', 'http://help-f.akamaihd.net'),
-    'indian': ('76455', 'http://indian-f.akamaihd.net'),
-    'intel': ('76456', 'http://intel-f.akamaihd.net'),
-    'intlnarc': ('76457', 'http://intlnarc-f.akamaihd.net'),
-    'jccic': ('85180', 'http://jccic-f.akamaihd.net'),
-    'jec': ('76458', 'http://jec-f.akamaihd.net'),
-    'judiciary': ('76459', 'http://judiciary-f.akamaihd.net'),
-    'rpc': ('76591', 'http://rpc-f.akamaihd.net'),
-    'rules': ('76460', 'http://rules-f.akamaihd.net'),
-    'saa': ('76489', 'http://srs-f.akamaihd.net'),
-    'smbiz': ('76461', 'http://smbiz-f.akamaihd.net'),
-    'srs': ('75229', 'http://srs-f.akamaihd.net'),
-    'uscc': ('76487', 'http://srs-f.akamaihd.net'),
-    'vetaff': ('76462', 'http://vetaff-f.akamaihd.net'),
+    'ag': ('76440', 'https://ag-f.akamaihd.net', '2036803', 'agriculture'),
+    'aging': ('76442', 'https://aging-f.akamaihd.net', '2036801', 'aging'),
+    'approps': ('76441', 'https://approps-f.akamaihd.net', '2036802', 'appropriations'),
+    'arch': ('', 'https://ussenate-f.akamaihd.net/', '', 'arch'),
+    'armed': ('76445', 'https://armed-f.akamaihd.net', '2036800', 'armedservices'),
+    'banking': ('76446', 'https://banking-f.akamaihd.net', '2036799', 'banking'),
+    'budget': ('76447', 'https://budget-f.akamaihd.net', '2036798', 'budget'),
+    'cecc': ('76486', 'https://srs-f.akamaihd.net', '2036782', 'srs_cecc'),
+    'commerce': ('80177', 'https://commerce1-f.akamaihd.net', '2036779', 'commerce'),
+    'csce': ('75229', 'https://srs-f.akamaihd.net', '2036777', 'srs_srs'),
+    'dpc': ('76590', 'https://dpc-f.akamaihd.net', '', 'dpc'),
+    'energy': ('76448', 'https://energy-f.akamaihd.net', '2036797', 'energy'),
+    'epw': ('76478', 'https://epw-f.akamaihd.net', '2036783', 'environment'),
+    'ethics': ('76449', 'https://ethics-f.akamaihd.net', '2036796', 'ethics'),
+    'finance': ('76450', 'https://finance-f.akamaihd.net', '2036795', 'finance_finance'),
+    'foreign': ('76451', 'https://foreign-f.akamaihd.net', '2036794', 'foreignrelations'),
+    'govtaff': ('76453', 'https://govtaff-f.akamaihd.net', '2036792', 'hsgac'),
+    'help': ('76452', 'https://help-f.akamaihd.net', '2036793', 'help'),
+    'indian': ('76455', 'https://indian-f.akamaihd.net', '2036791', 'indianaffairs'),
+    'intel': ('76456', 'https://intel-f.akamaihd.net', '2036790', 'intelligence'),
+    'intlnarc': ('76457', 'https://intlnarc-f.akamaihd.net', '', 'internationalnarcoticscaucus'),
+    'jccic': ('85180', 'https://jccic-f.akamaihd.net', '2036778', 'jccic'),
+    'jec': ('76458', 'https://jec-f.akamaihd.net', '2036789', 'jointeconomic'),
+    'judiciary': ('76459', 'https://judiciary-f.akamaihd.net', '2036788', 'judiciary'),
+    'rpc': ('76591', 'https://rpc-f.akamaihd.net', '', 'rpc'),
+    'rules': ('76460', 'https://rules-f.akamaihd.net', '2036787', 'rules'),
+    'saa': ('76489', 'https://srs-f.akamaihd.net', '2036780', 'srs_saa'),
+    'smbiz': ('76461', 'https://smbiz-f.akamaihd.net', '2036786', 'smallbusiness'),
+    'srs': ('75229', 'https://srs-f.akamaihd.net', '2031966', 'srs_srs'),
+    'uscc': ('76487', 'https://srs-f.akamaihd.net', '2036781', 'srs_uscc'),
+    'vetaff': ('76462', 'https://vetaff-f.akamaihd.net', '2036785', 'veteransaffairs'),
 }


@ -176,15 +176,23 @@ class SenateGovIE(InfoExtractor):
    def _real_extract(self, url):
        display_id = self._generic_id(url)
        webpage = self._download_webpage(url, display_id)
-        parse_info = parse_qs(self._search_regex(
-            r'<iframe class="[^>"]*streaminghearing[^>"]*"\s[^>]*\bsrc="([^">]*)', webpage, 'hearing URL'))
-
-        stream_num, stream_domain = _COMMITTEES[parse_info['comm'][-1]]
+        iframe_src = self._search_regex(
+            (r'<iframe class="[^>"]*streaminghearing[^>"]*"\s[^>]*\bsrc="([^">]*)',
+             r'<iframe title="[^>"]*[^>"]*"\s[^>]*\bsrc="([^">]*)'),
+            webpage, 'hearing URL').replace('&amp;', '&')
+        parse_info = parse_qs(iframe_src)
+        comm = parse_info['comm'][-1]
+        stream_num, stream_domain, stream_id, msl3 = _COMMITTEES[comm]
        filename = parse_info['filename'][-1]

-        formats = self._extract_m3u8_formats(
-            f'{stream_domain}/i/{filename}_1@{stream_num}/master.m3u8',
-            display_id, ext='mp4')
+        urls_alternatives = [f'https://www-senate-gov-media-srs.akamaized.net/hls/live/{stream_id}/{comm}/{filename}/master.m3u8',
+                             f'https://www-senate-gov-msl3archive.akamaized.net/{msl3}/{filename}_1/master.m3u8',
+                             f'{stream_domain}/i/{filename}_1@{stream_num}/master.m3u8',
+                             f'{stream_domain}/i/{filename}.mp4/master.m3u8']
+        for video_url in urls_alternatives:
+            formats = self._extract_m3u8_formats(video_url, display_id, ext='mp4', fatal=False)
+            if formats:
+                break

        title = self._html_search_regex(
            (*self._og_regexes('title'), r'(?s)<title>([^<]*?)</title>'), webpage, 'video title')
Autor	SHA1	Wiadomość	Data
Grabien	7fa89c7adc	Merge `fe354fa548` into `e3b42d8b1b`	2024-04-21 04:46:31 +05:30
bashonly	e3b42d8b1b	[ie/facebook] Fix DASH formats extraction (#9734 ) Closes #9720 Authored by: bashonly	2024-04-20 10:23:12 +00:00
bashonly	c9ce57d9bf	[ie/patreon] Fix Vimeo embed extraction (#9712 ) Fixes regression in `36b240f9a7` Closes #9709 Authored by: bashonly	2024-04-18 23:18:56 +00:00
bashonly	02483bea1c	[build] Normalize `curl_cffi` group to `curl-cffi` (#9698 ) Closes #9682 Authored by: bashonly	2024-04-18 23:11:12 +00:00
Grabien	fe354fa548	Merge branch 'master' of https://github.com/Grabien/yt-dlp	2024-03-20 18:51:52 +02:00
Grabien	35b614b598	Merge branch 'yt-dlp:master' into master	2024-03-20 18:51:42 +02:00
Grabien	974babf965	[ie/senategov] URL parsing fix	2024-03-20 18:50:50 +02:00
pukkandan	c6f0d05213	Update yt_dlp/extractor/senategov.py	2024-03-05 00:59:33 +05:30
Grabien	2f54918f30	[senategod] Committees data update	2024-03-04 19:11:26 +02:00
Grabien	424dd9e061	[senategov] Support for new video paths	2024-03-04 19:01:18 +02:00