Skip to content
New issue

Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.

By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.

Already on GitHub? Sign in to your account

[ie/orf:on] Improve extraction #9677

Merged
merged 13 commits into from
May 23, 2024
50 changes: 41 additions & 9 deletions yt_dlp/extractor/orf.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
url_or_none,
)
from ..utils.traversal import traverse_obj
from ..utils import parse_age_limit
TuxCoder marked this conversation as resolved.
Show resolved Hide resolved


class ORFTVthekIE(InfoExtractor):
Expand Down Expand Up @@ -569,7 +570,7 @@ def _real_extract(self, url):

class ORFONIE(InfoExtractor):
IE_NAME = 'orf:on'
_VALID_URL = r'https?://on\.orf\.at/video/(?P<id>\d{8})/(?P<slug>[\w-]+)'
_VALID_URL = r'https?://on\.orf\.at/video/(?P<id>\d+)'
_TESTS = [{
'url': 'https://on.orf.at/video/14210000/school-of-champions-48',
'info_dict': {
Expand All @@ -583,32 +584,63 @@ class ORFONIE(InfoExtractor):
'timestamp': 1706472362,
'upload_date': '20240128',
}
}, {
'url': 'https://on.orf.at/video/3220355',
seproDev marked this conversation as resolved.
Show resolved Hide resolved
'md5': 'f94d98e667cf9a3851317efb4e136662',
'info_dict': {
'id': '3220355',
'ext': 'mp4',
'duration': 445.04,
'thumbnail': 'https://api-tvthek.orf.at/assets/segments/0002/60/thumb_159573_segments_highlight_teaser.png',
'title': '50 Jahre Burgenland: Der Festumzug',
'description': 'md5:1560bf855119544ee8c4fa5376a2a6b0',
'media_type': 'episode',
'timestamp': 52916400,
'upload_date': '19710905',
}
}]

def _extract_video(self, video_id, display_id):
def _extract_video(self, video_id):
encrypted_id = base64.b64encode(f'3dSlfek03nsLKdj4Jsd{video_id}'.encode()).decode()
api_json = self._download_json(
f'https://api-tvthek.orf.at/api/v4.3/public/episode/encrypted/{encrypted_id}', display_id)
f'https://api-tvthek.orf.at/api/v4.3/public/episode/encrypted/{encrypted_id}', video_id)

if api_json.get('is_drm_protected'):
seproDev marked this conversation as resolved.
Show resolved Hide resolved
self.report_drm(video_id)

formats, subtitles = [], {}
for manifest_type in traverse_obj(api_json, ('sources', {dict.keys}, ...)):
for manifest_url in traverse_obj(api_json, ('sources', manifest_type, ..., 'src', {url_or_none})):
if manifest_type == 'hls':
fmts, subs = self._extract_m3u8_formats_and_subtitles(
manifest_url, display_id, fatal=False, m3u8_id='hls')
manifest_url, video_id, fatal=False, m3u8_id='hls')
elif manifest_type == 'dash':
fmts, subs = self._extract_mpd_formats_and_subtitles(
manifest_url, display_id, fatal=False, mpd_id='dash')
manifest_url, video_id, fatal=False, mpd_id='dash')
else:
continue
formats.extend(fmts)
self._merge_subtitles(subs, target=subtitles)

for subtitle_type in ['vtt']: # not working formats 'xml', 'srt', 'sami', 'ttml', 'stl'
seproDev marked this conversation as resolved.
Show resolved Hide resolved
subtitle_url = traverse_obj(api_json, ('_embedded', 'subtitle', f'{subtitle_type}_url'), {str})
if subtitle_url is None:
continue
self._merge_subtitles({
'de': [
{
'url': subtitle_url,
'ext': f'{subtitle_type}',
}
],
}, target=subtitles)

return {
'id': video_id,
'formats': formats,
'subtitles': subtitles,
**traverse_obj(api_json, {
'age_limit': ('age_classification', {parse_age_limit}),
'duration': ('duration_second', {float_or_none}),
'title': (('title', 'headline'), {str}),
'description': (('description', 'teaser_text'), {str}),
Expand All @@ -617,14 +649,14 @@ def _extract_video(self, video_id, display_id):
}

def _real_extract(self, url):
video_id, display_id = self._match_valid_url(url).group('id', 'slug')
webpage = self._download_webpage(url, display_id)
video_id = self._match_valid_url(url).group('id')
TuxCoder marked this conversation as resolved.
Show resolved Hide resolved
webpage = self._download_webpage(url, video_id)

return {
'id': video_id,
'title': self._html_search_meta(['og:title', 'twitter:title'], webpage, default=None),
'description': self._html_search_meta(
['description', 'og:description', 'twitter:description'], webpage, default=None),
**self._search_json_ld(webpage, display_id, fatal=False),
**self._extract_video(video_id, display_id),
**self._search_json_ld(webpage, video_id, fatal=False),
**self._extract_video(video_id),
}
Loading