mirror of
https://github.com/yt-dlp/yt-dlp.git
synced 2024-11-21 20:46:36 -05:00
[teachable] Extract chapter metadata (closes #24421)
This commit is contained in:
parent
63dce3094b
commit
4560adc820
1 changed files with 25 additions and 0 deletions
|
@ -7,7 +7,9 @@
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
int_or_none,
|
||||||
get_element_by_class,
|
get_element_by_class,
|
||||||
|
strip_or_none,
|
||||||
urlencode_postdata,
|
urlencode_postdata,
|
||||||
urljoin,
|
urljoin,
|
||||||
)
|
)
|
||||||
|
@ -173,11 +175,34 @@ def _real_extract(self, url):
|
||||||
|
|
||||||
title = self._og_search_title(webpage, default=None)
|
title = self._og_search_title(webpage, default=None)
|
||||||
|
|
||||||
|
chapter = None
|
||||||
|
chapter_number = None
|
||||||
|
section_item = self._search_regex(
|
||||||
|
r'(?s)(?P<li><li[^>]+\bdata-lecture-id=["\']%s[^>]+>.+?</li>)' % video_id,
|
||||||
|
webpage, 'section item', default=None, group='li')
|
||||||
|
if section_item:
|
||||||
|
chapter_number = int_or_none(self._search_regex(
|
||||||
|
r'data-ss-position=["\'](\d+)', section_item, 'section id',
|
||||||
|
default=None))
|
||||||
|
if chapter_number is not None:
|
||||||
|
sections = []
|
||||||
|
for s in re.findall(
|
||||||
|
r'(?s)<div[^>]+\bclass=["\']section-title[^>]+>(.+?)</div>', webpage):
|
||||||
|
section = strip_or_none(clean_html(s))
|
||||||
|
if not section:
|
||||||
|
sections = []
|
||||||
|
break
|
||||||
|
sections.append(section)
|
||||||
|
if chapter_number <= len(sections):
|
||||||
|
chapter = sections[chapter_number - 1]
|
||||||
|
|
||||||
entries = [{
|
entries = [{
|
||||||
'_type': 'url_transparent',
|
'_type': 'url_transparent',
|
||||||
'url': wistia_url,
|
'url': wistia_url,
|
||||||
'ie_key': WistiaIE.ie_key(),
|
'ie_key': WistiaIE.ie_key(),
|
||||||
'title': title,
|
'title': title,
|
||||||
|
'chapter': chapter,
|
||||||
|
'chapter_number': chapter_number,
|
||||||
} for wistia_url in wistia_urls]
|
} for wistia_url in wistia_urls]
|
||||||
|
|
||||||
return self.playlist_result(entries, video_id, title)
|
return self.playlist_result(entries, video_id, title)
|
||||||
|
|
Loading…
Reference in a new issue