From 8230018c20595a22e636b834ebb522a6a85d0d8b Mon Sep 17 00:00:00 2001 From: Gabriel Schubiner Date: Sun, 19 Oct 2014 22:47:05 -0700 Subject: [PATCH 1/2] Added extractor for crunchyroll 'playlists' i.e. series. so that one can, e.g. download all episodes of a series --- youtube_dl/extractor/__init__.py | 5 ++++- youtube_dl/extractor/crunchyroll.py | 35 +++++++++++++++++++++++++++++ 2 files changed, 39 insertions(+), 1 deletion(-) diff --git a/youtube_dl/extractor/__init__.py b/youtube_dl/extractor/__init__.py index 070f9ff19..0dd763006 100644 --- a/youtube_dl/extractor/__init__.py +++ b/youtube_dl/extractor/__init__.py @@ -60,7 +60,10 @@ from .condenast import CondeNastIE from .cracked import CrackedIE from .criterion import CriterionIE -from .crunchyroll import CrunchyrollIE +from .crunchyroll import ( + CrunchyrollIE, + CrunchyrollShowPlaylistIE +) from .cspan import CSpanIE from .d8 import D8IE from .dailymotion import ( diff --git a/youtube_dl/extractor/crunchyroll.py b/youtube_dl/extractor/crunchyroll.py index f99888ecc..414c46b0d 100644 --- a/youtube_dl/extractor/crunchyroll.py +++ b/youtube_dl/extractor/crunchyroll.py @@ -24,6 +24,7 @@ aes_cbc_decrypt, inc, ) +from .common import InfoExtractor class CrunchyrollIE(SubtitlesInfoExtractor): @@ -285,3 +286,37 @@ def _real_extract(self,url): 'subtitles': subtitles, 'formats': formats, } + + +class CrunchyrollShowPlaylistIE(InfoExtractor): + IE_NAME = "crunchyroll:playlist" + _VALID_URL = r'https?://(?:(?Pwww|m)\.)?(?Pcrunchyroll\.com/(?!(?:news|anime-news|library|forum|launchcalendar|lineup|store|comics|freetrial|login))(?P[\w\-]+))/?$' + _TITLE_EXTR = r'\s*(?P[\w\s]+)' + + _TESTS = [{ + 'url' : 'http://www.crunchyroll.com/attack-on-titan', + 'info_dict' : { + 'title' : 'Attack on Titan' + }, + 'playlist_count' : 15 + }] + + def _extract_title_entries(self,id,webpage): + _EPISODE_ID_EXTR = r'id="showview_videos_media_(?P\d+)".*?href="/{0}/(?P[\w\-]+-(?P=vidid))"'.format(id) + title = self._html_search_regex(self._TITLE_EXTR,webpage,"title",flags=re.UNICODE|re.MULTILINE) + episode_urls = [self.url_result('http://www.crunchyroll.com/{0}/{1}'.format(id, showmatch[1])) for + showmatch in re.findall(_EPISODE_ID_EXTR, webpage,re.UNICODE|re.MULTILINE|re.DOTALL)] + return title, episode_urls + + + def _real_extract(self, url): + url_match = re.match(self._VALID_URL,url) + show_id = url_match.group('show') + webpage = self._download_webpage(url,show_id) + (title,entries) = self._extract_title_entries(show_id,webpage) + return { + '_type' : 'playlist', + 'id' : show_id, + 'title' : title, + 'entries' : entries + } \ No newline at end of file From 1b10a011ec7544f49159ed60642128720333b8aa Mon Sep 17 00:00:00 2001 From: Gabriel Schubiner Date: Mon, 20 Oct 2014 18:38:42 -0700 Subject: [PATCH 2/2] Forgot to reverse extracted video urls so they are in correct order for video selection args --- youtube_dl/extractor/crunchyroll.py | 1 + 1 file changed, 1 insertion(+) diff --git a/youtube_dl/extractor/crunchyroll.py b/youtube_dl/extractor/crunchyroll.py index 414c46b0d..9ac86c2be 100644 --- a/youtube_dl/extractor/crunchyroll.py +++ b/youtube_dl/extractor/crunchyroll.py @@ -306,6 +306,7 @@ def _extract_title_entries(self,id,webpage): title = self._html_search_regex(self._TITLE_EXTR,webpage,"title",flags=re.UNICODE|re.MULTILINE) episode_urls = [self.url_result('http://www.crunchyroll.com/{0}/{1}'.format(id, showmatch[1])) for showmatch in re.findall(_EPISODE_ID_EXTR, webpage,re.UNICODE|re.MULTILINE|re.DOTALL)] + episode_urls.reverse() return title, episode_urls