From 64102296818f94d3814a8183daa5d92cbdd952fd Mon Sep 17 00:00:00 2001 From: newtonelectron Date: Sun, 5 Apr 2015 12:50:21 -0700 Subject: [PATCH] [SpankBang] Add new extractor --- youtube_dl/extractor/__init__.py | 1 + youtube_dl/extractor/spankbang.py | 38 +++++++++++++++++++++++++++++++ 2 files changed, 39 insertions(+) create mode 100644 youtube_dl/extractor/spankbang.py diff --git a/youtube_dl/extractor/__init__.py b/youtube_dl/extractor/__init__.py index 0f7d44616..e6fdf1297 100644 --- a/youtube_dl/extractor/__init__.py +++ b/youtube_dl/extractor/__init__.py @@ -471,6 +471,7 @@ from .southpark import ( SouthparkDeIE, ) from .space import SpaceIE +from .spankbang import SpankBangIE from .spankwire import SpankwireIE from .spiegel import SpiegelIE, SpiegelArticleIE from .spiegeltv import SpiegeltvIE diff --git a/youtube_dl/extractor/spankbang.py b/youtube_dl/extractor/spankbang.py new file mode 100644 index 000000000..8e845ef26 --- /dev/null +++ b/youtube_dl/extractor/spankbang.py @@ -0,0 +1,38 @@ +# coding: utf-8 +from __future__ import unicode_literals + +from .common import InfoExtractor +import re + +class SpankBangIE(InfoExtractor): + """Extractor for http://spankbang.com""" + + _VALID_URL = r"https?://(?:www\.)?spankbang\.com/(?P\w+)/video/.*" + + def _real_extract(self, url): + video_id = self._match_id(url) + webpage = self._download_webpage(url, video_id) + + title = self._html_search_regex(r"

(?:)?(.*?)

", webpage, "title") + + stream_key = self._html_search_regex(r"""var\s+stream_key\s*[=]\s*['"](.+?)['"]\s*;""", webpage, "stream_key") + + qualities = re.findall(r"([0-9]+p).*?", webpage) + + formats = [] + for q in sorted(qualities): + formats.append({ + "format_id": q, + "format": q, + "ext": "mp4", + "url": "http://spankbang.com/_{}/{}/title/{}__mp4".format(video_id, stream_key, q) + }) + + return { + "id": video_id, + "title": title, + "description": self._og_search_description(webpage), + "formats": formats + } + +# vim: tabstop=4 expandtab