about summary refs log tree commit diff
path: root/youtube_dl/extractor/infoq.py
diff options
context:
space:
mode:
authorPhilipp Hagemeister <phihag@phihag.de>2013-06-23 21:14:19 +0200
committerPhilipp Hagemeister <phihag@phihag.de>2013-06-23 21:14:19 +0200
commitfda7d31aa0d002b38418ed5c9f32ae211a6585ce (patch)
tree0c11595f7e92c431eb5ab8ebc68d55d9af2241f3 /youtube_dl/extractor/infoq.py
parentcbf46c737c3f4156dee019b70521dcd3194877ac (diff)
downloadyoutube-dl-fda7d31aa0d002b38418ed5c9f32ae211a6585ce.tar.gz
youtube-dl-fda7d31aa0d002b38418ed5c9f32ae211a6585ce.tar.xz
youtube-dl-fda7d31aa0d002b38418ed5c9f32ae211a6585ce.zip
Move infoq into its own file
Diffstat (limited to 'youtube_dl/extractor/infoq.py')
-rw-r--r--youtube_dl/extractor/infoq.py50
1 files changed, 50 insertions, 0 deletions
diff --git a/youtube_dl/extractor/infoq.py b/youtube_dl/extractor/infoq.py
new file mode 100644
index 000000000..905674282
--- /dev/null
+++ b/youtube_dl/extractor/infoq.py
@@ -0,0 +1,50 @@
+import base64
+import re
+
+from .common import InfoExtractor
+from ..utils import (
+    compat_urllib_parse,
+
+    ExtractorError,
+)
+
+
+class InfoQIE(InfoExtractor):
+    _VALID_URL = r'^(?:https?://)?(?:www\.)?infoq\.com/[^/]+/[^/]+$'
+
+    def _real_extract(self, url):
+        mobj = re.match(self._VALID_URL, url)
+
+        webpage = self._download_webpage(url, video_id=url)
+        self.report_extraction(url)
+
+        # Extract video URL
+        mobj = re.search(r"jsclassref ?= ?'([^']*)'", webpage)
+        if mobj is None:
+            raise ExtractorError(u'Unable to extract video url')
+        real_id = compat_urllib_parse.unquote(base64.b64decode(mobj.group(1).encode('ascii')).decode('utf-8'))
+        video_url = 'rtmpe://video.infoq.com/cfx/st/' + real_id
+
+        # Extract title
+        video_title = self._search_regex(r'contentTitle = "(.*?)";',
+            webpage, u'title')
+
+        # Extract description
+        video_description = self._html_search_regex(r'<meta name="description" content="(.*)"(?:\s*/)?>',
+            webpage, u'description', fatal=False)
+
+        video_filename = video_url.split('/')[-1]
+        video_id, extension = video_filename.split('.')
+
+        info = {
+            'id': video_id,
+            'url': video_url,
+            'uploader': None,
+            'upload_date': None,
+            'title': video_title,
+            'ext': extension, # Extension is always(?) mp4, but seems to be flv
+            'thumbnail': None,
+            'description': video_description,
+        }
+
+        return [info]
\ No newline at end of file