extract_item_info: Don't extract author, author_id, etc. for channel items

Philosophically, a channel doesn't create itself.
2019-12-24 13:11:21 -08:00
parent 3200d66d88
commit f706689a56
1 changed files with 8 additions and 7 deletions
--- a/youtube/yt_data_extract/common.py
+++ b/youtube/yt_data_extract/common.py
@@ -218,13 +218,14 @@ def extract_item_info(item, additional_info={}):
        info['type'] = 'unsupported'

    info['title'] = extract_str(item.get('title'))
-    info['author'] = extract_str(multi_get(item, 'longBylineText', 'shortBylineText', 'ownerText'))
-    info['author_id'] = extract_str(multi_deep_get(item,
-        ['longBylineText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId'],
-        ['shortBylineText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId'],
-        ['ownerText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId']
-    ))
-    info['author_url'] = ('https://www.youtube.com/channel/' + info['author_id']) if info['author_id'] else None
+    if primary_type != 'channel':
+        info['author'] = extract_str(multi_get(item, 'longBylineText', 'shortBylineText', 'ownerText'))
+        info['author_id'] = extract_str(multi_deep_get(item,
+            ['longBylineText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId'],
+            ['shortBylineText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId'],
+            ['ownerText', 'runs', 0, 'navigationEndpoint', 'browseEndpoint', 'browseId']
+        ))
+        info['author_url'] = ('https://www.youtube.com/channel/' + info['author_id']) if info['author_id'] else None
    info['description'] = extract_formatted_text(multi_get(item, 'descriptionSnippet', 'descriptionText'))
    info['thumbnail'] = multi_deep_get(item,
        ['thumbnail', 'thumbnails', 0, 'url'],      # videos