Created
August 26, 2017 02:10
-
-
Save pclose/d3e5ab47db996524f4c2f2863c657902 to your computer and use it in GitHub Desktop.
Youtube conversion tool -pete 2017-08-01
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/python | |
| ''' | |
| usage: yt-convert.py [-h] [--input INPUT] [--output OUTPUT] [--format FORMAT] | |
| [--debug] [--url URL] | |
| {channel,convert,query} | |
| Youtube conversion tool -pete 2017-08-01 | |
| positional arguments: | |
| {channel,convert,query} | |
| optional arguments: | |
| -h, --help show this help message and exit | |
| --input INPUT, -i INPUT | |
| Input file (defaults to stdin) | |
| --output OUTPUT, -o OUTPUT | |
| Output file (defaults to stdout) | |
| --format FORMAT, -f FORMAT | |
| Format to convert to | |
| --debug, -d Print DEBUG info | |
| --url URL, -u URL URL of Youtube link eg: | |
| https://www.youtube.com/watch?v=StTqXEQ2l-Y | |
| Example usages: | |
| 1) Convert all videos uploaded to a given channel | |
| eg: `python yt-convert.py channel -u https://www.youtube.com/user/enj01n| \ | |
| python yt-convert.py query | \ | |
| python yt-convert.py convert` | |
| 2) Convert media from individual video url | |
| eg: `python yt-convert.py query -u https://www.youtube.com/watch?v=StTqXEQ2l-Y | \ | |
| python yt-convert.py convert` | |
| 3) Query media info from individual video url | |
| eg: `python yt-convert.py -u "https://www.youtube.com/watch?v=StTqXEQ2l-Y" query` | |
| 4) Query media info from list of video IDs | |
| eg: `echo "StTqXEQ2l-Y" | python yt-convert.py query` | |
| 5) Save CSV of media info from channel | |
| eg: `python yt-convert.py channel -u https://www.youtube.com/user/enj01n | \ | |
| python yt-convert.py query > out.txt` | |
| ''' | |
| from requests import Request, Session | |
| import sys | |
| import json | |
| import re | |
| import traceback | |
| #import brotli | |
| import urllib | |
| import HTMLParser | |
| import subprocess | |
| import csv | |
| UA = "Mozilla/5.0 (X11; Linux x86_64; rv:45.0) Gecko/20100101 Firefox/45.0" | |
| YT_URL = "https://www.youtube.com" | |
| VIDEO_LIST_URL = "https://www.youtube.com/channel/{}/videos" | |
| WATCH_URL = "https://www.youtube.com/watch?v={}&spf=navigate" | |
| EPISODE_MATCH='href="/watch?v=' | |
| def pr_full(req,res): | |
| print('===========Request {}================'.format(req.url)) | |
| print( '-----------Request HEADERS-----------') | |
| print( req.method + ' ' + req.url) | |
| print('\n'.join('{}: {}'.format(k, v) for k, v in req.headers.items())) | |
| print( '-----------Request BODY-----------') | |
| print(req.body) | |
| print("") | |
| print("") | |
| print('-----------Response HEADERS-----------') | |
| print(res.status_code) | |
| print('\n'.join('{}: {}'.format(k, v) for k, v in res.headers.items())) | |
| print('-----------Response BODY-----------') | |
| print(res.content.encode('utf-8')) | |
| def find_media(debug, watch_url): | |
| session = Session() | |
| session.headers = { | |
| "User-Agent": UA, | |
| "Accept-Encoding": "gzip" | |
| #"Accept-Encoding": "gzip, br", # Include brotli for higher definition | |
| } | |
| req = session.prepare_request(Request("GET",YT_URL)) | |
| res = session.send(req) | |
| req = session.prepare_request(Request("GET",WATCH_URL.format(watch_url))) | |
| res = session.send(req) | |
| #content = brotli.decompress(res.content).decode('utf-8') | |
| content = res.content.decode('utf-8') | |
| #if debug: pr_full(req, res) | |
| try: | |
| data = json.loads(content) | |
| dat = data[2]["data"]["swfcfg"]["args"] | |
| if "adaptive_fmts" in dat: | |
| formats_strings = dat["adaptive_fmts"].split(",") | |
| elif "url_encoded_fmt_stream_map" in dat: | |
| formats_strings = dat["url_encoded_fmt_stream_map"].split(",") | |
| video_title = dat["title"] | |
| formats = {} | |
| if debug: print video_title.encode('utf-8') | |
| for e in formats_strings: | |
| obj = {} | |
| for x in e.split("&"): | |
| d = x.split("=") | |
| obj[d[0]] = urllib.unquote(d[1]) | |
| formats[obj["itag"]] = obj | |
| if debug: print ("-"*20+"\n"+"\n".join(["{}\t{}".format(x[0],x[1]) for x in obj.items()])) | |
| audio_only = filter(lambda x: "audio" in formats[x]["type"], formats) | |
| if len(audio_only) > 0: | |
| audio = {} | |
| for x in audio_only: | |
| audio[x] = formats[x] | |
| formats = audio | |
| fmt = formats[max(formats.keys())] # Highest quality | |
| #fmt = formats[min(formats.keys())] # Lowest quality | |
| return [watch_url, video_title, fmt["url"]] | |
| except Exception as e: | |
| if debug: pr_full(req, res) | |
| print >>sys.stderr, "FAIL!: unable to parse media" | |
| traceback.print_exc(file=sys.stderr) | |
| return [watch_url, "n/a", ""] | |
| # Returns a list of video IDs from a channel or user landing page | |
| def find_episodes(debug, channel_url): | |
| result=[] | |
| session = Session() | |
| req = session.prepare_request(Request("GET",channel_url)) | |
| res = session.send(req) | |
| content = res.content.decode('utf-8') | |
| re_search = re.search("yt.setConfig\(\'CHANNEL_ID\', \"(.*?)\"", content) | |
| if re_search: channel_id = re_search.groups()[0] | |
| else: return | |
| req = session.prepare_request(Request("GET",VIDEO_LIST_URL.format(channel_id))) | |
| res = session.send(req) | |
| content = res.content.decode('utf-8') | |
| for e in content.split("\n"): | |
| i = e.find(EPISODE_MATCH) | |
| if i >= 0: | |
| val = e[i+len(EPISODE_MATCH):e.index('"', i+len(EPISODE_MATCH))] | |
| if val not in result: result.append(val) | |
| unescape = HTMLParser.HTMLParser().unescape | |
| re_search = re.search("data-uix-load-more-href=\"(.*?)\"", content) | |
| if re_search: load_url = unescape(re_search.groups()[0]) | |
| else: load_url = None | |
| while load_url: | |
| req = session.prepare_request(Request("GET",YT_URL+load_url)) | |
| res = session.send(req) | |
| content = res.content.decode('utf-8') | |
| data = json.loads(content) | |
| load_url = None | |
| for e in data["load_more_widget_html"]: | |
| re_search = re.search("data-uix-load-more-href=\"(.*?)\"", data["load_more_widget_html"]) | |
| if re_search: load_url = unescape(re_search.groups()[0]) | |
| for e in data["content_html"].split("\n"): | |
| i = e.find(EPISODE_MATCH) | |
| if i >= 0: | |
| val = e[i+len(EPISODE_MATCH):e.index('"', i+len(EPISODE_MATCH))] | |
| if val not in result: result.append(val) | |
| return result | |
| def convert_media(debug, url, title): | |
| print "Processing "+title," ..", | |
| import tempfile | |
| errfh = tempfile.TemporaryFile() | |
| cmd1 = ["wget", url, "-O", "-"] | |
| #cmd2 = ["ffmpeg", "-y", "-i", "-", "-vn", "-ab", "24k", "-ar", "22050", title] # If you want to downsample the audio a bit | |
| cmd2 = ["ffmpeg", "-y", "-i", "-", "-vn", title] # Just copy audio from media | |
| if debug: print " ".join(cmd1) | |
| p1 = subprocess.Popen(cmd1, stderr=errfh, stdout=subprocess.PIPE) | |
| if debug: print " ".join(cmd2) | |
| p2 = subprocess.Popen(cmd2, stdin=p1.stdout, stderr=errfh, stdout=errfh) | |
| p1.stdout.close() | |
| stdout = p2.communicate() | |
| errfh.seek(0) | |
| if debug: print "".join(errfh.readlines()) | |
| if p2.returncode != 0: | |
| print "ERROR!" | |
| else: | |
| print "Done." | |
| if __name__ == '__main__': | |
| usage_examples=''' | |
| Example usages: | |
| 1) Convert all videos uploaded to a given channel | |
| `python yt-convert.py channel -u https://www.youtube.com/user/enj01n| \\ | |
| python yt-convert.py query | \\ | |
| python yt-convert.py convert` | |
| 2) Convert media from individual video url | |
| `python yt-convert.py query -u https://www.youtube.com/watch?v=StTqXEQ2l-Y | \\ | |
| python yt-convert.py convert` | |
| 3) Query media info from individual video url | |
| `python yt-convert.py query -u "https://www.youtube.com/watch?v=StTqXEQ2l-Y"` | |
| 4) Query media info from list of video IDs | |
| `echo "StTqXEQ2l-Y" | python yt-convert.py query` | |
| 5) Save CSV of media info from channel | |
| `python yt-convert.py channel -u https://www.youtube.com/user/enj01n | \\ | |
| python yt-convert.py query > out.csv` | |
| ''' | |
| import argparse | |
| parser = argparse.ArgumentParser(description="Youtube conversion tool -pete 2017-08-01", epilog=usage_examples, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| parser.add_argument("command", choices=["channel", "convert", "query"], help="") | |
| parser.add_argument("--input", "-i", default=sys.stdin, help="Input file (defaults to stdin)") | |
| parser.add_argument("--output", "-o", default=sys.stdout, help="Output file (defaults to stdout)") | |
| parser.add_argument("--format", "-f", default="mp3", help="Format to convert to") | |
| parser.add_argument("--debug", "-d", help="Print DEBUG info", action="store_true") | |
| parser.add_argument("--url", "-u", help="URL of Youtube link eg: https://www.youtube.com/watch?v=StTqXEQ2l-Y") | |
| args = parser.parse_args() | |
| if hasattr(args.output, 'write'): | |
| outpfh = args.output | |
| else: outpfh = open(args.output, "wb") | |
| if hasattr(args.input, 'read'): | |
| inputfh = args.input | |
| else: inputfh = open(args.read, "rb") | |
| if args.command=="channel": | |
| if not args.url: | |
| print("Please provide a --url to query") | |
| parser.print_help() | |
| sys.exit(1) | |
| episodes = find_episodes(args.debug, args.url) | |
| for e in episodes: | |
| outpfh.write(e+"\n") | |
| elif args.command=="query": | |
| episodes = [] | |
| if args.url: | |
| regex = re.match("(?:http.://)www.youtube.com/watch\?v=(.*)?", args.url) | |
| if regex: | |
| episodes.append(find_media(args.debug, regex.groups()[0])) | |
| else: | |
| print("Please provide a valid --url parameter") | |
| parser.print_help() | |
| sys.exit(1) | |
| else: | |
| print >>sys.stderr, "Parsing list of Youtube media from {}".format(args.input) | |
| for e in inputfh: | |
| e = e.replace("\n","").replace("\r","") | |
| episodes.append(find_media(args.debug, e)) | |
| wr = csv.writer(outpfh) | |
| for e in episodes: | |
| wr.writerow([x.encode('utf-8') for x in e]) | |
| elif args.command=="convert": | |
| if args.url: | |
| regex = re.match("(?:http.://)www.youtube.com/watch\?v=(.*)?", args.url) | |
| if regex: | |
| seq = find_media(args.debug, regex.groups()[0]) | |
| convert_media(args.debug, args.url, args.format) | |
| else: | |
| print >>sys.stderr, "Parsing csv of Youtube media from {}".format(args.input) | |
| for e in csv.reader(inputfh): | |
| convert_media(args.debug, e[2], "{}.{}".format(e[1], args.format)) | |
| else: | |
| parser.print_help() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment