]>
code.delx.au - webdl/blob - common.py
18 USER_AGENT
= "Mozilla/5.0 (X11; Linux x86_64; rv:74.0) Gecko/20100101 Firefox/74.0"
22 autosocks
.try_autosocks()
28 format
= "%(levelname)s %(message)s",
29 level
= logging
.INFO
if os
.environ
.get("DEBUG", None) is None else logging
.DEBUG
,
33 CACHE_FILE
= os
.path
.join(
34 os
.environ
.get("XDG_CACHE_HOME", os
.path
.expanduser("~/.cache")),
38 if not os
.path
.isdir(os
.path
.dirname(CACHE_FILE
)):
39 os
.makedirs(os
.path
.dirname(CACHE_FILE
))
41 requests_cache
.install_cache(CACHE_FILE
, backend
='sqlite', expire_after
=3600)
45 def __init__(self
, title
, parent
=None):
48 parent
.children
.append(self
)
51 self
.can_download
= False
53 def get_children(self
):
56 self
.children
= natural_sort(self
.children
, key
=lambda node
: node
.title
)
59 def fill_children(self
):
67 root_node
= Node("Root")
70 iview
.fill_nodes(root_node
)
73 sbs
.fill_nodes(root_node
)
76 ten
.fill_nodes(root_node
)
80 valid_chars
= frozenset("-_.()!@#%^ abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789")
81 def sanify_filename(filename
):
82 filename
= "".join(c
for c
in filename
if c
in valid_chars
)
83 assert len(filename
) > 0
86 def ensure_scheme(url
):
87 parts
= urllib
.parse
.urlparse(url
)
92 return urllib
.parse
.urlunparse(parts
)
94 http_session
= requests
.Session()
95 http_session
.headers
["User-Agent"] = USER_AGENT
98 logging
.debug("grab_text(%r)", url
)
99 request
= http_session
.prepare_request(requests
.Request("GET", url
))
100 response
= http_session
.send(request
)
104 logging
.debug("grab_html(%r)", url
)
105 request
= http_session
.prepare_request(requests
.Request("GET", url
))
106 response
= http_session
.send(request
, stream
=True)
107 doc
= lxml
.html
.parse(io
.BytesIO(response
.content
), lxml
.html
.HTMLParser(encoding
="utf-8", recover
=True))
112 logging
.debug("grab_xml(%r)", url
)
113 request
= http_session
.prepare_request(requests
.Request("GET", url
))
114 response
= http_session
.send(request
, stream
=True)
115 doc
= lxml
.etree
.parse(io
.BytesIO(response
.content
), lxml
.etree
.XMLParser(encoding
="utf-8", recover
=True))
120 logging
.debug("grab_json(%r)", url
)
121 request
= http_session
.prepare_request(requests
.Request("GET", url
))
122 response
= http_session
.send(request
)
123 return response
.json()
125 def exec_subprocess(cmd
):
126 logging
.debug("Executing: %s", cmd
)
128 p
= subprocess
.Popen(cmd
)
131 logging
.error("%s exited with error code: %s", cmd
[0], ret
)
136 logging
.error("Failed to run: %s -- %s", cmd
[0], e
)
137 except KeyboardInterrupt:
138 logging
.info("Cancelled: %s", cmd
)
142 except KeyboardInterrupt:
143 p
.send_signal(signal
.SIGKILL
)
148 def check_command_exists(cmd
):
150 subprocess
.check_output(cmd
, stderr
=subprocess
.STDOUT
)
156 if check_command_exists(["ffmpeg", "--help"]):
159 if check_command_exists(["avconv", "--help"]):
160 logging
.warn("Detected libav-tools! ffmpeg is recommended")
163 raise Exception("You must install ffmpeg or libav-tools")
166 if check_command_exists(["ffprobe", "--help"]):
169 if check_command_exists(["avprobe", "--help"]):
170 logging
.warn("Detected libav-tools! ffmpeg is recommended")
173 raise Exception("You must install ffmpeg or libav-tools")
175 def get_duration(filename
):
176 ffprobe
= find_ffprobe()
181 "-show_format_entry", "duration",
184 output
= subprocess
.check_output(cmd
).decode("utf-8")
185 for line
in output
.split("\n"):
186 m
= re
.search(R
"([0-9]+)", line
)
189 duration
= m
.group(1)
190 if duration
.isdigit():
194 logging
.debug("Falling back to full decode to find duration: %s % filename")
196 ffmpeg
= find_ffmpeg()
203 output
= subprocess
.check_output(cmd
, stderr
=subprocess
.STDOUT
).decode("utf-8")
205 for line
in re
.split(R
"[\r\n]", output
):
206 m
= re
.search(R
"time=([0-9:]*)\.", line
)
209 [h
, m
, s
] = m
.group(1).split(":")
210 # ffmpeg prints the duration as it reads the file, we want the last one
211 duration
= int(h
) * 3600 + int(m
) * 60 + int(s
)
216 raise Exception("Unable to determine video duration of " + filename
)
218 def check_video_durations(flv_filename
, mp4_filename
):
219 flv_duration
= get_duration(flv_filename
)
220 mp4_duration
= get_duration(mp4_filename
)
222 if abs(flv_duration
- mp4_duration
) > 1:
224 "The duration of %s is suspicious, did the remux fail? Expected %s == %s",
225 mp4_filename
, flv_duration
, mp4_duration
231 def remux(infile
, outfile
):
232 logging
.info("Converting %s to mp4", infile
)
234 ffmpeg
= find_ffmpeg()
238 "-bsf:a", "aac_adtstoasc",
244 if not exec_subprocess(cmd
):
247 if not check_video_durations(infile
, outfile
):
253 def convert_to_mp4(filename
):
254 with
open(filename
, "rb") as f
:
256 basename
, ext
= os
.path
.splitext(filename
)
258 if ext
== ".mp4" and fourcc
== b
"FLV\x01":
259 os
.rename(filename
, basename
+ ".flv")
261 filename
= basename
+ ext
263 if ext
in (".flv", ".ts"):
264 filename_mp4
= basename
+ ".mp4"
265 return remux(filename
, filename_mp4
)
270 def download_hds(filename
, video_url
, pvswf
=None):
271 filename
= sanify_filename(filename
)
272 logging
.info("Downloading: %s", filename
)
274 video_url
= "hds://" + video_url
276 param
= "%s pvswf=%s" % (video_url
, pvswf
)
283 "--output", filename
,
287 if exec_subprocess(cmd
):
288 return convert_to_mp4(filename
)
292 def download_hls(filename
, video_url
):
293 filename
= sanify_filename(filename
)
294 video_url
= "hlsvariant://" + video_url
295 logging
.info("Downloading: %s", filename
)
299 "--http-header", "User-Agent=" + USER_AGENT
,
301 "--output", filename
,
305 if exec_subprocess(cmd
):
306 return convert_to_mp4(filename
)
310 def download_mpd(filename
, video_url
):
311 filename
= sanify_filename(filename
)
312 video_url
= "dash://" + video_url
313 logging
.info("Downloading: %s", filename
)
318 "--output", filename
,
322 if exec_subprocess(cmd
):
323 return convert_to_mp4(filename
)
327 def download_http(filename
, video_url
):
328 filename
= sanify_filename(filename
)
329 logging
.info("Downloading: %s", filename
)
333 "--fail", "--retry", "3",
337 if exec_subprocess(cmd
):
338 return convert_to_mp4(filename
)
342 def natural_sort(l
, key
=None):
343 ignore_list
= ["a", "the"]
349 for c
in re
.split("([0-9]+)", k
):
352 newk
.append(c
.zfill(5))
354 for subc
in c
.split():
355 if subc
not in ignore_list
:
359 return sorted(l
, key
=key_func
)
361 def append_to_qs(url
, params
):
362 r
= list(urllib
.parse
.urlsplit(url
))
363 qs
= urllib
.parse
.parse_qs(r
[3])
364 for k
, v
in params
.items():
369 r
[3] = urllib
.parse
.urlencode(sorted(qs
.items()), True)
370 url
= urllib
.parse
.urlunsplit(r
)