Merge pull request #218 from nrenzoni/fix/214.2
fix extracting img urls (thanks @nrenzoni)
This commit is contained in:
commit
2a7df72bd1
1 changed files with 17 additions and 8 deletions
|
|
@ -1,5 +1,6 @@
|
||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# coding: utf-8
|
# coding: utf-8
|
||||||
|
import pathlib
|
||||||
import re
|
import re
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
|
@ -348,6 +349,8 @@ class SafariBooks:
|
||||||
self.display.info("Retrieving book chapters...")
|
self.display.info("Retrieving book chapters...")
|
||||||
self.book_chapters = self.get_book_chapters()
|
self.book_chapters = self.get_book_chapters()
|
||||||
|
|
||||||
|
self.images = self.extract_image_links(self.book_chapters)
|
||||||
|
|
||||||
self.chapters_queue = self.book_chapters[:]
|
self.chapters_queue = self.book_chapters[:]
|
||||||
|
|
||||||
if len(self.book_chapters) > sys.getrecursionlimit():
|
if len(self.book_chapters) > sys.getrecursionlimit():
|
||||||
|
|
@ -373,7 +376,6 @@ class SafariBooks:
|
||||||
self.filename = ""
|
self.filename = ""
|
||||||
self.chapter_stylesheets = []
|
self.chapter_stylesheets = []
|
||||||
self.css = []
|
self.css = []
|
||||||
self.images = []
|
|
||||||
|
|
||||||
self.display.info("Downloading book contents... (%s chapters)" % len(self.book_chapters), state=True)
|
self.display.info("Downloading book contents... (%s chapters)" % len(self.book_chapters), state=True)
|
||||||
self.BASE_HTML = self.BASE_01_HTML + (self.KINDLE_HTML if not args.no_kindle else "") + self.BASE_02_HTML
|
self.BASE_HTML = self.BASE_01_HTML + (self.KINDLE_HTML if not args.no_kindle else "") + self.BASE_02_HTML
|
||||||
|
|
@ -609,16 +611,15 @@ class SafariBooks:
|
||||||
def url_is_absolute(url):
|
def url_is_absolute(url):
|
||||||
return bool(urlparse(url).netloc)
|
return bool(urlparse(url).netloc)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def is_image_link(url: str):
|
||||||
|
return pathlib.Path(url).suffix[1:] in ["jpg", "peg", "png", "gif"]
|
||||||
|
|
||||||
def link_replace(self, link):
|
def link_replace(self, link):
|
||||||
if link and not link.startswith("mailto"):
|
if link and not link.startswith("mailto"):
|
||||||
if not self.url_is_absolute(link):
|
if not self.url_is_absolute(link):
|
||||||
if "cover" in link or "images" in link or "graphics" in link or \
|
if any(x in link for x in ["cover", "images", "graphics"]) or \
|
||||||
link[-3:] in ["jpg", "peg", "png", "gif"]:
|
self.is_image_link(link):
|
||||||
link = urljoin(self.base_url, link)
|
|
||||||
if link not in self.images:
|
|
||||||
self.images.append(link)
|
|
||||||
self.display.log("Crawler: found a new image at %s" % link)
|
|
||||||
|
|
||||||
image = link.split("/")[-1]
|
image = link.split("/")[-1]
|
||||||
return "Images/" + image
|
return "Images/" + image
|
||||||
|
|
||||||
|
|
@ -1044,6 +1045,14 @@ class SafariBooks:
|
||||||
shutil.make_archive(zip_file, 'zip', self.BOOK_PATH)
|
shutil.make_archive(zip_file, 'zip', self.BOOK_PATH)
|
||||||
os.rename(zip_file + ".zip", os.path.join(self.BOOK_PATH, self.book_id) + ".epub")
|
os.rename(zip_file + ".zip", os.path.join(self.BOOK_PATH, self.book_id) + ".epub")
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def extract_image_links(chapters):
|
||||||
|
imgs = []
|
||||||
|
for chapter in chapters:
|
||||||
|
chapter_imgs = [urljoin(chapter['asset_base_url'], img_url) for img_url in chapter['images']]
|
||||||
|
imgs.extend(chapter_imgs)
|
||||||
|
return imgs
|
||||||
|
|
||||||
|
|
||||||
# MAIN
|
# MAIN
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue