General improvement

Fixed #3 (Again)
This commit is contained in:
Lorenzo Di Fuccia 2018-02-27 15:13:12 +01:00
commit a7b95ebd5b

View file

@ -290,15 +290,6 @@ class SafariBooks:
self.display.info("Retrieving book chapters...") self.display.info("Retrieving book chapters...")
self.book_chapters = self.get_book_chapters() self.book_chapters = self.get_book_chapters()
self.no_cover = False
if "cover" not in self.book_chapters[0]["filename"] or "cover" not in self.book_chapters[0]["title"]:
self.book_chapters = [{
"filename": "cover",
"title": "Cover",
"web_url": self.book_info["cover"]
}] + self.book_chapters
self.no_cover = True
self.chapters_queue = self.book_chapters[:] self.chapters_queue = self.book_chapters[:]
if len(self.book_chapters) > sys.getrecursionlimit(): if len(self.book_chapters) > sys.getrecursionlimit():
@ -309,6 +300,8 @@ class SafariBooks:
self.clean_book_title = self.clean_dirname(self.book_title) self.clean_book_title = self.clean_dirname(self.book_title)
self.BOOK_PATH = os.path.join(PATH, "Books", self.clean_book_title) self.BOOK_PATH = os.path.join(PATH, "Books", self.clean_book_title)
self.css_path = ""
self.images_path = ""
self.create_dirs() self.create_dirs()
self.display.info("Output directory:\n %s" % self.BOOK_PATH) self.display.info("Output directory:\n %s" % self.BOOK_PATH)
@ -320,11 +313,21 @@ class SafariBooks:
self.display.info("Downloading book contents... (%s chapters)" % len(self.book_chapters), state=True) self.display.info("Downloading book contents... (%s chapters)" % len(self.book_chapters), state=True)
self.BASE_HTML = self.BASE_01_HTML + (self.KINDLE_HTML if not args.no_kindle else "") + self.BASE_02_HTML self.BASE_HTML = self.BASE_01_HTML + (self.KINDLE_HTML if not args.no_kindle else "") + self.BASE_02_HTML
self.cover = False
self.get() self.get()
if not self.cover:
self.cover = self.get_default_cover()
cover_html = self.parse_html(
html.fromstring("<div id=\"sbo-rt-content\"><img src=\"Images/{0}\"></div>".format(self.cover)), True
)
self.css_path = "" self.book_chapters = [{
self.images_path = "" "filename": "default_cover.xhtml",
self.cover = "" "title": "Cover"
}] + self.book_chapters
self.filename = self.book_chapters[0]["filename"]
self.save_page_html(cover_html)
self.css_done_queue = Queue(0) if "win" not in sys.platform else WinQueue() self.css_done_queue = Queue(0) if "win" not in sys.platform else WinQueue()
self.display.info("Downloading book CSSs... (%s files)" % len(self.css), state=True) self.display.info("Downloading book CSSs... (%s files)" % len(self.css), state=True)
@ -473,13 +476,26 @@ class SafariBooks:
sys.setrecursionlimit(response["count"]) sys.setrecursionlimit(response["count"])
result = [] result = []
result.extend([c for c in response["results"] if "cover." in c["filename"]]) result.extend([c for c in response["results"] if "cover" in c["filename"] or "cover" in c["title"]])
for c in result: for c in result:
del response["results"][response["results"].index(c)] del response["results"][response["results"].index(c)]
result += response["results"] result += response["results"]
return result + (self.get_book_chapters(page + 1) if response["next"] else []) return result + (self.get_book_chapters(page + 1) if response["next"] else [])
def get_default_cover(self):
response = self.requests_provider(self.book_info["cover"], update_cookies=False, stream=True)
if response == 0:
self.display.error("Error trying to retrieve the cover: %s" % self.book_info["cover"])
return False
file_ext = response.headers["Content-Type"].split("/")[-1]
with open(os.path.join(self.images_path, "default_cover." + file_ext), 'wb') as i:
for chunk in response.iter_content(1024):
i.write(chunk)
return "default_cover." + file_ext
def get_html(self, url): def get_html(self, url):
response = self.requests_provider(url) response = self.requests_provider(url)
if response == 0: if response == 0:
@ -508,9 +524,9 @@ class SafariBooks:
def link_replace(self, link): def link_replace(self, link):
if link: if link:
if not self.url_is_absolute(link): if not self.url_is_absolute(link):
link = urljoin(self.base_url, link)
if "cover" in link or "images" in link or "graphics" in link or \ if "cover" in link or "images" in link or "graphics" in link or \
link[-3:] in ["jpg", "peg", "png", "gif"]: link[-3:] in ["jpg", "peg", "png", "gif"]:
link = urljoin(self.base_url, link)
if link not in self.images: if link not in self.images:
self.images.append(link) self.images.append(link)
self.display.log("Crawler: found a new image at %s" % link) self.display.log("Crawler: found a new image at %s" % link)
@ -520,9 +536,31 @@ class SafariBooks:
return link.replace(".html", ".xhtml") return link.replace(".html", ".xhtml")
else:
if self.book_id in link:
return self.link_replace(link.split(self.book_id)[-1])
return link return link
def parse_html(self, root, is_cover=False): @staticmethod
def get_cover(html_root):
images = html_root.xpath("//img[contains(@id, 'cover') or "
"contains(@name, 'cover') or contains(@src, 'cover')]")
if len(images):
return images[0]
divs = html_root.xpath("//div[contains(@id, 'cover') or "
"contains(@name, 'cover') or contains(@src, 'cover')]//img")
if len(divs):
return divs[0]
a = html_root.xpath("//a[contains(@id, 'cover') or contains(@name, 'cover') or contains(@src, 'cover')]//img")
if len(a):
return a[0]
return None
def parse_html(self, root, first_page=False):
if random() > 0.5: if random() > 0.5:
if len(root.xpath("//div[@class='controls']/a/text()")): if len(root.xpath("//div[@class='controls']/a/text()")):
self.display.exit(self.display.api_error(" ")) self.display.exit(self.display.api_error(" "))
@ -585,22 +623,23 @@ class SafariBooks:
xhtml = None xhtml = None
try: try:
if is_cover: if first_page:
page_css = "<style>" \ is_cover = self.get_cover(book_content)
"body{display:table;position:absolute;margin:0!important;height:100%;width:100%;}" \ if is_cover is not None:
"#Cover{display:table-cell;vertical-align:middle;text-align:center;}" \ page_css = "<style>" \
"img{height:90vh;margin-left:auto;margin-right:auto;}" \ "body{display:table;position:absolute;margin:0!important;height:100%;width:100%;}" \
"</style>" "#Cover{display:table-cell;vertical-align:middle;text-align:center;}" \
"img{height:90vh;margin-left:auto;margin-right:auto;}" \
cover_html = html.fromstring("<div id=\"Cover\"></div>") "</style>"
cover_div = cover_html.xpath("//div")[0] cover_html = html.fromstring("<div id=\"Cover\"></div>")
cover_div = cover_html.xpath("//div")[0]
if len(book_content.xpath("//img")):
cover_img = cover_div.makeelement("img") cover_img = cover_div.makeelement("img")
cover_img.attrib.update({"src": book_content.xpath("//img")[0].attrib["src"]}) cover_img.attrib.update({"src": is_cover.attrib["src"]})
cover_div.append(cover_img) cover_div.append(cover_img)
book_content = cover_html book_content = cover_html
self.cover = is_cover.attrib["src"]
xhtml = html.tostring(book_content, method="xml", encoding='unicode') xhtml = html.tostring(book_content, method="xml", encoding='unicode')
except (html.etree.ParseError, html.etree.ParserError) as parsing_error: except (html.etree.ParseError, html.etree.ParserError) as parsing_error:
@ -620,7 +659,7 @@ class SafariBooks:
for ch in ['\\', '/', '<', '>', '`', '\'', '"', '*', '?', '|']: for ch in ['\\', '/', '<', '>', '`', '\'', '"', '*', '?', '|']:
if ch in dirname: if ch in dirname:
dirname = dirname.replace(ch, "_") dirname = dirname.replace(ch, "")
return dirname return dirname
@ -636,6 +675,22 @@ class SafariBooks:
self.display.book_ad_info = True self.display.book_ad_info = True
os.makedirs(oebps) os.makedirs(oebps)
self.css_path = os.path.join(oebps, "Styles")
if os.path.isdir(self.css_path):
self.display.log("CSSs directory already exists: %s" % self.css_path)
else:
os.makedirs(self.css_path)
self.display.css_ad_info.value = 1
self.images_path = os.path.join(oebps, "Images")
if os.path.isdir(self.images_path):
self.display.log("Images directory already exists: %s" % self.images_path)
else:
os.makedirs(self.images_path)
self.display.images_ad_info.value = 1
def save_page_html(self, contents): def save_page_html(self, contents):
self.filename = self.filename.replace(".html", ".xhtml") self.filename = self.filename.replace(".html", ".xhtml")
open(os.path.join(self.BOOK_PATH, "OEBPS", self.filename), "wb")\ open(os.path.join(self.BOOK_PATH, "OEBPS", self.filename), "wb")\
@ -645,11 +700,11 @@ class SafariBooks:
def get(self): def get(self):
len_books = len(self.book_chapters) len_books = len(self.book_chapters)
for _ in self.book_chapters: for _ in range(len_books):
if not len(self.chapters_queue): if not len(self.chapters_queue):
return return
is_cover = len_books == len(self.chapters_queue) first_page = len_books == len(self.chapters_queue)
next_chapter = self.chapters_queue.pop(0) next_chapter = self.chapters_queue.pop(0)
self.chapter_title = next_chapter["title"] self.chapter_title = next_chapter["title"]
@ -671,25 +726,7 @@ class SafariBooks:
self.display.book_ad_info = 2 self.display.book_ad_info = 2
else: else:
if is_cover and self.no_cover: self.save_page_html(self.parse_html(self.get_html(next_chapter["web_url"]), first_page))
response = self.requests_provider(next_chapter["web_url"], update_cookies=False, stream=True)
if response != 0:
with open(os.path.join(self.BOOK_PATH, "OEBPS", "Images",
self.filename + "." + response.headers["Content-Type"].split("/")[-1]), 'wb') as s:
for chunk in response.iter_content(1024):
s.write(chunk)
cover_html = self.parse_html(html.fromstring(
"<div id=\"sbo-rt-content\"><img src=\"Images/{0}\"></div>".format(
self.filename + "." + response.headers["Content-Type"].split("/")[-1]
)
), is_cover)
self.filename += ".xhtml"
self.book_chapters[0]["filename"] += ".xhtml"
self.save_page_html(cover_html)
continue
self.save_page_html(self.parse_html(self.get_html(next_chapter["web_url"]), is_cover))
self.display.state(len_books, len_books - len(self.chapters_queue)) self.display.state(len_books, len_books - len(self.chapters_queue))
@ -756,14 +793,6 @@ class SafariBooks:
proc.join() proc.join()
def collect_css(self): def collect_css(self):
self.css_path = os.path.join(self.BOOK_PATH, "OEBPS", "Styles")
if os.path.isdir(self.css_path):
self.display.log("CSSs directory already exists: %s" % self.css_path)
else:
os.makedirs(self.css_path)
self.display.css_ad_info.value = 1
self.display.state_status.value = -1 self.display.state_status.value = -1
if "win" in sys.platform: if "win" in sys.platform:
@ -775,14 +804,6 @@ class SafariBooks:
self._start_multiprocessing(self._thread_download_css, self.css) self._start_multiprocessing(self._thread_download_css, self.css)
def collect_images(self): def collect_images(self):
self.images_path = os.path.join(self.BOOK_PATH, "OEBPS", "Images")
if os.path.isdir(self.images_path):
self.display.log("Images directory already exists: %s" % self.images_path)
else:
os.makedirs(self.images_path)
self.display.images_ad_info.value = 1
if self.display.book_ad_info == 2: if self.display.book_ad_info == 2:
self.display.info("Some of the book contents were already downloaded.\n" self.display.info("Some of the book contents were already downloaded.\n"
" If you want to be sure that all the images will be downloaded,\n" " If you want to be sure that all the images will be downloaded,\n"
@ -799,7 +820,6 @@ class SafariBooks:
self._start_multiprocessing(self._thread_download_images, self.images) self._start_multiprocessing(self._thread_download_images, self.images)
def create_content_opf(self): def create_content_opf(self):
self.cover = self.images[0] if len(self.images) else ""
self.css = next(os.walk(self.css_path))[2] self.css = next(os.walk(self.css_path))[2]
self.images = next(os.walk(self.images_path))[2] self.images = next(os.walk(self.images_path))[2]
@ -813,7 +833,6 @@ class SafariBooks:
)) ))
spine.append("<itemref idref=\"{0}\"/>".format(item_id)) spine.append("<itemref idref=\"{0}\"/>".format(item_id))
alt_cover_id = False
for i in set(self.images): for i in set(self.images):
dot_split = i.split(".") dot_split = i.split(".")
head = "img_" + escape("".join(dot_split[:-1])) head = "img_" + escape("".join(dot_split[:-1]))
@ -822,9 +841,6 @@ class SafariBooks:
head, i, "jpeg" if "jp" in extension else extension head, i, "jpeg" if "jp" in extension else extension
)) ))
if not alt_cover_id:
alt_cover_id = head
for i in range(len(self.css)): for i in range(len(self.css)):
manifest.append("<item id=\"style_{0:0>2}\" href=\"Styles/Style{0:0>2}.css\" " manifest.append("<item id=\"style_{0:0>2}\" href=\"Styles/Style{0:0>2}.css\" "
"media-type=\"text/css\" />".format(i)) "media-type=\"text/css\" />".format(i))
@ -845,7 +861,7 @@ class SafariBooks:
", ".join(escape(pub["name"]) for pub in self.book_info["publishers"]), ", ".join(escape(pub["name"]) for pub in self.book_info["publishers"]),
escape(self.book_info["rights"]), escape(self.book_info["rights"]),
self.book_info["issued"], self.book_info["issued"],
self.cover if self.cover else alt_cover_id, self.cover,
"\n".join(manifest), "\n".join(manifest),
"\n".join(spine), "\n".join(spine),
self.book_chapters[0]["filename"].replace(".html", ".xhtml") self.book_chapters[0]["filename"].replace(".html", ".xhtml")