diff --git a/safaribooks.py b/safaribooks.py index 97ca52b..b646174 100644 --- a/safaribooks.py +++ b/safaribooks.py @@ -290,15 +290,6 @@ class SafariBooks: self.display.info("Retrieving book chapters...") self.book_chapters = self.get_book_chapters() - self.no_cover = False - if "cover" not in self.book_chapters[0]["filename"] or "cover" not in self.book_chapters[0]["title"]: - self.book_chapters = [{ - "filename": "cover", - "title": "Cover", - "web_url": self.book_info["cover"] - }] + self.book_chapters - self.no_cover = True - self.chapters_queue = self.book_chapters[:] if len(self.book_chapters) > sys.getrecursionlimit(): @@ -309,6 +300,8 @@ class SafariBooks: self.clean_book_title = self.clean_dirname(self.book_title) self.BOOK_PATH = os.path.join(PATH, "Books", self.clean_book_title) + self.css_path = "" + self.images_path = "" self.create_dirs() self.display.info("Output directory:\n %s" % self.BOOK_PATH) @@ -320,11 +313,21 @@ class SafariBooks: self.display.info("Downloading book contents... (%s chapters)" % len(self.book_chapters), state=True) self.BASE_HTML = self.BASE_01_HTML + (self.KINDLE_HTML if not args.no_kindle else "") + self.BASE_02_HTML + self.cover = False self.get() + if not self.cover: + self.cover = self.get_default_cover() + cover_html = self.parse_html( + html.fromstring("
".format(self.cover)), True + ) - self.css_path = "" - self.images_path = "" - self.cover = "" + self.book_chapters = [{ + "filename": "default_cover.xhtml", + "title": "Cover" + }] + self.book_chapters + + self.filename = self.book_chapters[0]["filename"] + self.save_page_html(cover_html) self.css_done_queue = Queue(0) if "win" not in sys.platform else WinQueue() self.display.info("Downloading book CSSs... (%s files)" % len(self.css), state=True) @@ -473,13 +476,26 @@ class SafariBooks: sys.setrecursionlimit(response["count"]) result = [] - result.extend([c for c in response["results"] if "cover." in c["filename"]]) + result.extend([c for c in response["results"] if "cover" in c["filename"] or "cover" in c["title"]]) for c in result: del response["results"][response["results"].index(c)] result += response["results"] return result + (self.get_book_chapters(page + 1) if response["next"] else []) + def get_default_cover(self): + response = self.requests_provider(self.book_info["cover"], update_cookies=False, stream=True) + if response == 0: + self.display.error("Error trying to retrieve the cover: %s" % self.book_info["cover"]) + return False + + file_ext = response.headers["Content-Type"].split("/")[-1] + with open(os.path.join(self.images_path, "default_cover." + file_ext), 'wb') as i: + for chunk in response.iter_content(1024): + i.write(chunk) + + return "default_cover." + file_ext + def get_html(self, url): response = self.requests_provider(url) if response == 0: @@ -508,9 +524,9 @@ class SafariBooks: def link_replace(self, link): if link: if not self.url_is_absolute(link): - link = urljoin(self.base_url, link) if "cover" in link or "images" in link or "graphics" in link or \ link[-3:] in ["jpg", "peg", "png", "gif"]: + link = urljoin(self.base_url, link) if link not in self.images: self.images.append(link) self.display.log("Crawler: found a new image at %s" % link) @@ -520,9 +536,31 @@ class SafariBooks: return link.replace(".html", ".xhtml") + else: + if self.book_id in link: + return self.link_replace(link.split(self.book_id)[-1]) + return link - def parse_html(self, root, is_cover=False): + @staticmethod + def get_cover(html_root): + images = html_root.xpath("//img[contains(@id, 'cover') or " + "contains(@name, 'cover') or contains(@src, 'cover')]") + if len(images): + return images[0] + + divs = html_root.xpath("//div[contains(@id, 'cover') or " + "contains(@name, 'cover') or contains(@src, 'cover')]//img") + if len(divs): + return divs[0] + + a = html_root.xpath("//a[contains(@id, 'cover') or contains(@name, 'cover') or contains(@src, 'cover')]//img") + if len(a): + return a[0] + + return None + + def parse_html(self, root, first_page=False): if random() > 0.5: if len(root.xpath("//div[@class='controls']/a/text()")): self.display.exit(self.display.api_error(" ")) @@ -585,22 +623,23 @@ class SafariBooks: xhtml = None try: - if is_cover: - page_css = "" - - cover_html = html.fromstring("
") - cover_div = cover_html.xpath("//div")[0] - - if len(book_content.xpath("//img")): + if first_page: + is_cover = self.get_cover(book_content) + if is_cover is not None: + page_css = "" + cover_html = html.fromstring("
") + cover_div = cover_html.xpath("//div")[0] cover_img = cover_div.makeelement("img") - cover_img.attrib.update({"src": book_content.xpath("//img")[0].attrib["src"]}) + cover_img.attrib.update({"src": is_cover.attrib["src"]}) cover_div.append(cover_img) book_content = cover_html + self.cover = is_cover.attrib["src"] + xhtml = html.tostring(book_content, method="xml", encoding='unicode') except (html.etree.ParseError, html.etree.ParserError) as parsing_error: @@ -620,7 +659,7 @@ class SafariBooks: for ch in ['\\', '/', '<', '>', '`', '\'', '"', '*', '?', '|']: if ch in dirname: - dirname = dirname.replace(ch, "_") + dirname = dirname.replace(ch, "") return dirname @@ -636,6 +675,22 @@ class SafariBooks: self.display.book_ad_info = True os.makedirs(oebps) + self.css_path = os.path.join(oebps, "Styles") + if os.path.isdir(self.css_path): + self.display.log("CSSs directory already exists: %s" % self.css_path) + + else: + os.makedirs(self.css_path) + self.display.css_ad_info.value = 1 + + self.images_path = os.path.join(oebps, "Images") + if os.path.isdir(self.images_path): + self.display.log("Images directory already exists: %s" % self.images_path) + + else: + os.makedirs(self.images_path) + self.display.images_ad_info.value = 1 + def save_page_html(self, contents): self.filename = self.filename.replace(".html", ".xhtml") open(os.path.join(self.BOOK_PATH, "OEBPS", self.filename), "wb")\ @@ -645,11 +700,11 @@ class SafariBooks: def get(self): len_books = len(self.book_chapters) - for _ in self.book_chapters: + for _ in range(len_books): if not len(self.chapters_queue): return - is_cover = len_books == len(self.chapters_queue) + first_page = len_books == len(self.chapters_queue) next_chapter = self.chapters_queue.pop(0) self.chapter_title = next_chapter["title"] @@ -671,25 +726,7 @@ class SafariBooks: self.display.book_ad_info = 2 else: - if is_cover and self.no_cover: - response = self.requests_provider(next_chapter["web_url"], update_cookies=False, stream=True) - if response != 0: - with open(os.path.join(self.BOOK_PATH, "OEBPS", "Images", - self.filename + "." + response.headers["Content-Type"].split("/")[-1]), 'wb') as s: - for chunk in response.iter_content(1024): - s.write(chunk) - - cover_html = self.parse_html(html.fromstring( - "
".format( - self.filename + "." + response.headers["Content-Type"].split("/")[-1] - ) - ), is_cover) - self.filename += ".xhtml" - self.book_chapters[0]["filename"] += ".xhtml" - self.save_page_html(cover_html) - continue - - self.save_page_html(self.parse_html(self.get_html(next_chapter["web_url"]), is_cover)) + self.save_page_html(self.parse_html(self.get_html(next_chapter["web_url"]), first_page)) self.display.state(len_books, len_books - len(self.chapters_queue)) @@ -756,14 +793,6 @@ class SafariBooks: proc.join() def collect_css(self): - self.css_path = os.path.join(self.BOOK_PATH, "OEBPS", "Styles") - if os.path.isdir(self.css_path): - self.display.log("CSSs directory already exists: %s" % self.css_path) - - else: - os.makedirs(self.css_path) - self.display.css_ad_info.value = 1 - self.display.state_status.value = -1 if "win" in sys.platform: @@ -775,14 +804,6 @@ class SafariBooks: self._start_multiprocessing(self._thread_download_css, self.css) def collect_images(self): - self.images_path = os.path.join(self.BOOK_PATH, "OEBPS", "Images") - if os.path.isdir(self.images_path): - self.display.log("Images directory already exists: %s" % self.images_path) - - else: - os.makedirs(self.images_path) - self.display.images_ad_info.value = 1 - if self.display.book_ad_info == 2: self.display.info("Some of the book contents were already downloaded.\n" " If you want to be sure that all the images will be downloaded,\n" @@ -799,7 +820,6 @@ class SafariBooks: self._start_multiprocessing(self._thread_download_images, self.images) def create_content_opf(self): - self.cover = self.images[0] if len(self.images) else "" self.css = next(os.walk(self.css_path))[2] self.images = next(os.walk(self.images_path))[2] @@ -813,7 +833,6 @@ class SafariBooks: )) spine.append("".format(item_id)) - alt_cover_id = False for i in set(self.images): dot_split = i.split(".") head = "img_" + escape("".join(dot_split[:-1])) @@ -822,9 +841,6 @@ class SafariBooks: head, i, "jpeg" if "jp" in extension else extension )) - if not alt_cover_id: - alt_cover_id = head - for i in range(len(self.css)): manifest.append("2}\" href=\"Styles/Style{0:0>2}.css\" " "media-type=\"text/css\" />".format(i)) @@ -845,7 +861,7 @@ class SafariBooks: ", ".join(escape(pub["name"]) for pub in self.book_info["publishers"]), escape(self.book_info["rights"]), self.book_info["issued"], - self.cover if self.cover else alt_cover_id, + self.cover, "\n".join(manifest), "\n".join(spine), self.book_chapters[0]["filename"].replace(".html", ".xhtml")