diff --git a/README.md b/README.md index 92fb558..964513e 100644 --- a/README.md +++ b/README.md @@ -72,8 +72,9 @@ optional arguments: --help Show this help message. ``` -The first time you use the program, you'll have to specify your Safari Books Online account credentials (look [`here`](/../../issues/15) for special character). -The next times you'll download a book, before session expires, you can omit the credential, because the program save your session cookies in a file called `cookies.json` (for **SSO** look the file format [`here`](/../../issues/2#issuecomment-367726544)). +The first time you use the program, you'll have to specify your Safari Books Online account credentials (look [`here`](/../../issues/15) for special character). +The next times you'll download a book, before session expires, you can omit the credential, because the program save your session cookies in a file called `cookies.json`. +For **SSO**, please use the `sso_cookies.py` program in order to create the `cookies.json` file from the SSO cookies retrieved by your browser session (please follow [`these steps`](/../../issues/150#issuecomment-555423085)). Pay attention if you use a shared PC, because everyone that has access to your files can steal your session. If you don't want to cache the cookies, just use the `--no-cookies` option and provide all time your `--cred` to perform `--login`. diff --git a/safaribooks.py b/safaribooks.py index 4146ef3..c8cf9e3 100755 --- a/safaribooks.py +++ b/safaribooks.py @@ -30,6 +30,10 @@ SAFARI_BASE_URL = "https://" + SAFARI_BASE_HOST API_ORIGIN_URL = "https://" + API_ORIGIN_HOST PROFILE_URL = SAFARI_BASE_URL + "/profile/" +# DEBUG +USE_PROXY = True +PROXIES = {"https": "https://127.0.0.1:8080"} + class Display: BASE_FORMAT = logging.Formatter( @@ -312,13 +316,17 @@ class SafariBooks: self.display.intro() self.session = requests.Session() + if USE_PROXY: # DEBUG + self.session.proxies = PROXIES + self.session.verify = False + self.session.headers.update(self.HEADERS) self.jwt = {} if not args.cred: if not os.path.isfile(COOKIES_FILE): - self.display.exit("Login: unable to find cookies file.\n" + self.display.exit("Login: unable to find `cookies.json` file.\n" " Please use the `--cred` or `--login` options to perform the login.") self.session.cookies.update(json.load(open(COOKIES_FILE))) @@ -364,6 +372,7 @@ class SafariBooks: self.chapter_title = "" self.filename = "" + self.chapter_stylesheets = [] self.css = [] self.images = [] @@ -655,9 +664,17 @@ class SafariBooks: ) page_css = "" + if len(self.chapter_stylesheets): + for chapter_css_url in self.chapter_stylesheets: + if chapter_css_url not in self.css: + self.css.append(chapter_css_url) + self.display.log("Crawler: found a new CSS at %s" % chapter_css_url) + + page_css += "2}.css\" " \ + "rel=\"stylesheet\" type=\"text/css\" />\n".format(self.css.index(chapter_css_url)) + stylesheet_links = root.xpath("//link[@rel='stylesheet']") if len(stylesheet_links): - stylesheet_count = 0 for s in stylesheet_links: css_url = urljoin("https:", s.attrib["href"]) if s.attrib["href"][:2] == "//" \ else urljoin(self.base_url, s.attrib["href"]) @@ -667,8 +684,7 @@ class SafariBooks: self.display.log("Crawler: found a new CSS at %s" % css_url) page_css += "2}.css\" " \ - "rel=\"stylesheet\" type=\"text/css\" />\n".format(stylesheet_count) - stylesheet_count += 1 + "rel=\"stylesheet\" type=\"text/css\" />\n".format(self.css.index(css_url)) stylesheets = root.xpath("//style") if len(stylesheets): @@ -795,6 +811,14 @@ class SafariBooks: self.chapter_title = next_chapter["title"] self.filename = next_chapter["filename"] + # Stylesheets + self.chapter_stylesheets = [] + if "stylesheets" in next_chapter and len(next_chapter["stylesheets"]): + self.chapter_stylesheets.extend(x["url"] for x in next_chapter["stylesheets"]) + + if "site_styles" in next_chapter and len(next_chapter["site_styles"]): + self.chapter_stylesheets.extend(next_chapter["site_styles"]) + if os.path.isfile(os.path.join(self.BOOK_PATH, "OEBPS", self.filename.replace(".html", ".xhtml"))): if not self.display.book_ad_info and \ next_chapter not in self.book_chapters[:self.book_chapters.index(next_chapter)]: @@ -879,13 +903,9 @@ class SafariBooks: def collect_css(self): self.display.state_status.value = -1 - if "win" in sys.platform: - # TODO - for css_url in self.css: - self._thread_download_css(css_url) - - else: - self._start_multiprocessing(self._thread_download_css, self.css) + # "self._start_multiprocessing" seems to cause problem. Switching to mono-thread download. + for css_url in self.css: + self._thread_download_css(css_url) def collect_images(self): if self.display.book_ad_info == 2: @@ -896,13 +916,9 @@ class SafariBooks: self.display.state_status.value = -1 - if "win" in sys.platform: - # TODO - for image_url in self.images: - self._thread_download_images(image_url) - - else: - self._start_multiprocessing(self._thread_download_images, self.images) + # "self._start_multiprocessing" seems to cause problem. Switching to mono-thread download. + for image_url in self.images: + self._thread_download_images(image_url) def create_content_opf(self): self.css = next(os.walk(self.css_path))[2] diff --git a/sso_cookies.py b/sso_cookies.py new file mode 100644 index 0000000..a408f84 --- /dev/null +++ b/sso_cookies.py @@ -0,0 +1,43 @@ +""" +Script for SSO support, saves and converts the cookie string retrieved by the browser. +Please follow: +- https://github.com/lorenzodifuccia/safaribooks/issues/26 +- https://github.com/lorenzodifuccia/safaribooks/issues/150#issuecomment-555423085 +- https://github.com/lorenzodifuccia/safaribooks/issues/2#issuecomment-367726544 + + +Thanks: @elrob, @noxymon +""" + +import json +import safaribooks + + +def transform(cookies_string): + cookies = {} + for cookie in cookies_string.split(";"): + cookie = cookie.strip() + key, value = cookie.split("=", 1) + cookies[key] = value + + print(cookies) + json.dump(cookies, open(safaribooks.COOKIES_FILE, 'w')) + print("\n\nDone! Cookie Jar saved into `cookies.json`. " + "Now you can run `safaribooks.py` without the `--cred` argument...") + + +USAGE = "\n\n[*] Please use this command putting as argument the cookies retrieved by your browser.\n" + \ + "[+] In order to do so, please follow these steps: \n" + \ + "https://github.com/lorenzodifuccia/safaribooks/issues/150#issuecomment-555423085\n" + +if __name__ == "__main__": + import sys + if len(sys.argv) < 2: + print("[!] Error: too few arguments." + USAGE) + exit(1) + + elif len(sys.argv) > 2: + print("[!] Error: too much arguments, try to enclose the string with quote '\"'." + USAGE) + exit(1) + + transform(sys.argv[1])