Fix CSS issues and Login (thanks @jp-13 and @noxymon)

Closes #198
Closes #200
This commit is contained in:
Lorenzo Di Fuccia 2020-04-18 23:00:55 +02:00
commit d339d9dd0f
3 changed files with 80 additions and 20 deletions

View file

@ -72,8 +72,9 @@ optional arguments:
--help Show this help message.
```
The first time you use the program, you'll have to specify your Safari Books Online account credentials (look [`here`](/../../issues/15) for special character).
The next times you'll download a book, before session expires, you can omit the credential, because the program save your session cookies in a file called `cookies.json` (for **SSO** look the file format [`here`](/../../issues/2#issuecomment-367726544)).
The first time you use the program, you'll have to specify your Safari Books Online account credentials (look [`here`](/../../issues/15) for special character).
The next times you'll download a book, before session expires, you can omit the credential, because the program save your session cookies in a file called `cookies.json`.
For **SSO**, please use the `sso_cookies.py` program in order to create the `cookies.json` file from the SSO cookies retrieved by your browser session (please follow [`these steps`](/../../issues/150#issuecomment-555423085)).
Pay attention if you use a shared PC, because everyone that has access to your files can steal your session.
If you don't want to cache the cookies, just use the `--no-cookies` option and provide all time your `--cred` to perform `--login`.

View file

@ -30,6 +30,10 @@ SAFARI_BASE_URL = "https://" + SAFARI_BASE_HOST
API_ORIGIN_URL = "https://" + API_ORIGIN_HOST
PROFILE_URL = SAFARI_BASE_URL + "/profile/"
# DEBUG
USE_PROXY = True
PROXIES = {"https": "https://127.0.0.1:8080"}
class Display:
BASE_FORMAT = logging.Formatter(
@ -312,13 +316,17 @@ class SafariBooks:
self.display.intro()
self.session = requests.Session()
if USE_PROXY: # DEBUG
self.session.proxies = PROXIES
self.session.verify = False
self.session.headers.update(self.HEADERS)
self.jwt = {}
if not args.cred:
if not os.path.isfile(COOKIES_FILE):
self.display.exit("Login: unable to find cookies file.\n"
self.display.exit("Login: unable to find `cookies.json` file.\n"
" Please use the `--cred` or `--login` options to perform the login.")
self.session.cookies.update(json.load(open(COOKIES_FILE)))
@ -364,6 +372,7 @@ class SafariBooks:
self.chapter_title = ""
self.filename = ""
self.chapter_stylesheets = []
self.css = []
self.images = []
@ -655,9 +664,17 @@ class SafariBooks:
)
page_css = ""
if len(self.chapter_stylesheets):
for chapter_css_url in self.chapter_stylesheets:
if chapter_css_url not in self.css:
self.css.append(chapter_css_url)
self.display.log("Crawler: found a new CSS at %s" % chapter_css_url)
page_css += "<link href=\"Styles/Style{0:0>2}.css\" " \
"rel=\"stylesheet\" type=\"text/css\" />\n".format(self.css.index(chapter_css_url))
stylesheet_links = root.xpath("//link[@rel='stylesheet']")
if len(stylesheet_links):
stylesheet_count = 0
for s in stylesheet_links:
css_url = urljoin("https:", s.attrib["href"]) if s.attrib["href"][:2] == "//" \
else urljoin(self.base_url, s.attrib["href"])
@ -667,8 +684,7 @@ class SafariBooks:
self.display.log("Crawler: found a new CSS at %s" % css_url)
page_css += "<link href=\"Styles/Style{0:0>2}.css\" " \
"rel=\"stylesheet\" type=\"text/css\" />\n".format(stylesheet_count)
stylesheet_count += 1
"rel=\"stylesheet\" type=\"text/css\" />\n".format(self.css.index(css_url))
stylesheets = root.xpath("//style")
if len(stylesheets):
@ -795,6 +811,14 @@ class SafariBooks:
self.chapter_title = next_chapter["title"]
self.filename = next_chapter["filename"]
# Stylesheets
self.chapter_stylesheets = []
if "stylesheets" in next_chapter and len(next_chapter["stylesheets"]):
self.chapter_stylesheets.extend(x["url"] for x in next_chapter["stylesheets"])
if "site_styles" in next_chapter and len(next_chapter["site_styles"]):
self.chapter_stylesheets.extend(next_chapter["site_styles"])
if os.path.isfile(os.path.join(self.BOOK_PATH, "OEBPS", self.filename.replace(".html", ".xhtml"))):
if not self.display.book_ad_info and \
next_chapter not in self.book_chapters[:self.book_chapters.index(next_chapter)]:
@ -879,13 +903,9 @@ class SafariBooks:
def collect_css(self):
self.display.state_status.value = -1
if "win" in sys.platform:
# TODO
for css_url in self.css:
self._thread_download_css(css_url)
else:
self._start_multiprocessing(self._thread_download_css, self.css)
# "self._start_multiprocessing" seems to cause problem. Switching to mono-thread download.
for css_url in self.css:
self._thread_download_css(css_url)
def collect_images(self):
if self.display.book_ad_info == 2:
@ -896,13 +916,9 @@ class SafariBooks:
self.display.state_status.value = -1
if "win" in sys.platform:
# TODO
for image_url in self.images:
self._thread_download_images(image_url)
else:
self._start_multiprocessing(self._thread_download_images, self.images)
# "self._start_multiprocessing" seems to cause problem. Switching to mono-thread download.
for image_url in self.images:
self._thread_download_images(image_url)
def create_content_opf(self):
self.css = next(os.walk(self.css_path))[2]

43
sso_cookies.py Normal file
View file

@ -0,0 +1,43 @@
"""
Script for SSO support, saves and converts the cookie string retrieved by the browser.
Please follow:
- https://github.com/lorenzodifuccia/safaribooks/issues/26
- https://github.com/lorenzodifuccia/safaribooks/issues/150#issuecomment-555423085
- https://github.com/lorenzodifuccia/safaribooks/issues/2#issuecomment-367726544
Thanks: @elrob, @noxymon
"""
import json
import safaribooks
def transform(cookies_string):
cookies = {}
for cookie in cookies_string.split(";"):
cookie = cookie.strip()
key, value = cookie.split("=", 1)
cookies[key] = value
print(cookies)
json.dump(cookies, open(safaribooks.COOKIES_FILE, 'w'))
print("\n\nDone! Cookie Jar saved into `cookies.json`. "
"Now you can run `safaribooks.py` without the `--cred` argument...")
USAGE = "\n\n[*] Please use this command putting as argument the cookies retrieved by your browser.\n" + \
"[+] In order to do so, please follow these steps: \n" + \
"https://github.com/lorenzodifuccia/safaribooks/issues/150#issuecomment-555423085\n"
if __name__ == "__main__":
import sys
if len(sys.argv) < 2:
print("[!] Error: too few arguments." + USAGE)
exit(1)
elif len(sys.argv) > 2:
print("[!] Error: too much arguments, try to enclose the string with quote '\"'." + USAGE)
exit(1)
transform(sys.argv[1])