commit 4f5f71bdafebab65c18c4f500d12cca2e2a698f8 Author: PixelMelt <44953835+PixelMelt@users.noreply.github.com> Date: Sun Oct 12 19:20:43 2025 -0400 first commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..8987337 --- /dev/null +++ b/.gitignore @@ -0,0 +1,7 @@ +/BLOG.MD +/__pycache__ +/archive +/downloads +/headers.json +/renderer.js +/ttf_character_mapping.json \ No newline at end of file diff --git a/create_epub.py b/create_epub.py new file mode 100644 index 0000000..abe0332 --- /dev/null +++ b/create_epub.py @@ -0,0 +1,401 @@ +#!/usr/bin/env python3 +""" +Create an EPUB file from the decoded Amazon book data with proper formatting. +""" + +import json +from pathlib import Path +import sys +from ebooklib import epub + + +def main(): + if len(sys.argv) < 2: + print("Usage: python3 create_epub_new.py ") + sys.exit(1) + + book_dir = Path(sys.argv[1]) + + # Load the TTF character mapping + mapping_file = Path("ttf_character_mapping.json") + if not mapping_file.exists(): + print(f"Mapping file not found: {mapping_file}") + print("Run match_ttf_to_glyphs.py first!") + return + + with open(mapping_file) as f: + char_mapping = json.load(f) + + print(f"Loaded character mapping: {len(char_mapping)} glyphs") + + # Load metadata + metadata_file = book_dir / 'batch_0' / 'metadata.json' + with open(metadata_file) as f: + metadata = json.load(f) + + # Load TOC + toc_file = book_dir / 'batch_0' / 'toc.json' + with open(toc_file) as f: + toc_data = json.load(f) + + print(f"Book: {metadata['bookTitle']}") + print(f"Author: {metadata['authors'][0]}") + print(f"TOC entries: {len(toc_data)}") + + # Load all_glyphs + all_glyphs_file = book_dir / 'hash_mapping' / 'all_glyphs.json' + if not all_glyphs_file.exists(): + print(f"Book file not found: {all_glyphs_file}") + return + + with open(all_glyphs_file) as f: + all_glyphs = json.load(f) + + print(f"Loaded {len(all_glyphs)} glyphs from all_glyphs.json") + + # Build line ending info (where newlines go) - same as decode_book_with_newlines.py + print("Building line ending positions...") + batch_dirs = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_')], + key=lambda x: int(x.name.split('_')[1])) + + line_info = {} # Index in all_glyphs -> formatting info + current_index = 0 + + # Get page dimensions from actual page data + first_page_file = batch_dirs[0] / sorted(batch_dirs[0].glob('page_data_*.json'))[0].name + with open(first_page_file) as f: + first_page_data = json.load(f) + page_width = first_page_data[0]['width'] + page_height = first_page_data[0]['height'] + + print(f"Page dimensions: {page_width}x{page_height}") + + prev_y = None # Track Y coordinate to detect line breaks + + for batch_dir in batch_dirs: + page_files = sorted(batch_dir.glob('page_data_*.json')) + for page_file in page_files: + with open(page_file) as f: + pages = json.load(f) + + for page in pages: + for run in page.get('children', []): + if 'glyphs' not in run: + continue + + num_glyphs = len(run['glyphs']) + + # Extract formatting info with transform applied + rect = run.get('rect', {}) + transform = run.get('transform', [1, 0, 0, 1, 0, 0]) + tx = transform[4] if len(transform) >= 6 else 0 + ty = transform[5] if len(transform) >= 6 else 0 + + left = rect.get('left', 0) + tx + right = rect.get('right', 0) + tx + top = rect.get('top', 0) + ty + font_style = run.get('fontStyle', 'normal') + font_weight = run.get('fontWeight', 400) + font_size = run.get('fontSize', 8.91) # Default from downloader.py + has_link = 'link' in run + + # Detect alignment type using relative thresholds + center = (left + right) / 2 + page_center = page_width / 2 + text_width = right - left + alignment = 'left' + + # Use relative thresholds based on page width + center_tolerance = page_width * 0.05 # 5% of page width + edge_tolerance = page_width * 0.05 # 5% tolerance for edges + min_side_margin = page_width * 0.1 # 10% margin on each side for center + min_left_margin_right = page_width * 0.2 # 20% left margin for right-align + min_indent = page_width * 0.05 # 5% indent + max_indent = page_width * 0.15 # 15% max for paragraph indent + min_text_width = page_width * 0.3 # 30% minimum text width + + # Check if centered: text center near page center AND margins on both sides + if abs(center - page_center) < center_tolerance and left > min_side_margin and (page_width - right) > min_side_margin: + alignment = 'center' + # Check if right-aligned: close to right edge with significant left margin + elif abs(right - page_width) < edge_tolerance and left > min_left_margin_right: + alignment = 'right' + # Check for indented paragraphs: moderate left margin with substantial text + elif min_indent < left < max_indent and text_width > min_text_width: + alignment = 'indent' + + # Determine if this is a new line (Y coordinate changed significantly) + is_new_line = prev_y is None or abs(top - prev_y) > 5 + + # Store info for each glyph position in this run + for i in range(num_glyphs): + line_info[current_index + i] = { + 'font_style': font_style, + 'font_weight': font_weight, + 'font_size': font_size, + 'has_link': has_link, + 'left': left, + 'alignment': alignment + } + + # Only mark line break if this run is on a NEW line + if is_new_line and current_index > 0: + # Mark line break at the END of the PREVIOUS run + line_info[current_index - 1]['line_break'] = True + + current_index += num_glyphs + prev_y = top + + print(f"Processed {current_index} glyphs with line break info") + + # Create EPUB + print("Creating EPUB...") + book = epub.EpubBook() + + # Set metadata + book.set_identifier(metadata.get('asin', 'unknown')) + book.set_title(metadata['bookTitle']) + book.set_language(metadata.get('lang', 'en')) + + for author in metadata.get('authors', ['Unknown']): + book.add_author(author) + + + # Add CSS for styling - match Kindle rendering parameters + # Based on downloader.py: fontFamily='Bookerly', fontSize='8.91', lineHeight='1.4' + style = ''' + body { + font-family: Bookerly, Georgia, serif; + font-size: 8pt; + line-height: 1.0; + margin: 0 auto; + padding: 0; + max-width: 1000px; + background-color: #ffffff; + color: #000000; + } + p { + margin: 0; + padding: 0; + line-height: 1.0; + } + p.center { + text-align: center; + } + p.right { + text-align: right; + } + p.indent { + text-indent: 2em; + } + p.break { + margin-top: 0.8em; + } + .italic { font-style: italic; } + .bold { font-weight: bold; } + .link { + color: #0066cc; + text-decoration: underline; + } + h1 { + font-size: 1.8em; + margin: 1em 0 0.5em 0; + font-weight: bold; + } + h2 { + font-size: 1.4em; + margin: 0.8em 0 0.4em 0; + font-weight: bold; + } + ''' + + default_css = epub.EpubItem( + uid="style_default", + file_name="style/default.css", + media_type="text/css", + content=style + ) + book.add_item(default_css) + + # Map position IDs to glyph indices + print("Mapping TOC positions to glyph indices...") + position_to_glyph_idx = {} + current_glyph_idx = 0 + + for batch_dir in batch_dirs: + page_files = sorted(batch_dir.glob('page_data_*.json')) + for page_file in page_files: + with open(page_file) as f: + pages = json.load(f) + + for page in pages: + for run in page.get('children', []): + if 'glyphs' not in run: + continue + + # Check if this run has position info + start_pos_id = run.get('startPositionId') + if start_pos_id is not None: + position_to_glyph_idx[start_pos_id] = current_glyph_idx + + current_glyph_idx += len(run['glyphs']) + + # Map TOC entries to glyph indices + toc_chapters = [] + for i, toc_entry in enumerate(toc_data): + pos_id = toc_entry['tocPositionId'] + if pos_id in position_to_glyph_idx: + toc_chapters.append({ + 'label': toc_entry['label'], + 'glyph_idx': position_to_glyph_idx[pos_id], + 'chapter_num': i + }) + + print(f"Found {len(toc_chapters)} TOC entries with positions") + + # Build chapters based on TOC structure + print("Building chapters with formatting...") + import html + chapters = [] + chapter_contents = {} # chapter_num -> content list + current_chapter_num = -1 # Start before first chapter + current_span_classes = [] + consecutive_line_breaks = 0 + + for idx, glyph_id in enumerate(all_glyphs): + # Check if we're at a new chapter start + for toc_ch in toc_chapters: + if toc_ch['glyph_idx'] == idx: + current_chapter_num = toc_ch['chapter_num'] + if current_chapter_num not in chapter_contents: + chapter_contents[current_chapter_num] = ['

'] + break + + # Skip content before first chapter + if current_chapter_num == -1: + continue + + # Decode this glyph + glyph_key = str(glyph_id) + if glyph_key in char_mapping: + char = char_mapping[glyph_key]["character"] + else: + char = f"[{glyph_id}]" + + # Get formatting for this position + info = line_info.get(idx, {}) + font_style = info.get('font_style', 'normal') + font_weight = info.get('font_weight', 400) + font_size = info.get('font_size', 8.91) + has_link = info.get('has_link', False) + alignment = info.get('alignment', 'left') + + # Determine classes and inline styles needed + classes = [] + if font_style == 'italic': + classes.append('italic') + if font_weight >= 700: + classes.append('bold') + if has_link: + classes.append('link') + + # Add font size as inline style if it differs significantly from base (8.91pt) + font_size_style = '' + if abs(font_size - 8.91) > 1.0: # More than 1pt difference + # Convert to relative em size (base is 8pt in CSS) + em_size = font_size / 8.0 + font_size_style = f'font-size: {em_size:.2f}em' + + # If classes changed, close previous span and open new one + if classes != current_span_classes: + if current_span_classes: + chapter_contents[current_chapter_num].append('') + if classes: + class_attr = f' class="{" ".join(classes)}"' + style_attr = f' style="{font_size_style}"' if font_size_style else '' + chapter_contents[current_chapter_num].append(f'') + elif font_size_style: + # Font size change without class changes + chapter_contents[current_chapter_num].append(f'') + current_span_classes = classes + + # Add the character + chapter_contents[current_chapter_num].append(html.escape(char)) + + # Check if this is a line break position + if info.get('line_break', False): + # Detect bullet point context to keep bullets with their text + is_current_bullet = char in ['•', '◦', '●'] + + prev_is_bullet = False + for look_back in range(1, min(5, idx + 1)): + prev_char = char_mapping.get(str(all_glyphs[idx - look_back]), {}).get("character", "") + if prev_char in ['•', '◦', '●']: + prev_is_bullet = True + break + elif prev_char != ' ': + break + + if current_span_classes: + chapter_contents[current_chapter_num].append('') + current_span_classes = [] + + # Suppress line breaks after bullets to keep them with their text + if is_current_bullet or prev_is_bullet: + consecutive_line_breaks = 0 + else: + consecutive_line_breaks += 1 + + next_alignment = 'left' + if idx + 1 < len(all_glyphs): + next_info = line_info.get(idx + 1, {}) + next_alignment = next_info.get('alignment', 'left') + + classes = [] + if consecutive_line_breaks >= 2: + classes.append('break') + consecutive_line_breaks = 0 + if next_alignment in ['center', 'right', 'indent']: + classes.append(next_alignment) + + class_str = f' class="{" ".join(classes)}"' if classes else '' + chapter_contents[current_chapter_num].append(f'

\n') + else: + consecutive_line_breaks = 0 + + # Create EPUB chapters + print("Creating EPUB chapters...") + for toc_ch in toc_chapters: + ch_num = toc_ch['chapter_num'] + if ch_num in chapter_contents: + chapter_contents[ch_num].append('

') + + chapter = epub.EpubHtml( + title=toc_ch['label'], + file_name=f'chap_{ch_num:03d}.xhtml', + lang=metadata.get('lang', 'en') + ) + chapter.content = ''.join(chapter_contents[ch_num]) + chapter.add_item(default_css) + book.add_item(chapter) + chapters.append(chapter) + + # Define Table of Contents + book.toc = tuple(chapters) + + # Add navigation files + book.add_item(epub.EpubNcx()) + book.add_item(epub.EpubNav()) + + # Define spine + book.spine = ['nav'] + chapters + + # Save EPUB + output_file = Path("decoded_book.epub") + epub.write_epub(str(output_file), book) + + print(f"\nEPUB created successfully: {output_file}") + print(f"Total chapters: {len(chapters)}") + +if __name__ == "__main__": + main() diff --git a/decode_glyphs_complete.py b/decode_glyphs_complete.py new file mode 100644 index 0000000..8d06c54 --- /dev/null +++ b/decode_glyphs_complete.py @@ -0,0 +1,740 @@ +#!/usr/bin/env python3 +""" +Complete Glyph Decoding Pipeline + +This script combines hash-based glyph normalization with TTF character matching: +1. Hash-Based Normalization: Renders glyphs from all batches and groups by perceptual hash +2. TTF Matching: Matches unique glyphs to TTF characters using progressive SSIM + +Usage: + python3 decode_glyphs_complete.py [--fast] [--full] [--progressive] + +Options: + --fast Early exit on good SSIM matches + --full Check all characters in font (not just alphanumeric) + --progressive Use multi-stage filtering (32→64→128→256→512→1024px) +""" + +import json +import sys +from pathlib import Path +from collections import defaultdict, Counter +import io +from multiprocessing import Pool, cpu_count +import string +import time +import numpy as np + +try: + from PIL import Image, ImageOps + import imagehash + import cairosvg + from svgpathtools import parse_path + from fontTools.ttLib import TTFont + from fontTools.pens.svgPathPen import SVGPathPen + from fontTools.pens.boundsPen import BoundsPen + from fontTools.misc.transform import Transform + from fontTools.pens.transformPen import TransformPen + from skimage.metrics import structural_similarity as ssim + from tqdm import tqdm +except ImportError as e: + print("[!] Missing dependencies! Install with:") + print(" pip install pillow cairosvg imagehash svgpathtools fonttools scikit-image tqdm") + sys.exit(1) + + +# ============================================================================ +# PART 1: HASH-BASED GLYPH NORMALIZATION +# ============================================================================ + +class GlyphHasher: + """Renders SVG glyphs and computes perceptual hashes""" + + def __init__(self, size=128): + self.size = size + + def render_glyph(self, glyph_data): + """Render SVG path as filled shape""" + path_str = glyph_data.get('path', '') + if not path_str or path_str.strip() == '': + return None + + try: + # Parse path to get bounding box + path = parse_path(path_str) + if len(path) == 0: + return None + + xmin, xmax, ymin, ymax = path.bbox() + width = xmax - xmin + height = ymax - ymin + + if width == 0 or height == 0: + return None + + # Use font metrics for consistent viewbox + units_per_em = glyph_data.get('unitsPerEm', 1000) + ascent = glyph_data.get('ascent', 800) + descent = glyph_data.get('descent', -200) + + # Center glyph both horizontally and vertically + glyph_center_x = (xmin + xmax) / 2 + glyph_center_y = (ymin + ymax) / 2 + + half_width = units_per_em / 2 + font_height = ascent - descent + half_height = font_height / 2 + + viewbox_x = glyph_center_x - half_width + viewbox_y = glyph_center_y - half_height + viewbox = f"{viewbox_x} {viewbox_y} {units_per_em} {font_height}" + + # Create SVG document + svg = f''' + + +''' + + # Render using cairosvg + png_bytes = cairosvg.svg2png(bytestring=svg.encode('utf-8'), + output_width=self.size, + output_height=self.size) + + # Load as RGBA + img_rgba = Image.open(io.BytesIO(png_bytes)) + + # Create white background and composite + img = Image.new('L', (self.size, self.size), 255) + if img_rgba.mode == 'RGBA': + alpha = img_rgba.split()[3] + inverted = ImageOps.invert(alpha) + img.paste(0, mask=inverted) + + return img + + except Exception: + return None + + def compute_hash(self, img): + """Compute hash - use multiple hash types for better precision""" + if img is None: + return None + # Use average hash (more precise) + dhash (directional) for better uniqueness + # Larger hash_size = more precision, less chance of collisions + ahash = str(imagehash.average_hash(img, hash_size=16)) + dhash = str(imagehash.dhash(img, hash_size=16)) + return f"{ahash}_{dhash}" # Combine both for maximum uniqueness + + +def process_batch(args): + """Process a single batch (for multiprocessing)""" + book_dir, batch_num, save_images = args + batch_dir = Path(book_dir) / f'batch_{batch_num}' + glyphs_file = batch_dir / 'glyphs.json' + + if not glyphs_file.exists(): + return None + + hasher = GlyphHasher() + batch_results = { + 'batch_num': batch_num, + 'glyph_to_hash': {}, + 'glyphs_in_text': [], + 'images': {} + } + + # Load glyphs and compute hashes + with open(glyphs_file) as f: + glyph_data = json.load(f) + + for font_data in glyph_data: + font_family = font_data['fontFamily'] + glyphs = font_data.get('glyphs', {}) + + for glyph_id, glyph_info in glyphs.items(): + # Add font metrics + glyph_info['unitsPerEm'] = font_data.get('unitsPerEm', 1000) + glyph_info['ascent'] = font_data.get('ascent', 800) + glyph_info['descent'] = font_data.get('descent', -200) + + # Render and hash + img = hasher.render_glyph(glyph_info) + if img is not None: + phash = hasher.compute_hash(img) + batch_results['glyph_to_hash'][int(glyph_id)] = { + 'hash': phash, + 'font': font_family + } + + if save_images and phash not in batch_results['images']: + batch_results['images'][phash] = img + + # Load text and extract all glyph IDs used + page_files = sorted(batch_dir.glob('page_data_*.json')) + for page_file in page_files: + with open(page_file) as f: + pages = json.load(f) + + for page in pages: + for run in page.get('children', []): + if 'glyphs' in run: + batch_results['glyphs_in_text'].extend(run['glyphs']) + + return batch_results + + +def create_hash_mapping(book_dir): + """Phase 1: Create hash-based mapping of all glyphs""" + print(f"\n{'='*80}") + print(f"PHASE 1: HASH-BASED GLYPH NORMALIZATION") + print(f"{'='*80}\n") + print(f"Book directory: {book_dir}") + print(f"Using all {cpu_count()} CPU cores") + + # Find all batches + batch_dirs = [] + front_batches = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_front')]) + batch_dirs.extend(front_batches) + numbered_batches = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_') and not d.name.startswith('batch_front')], + key=lambda x: int(x.name.split('_')[1])) + batch_dirs.extend(numbered_batches) + batch_nums = list(range(len(batch_dirs))) + + print(f"\n[*] Found {len(batch_dirs)} batches") + + # Process all batches in parallel + print(f"[*] Processing batches (rendering all glyphs)...") + with Pool(cpu_count()) as pool: + batch_args = [(str(book_dir), batch_num, True) for batch_num in batch_nums] + results = list(pool.imap_unordered(process_batch, batch_args)) + + results = [r for r in results if r is not None] + print(f"[✓] Processed {len(results)} batches") + + # Build hash -> unique_id mapping + print(f"\n[*] Building hash-based mapping...") + hash_to_id = {} + hash_counter = 0 + hash_fonts = {} + hash_samples = {} + hash_images = {} + + for result in results: + batch_num = result['batch_num'] + + for phash, img in result.get('images', {}).items(): + if phash not in hash_images: + hash_images[phash] = img + + for local_glyph_id, glyph_info in result['glyph_to_hash'].items(): + phash = glyph_info['hash'] + font = glyph_info['font'] + + if phash not in hash_to_id: + hash_to_id[phash] = hash_counter + hash_fonts[hash_counter] = font + hash_samples[hash_counter] = (batch_num, local_glyph_id) + hash_counter += 1 + + print(f"[✓] Found {len(hash_to_id)} unique glyphs") + + # Verify no hash collisions + print(f"\n[*] Verifying hash uniqueness...") + from collections import Counter + hash_counts = Counter(hash_to_id.keys()) + collisions = {h: count for h, count in hash_counts.items() if count > 1} + + if collisions: + print(f"⚠ WARNING: Found {len(collisions)} hash collisions!") + print(f"This means some distinct glyphs are being merged together.") + print(f"First few collisions:") + for h, count in list(collisions.items())[:5]: + print(f" Hash {h}: {count} glyphs") + print(f"\n⚠ This will cause incorrect decoding. Please report this issue.") + else: + print(f"✓ No hash collisions - each glyph has a unique hash") + + # Normalize all text + print(f"\n[*] Normalizing all text...") + all_normalized_glyphs = [] + + for result in sorted(results, key=lambda r: r['batch_num']): + batch_mapping = {} + for local_glyph_id, glyph_info in result['glyph_to_hash'].items(): + phash = glyph_info['hash'] + batch_mapping[local_glyph_id] = hash_to_id[phash] + + for glyph_id in result['glyphs_in_text']: + unique_id = batch_mapping.get(glyph_id, -1) + all_normalized_glyphs.append(unique_id) + + print(f"[✓] Normalized {len(all_normalized_glyphs):,} glyphs") + + # Save results + output_dir = book_dir / 'hash_mapping' + output_dir.mkdir(exist_ok=True) + + hash_info = { + 'total_unique_glyphs': len(hash_to_id), + 'hash_to_id': hash_to_id, + 'id_to_font': {str(k): v for k, v in hash_fonts.items()}, + 'id_samples': {str(k): {'batch': v[0], 'glyph': v[1]} for k, v in hash_samples.items()} + } + + with open(output_dir / 'hash_info.json', 'w') as f: + json.dump(hash_info, f, indent=2) + + with open(output_dir / 'all_glyphs.json', 'w') as f: + json.dump(all_normalized_glyphs, f) + + # Save glyph images + images_dir = output_dir / 'glyph_images' + images_dir.mkdir(exist_ok=True) + + for phash, unique_id in hash_to_id.items(): + if phash in hash_images: + img = hash_images[phash] + font = hash_fonts[unique_id] + img.save(images_dir / f'id_{unique_id:03d}_{font}.png') + + print(f"[✓] Saved to {output_dir}/") + + # Show frequency + freq = Counter(all_normalized_glyphs) + print(f"\n[*] Top 20 most frequent glyphs:") + for unique_id, count in freq.most_common(20): + pct = count / len(all_normalized_glyphs) * 100 + font = hash_fonts.get(unique_id, 'unknown') + print(f" ID {unique_id:3d} ({font:12s}): {count:7,} ({pct:5.2f}%)") + + return output_dir, hash_info + + +# ============================================================================ +# PART 2: TTF CHARACTER MATCHING +# ============================================================================ + +def render_glyph_by_name(tt, glyph_name, size=128): + """Render a glyph by name from TTF""" + glyph_set = tt.getGlyphSet() + if glyph_name not in glyph_set: + return None + + glyph = glyph_set[glyph_name] + + # Get font metrics + head = tt['head'] + units_per_em = head.unitsPerEm + hhea = tt['hhea'] + ascent = hhea.ascent + descent = hhea.descent + + # Get bounding box + bounds_pen = BoundsPen(glyph_set) + glyph.draw(bounds_pen) + if bounds_pen.bounds is None: + return None + xmin, ymin, xmax, ymax = bounds_pen.bounds + + # Apply Y-flip to bbox + ymin_svg = -ymax + ymax_svg = -ymin + + # Extract SVG path with Y-flip + svg_pen = SVGPathPen(glyph_set) + transform_pen = TransformPen(svg_pen, Transform(1, 0, 0, -1, 0, 0)) + glyph.draw(transform_pen) + path_data = svg_pen.getCommands() + + if not path_data or path_data.strip() == '': + return None + + # Center glyph + glyph_center_x = (xmin + xmax) / 2 + glyph_center_y = (ymin_svg + ymax_svg) / 2 + font_height = ascent - descent + viewbox_x = glyph_center_x - units_per_em / 2 + viewbox_y = glyph_center_y - font_height / 2 + viewbox = f"{viewbox_x} {viewbox_y} {units_per_em} {font_height}" + + # Create SVG + svg = f''' + + +''' + + # Render + try: + png_bytes = cairosvg.svg2png( + bytestring=svg.encode('utf-8'), + output_width=size, + output_height=size + ) + img_rgba = Image.open(io.BytesIO(png_bytes)) + img = Image.new('L', (size, size), 255) + alpha = img_rgba.split()[3] + inverted = ImageOps.invert(alpha) + img.paste(0, mask=inverted) + return img + except Exception: + return None + + +def render_char_from_ttf(tt, char, size=128): + """Render a character from TTF""" + cmap = tt.getBestCmap() + if ord(char) not in cmap: + return None + glyph_name = cmap[ord(char)] + return render_glyph_by_name(tt, glyph_name, size) + + +def compare_images_ssim(img1, img2): + """Compare two images using SSIM. Returns distance (0=identical)""" + arr1 = np.array(img1) + arr2 = np.array(img2) + similarity = ssim(arr1, arr2) + distance = (1 - similarity) * 10 + return distance + + +def match_single_glyph(args): + """Match a single glyph (for parallel processing)""" + unique_id, glyph_images_dir, ttf_library_items, fast_mode, progressive_mode = args + + # Load Amazon glyph image + glyph_image_files = list(glyph_images_dir.glob(f'id_{unique_id:03d}_*.png')) + if not glyph_image_files: + return (unique_id, None, float('inf')) + + amazon_img = Image.open(glyph_image_files[0]) + + if not progressive_mode: + # Original single-pass approach + best_match = None + best_distance = float('inf') + early_exit_threshold = 0.05 if fast_mode else -1 + + for (char, font_name, style), ttf_img in ttf_library_items: + distance = compare_images_ssim(amazon_img, ttf_img) + if distance < best_distance: + best_distance = distance + best_match = (char, font_name, style) + + if fast_mode and distance <= early_exit_threshold: + break + + return (unique_id, best_match, best_distance) + + # Progressive resolution approach + # Stage 1: 128x128 - Quick filter + amazon_128 = amazon_img.resize((128, 128), Image.LANCZOS) + candidates_128 = [] + + for (char, font_name, style), ttf_img in ttf_library_items: + ttf_128 = ttf_img.resize((128, 128), Image.LANCZOS) + distance = compare_images_ssim(amazon_128, ttf_128) + candidates_128.append(((char, font_name, style), ttf_img, distance)) + + # Sort and keep top 30 candidates only + candidates_128.sort(key=lambda x: x[2]) + candidates_128 = candidates_128[:30] + + # Stage 2: 256x256 - Narrow down + amazon_256 = amazon_img.resize((256, 256), Image.LANCZOS) + candidates_256 = [] + + for (char, font_name, style), ttf_img, _ in candidates_128: + ttf_256 = ttf_img.resize((256, 256), Image.LANCZOS) + distance = compare_images_ssim(amazon_256, ttf_256) + candidates_256.append(((char, font_name, style), ttf_img, distance)) + + # Sort and keep top 10 + candidates_256.sort(key=lambda x: x[2]) + candidates_256 = candidates_256[:10] + + # Stage 3: 512x512 - Final decision + amazon_512 = amazon_img.resize((512, 512), Image.LANCZOS) + best_match = None + best_distance = float('inf') + + for (char, font_name, style), ttf_img, _ in candidates_256: + ttf_512 = ttf_img.resize((512, 512), Image.LANCZOS) + distance = compare_images_ssim(amazon_512, ttf_512) + if distance < best_distance: + best_distance = distance + best_match = (char, font_name, style) + + # Early exit if very confident + if distance < 0.05: + break + + return (unique_id, best_match, best_distance) + + +def match_ttf_characters(hash_mapping_dir, fast_mode, full_mode, progressive_mode): + """Phase 2: Match unique glyphs to TTF characters""" + print(f"\n{'='*80}") + print(f"PHASE 2: TTF CHARACTER MATCHING") + print(f"{'='*80}\n") + + hash_info_file = hash_mapping_dir / 'hash_info.json' + glyph_images_dir = hash_mapping_dir / 'glyph_images' + + # Load hash info + with open(hash_info_file) as f: + hash_info = json.load(f) + + id_to_font = {int(k): v for k, v in hash_info['id_to_font'].items()} + + # Find all font files (check multiple directories) + font_dirs = [Path('fonts'), Path('.')] + font_files = [] + for font_dir in font_dirs: + if font_dir.exists(): + font_files.extend(font_dir.glob('*.ttf')) + + font_files = sorted(set(font_files)) # Remove duplicates + print(f"Found {len(font_files)} font files") + + # Check which fonts we have vs what the book needs + found_font_names = {f.stem.lower() for f in font_files} + needed_fonts = set(id_to_font.values()) + missing_fonts = needed_fonts - found_font_names + + if missing_fonts: + print(f"\n⚠ WARNING: Book uses fonts not in font directory:") + for font in missing_fonts: + glyph_count = sum(1 for f in id_to_font.values() if f == font) + print(f" - {font}: {glyph_count} glyphs") + print(f"\nGlyphs using these fonts will be matched against available fonts (may be inaccurate)") + else: + print(f"✓ All required fonts available") + + # Characters to test + if full_mode: + chars_to_test = [] + print("Full mode: Will check ALL characters in font") + else: + # Standard ASCII characters + chars_to_test = string.ascii_letters + string.digits + string.punctuation + " " + + # Add common special characters that appear in books + special_chars = [ + '\u2022', # • BULLET + '\u2023', # ‣ TRIANGULAR BULLET + '\u2043', # ⁃ HYPHEN BULLET + '\u00B7', # · MIDDLE DOT + '\u25E6', # ◦ WHITE BULLET + '\u2219', # ∙ BULLET OPERATOR + '\u00A0', # Non-breaking space + '\u00A9', # © COPYRIGHT + '\u00AE', # ® REGISTERED + '\u2122', # ™ TRADEMARK + '\u00AB', # « LEFT DOUBLE ANGLE QUOTE + '\u00BB', # » RIGHT DOUBLE ANGLE QUOTE + '\u2018', # ' LEFT SINGLE QUOTE (already in ligatures but add anyway) + '\u2019', # ' RIGHT SINGLE QUOTE + '\u201A', # ‚ SINGLE LOW-9 QUOTE + '\u201B', # ‛ SINGLE HIGH-REVERSED-9 QUOTE + '\u2032', # ′ PRIME + '\u2033', # ″ DOUBLE PRIME + ] + chars_to_test += ''.join(special_chars) + print(f"Standard mode: Checking {len(chars_to_test)} predefined characters (including special chars)") + + # Ligatures and special glyphs + ligature_glyphs = { + 'f_f': 'ff', 'f_i': 'fi', 'f_l': 'fl', 'f_f_i': 'ffi', 'f_f_l': 'ffl', + 'uniFB00': 'ff', 'uniFB01': 'fi', 'uniFB02': 'fl', 'uniFB03': 'ffi', 'uniFB04': 'ffl', + 'space': ' ', + 'endash': chr(0x2013), 'emdash': chr(0x2014), + 'quotedblleft': chr(0x201C), 'quotedblright': chr(0x201D), + 'quoteleft': chr(0x2018), 'quoteright': chr(0x2019), + 'ellipsis': chr(0x2026), + } + + # Build TTF character library + print("=" * 60) + print("Building TTF character library...") + print("=" * 60) + + ttf_library = {} + + for font_path in font_files: + font_name = font_path.stem + print(f"\nProcessing: {font_name}") + + font_style = "normal" + if "Bold" in font_name and "Italic" in font_name: + font_style = "bold-italic" + elif "Bold" in font_name: + font_style = "bold" + elif "Italic" in font_name: + font_style = "italic" + + try: + tt = TTFont(font_path) + rendered_count = 0 + + if full_mode: + cmap = tt.getBestCmap() + if cmap: + for codepoint, glyph_name in cmap.items(): + char = chr(codepoint) + img = render_char_from_ttf(tt, char) + if img is not None: + ttf_library[(char, font_name, font_style)] = img + rendered_count += 1 + else: + for char in chars_to_test: + img = render_char_from_ttf(tt, char) + if img is not None: + ttf_library[(char, font_name, font_style)] = img + rendered_count += 1 + + # Render ligatures and special characters + glyph_set = tt.getGlyphSet() + for glyph_name, char in ligature_glyphs.items(): + if glyph_name in glyph_set: + img = render_glyph_by_name(tt, glyph_name) + if img is not None: + ttf_library[(char, font_name, font_style)] = img + rendered_count += 1 + + print(f" Rendered {rendered_count} glyphs") + + except Exception as e: + print(f" Error: {e}") + + print(f"\n[✓] TTF library built: {len(ttf_library)} glyphs") + + # Match glyphs + print("\n" + "=" * 60) + mode_parts = [] + if progressive_mode: + mode_parts.append("PROGRESSIVE MODE - 3-stage filtering (128→256→512px)") + elif fast_mode: + mode_parts.append("FAST MODE - early exit on good matches") + else: + mode_parts.append("FULL MODE - exhaustive search") + + print(f"Matching Amazon glyphs to TTF characters (using SSIM, {cpu_count()} threads)") + print(f"{' | '.join(mode_parts)}") + print("=" * 60) + + # Prepare arguments + ttf_library_items = list(ttf_library.items()) + glyph_ids = sorted(id_to_font.keys()) + args_list = [(gid, glyph_images_dir, ttf_library_items, fast_mode, progressive_mode) for gid in glyph_ids] + + # Process in parallel + matches = {} + no_match_count = 0 + + start_time = time.time() + with Pool(cpu_count()) as pool: + results = list(tqdm(pool.imap(match_single_glyph, args_list), total=len(args_list), desc="Matching glyphs")) + elapsed_time = time.time() - start_time + + for unique_id, best_match, best_distance in results: + if best_match and best_distance <= 1.0: + matches[unique_id] = (*best_match, best_distance) + # Highlight potential mismatches + if best_match[0] in [',', "'", '"', '`'] and best_distance > 0.3: + print(f"⚠ Glyph {unique_id:3d} → '{best_match[0]}' (distance={best_distance:.3f}, font={best_match[1]}) [UNCERTAIN]") + else: + print(f"✓ Glyph {unique_id:3d} → '{best_match[0]}' (distance={best_distance:.3f}, font={best_match[1]})") + else: + no_match_count += 1 + print(f"✗ Glyph {unique_id:3d} → NO MATCH (best distance={best_distance:.3f})") + + # Add special case for space + matches[-1] = (' ', 'special', 'normal', 0) + + print("\n" + "=" * 60) + print("RESULTS") + print("=" * 60) + print(f"Matched: {len(matches)-1}/{len(id_to_font)} glyphs ({100*(len(matches)-1)/len(id_to_font):.0f}%)") + print(f"No match: {no_match_count} glyphs") + print(f"Time taken: {elapsed_time:.2f} seconds") + print(f"\nUnmatched glyph IDs: {[k for k in sorted(id_to_font.keys()) if k not in matches]}") + + # Save mapping + output_file = Path('ttf_character_mapping.json') + mapping_output = { + str(glyph_id): { + "character": char, + "font": font, + "style": style, + "distance": dist + } + for glyph_id, (char, font, style, dist) in matches.items() + } + + with open(output_file, 'w', encoding='utf-8') as f: + json.dump(mapping_output, f, indent=2, ensure_ascii=False) + + print(f"\nMapping saved to: {output_file}") + + # Show character frequency + char_counts = defaultdict(int) + style_counts = defaultdict(int) + for char, _font, style, _dist in matches.values(): + char_counts[char] += 1 + style_counts[style] += 1 + + print("\nMost common matched characters:") + for char, count in sorted(char_counts.items(), key=lambda x: -x[1])[:20]: + print(f" '{char}': {count} glyphs") + + print("\nMatches by style:") + for style, count in sorted(style_counts.items()): + print(f" {style}: {count} glyphs") + + return output_file + + +# ============================================================================ +# MAIN +# ============================================================================ + +def main(): + if len(sys.argv) < 2: + print("Usage: python3 decode_glyphs_complete.py [--fast] [--full] [--progressive]") + print("\nOptions:") + print(" --fast Early exit on good SSIM matches") + print(" --full Check all characters in font (not just alphanumeric)") + print(" --progressive Use multi-stage filtering (32→64→128→256→512→1024px)") + sys.exit(1) + + fast_mode = "--fast" in sys.argv + full_mode = "--full" in sys.argv + progressive_mode = "--progressive" in sys.argv + book_dir = Path(sys.argv[1]) + + print(f"\n{'='*80}") + print(f"COMPLETE GLYPH DECODING PIPELINE") + print(f"{'='*80}") + print(f"\nBook: {book_dir}") + print(f"Options:") + print(f" Fast mode: {fast_mode}") + print(f" Full character set: {full_mode}") + print(f" Progressive matching: {progressive_mode}") + + # Phase 1: Hash-based normalization + hash_mapping_dir, hash_info = create_hash_mapping(book_dir) + + # Phase 2: TTF character matching + mapping_file = match_ttf_characters(hash_mapping_dir, fast_mode, full_mode, progressive_mode) + + print(f"\n{'='*80}") + print(f"[✓] COMPLETE PIPELINE FINISHED!") + print(f"{'='*80}") + print(f"\nOutputs:") + print(f" Hash mapping: {hash_mapping_dir}/") + print(f" Character mapping: {mapping_file}") + + +if __name__ == '__main__': + main() diff --git a/decoded_book.epub b/decoded_book.epub new file mode 100644 index 0000000..5367200 Binary files /dev/null and b/decoded_book.epub differ diff --git a/download_full_book.py b/download_full_book.py new file mode 100755 index 0000000..719b1a6 --- /dev/null +++ b/download_full_book.py @@ -0,0 +1,160 @@ +#!/usr/bin/env python3 +""" +Download complete book by downloading 5 pages at a time in a single session. +This ensures all pages share the same font/glyph encoding. + +Strategy: +1. Download from start position (includes TOC) - 5 pages at a time +2. Keep downloading until we reach the end +3. All downloads in ONE session so fonts match +4. Use TOC from first download to build glyph mapping +5. Decode all pages using that single mapping +""" +import json +import sys +from pathlib import Path +from downloader import KindleDownloader + +def main(): + if len(sys.argv) < 2: + print("Usage: python3 download_full_book.py [--yes]") + sys.exit(1) + + asin = sys.argv[1] + auto_confirm = '--yes' in sys.argv or '-y' in sys.argv + output_base = Path(f'downloads/{asin}') + output_base.mkdir(parents=True, exist_ok=True) + + # Load credentials + headers_file = Path('headers.json') + if not headers_file.exists(): + print("[✗] headers.json not found!") + sys.exit(1) + + with open(headers_file) as f: + headers_data = json.load(f) + + cookies = headers_data.get('cookies', '') + adp_token = headers_data['headers'].get('x-adp-session-token') if 'headers' in headers_data else None + + # Initialize downloader (single session for entire book) + print(f"\n{'='*80}") + print(f"DOWNLOADING COMPLETE BOOK: {asin}") + print(f"{'='*80}\n") + + downloader = KindleDownloader(cookies, adp_token) + + # Get book metadata + print("[*] Getting book metadata...") + metadata = downloader.start_reading(asin) + + title = metadata.get('deliveredAsin', asin) + revision = metadata.get('contentVersion', '') + start_pos = metadata.get('srl', 0) + + print(f"[*] Title: {title}") + print(f"[*] Revision: {revision}") + print(f"[*] Default start position (srl): {start_pos}") + print(f"[*] Downloading from position 0 to include front matter (TOC, cover, etc)") + + # Save karamelToken for image decryption + if 'karamelToken' in metadata: + karamel_token = { + 'token': metadata['karamelToken']['token'], + 'expiresAt': metadata['karamelToken']['expiresAt'] + } + token_file = output_base / 'karamel_token.json' + with open(token_file, 'w') as f: + json.dump(karamel_token, f, indent=2) + print(f"[✓] Saved karamelToken to {token_file}") + + # Download from position 0 to get the complete book including front matter + print(f"\n[*] Batch 0: position 0...") + first_tar = downloader.render_pages(asin, revision, start_position=0, num_pages=5) + first_files = downloader.extract_tar(first_tar, output_base / 'batch_0') + + # Get position range from batch 0 + page_data_file = list((output_base / 'batch_0').glob('page_data_*.json'))[0] + with open(page_data_file) as f: + first_pages = json.load(f) + + batch_0_start = first_pages[0]['startPositionId'] + batch_0_end = first_pages[-1]['endPositionId'] + print(f"[✓] Batch 0: {batch_0_start} to {batch_0_end} ({len(first_files)} files)") + + # Load TOC to estimate book length + toc_file = output_base / 'batch_0' / 'toc.json' + with open(toc_file) as f: + toc = json.load(f) + + last_toc_pos = max(entry['tocPositionId'] for entry in toc) + print(f"[*] Book ends around position {last_toc_pos}") + + # Estimate number of batches + positions_per_batch = batch_0_end - batch_0_start + estimated_batches = int((last_toc_pos - start_pos) / positions_per_batch) + 1 + + print(f"[*] Estimated {estimated_batches} batches needed (~{positions_per_batch} positions per 5 pages)") + print(f"\n[!] WARNING: This will download the entire book!") + print(f"[!] Estimated total: {estimated_batches * 5} pages") + + if not auto_confirm: + response = input(f"\nContinue? [y/N]: ") + if response.lower() != 'y': + print("[*] Aborted") + sys.exit(0) + else: + print("[*] Auto-confirmed with --yes flag") + + # Download remaining batches starting from where batch_0 ended + current_pos = batch_0_end + 1 + batch_num = 1 + + print(f"\n[*] Downloading remaining batches...") + + while current_pos < last_toc_pos: + try: + print(f"\n[*] Batch {batch_num}: position {current_pos}...") + tar_data = downloader.render_pages(asin, revision, start_position=current_pos, num_pages=5) + files = downloader.extract_tar(tar_data, output_base / f'batch_{batch_num}') + + # Get end position from this batch + page_file = list((output_base / f'batch_{batch_num}').glob('page_data_*.json'))[0] + with open(page_file) as f: + pages = json.load(f) + + if pages: + batch_end = pages[-1]['endPositionId'] + print(f"[✓] Batch {batch_num}: {pages[0]['startPositionId']} to {batch_end}") + current_pos = batch_end + 1 + else: + print(f"[!] Batch {batch_num}: No pages returned, stopping") + break + + batch_num += 1 + + except Exception as e: + print(f"[✗] Error downloading batch {batch_num}: {e}") + break + + print(f"\n{'='*80}") + print(f"[✓] DOWNLOAD COMPLETE") + print(f"[✓] Downloaded {batch_num} batches") + print(f"[✓] Saved to: {output_base}/") + print(f"{'='*80}\n") + + # Save download metadata + download_info = { + 'asin': asin, + 'revision': revision, + 'start_position': start_pos, + 'total_batches': batch_num, + 'pages_per_batch': 5, + 'estimated_positions': f'{start_pos} to {current_pos}' + } + + with open(output_base / 'download_info.json', 'w') as f: + json.dump(download_info, f, indent=2) + +if __name__ == '__main__': + main() diff --git a/downloader.py b/downloader.py new file mode 100644 index 0000000..7bd3bd9 --- /dev/null +++ b/downloader.py @@ -0,0 +1,282 @@ +#!/usr/bin/env python3 +""" +Kindle Book Downloader +Downloads raw page data from Kindle Cloud Reader (Stage 1) + +Usage: + python3 downloader.py [--pages N] [--output DIR] + +Example: + python3 downloader.py B0FLBTR2FS --pages 10 --output downloads/ +""" +import requests +import json +import tarfile +import io +import sys +import argparse +from pathlib import Path + +class KindleDownloader: + """Downloads raw encrypted book data from Kindle Cloud Reader""" + + def __init__(self, cookies_string, adp_session_token=None): + """ + Initialize with authentication credentials + + Args: + cookies_string: Cookie string from browser + adp_session_token: x-adp-session-token header value + """ + self.session = requests.Session() + + # Parse cookies + for cookie in cookies_string.split('; '): + if '=' in cookie: + name, value = cookie.split('=', 1) + self.session.cookies.set(name, value, domain='.amazon.com') + + self.adp_session_token = adp_session_token + self.rendering_token = None + self.token_expires = None + + def start_reading(self, asin): + """ + Initialize reading session and get rendering token + + Args: + asin: Book ASIN + + Returns: + dict: Book metadata including token, revision, srl + """ + url = 'https://read.amazon.com/service/mobile/reader/startReading' + params = { + 'asin': asin, + 'clientVersion': '20000100' + } + + headers = {} + if self.adp_session_token: + headers['x-adp-session-token'] = self.adp_session_token + + print(f"[*] Requesting reading session for {asin}...") + response = self.session.get(url, params=params, headers=headers) + response.raise_for_status() + + data = response.json() + + # Store token + if 'karamelToken' in data: + self.rendering_token = data['karamelToken']['token'] + self.token_expires = data['karamelToken']['expiresAt'] + print(f"[✓] Got rendering token (expires: {self.token_expires})") + + return data + + def render_pages(self, asin, revision, start_position=0, num_pages=2): + """ + Download raw page data from Kindle renderer + + Args: + asin: Book ASIN + revision: Content revision ID + start_position: Starting position ID + num_pages: Number of pages to fetch + + Returns: + bytes: Raw TAR archive containing page data + """ + url = 'https://read.amazon.com/renderer/render' + + params = { + 'version': '3.0', + 'asin': asin, + 'contentType': 'FullBook', + 'revision': revision, + 'fontFamily': 'Bookerly', + 'fontSize': '8.91', + 'lineHeight': '1.4', + 'dpi': '160', + 'height': '1600', + 'width': '1000', + 'marginBottom': '0', + 'marginLeft': '9', + 'marginRight': '9', + 'marginTop': '0', + 'maxNumberColumns': '1', + 'theme': 'dark', + 'locationMap': 'false', + 'packageType': 'TAR', + 'encryptionVersion': 'NONE', + 'numPage': str(num_pages), + 'skipPageCount': '0', + 'startingPosition': str(start_position), + 'bundleImages': 'false' + } + + headers = { + 'x-amz-rendering-token': self.rendering_token + } + + print(f"[*] Downloading {num_pages} pages from position {start_position}...") + response = self.session.get(url, params=params, headers=headers) + + if response.status_code != 200: + print(f"[✗] Error {response.status_code}: {response.text[:200]}") + + response.raise_for_status() + + return response.content + + def extract_tar(self, tar_bytes, output_dir): + """ + Extract TAR archive to directory + + Args: + tar_bytes: Raw TAR data + output_dir: Directory to extract to + + Returns: + list: Names of extracted files + """ + output_path = Path(output_dir) + output_path.mkdir(parents=True, exist_ok=True) + + extracted_files = [] + + with tarfile.open(fileobj=io.BytesIO(tar_bytes)) as tar: + for member in tar.getmembers(): + if member.isfile(): + content = tar.extractfile(member).read() + file_path = output_path / member.name + # Create parent directories if they don't exist + file_path.parent.mkdir(parents=True, exist_ok=True) + file_path.write_bytes(content) + extracted_files.append(member.name) + + return extracted_files + + def download(self, asin, num_pages=2, output_dir=None): + """ + Download book pages and save raw data + + Args: + asin: Book ASIN + num_pages: Number of pages to download + output_dir: Output directory (default: downloads//) + + Returns: + dict: Download metadata + """ + print(f"\n{'='*80}") + print(f"KINDLE DOWNLOADER") + print(f"{'='*80}\n") + + # Get metadata and token + metadata = self.start_reading(asin) + + title = metadata.get('deliveredAsin', asin) + revision = metadata.get('contentVersion', '') + srl = metadata.get('srl', 0) + + print(f"[*] ASIN: {title}") + print(f"[*] Revision: {revision}") + print(f"[*] SRL (start position): {srl}") + + # Download pages + tar_data = self.render_pages(asin, revision, start_position=srl, num_pages=num_pages) + print(f"[✓] Downloaded {len(tar_data)} bytes") + + # Extract to directory + if output_dir is None: + output_dir = f"downloads/{asin}" + + print(f"[*] Extracting to {output_dir}/...") + extracted_files = self.extract_tar(tar_data, output_dir) + print(f"[✓] Extracted {len(extracted_files)} files:") + for filename in extracted_files: + print(f" - {filename}") + + # Save metadata + metadata_file = Path(output_dir) / 'download_metadata.json' + download_info = { + 'asin': asin, + 'revision': revision, + 'srl': srl, + 'start_position': srl, + 'num_pages': num_pages, + 'extracted_files': extracted_files + } + metadata_file.write_text(json.dumps(download_info, indent=2)) + print(f"[✓] Saved metadata to {metadata_file}") + + print(f"\n{'='*80}") + print(f"[✓] DOWNLOAD COMPLETE") + print(f"[✓] Data saved to: {output_dir}/") + print(f"{'='*80}\n") + + return download_info + +def main(): + parser = argparse.ArgumentParser( + description='Download raw page data from Kindle Cloud Reader', + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=""" +Examples: + python3 downloader.py B0FLBTR2FS + python3 downloader.py B0FLBTR2FS --pages 10 + python3 downloader.py B0FLBTR2FS --output my_books/ + """ + ) + parser.add_argument('asin', help='Book ASIN to download') + parser.add_argument('--pages', type=int, default=2, help='Number of pages to download (default: 2)') + parser.add_argument('--output', help='Output directory (default: downloads//)') + parser.add_argument('--start-position', type=int, help='Override start position (default: use SRL from metadata)') + + args = parser.parse_args() + + # Load credentials from headers.json + headers_file = Path('headers.json') + if not headers_file.exists(): + print("[✗] headers.json not found!") + print("\nCreate headers.json with:") + print(' {') + print(' "headers": {"x-adp-session-token": "..."},') + print(' "cookies": "session-id=...; ..."') + print(' }') + sys.exit(1) + + with open(headers_file) as f: + headers_data = json.load(f) + + cookies = headers_data.get('cookies', '') + if not cookies: + print("[✗] No cookies found in headers.json!") + sys.exit(1) + + adp_token = None + if 'headers' in headers_data: + adp_token = headers_data['headers'].get('x-adp-session-token') + + # Download + downloader = KindleDownloader(cookies, adp_token) + + # Override start position if specified + if args.start_position is not None: + metadata = downloader.start_reading(args.asin) + revision = metadata.get('contentVersion', '') + + # Download from custom position + tar_data = downloader.render_pages(args.asin, revision, start_position=args.start_position, num_pages=args.pages) + + # Extract + output_dir = args.output or f"downloads/{args.asin}" + print(f"[*] Extracting to {output_dir}/...") + extracted_files = downloader.extract_tar(tar_data, output_dir) + print(f"[✓] Extracted {len(extracted_files)} files") + else: + downloader.download(args.asin, num_pages=args.pages, output_dir=args.output) + +if __name__ == '__main__': + main() diff --git a/fonts/Bookerly Bold Italic.ttf b/fonts/Bookerly Bold Italic.ttf new file mode 100644 index 0000000..1da55cb Binary files /dev/null and b/fonts/Bookerly Bold Italic.ttf differ diff --git a/fonts/Bookerly Bold.ttf b/fonts/Bookerly Bold.ttf new file mode 100644 index 0000000..0674551 Binary files /dev/null and b/fonts/Bookerly Bold.ttf differ diff --git a/fonts/Bookerly Display Bold Italic.ttf b/fonts/Bookerly Display Bold Italic.ttf new file mode 100644 index 0000000..db79105 Binary files /dev/null and b/fonts/Bookerly Display Bold Italic.ttf differ diff --git a/fonts/Bookerly Display Bold.ttf b/fonts/Bookerly Display Bold.ttf new file mode 100644 index 0000000..dc4e207 Binary files /dev/null and b/fonts/Bookerly Display Bold.ttf differ diff --git a/fonts/Bookerly Display Italic.ttf b/fonts/Bookerly Display Italic.ttf new file mode 100644 index 0000000..7470fef Binary files /dev/null and b/fonts/Bookerly Display Italic.ttf differ diff --git a/fonts/Bookerly Display.ttf b/fonts/Bookerly Display.ttf new file mode 100644 index 0000000..b3b634b Binary files /dev/null and b/fonts/Bookerly Display.ttf differ diff --git a/fonts/Bookerly Italic.ttf b/fonts/Bookerly Italic.ttf new file mode 100644 index 0000000..68c9a81 Binary files /dev/null and b/fonts/Bookerly Italic.ttf differ diff --git a/fonts/Bookerly LCD Italic.ttf b/fonts/Bookerly LCD Italic.ttf new file mode 100644 index 0000000..55adbf5 Binary files /dev/null and b/fonts/Bookerly LCD Italic.ttf differ diff --git a/fonts/Bookerly LCD Light Italic.ttf b/fonts/Bookerly LCD Light Italic.ttf new file mode 100644 index 0000000..a433c78 Binary files /dev/null and b/fonts/Bookerly LCD Light Italic.ttf differ diff --git a/fonts/Bookerly Light Italic.ttf b/fonts/Bookerly Light Italic.ttf new file mode 100644 index 0000000..a433c78 Binary files /dev/null and b/fonts/Bookerly Light Italic.ttf differ diff --git a/fonts/Bookerly Light.ttf b/fonts/Bookerly Light.ttf new file mode 100644 index 0000000..c309853 Binary files /dev/null and b/fonts/Bookerly Light.ttf differ diff --git a/fonts/Bookerly.ttf b/fonts/Bookerly.ttf new file mode 100644 index 0000000..62350dd Binary files /dev/null and b/fonts/Bookerly.ttf differ