first commit
This commit is contained in:
commit
4f5f71bdaf
18 changed files with 1590 additions and 0 deletions
7
.gitignore
vendored
Normal file
7
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
/BLOG.MD
|
||||
/__pycache__
|
||||
/archive
|
||||
/downloads
|
||||
/headers.json
|
||||
/renderer.js
|
||||
/ttf_character_mapping.json
|
||||
401
create_epub.py
Normal file
401
create_epub.py
Normal file
|
|
@ -0,0 +1,401 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Create an EPUB file from the decoded Amazon book data with proper formatting.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
import sys
|
||||
from ebooklib import epub
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 create_epub_new.py <book_dir>")
|
||||
sys.exit(1)
|
||||
|
||||
book_dir = Path(sys.argv[1])
|
||||
|
||||
# Load the TTF character mapping
|
||||
mapping_file = Path("ttf_character_mapping.json")
|
||||
if not mapping_file.exists():
|
||||
print(f"Mapping file not found: {mapping_file}")
|
||||
print("Run match_ttf_to_glyphs.py first!")
|
||||
return
|
||||
|
||||
with open(mapping_file) as f:
|
||||
char_mapping = json.load(f)
|
||||
|
||||
print(f"Loaded character mapping: {len(char_mapping)} glyphs")
|
||||
|
||||
# Load metadata
|
||||
metadata_file = book_dir / 'batch_0' / 'metadata.json'
|
||||
with open(metadata_file) as f:
|
||||
metadata = json.load(f)
|
||||
|
||||
# Load TOC
|
||||
toc_file = book_dir / 'batch_0' / 'toc.json'
|
||||
with open(toc_file) as f:
|
||||
toc_data = json.load(f)
|
||||
|
||||
print(f"Book: {metadata['bookTitle']}")
|
||||
print(f"Author: {metadata['authors'][0]}")
|
||||
print(f"TOC entries: {len(toc_data)}")
|
||||
|
||||
# Load all_glyphs
|
||||
all_glyphs_file = book_dir / 'hash_mapping' / 'all_glyphs.json'
|
||||
if not all_glyphs_file.exists():
|
||||
print(f"Book file not found: {all_glyphs_file}")
|
||||
return
|
||||
|
||||
with open(all_glyphs_file) as f:
|
||||
all_glyphs = json.load(f)
|
||||
|
||||
print(f"Loaded {len(all_glyphs)} glyphs from all_glyphs.json")
|
||||
|
||||
# Build line ending info (where newlines go) - same as decode_book_with_newlines.py
|
||||
print("Building line ending positions...")
|
||||
batch_dirs = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_')],
|
||||
key=lambda x: int(x.name.split('_')[1]))
|
||||
|
||||
line_info = {} # Index in all_glyphs -> formatting info
|
||||
current_index = 0
|
||||
|
||||
# Get page dimensions from actual page data
|
||||
first_page_file = batch_dirs[0] / sorted(batch_dirs[0].glob('page_data_*.json'))[0].name
|
||||
with open(first_page_file) as f:
|
||||
first_page_data = json.load(f)
|
||||
page_width = first_page_data[0]['width']
|
||||
page_height = first_page_data[0]['height']
|
||||
|
||||
print(f"Page dimensions: {page_width}x{page_height}")
|
||||
|
||||
prev_y = None # Track Y coordinate to detect line breaks
|
||||
|
||||
for batch_dir in batch_dirs:
|
||||
page_files = sorted(batch_dir.glob('page_data_*.json'))
|
||||
for page_file in page_files:
|
||||
with open(page_file) as f:
|
||||
pages = json.load(f)
|
||||
|
||||
for page in pages:
|
||||
for run in page.get('children', []):
|
||||
if 'glyphs' not in run:
|
||||
continue
|
||||
|
||||
num_glyphs = len(run['glyphs'])
|
||||
|
||||
# Extract formatting info with transform applied
|
||||
rect = run.get('rect', {})
|
||||
transform = run.get('transform', [1, 0, 0, 1, 0, 0])
|
||||
tx = transform[4] if len(transform) >= 6 else 0
|
||||
ty = transform[5] if len(transform) >= 6 else 0
|
||||
|
||||
left = rect.get('left', 0) + tx
|
||||
right = rect.get('right', 0) + tx
|
||||
top = rect.get('top', 0) + ty
|
||||
font_style = run.get('fontStyle', 'normal')
|
||||
font_weight = run.get('fontWeight', 400)
|
||||
font_size = run.get('fontSize', 8.91) # Default from downloader.py
|
||||
has_link = 'link' in run
|
||||
|
||||
# Detect alignment type using relative thresholds
|
||||
center = (left + right) / 2
|
||||
page_center = page_width / 2
|
||||
text_width = right - left
|
||||
alignment = 'left'
|
||||
|
||||
# Use relative thresholds based on page width
|
||||
center_tolerance = page_width * 0.05 # 5% of page width
|
||||
edge_tolerance = page_width * 0.05 # 5% tolerance for edges
|
||||
min_side_margin = page_width * 0.1 # 10% margin on each side for center
|
||||
min_left_margin_right = page_width * 0.2 # 20% left margin for right-align
|
||||
min_indent = page_width * 0.05 # 5% indent
|
||||
max_indent = page_width * 0.15 # 15% max for paragraph indent
|
||||
min_text_width = page_width * 0.3 # 30% minimum text width
|
||||
|
||||
# Check if centered: text center near page center AND margins on both sides
|
||||
if abs(center - page_center) < center_tolerance and left > min_side_margin and (page_width - right) > min_side_margin:
|
||||
alignment = 'center'
|
||||
# Check if right-aligned: close to right edge with significant left margin
|
||||
elif abs(right - page_width) < edge_tolerance and left > min_left_margin_right:
|
||||
alignment = 'right'
|
||||
# Check for indented paragraphs: moderate left margin with substantial text
|
||||
elif min_indent < left < max_indent and text_width > min_text_width:
|
||||
alignment = 'indent'
|
||||
|
||||
# Determine if this is a new line (Y coordinate changed significantly)
|
||||
is_new_line = prev_y is None or abs(top - prev_y) > 5
|
||||
|
||||
# Store info for each glyph position in this run
|
||||
for i in range(num_glyphs):
|
||||
line_info[current_index + i] = {
|
||||
'font_style': font_style,
|
||||
'font_weight': font_weight,
|
||||
'font_size': font_size,
|
||||
'has_link': has_link,
|
||||
'left': left,
|
||||
'alignment': alignment
|
||||
}
|
||||
|
||||
# Only mark line break if this run is on a NEW line
|
||||
if is_new_line and current_index > 0:
|
||||
# Mark line break at the END of the PREVIOUS run
|
||||
line_info[current_index - 1]['line_break'] = True
|
||||
|
||||
current_index += num_glyphs
|
||||
prev_y = top
|
||||
|
||||
print(f"Processed {current_index} glyphs with line break info")
|
||||
|
||||
# Create EPUB
|
||||
print("Creating EPUB...")
|
||||
book = epub.EpubBook()
|
||||
|
||||
# Set metadata
|
||||
book.set_identifier(metadata.get('asin', 'unknown'))
|
||||
book.set_title(metadata['bookTitle'])
|
||||
book.set_language(metadata.get('lang', 'en'))
|
||||
|
||||
for author in metadata.get('authors', ['Unknown']):
|
||||
book.add_author(author)
|
||||
|
||||
|
||||
# Add CSS for styling - match Kindle rendering parameters
|
||||
# Based on downloader.py: fontFamily='Bookerly', fontSize='8.91', lineHeight='1.4'
|
||||
style = '''
|
||||
body {
|
||||
font-family: Bookerly, Georgia, serif;
|
||||
font-size: 8pt;
|
||||
line-height: 1.0;
|
||||
margin: 0 auto;
|
||||
padding: 0;
|
||||
max-width: 1000px;
|
||||
background-color: #ffffff;
|
||||
color: #000000;
|
||||
}
|
||||
p {
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
line-height: 1.0;
|
||||
}
|
||||
p.center {
|
||||
text-align: center;
|
||||
}
|
||||
p.right {
|
||||
text-align: right;
|
||||
}
|
||||
p.indent {
|
||||
text-indent: 2em;
|
||||
}
|
||||
p.break {
|
||||
margin-top: 0.8em;
|
||||
}
|
||||
.italic { font-style: italic; }
|
||||
.bold { font-weight: bold; }
|
||||
.link {
|
||||
color: #0066cc;
|
||||
text-decoration: underline;
|
||||
}
|
||||
h1 {
|
||||
font-size: 1.8em;
|
||||
margin: 1em 0 0.5em 0;
|
||||
font-weight: bold;
|
||||
}
|
||||
h2 {
|
||||
font-size: 1.4em;
|
||||
margin: 0.8em 0 0.4em 0;
|
||||
font-weight: bold;
|
||||
}
|
||||
'''
|
||||
|
||||
default_css = epub.EpubItem(
|
||||
uid="style_default",
|
||||
file_name="style/default.css",
|
||||
media_type="text/css",
|
||||
content=style
|
||||
)
|
||||
book.add_item(default_css)
|
||||
|
||||
# Map position IDs to glyph indices
|
||||
print("Mapping TOC positions to glyph indices...")
|
||||
position_to_glyph_idx = {}
|
||||
current_glyph_idx = 0
|
||||
|
||||
for batch_dir in batch_dirs:
|
||||
page_files = sorted(batch_dir.glob('page_data_*.json'))
|
||||
for page_file in page_files:
|
||||
with open(page_file) as f:
|
||||
pages = json.load(f)
|
||||
|
||||
for page in pages:
|
||||
for run in page.get('children', []):
|
||||
if 'glyphs' not in run:
|
||||
continue
|
||||
|
||||
# Check if this run has position info
|
||||
start_pos_id = run.get('startPositionId')
|
||||
if start_pos_id is not None:
|
||||
position_to_glyph_idx[start_pos_id] = current_glyph_idx
|
||||
|
||||
current_glyph_idx += len(run['glyphs'])
|
||||
|
||||
# Map TOC entries to glyph indices
|
||||
toc_chapters = []
|
||||
for i, toc_entry in enumerate(toc_data):
|
||||
pos_id = toc_entry['tocPositionId']
|
||||
if pos_id in position_to_glyph_idx:
|
||||
toc_chapters.append({
|
||||
'label': toc_entry['label'],
|
||||
'glyph_idx': position_to_glyph_idx[pos_id],
|
||||
'chapter_num': i
|
||||
})
|
||||
|
||||
print(f"Found {len(toc_chapters)} TOC entries with positions")
|
||||
|
||||
# Build chapters based on TOC structure
|
||||
print("Building chapters with formatting...")
|
||||
import html
|
||||
chapters = []
|
||||
chapter_contents = {} # chapter_num -> content list
|
||||
current_chapter_num = -1 # Start before first chapter
|
||||
current_span_classes = []
|
||||
consecutive_line_breaks = 0
|
||||
|
||||
for idx, glyph_id in enumerate(all_glyphs):
|
||||
# Check if we're at a new chapter start
|
||||
for toc_ch in toc_chapters:
|
||||
if toc_ch['glyph_idx'] == idx:
|
||||
current_chapter_num = toc_ch['chapter_num']
|
||||
if current_chapter_num not in chapter_contents:
|
||||
chapter_contents[current_chapter_num] = ['<p>']
|
||||
break
|
||||
|
||||
# Skip content before first chapter
|
||||
if current_chapter_num == -1:
|
||||
continue
|
||||
|
||||
# Decode this glyph
|
||||
glyph_key = str(glyph_id)
|
||||
if glyph_key in char_mapping:
|
||||
char = char_mapping[glyph_key]["character"]
|
||||
else:
|
||||
char = f"[{glyph_id}]"
|
||||
|
||||
# Get formatting for this position
|
||||
info = line_info.get(idx, {})
|
||||
font_style = info.get('font_style', 'normal')
|
||||
font_weight = info.get('font_weight', 400)
|
||||
font_size = info.get('font_size', 8.91)
|
||||
has_link = info.get('has_link', False)
|
||||
alignment = info.get('alignment', 'left')
|
||||
|
||||
# Determine classes and inline styles needed
|
||||
classes = []
|
||||
if font_style == 'italic':
|
||||
classes.append('italic')
|
||||
if font_weight >= 700:
|
||||
classes.append('bold')
|
||||
if has_link:
|
||||
classes.append('link')
|
||||
|
||||
# Add font size as inline style if it differs significantly from base (8.91pt)
|
||||
font_size_style = ''
|
||||
if abs(font_size - 8.91) > 1.0: # More than 1pt difference
|
||||
# Convert to relative em size (base is 8pt in CSS)
|
||||
em_size = font_size / 8.0
|
||||
font_size_style = f'font-size: {em_size:.2f}em'
|
||||
|
||||
# If classes changed, close previous span and open new one
|
||||
if classes != current_span_classes:
|
||||
if current_span_classes:
|
||||
chapter_contents[current_chapter_num].append('</span>')
|
||||
if classes:
|
||||
class_attr = f' class="{" ".join(classes)}"'
|
||||
style_attr = f' style="{font_size_style}"' if font_size_style else ''
|
||||
chapter_contents[current_chapter_num].append(f'<span{class_attr}{style_attr}>')
|
||||
elif font_size_style:
|
||||
# Font size change without class changes
|
||||
chapter_contents[current_chapter_num].append(f'<span style="{font_size_style}">')
|
||||
current_span_classes = classes
|
||||
|
||||
# Add the character
|
||||
chapter_contents[current_chapter_num].append(html.escape(char))
|
||||
|
||||
# Check if this is a line break position
|
||||
if info.get('line_break', False):
|
||||
# Detect bullet point context to keep bullets with their text
|
||||
is_current_bullet = char in ['•', '◦', '●']
|
||||
|
||||
prev_is_bullet = False
|
||||
for look_back in range(1, min(5, idx + 1)):
|
||||
prev_char = char_mapping.get(str(all_glyphs[idx - look_back]), {}).get("character", "")
|
||||
if prev_char in ['•', '◦', '●']:
|
||||
prev_is_bullet = True
|
||||
break
|
||||
elif prev_char != ' ':
|
||||
break
|
||||
|
||||
if current_span_classes:
|
||||
chapter_contents[current_chapter_num].append('</span>')
|
||||
current_span_classes = []
|
||||
|
||||
# Suppress line breaks after bullets to keep them with their text
|
||||
if is_current_bullet or prev_is_bullet:
|
||||
consecutive_line_breaks = 0
|
||||
else:
|
||||
consecutive_line_breaks += 1
|
||||
|
||||
next_alignment = 'left'
|
||||
if idx + 1 < len(all_glyphs):
|
||||
next_info = line_info.get(idx + 1, {})
|
||||
next_alignment = next_info.get('alignment', 'left')
|
||||
|
||||
classes = []
|
||||
if consecutive_line_breaks >= 2:
|
||||
classes.append('break')
|
||||
consecutive_line_breaks = 0
|
||||
if next_alignment in ['center', 'right', 'indent']:
|
||||
classes.append(next_alignment)
|
||||
|
||||
class_str = f' class="{" ".join(classes)}"' if classes else ''
|
||||
chapter_contents[current_chapter_num].append(f'</p>\n<p{class_str}>')
|
||||
else:
|
||||
consecutive_line_breaks = 0
|
||||
|
||||
# Create EPUB chapters
|
||||
print("Creating EPUB chapters...")
|
||||
for toc_ch in toc_chapters:
|
||||
ch_num = toc_ch['chapter_num']
|
||||
if ch_num in chapter_contents:
|
||||
chapter_contents[ch_num].append('</p>')
|
||||
|
||||
chapter = epub.EpubHtml(
|
||||
title=toc_ch['label'],
|
||||
file_name=f'chap_{ch_num:03d}.xhtml',
|
||||
lang=metadata.get('lang', 'en')
|
||||
)
|
||||
chapter.content = ''.join(chapter_contents[ch_num])
|
||||
chapter.add_item(default_css)
|
||||
book.add_item(chapter)
|
||||
chapters.append(chapter)
|
||||
|
||||
# Define Table of Contents
|
||||
book.toc = tuple(chapters)
|
||||
|
||||
# Add navigation files
|
||||
book.add_item(epub.EpubNcx())
|
||||
book.add_item(epub.EpubNav())
|
||||
|
||||
# Define spine
|
||||
book.spine = ['nav'] + chapters
|
||||
|
||||
# Save EPUB
|
||||
output_file = Path("decoded_book.epub")
|
||||
epub.write_epub(str(output_file), book)
|
||||
|
||||
print(f"\nEPUB created successfully: {output_file}")
|
||||
print(f"Total chapters: {len(chapters)}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
740
decode_glyphs_complete.py
Normal file
740
decode_glyphs_complete.py
Normal file
|
|
@ -0,0 +1,740 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Complete Glyph Decoding Pipeline
|
||||
|
||||
This script combines hash-based glyph normalization with TTF character matching:
|
||||
1. Hash-Based Normalization: Renders glyphs from all batches and groups by perceptual hash
|
||||
2. TTF Matching: Matches unique glyphs to TTF characters using progressive SSIM
|
||||
|
||||
Usage:
|
||||
python3 decode_glyphs_complete.py <book_dir> [--fast] [--full] [--progressive]
|
||||
|
||||
Options:
|
||||
--fast Early exit on good SSIM matches
|
||||
--full Check all characters in font (not just alphanumeric)
|
||||
--progressive Use multi-stage filtering (32→64→128→256→512→1024px)
|
||||
"""
|
||||
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from collections import defaultdict, Counter
|
||||
import io
|
||||
from multiprocessing import Pool, cpu_count
|
||||
import string
|
||||
import time
|
||||
import numpy as np
|
||||
|
||||
try:
|
||||
from PIL import Image, ImageOps
|
||||
import imagehash
|
||||
import cairosvg
|
||||
from svgpathtools import parse_path
|
||||
from fontTools.ttLib import TTFont
|
||||
from fontTools.pens.svgPathPen import SVGPathPen
|
||||
from fontTools.pens.boundsPen import BoundsPen
|
||||
from fontTools.misc.transform import Transform
|
||||
from fontTools.pens.transformPen import TransformPen
|
||||
from skimage.metrics import structural_similarity as ssim
|
||||
from tqdm import tqdm
|
||||
except ImportError as e:
|
||||
print("[!] Missing dependencies! Install with:")
|
||||
print(" pip install pillow cairosvg imagehash svgpathtools fonttools scikit-image tqdm")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# PART 1: HASH-BASED GLYPH NORMALIZATION
|
||||
# ============================================================================
|
||||
|
||||
class GlyphHasher:
|
||||
"""Renders SVG glyphs and computes perceptual hashes"""
|
||||
|
||||
def __init__(self, size=128):
|
||||
self.size = size
|
||||
|
||||
def render_glyph(self, glyph_data):
|
||||
"""Render SVG path as filled shape"""
|
||||
path_str = glyph_data.get('path', '')
|
||||
if not path_str or path_str.strip() == '':
|
||||
return None
|
||||
|
||||
try:
|
||||
# Parse path to get bounding box
|
||||
path = parse_path(path_str)
|
||||
if len(path) == 0:
|
||||
return None
|
||||
|
||||
xmin, xmax, ymin, ymax = path.bbox()
|
||||
width = xmax - xmin
|
||||
height = ymax - ymin
|
||||
|
||||
if width == 0 or height == 0:
|
||||
return None
|
||||
|
||||
# Use font metrics for consistent viewbox
|
||||
units_per_em = glyph_data.get('unitsPerEm', 1000)
|
||||
ascent = glyph_data.get('ascent', 800)
|
||||
descent = glyph_data.get('descent', -200)
|
||||
|
||||
# Center glyph both horizontally and vertically
|
||||
glyph_center_x = (xmin + xmax) / 2
|
||||
glyph_center_y = (ymin + ymax) / 2
|
||||
|
||||
half_width = units_per_em / 2
|
||||
font_height = ascent - descent
|
||||
half_height = font_height / 2
|
||||
|
||||
viewbox_x = glyph_center_x - half_width
|
||||
viewbox_y = glyph_center_y - half_height
|
||||
viewbox = f"{viewbox_x} {viewbox_y} {units_per_em} {font_height}"
|
||||
|
||||
# Create SVG document
|
||||
svg = f'''<?xml version="1.0" encoding="UTF-8"?>
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="{viewbox}" width="{self.size}" height="{self.size}">
|
||||
<path d="{path_str}" fill="black"/>
|
||||
</svg>'''
|
||||
|
||||
# Render using cairosvg
|
||||
png_bytes = cairosvg.svg2png(bytestring=svg.encode('utf-8'),
|
||||
output_width=self.size,
|
||||
output_height=self.size)
|
||||
|
||||
# Load as RGBA
|
||||
img_rgba = Image.open(io.BytesIO(png_bytes))
|
||||
|
||||
# Create white background and composite
|
||||
img = Image.new('L', (self.size, self.size), 255)
|
||||
if img_rgba.mode == 'RGBA':
|
||||
alpha = img_rgba.split()[3]
|
||||
inverted = ImageOps.invert(alpha)
|
||||
img.paste(0, mask=inverted)
|
||||
|
||||
return img
|
||||
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def compute_hash(self, img):
|
||||
"""Compute hash - use multiple hash types for better precision"""
|
||||
if img is None:
|
||||
return None
|
||||
# Use average hash (more precise) + dhash (directional) for better uniqueness
|
||||
# Larger hash_size = more precision, less chance of collisions
|
||||
ahash = str(imagehash.average_hash(img, hash_size=16))
|
||||
dhash = str(imagehash.dhash(img, hash_size=16))
|
||||
return f"{ahash}_{dhash}" # Combine both for maximum uniqueness
|
||||
|
||||
|
||||
def process_batch(args):
|
||||
"""Process a single batch (for multiprocessing)"""
|
||||
book_dir, batch_num, save_images = args
|
||||
batch_dir = Path(book_dir) / f'batch_{batch_num}'
|
||||
glyphs_file = batch_dir / 'glyphs.json'
|
||||
|
||||
if not glyphs_file.exists():
|
||||
return None
|
||||
|
||||
hasher = GlyphHasher()
|
||||
batch_results = {
|
||||
'batch_num': batch_num,
|
||||
'glyph_to_hash': {},
|
||||
'glyphs_in_text': [],
|
||||
'images': {}
|
||||
}
|
||||
|
||||
# Load glyphs and compute hashes
|
||||
with open(glyphs_file) as f:
|
||||
glyph_data = json.load(f)
|
||||
|
||||
for font_data in glyph_data:
|
||||
font_family = font_data['fontFamily']
|
||||
glyphs = font_data.get('glyphs', {})
|
||||
|
||||
for glyph_id, glyph_info in glyphs.items():
|
||||
# Add font metrics
|
||||
glyph_info['unitsPerEm'] = font_data.get('unitsPerEm', 1000)
|
||||
glyph_info['ascent'] = font_data.get('ascent', 800)
|
||||
glyph_info['descent'] = font_data.get('descent', -200)
|
||||
|
||||
# Render and hash
|
||||
img = hasher.render_glyph(glyph_info)
|
||||
if img is not None:
|
||||
phash = hasher.compute_hash(img)
|
||||
batch_results['glyph_to_hash'][int(glyph_id)] = {
|
||||
'hash': phash,
|
||||
'font': font_family
|
||||
}
|
||||
|
||||
if save_images and phash not in batch_results['images']:
|
||||
batch_results['images'][phash] = img
|
||||
|
||||
# Load text and extract all glyph IDs used
|
||||
page_files = sorted(batch_dir.glob('page_data_*.json'))
|
||||
for page_file in page_files:
|
||||
with open(page_file) as f:
|
||||
pages = json.load(f)
|
||||
|
||||
for page in pages:
|
||||
for run in page.get('children', []):
|
||||
if 'glyphs' in run:
|
||||
batch_results['glyphs_in_text'].extend(run['glyphs'])
|
||||
|
||||
return batch_results
|
||||
|
||||
|
||||
def create_hash_mapping(book_dir):
|
||||
"""Phase 1: Create hash-based mapping of all glyphs"""
|
||||
print(f"\n{'='*80}")
|
||||
print(f"PHASE 1: HASH-BASED GLYPH NORMALIZATION")
|
||||
print(f"{'='*80}\n")
|
||||
print(f"Book directory: {book_dir}")
|
||||
print(f"Using all {cpu_count()} CPU cores")
|
||||
|
||||
# Find all batches
|
||||
batch_dirs = []
|
||||
front_batches = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_front')])
|
||||
batch_dirs.extend(front_batches)
|
||||
numbered_batches = sorted([d for d in book_dir.iterdir() if d.is_dir() and d.name.startswith('batch_') and not d.name.startswith('batch_front')],
|
||||
key=lambda x: int(x.name.split('_')[1]))
|
||||
batch_dirs.extend(numbered_batches)
|
||||
batch_nums = list(range(len(batch_dirs)))
|
||||
|
||||
print(f"\n[*] Found {len(batch_dirs)} batches")
|
||||
|
||||
# Process all batches in parallel
|
||||
print(f"[*] Processing batches (rendering all glyphs)...")
|
||||
with Pool(cpu_count()) as pool:
|
||||
batch_args = [(str(book_dir), batch_num, True) for batch_num in batch_nums]
|
||||
results = list(pool.imap_unordered(process_batch, batch_args))
|
||||
|
||||
results = [r for r in results if r is not None]
|
||||
print(f"[✓] Processed {len(results)} batches")
|
||||
|
||||
# Build hash -> unique_id mapping
|
||||
print(f"\n[*] Building hash-based mapping...")
|
||||
hash_to_id = {}
|
||||
hash_counter = 0
|
||||
hash_fonts = {}
|
||||
hash_samples = {}
|
||||
hash_images = {}
|
||||
|
||||
for result in results:
|
||||
batch_num = result['batch_num']
|
||||
|
||||
for phash, img in result.get('images', {}).items():
|
||||
if phash not in hash_images:
|
||||
hash_images[phash] = img
|
||||
|
||||
for local_glyph_id, glyph_info in result['glyph_to_hash'].items():
|
||||
phash = glyph_info['hash']
|
||||
font = glyph_info['font']
|
||||
|
||||
if phash not in hash_to_id:
|
||||
hash_to_id[phash] = hash_counter
|
||||
hash_fonts[hash_counter] = font
|
||||
hash_samples[hash_counter] = (batch_num, local_glyph_id)
|
||||
hash_counter += 1
|
||||
|
||||
print(f"[✓] Found {len(hash_to_id)} unique glyphs")
|
||||
|
||||
# Verify no hash collisions
|
||||
print(f"\n[*] Verifying hash uniqueness...")
|
||||
from collections import Counter
|
||||
hash_counts = Counter(hash_to_id.keys())
|
||||
collisions = {h: count for h, count in hash_counts.items() if count > 1}
|
||||
|
||||
if collisions:
|
||||
print(f"⚠ WARNING: Found {len(collisions)} hash collisions!")
|
||||
print(f"This means some distinct glyphs are being merged together.")
|
||||
print(f"First few collisions:")
|
||||
for h, count in list(collisions.items())[:5]:
|
||||
print(f" Hash {h}: {count} glyphs")
|
||||
print(f"\n⚠ This will cause incorrect decoding. Please report this issue.")
|
||||
else:
|
||||
print(f"✓ No hash collisions - each glyph has a unique hash")
|
||||
|
||||
# Normalize all text
|
||||
print(f"\n[*] Normalizing all text...")
|
||||
all_normalized_glyphs = []
|
||||
|
||||
for result in sorted(results, key=lambda r: r['batch_num']):
|
||||
batch_mapping = {}
|
||||
for local_glyph_id, glyph_info in result['glyph_to_hash'].items():
|
||||
phash = glyph_info['hash']
|
||||
batch_mapping[local_glyph_id] = hash_to_id[phash]
|
||||
|
||||
for glyph_id in result['glyphs_in_text']:
|
||||
unique_id = batch_mapping.get(glyph_id, -1)
|
||||
all_normalized_glyphs.append(unique_id)
|
||||
|
||||
print(f"[✓] Normalized {len(all_normalized_glyphs):,} glyphs")
|
||||
|
||||
# Save results
|
||||
output_dir = book_dir / 'hash_mapping'
|
||||
output_dir.mkdir(exist_ok=True)
|
||||
|
||||
hash_info = {
|
||||
'total_unique_glyphs': len(hash_to_id),
|
||||
'hash_to_id': hash_to_id,
|
||||
'id_to_font': {str(k): v for k, v in hash_fonts.items()},
|
||||
'id_samples': {str(k): {'batch': v[0], 'glyph': v[1]} for k, v in hash_samples.items()}
|
||||
}
|
||||
|
||||
with open(output_dir / 'hash_info.json', 'w') as f:
|
||||
json.dump(hash_info, f, indent=2)
|
||||
|
||||
with open(output_dir / 'all_glyphs.json', 'w') as f:
|
||||
json.dump(all_normalized_glyphs, f)
|
||||
|
||||
# Save glyph images
|
||||
images_dir = output_dir / 'glyph_images'
|
||||
images_dir.mkdir(exist_ok=True)
|
||||
|
||||
for phash, unique_id in hash_to_id.items():
|
||||
if phash in hash_images:
|
||||
img = hash_images[phash]
|
||||
font = hash_fonts[unique_id]
|
||||
img.save(images_dir / f'id_{unique_id:03d}_{font}.png')
|
||||
|
||||
print(f"[✓] Saved to {output_dir}/")
|
||||
|
||||
# Show frequency
|
||||
freq = Counter(all_normalized_glyphs)
|
||||
print(f"\n[*] Top 20 most frequent glyphs:")
|
||||
for unique_id, count in freq.most_common(20):
|
||||
pct = count / len(all_normalized_glyphs) * 100
|
||||
font = hash_fonts.get(unique_id, 'unknown')
|
||||
print(f" ID {unique_id:3d} ({font:12s}): {count:7,} ({pct:5.2f}%)")
|
||||
|
||||
return output_dir, hash_info
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# PART 2: TTF CHARACTER MATCHING
|
||||
# ============================================================================
|
||||
|
||||
def render_glyph_by_name(tt, glyph_name, size=128):
|
||||
"""Render a glyph by name from TTF"""
|
||||
glyph_set = tt.getGlyphSet()
|
||||
if glyph_name not in glyph_set:
|
||||
return None
|
||||
|
||||
glyph = glyph_set[glyph_name]
|
||||
|
||||
# Get font metrics
|
||||
head = tt['head']
|
||||
units_per_em = head.unitsPerEm
|
||||
hhea = tt['hhea']
|
||||
ascent = hhea.ascent
|
||||
descent = hhea.descent
|
||||
|
||||
# Get bounding box
|
||||
bounds_pen = BoundsPen(glyph_set)
|
||||
glyph.draw(bounds_pen)
|
||||
if bounds_pen.bounds is None:
|
||||
return None
|
||||
xmin, ymin, xmax, ymax = bounds_pen.bounds
|
||||
|
||||
# Apply Y-flip to bbox
|
||||
ymin_svg = -ymax
|
||||
ymax_svg = -ymin
|
||||
|
||||
# Extract SVG path with Y-flip
|
||||
svg_pen = SVGPathPen(glyph_set)
|
||||
transform_pen = TransformPen(svg_pen, Transform(1, 0, 0, -1, 0, 0))
|
||||
glyph.draw(transform_pen)
|
||||
path_data = svg_pen.getCommands()
|
||||
|
||||
if not path_data or path_data.strip() == '':
|
||||
return None
|
||||
|
||||
# Center glyph
|
||||
glyph_center_x = (xmin + xmax) / 2
|
||||
glyph_center_y = (ymin_svg + ymax_svg) / 2
|
||||
font_height = ascent - descent
|
||||
viewbox_x = glyph_center_x - units_per_em / 2
|
||||
viewbox_y = glyph_center_y - font_height / 2
|
||||
viewbox = f"{viewbox_x} {viewbox_y} {units_per_em} {font_height}"
|
||||
|
||||
# Create SVG
|
||||
svg = f'''<?xml version="1.0" encoding="UTF-8"?>
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="{viewbox}" width="{size}" height="{size}">
|
||||
<path d="{path_data}" fill="black"/>
|
||||
</svg>'''
|
||||
|
||||
# Render
|
||||
try:
|
||||
png_bytes = cairosvg.svg2png(
|
||||
bytestring=svg.encode('utf-8'),
|
||||
output_width=size,
|
||||
output_height=size
|
||||
)
|
||||
img_rgba = Image.open(io.BytesIO(png_bytes))
|
||||
img = Image.new('L', (size, size), 255)
|
||||
alpha = img_rgba.split()[3]
|
||||
inverted = ImageOps.invert(alpha)
|
||||
img.paste(0, mask=inverted)
|
||||
return img
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def render_char_from_ttf(tt, char, size=128):
|
||||
"""Render a character from TTF"""
|
||||
cmap = tt.getBestCmap()
|
||||
if ord(char) not in cmap:
|
||||
return None
|
||||
glyph_name = cmap[ord(char)]
|
||||
return render_glyph_by_name(tt, glyph_name, size)
|
||||
|
||||
|
||||
def compare_images_ssim(img1, img2):
|
||||
"""Compare two images using SSIM. Returns distance (0=identical)"""
|
||||
arr1 = np.array(img1)
|
||||
arr2 = np.array(img2)
|
||||
similarity = ssim(arr1, arr2)
|
||||
distance = (1 - similarity) * 10
|
||||
return distance
|
||||
|
||||
|
||||
def match_single_glyph(args):
|
||||
"""Match a single glyph (for parallel processing)"""
|
||||
unique_id, glyph_images_dir, ttf_library_items, fast_mode, progressive_mode = args
|
||||
|
||||
# Load Amazon glyph image
|
||||
glyph_image_files = list(glyph_images_dir.glob(f'id_{unique_id:03d}_*.png'))
|
||||
if not glyph_image_files:
|
||||
return (unique_id, None, float('inf'))
|
||||
|
||||
amazon_img = Image.open(glyph_image_files[0])
|
||||
|
||||
if not progressive_mode:
|
||||
# Original single-pass approach
|
||||
best_match = None
|
||||
best_distance = float('inf')
|
||||
early_exit_threshold = 0.05 if fast_mode else -1
|
||||
|
||||
for (char, font_name, style), ttf_img in ttf_library_items:
|
||||
distance = compare_images_ssim(amazon_img, ttf_img)
|
||||
if distance < best_distance:
|
||||
best_distance = distance
|
||||
best_match = (char, font_name, style)
|
||||
|
||||
if fast_mode and distance <= early_exit_threshold:
|
||||
break
|
||||
|
||||
return (unique_id, best_match, best_distance)
|
||||
|
||||
# Progressive resolution approach
|
||||
# Stage 1: 128x128 - Quick filter
|
||||
amazon_128 = amazon_img.resize((128, 128), Image.LANCZOS)
|
||||
candidates_128 = []
|
||||
|
||||
for (char, font_name, style), ttf_img in ttf_library_items:
|
||||
ttf_128 = ttf_img.resize((128, 128), Image.LANCZOS)
|
||||
distance = compare_images_ssim(amazon_128, ttf_128)
|
||||
candidates_128.append(((char, font_name, style), ttf_img, distance))
|
||||
|
||||
# Sort and keep top 30 candidates only
|
||||
candidates_128.sort(key=lambda x: x[2])
|
||||
candidates_128 = candidates_128[:30]
|
||||
|
||||
# Stage 2: 256x256 - Narrow down
|
||||
amazon_256 = amazon_img.resize((256, 256), Image.LANCZOS)
|
||||
candidates_256 = []
|
||||
|
||||
for (char, font_name, style), ttf_img, _ in candidates_128:
|
||||
ttf_256 = ttf_img.resize((256, 256), Image.LANCZOS)
|
||||
distance = compare_images_ssim(amazon_256, ttf_256)
|
||||
candidates_256.append(((char, font_name, style), ttf_img, distance))
|
||||
|
||||
# Sort and keep top 10
|
||||
candidates_256.sort(key=lambda x: x[2])
|
||||
candidates_256 = candidates_256[:10]
|
||||
|
||||
# Stage 3: 512x512 - Final decision
|
||||
amazon_512 = amazon_img.resize((512, 512), Image.LANCZOS)
|
||||
best_match = None
|
||||
best_distance = float('inf')
|
||||
|
||||
for (char, font_name, style), ttf_img, _ in candidates_256:
|
||||
ttf_512 = ttf_img.resize((512, 512), Image.LANCZOS)
|
||||
distance = compare_images_ssim(amazon_512, ttf_512)
|
||||
if distance < best_distance:
|
||||
best_distance = distance
|
||||
best_match = (char, font_name, style)
|
||||
|
||||
# Early exit if very confident
|
||||
if distance < 0.05:
|
||||
break
|
||||
|
||||
return (unique_id, best_match, best_distance)
|
||||
|
||||
|
||||
def match_ttf_characters(hash_mapping_dir, fast_mode, full_mode, progressive_mode):
|
||||
"""Phase 2: Match unique glyphs to TTF characters"""
|
||||
print(f"\n{'='*80}")
|
||||
print(f"PHASE 2: TTF CHARACTER MATCHING")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
hash_info_file = hash_mapping_dir / 'hash_info.json'
|
||||
glyph_images_dir = hash_mapping_dir / 'glyph_images'
|
||||
|
||||
# Load hash info
|
||||
with open(hash_info_file) as f:
|
||||
hash_info = json.load(f)
|
||||
|
||||
id_to_font = {int(k): v for k, v in hash_info['id_to_font'].items()}
|
||||
|
||||
# Find all font files (check multiple directories)
|
||||
font_dirs = [Path('fonts'), Path('.')]
|
||||
font_files = []
|
||||
for font_dir in font_dirs:
|
||||
if font_dir.exists():
|
||||
font_files.extend(font_dir.glob('*.ttf'))
|
||||
|
||||
font_files = sorted(set(font_files)) # Remove duplicates
|
||||
print(f"Found {len(font_files)} font files")
|
||||
|
||||
# Check which fonts we have vs what the book needs
|
||||
found_font_names = {f.stem.lower() for f in font_files}
|
||||
needed_fonts = set(id_to_font.values())
|
||||
missing_fonts = needed_fonts - found_font_names
|
||||
|
||||
if missing_fonts:
|
||||
print(f"\n⚠ WARNING: Book uses fonts not in font directory:")
|
||||
for font in missing_fonts:
|
||||
glyph_count = sum(1 for f in id_to_font.values() if f == font)
|
||||
print(f" - {font}: {glyph_count} glyphs")
|
||||
print(f"\nGlyphs using these fonts will be matched against available fonts (may be inaccurate)")
|
||||
else:
|
||||
print(f"✓ All required fonts available")
|
||||
|
||||
# Characters to test
|
||||
if full_mode:
|
||||
chars_to_test = []
|
||||
print("Full mode: Will check ALL characters in font")
|
||||
else:
|
||||
# Standard ASCII characters
|
||||
chars_to_test = string.ascii_letters + string.digits + string.punctuation + " "
|
||||
|
||||
# Add common special characters that appear in books
|
||||
special_chars = [
|
||||
'\u2022', # • BULLET
|
||||
'\u2023', # ‣ TRIANGULAR BULLET
|
||||
'\u2043', # ⁃ HYPHEN BULLET
|
||||
'\u00B7', # · MIDDLE DOT
|
||||
'\u25E6', # ◦ WHITE BULLET
|
||||
'\u2219', # ∙ BULLET OPERATOR
|
||||
'\u00A0', # Non-breaking space
|
||||
'\u00A9', # © COPYRIGHT
|
||||
'\u00AE', # ® REGISTERED
|
||||
'\u2122', # ™ TRADEMARK
|
||||
'\u00AB', # « LEFT DOUBLE ANGLE QUOTE
|
||||
'\u00BB', # » RIGHT DOUBLE ANGLE QUOTE
|
||||
'\u2018', # ' LEFT SINGLE QUOTE (already in ligatures but add anyway)
|
||||
'\u2019', # ' RIGHT SINGLE QUOTE
|
||||
'\u201A', # ‚ SINGLE LOW-9 QUOTE
|
||||
'\u201B', # ‛ SINGLE HIGH-REVERSED-9 QUOTE
|
||||
'\u2032', # ′ PRIME
|
||||
'\u2033', # ″ DOUBLE PRIME
|
||||
]
|
||||
chars_to_test += ''.join(special_chars)
|
||||
print(f"Standard mode: Checking {len(chars_to_test)} predefined characters (including special chars)")
|
||||
|
||||
# Ligatures and special glyphs
|
||||
ligature_glyphs = {
|
||||
'f_f': 'ff', 'f_i': 'fi', 'f_l': 'fl', 'f_f_i': 'ffi', 'f_f_l': 'ffl',
|
||||
'uniFB00': 'ff', 'uniFB01': 'fi', 'uniFB02': 'fl', 'uniFB03': 'ffi', 'uniFB04': 'ffl',
|
||||
'space': ' ',
|
||||
'endash': chr(0x2013), 'emdash': chr(0x2014),
|
||||
'quotedblleft': chr(0x201C), 'quotedblright': chr(0x201D),
|
||||
'quoteleft': chr(0x2018), 'quoteright': chr(0x2019),
|
||||
'ellipsis': chr(0x2026),
|
||||
}
|
||||
|
||||
# Build TTF character library
|
||||
print("=" * 60)
|
||||
print("Building TTF character library...")
|
||||
print("=" * 60)
|
||||
|
||||
ttf_library = {}
|
||||
|
||||
for font_path in font_files:
|
||||
font_name = font_path.stem
|
||||
print(f"\nProcessing: {font_name}")
|
||||
|
||||
font_style = "normal"
|
||||
if "Bold" in font_name and "Italic" in font_name:
|
||||
font_style = "bold-italic"
|
||||
elif "Bold" in font_name:
|
||||
font_style = "bold"
|
||||
elif "Italic" in font_name:
|
||||
font_style = "italic"
|
||||
|
||||
try:
|
||||
tt = TTFont(font_path)
|
||||
rendered_count = 0
|
||||
|
||||
if full_mode:
|
||||
cmap = tt.getBestCmap()
|
||||
if cmap:
|
||||
for codepoint, glyph_name in cmap.items():
|
||||
char = chr(codepoint)
|
||||
img = render_char_from_ttf(tt, char)
|
||||
if img is not None:
|
||||
ttf_library[(char, font_name, font_style)] = img
|
||||
rendered_count += 1
|
||||
else:
|
||||
for char in chars_to_test:
|
||||
img = render_char_from_ttf(tt, char)
|
||||
if img is not None:
|
||||
ttf_library[(char, font_name, font_style)] = img
|
||||
rendered_count += 1
|
||||
|
||||
# Render ligatures and special characters
|
||||
glyph_set = tt.getGlyphSet()
|
||||
for glyph_name, char in ligature_glyphs.items():
|
||||
if glyph_name in glyph_set:
|
||||
img = render_glyph_by_name(tt, glyph_name)
|
||||
if img is not None:
|
||||
ttf_library[(char, font_name, font_style)] = img
|
||||
rendered_count += 1
|
||||
|
||||
print(f" Rendered {rendered_count} glyphs")
|
||||
|
||||
except Exception as e:
|
||||
print(f" Error: {e}")
|
||||
|
||||
print(f"\n[✓] TTF library built: {len(ttf_library)} glyphs")
|
||||
|
||||
# Match glyphs
|
||||
print("\n" + "=" * 60)
|
||||
mode_parts = []
|
||||
if progressive_mode:
|
||||
mode_parts.append("PROGRESSIVE MODE - 3-stage filtering (128→256→512px)")
|
||||
elif fast_mode:
|
||||
mode_parts.append("FAST MODE - early exit on good matches")
|
||||
else:
|
||||
mode_parts.append("FULL MODE - exhaustive search")
|
||||
|
||||
print(f"Matching Amazon glyphs to TTF characters (using SSIM, {cpu_count()} threads)")
|
||||
print(f"{' | '.join(mode_parts)}")
|
||||
print("=" * 60)
|
||||
|
||||
# Prepare arguments
|
||||
ttf_library_items = list(ttf_library.items())
|
||||
glyph_ids = sorted(id_to_font.keys())
|
||||
args_list = [(gid, glyph_images_dir, ttf_library_items, fast_mode, progressive_mode) for gid in glyph_ids]
|
||||
|
||||
# Process in parallel
|
||||
matches = {}
|
||||
no_match_count = 0
|
||||
|
||||
start_time = time.time()
|
||||
with Pool(cpu_count()) as pool:
|
||||
results = list(tqdm(pool.imap(match_single_glyph, args_list), total=len(args_list), desc="Matching glyphs"))
|
||||
elapsed_time = time.time() - start_time
|
||||
|
||||
for unique_id, best_match, best_distance in results:
|
||||
if best_match and best_distance <= 1.0:
|
||||
matches[unique_id] = (*best_match, best_distance)
|
||||
# Highlight potential mismatches
|
||||
if best_match[0] in [',', "'", '"', '`'] and best_distance > 0.3:
|
||||
print(f"⚠ Glyph {unique_id:3d} → '{best_match[0]}' (distance={best_distance:.3f}, font={best_match[1]}) [UNCERTAIN]")
|
||||
else:
|
||||
print(f"✓ Glyph {unique_id:3d} → '{best_match[0]}' (distance={best_distance:.3f}, font={best_match[1]})")
|
||||
else:
|
||||
no_match_count += 1
|
||||
print(f"✗ Glyph {unique_id:3d} → NO MATCH (best distance={best_distance:.3f})")
|
||||
|
||||
# Add special case for space
|
||||
matches[-1] = (' ', 'special', 'normal', 0)
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("RESULTS")
|
||||
print("=" * 60)
|
||||
print(f"Matched: {len(matches)-1}/{len(id_to_font)} glyphs ({100*(len(matches)-1)/len(id_to_font):.0f}%)")
|
||||
print(f"No match: {no_match_count} glyphs")
|
||||
print(f"Time taken: {elapsed_time:.2f} seconds")
|
||||
print(f"\nUnmatched glyph IDs: {[k for k in sorted(id_to_font.keys()) if k not in matches]}")
|
||||
|
||||
# Save mapping
|
||||
output_file = Path('ttf_character_mapping.json')
|
||||
mapping_output = {
|
||||
str(glyph_id): {
|
||||
"character": char,
|
||||
"font": font,
|
||||
"style": style,
|
||||
"distance": dist
|
||||
}
|
||||
for glyph_id, (char, font, style, dist) in matches.items()
|
||||
}
|
||||
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(mapping_output, f, indent=2, ensure_ascii=False)
|
||||
|
||||
print(f"\nMapping saved to: {output_file}")
|
||||
|
||||
# Show character frequency
|
||||
char_counts = defaultdict(int)
|
||||
style_counts = defaultdict(int)
|
||||
for char, _font, style, _dist in matches.values():
|
||||
char_counts[char] += 1
|
||||
style_counts[style] += 1
|
||||
|
||||
print("\nMost common matched characters:")
|
||||
for char, count in sorted(char_counts.items(), key=lambda x: -x[1])[:20]:
|
||||
print(f" '{char}': {count} glyphs")
|
||||
|
||||
print("\nMatches by style:")
|
||||
for style, count in sorted(style_counts.items()):
|
||||
print(f" {style}: {count} glyphs")
|
||||
|
||||
return output_file
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# MAIN
|
||||
# ============================================================================
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 decode_glyphs_complete.py <book_dir> [--fast] [--full] [--progressive]")
|
||||
print("\nOptions:")
|
||||
print(" --fast Early exit on good SSIM matches")
|
||||
print(" --full Check all characters in font (not just alphanumeric)")
|
||||
print(" --progressive Use multi-stage filtering (32→64→128→256→512→1024px)")
|
||||
sys.exit(1)
|
||||
|
||||
fast_mode = "--fast" in sys.argv
|
||||
full_mode = "--full" in sys.argv
|
||||
progressive_mode = "--progressive" in sys.argv
|
||||
book_dir = Path(sys.argv[1])
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"COMPLETE GLYPH DECODING PIPELINE")
|
||||
print(f"{'='*80}")
|
||||
print(f"\nBook: {book_dir}")
|
||||
print(f"Options:")
|
||||
print(f" Fast mode: {fast_mode}")
|
||||
print(f" Full character set: {full_mode}")
|
||||
print(f" Progressive matching: {progressive_mode}")
|
||||
|
||||
# Phase 1: Hash-based normalization
|
||||
hash_mapping_dir, hash_info = create_hash_mapping(book_dir)
|
||||
|
||||
# Phase 2: TTF character matching
|
||||
mapping_file = match_ttf_characters(hash_mapping_dir, fast_mode, full_mode, progressive_mode)
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"[✓] COMPLETE PIPELINE FINISHED!")
|
||||
print(f"{'='*80}")
|
||||
print(f"\nOutputs:")
|
||||
print(f" Hash mapping: {hash_mapping_dir}/")
|
||||
print(f" Character mapping: {mapping_file}")
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
BIN
decoded_book.epub
Normal file
BIN
decoded_book.epub
Normal file
Binary file not shown.
160
download_full_book.py
Executable file
160
download_full_book.py
Executable file
|
|
@ -0,0 +1,160 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Download complete book by downloading 5 pages at a time in a single session.
|
||||
This ensures all pages share the same font/glyph encoding.
|
||||
|
||||
Strategy:
|
||||
1. Download from start position (includes TOC) - 5 pages at a time
|
||||
2. Keep downloading until we reach the end
|
||||
3. All downloads in ONE session so fonts match
|
||||
4. Use TOC from first download to build glyph mapping
|
||||
5. Decode all pages using that single mapping
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from downloader import KindleDownloader
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python3 download_full_book.py <ASIN> [--yes]")
|
||||
sys.exit(1)
|
||||
|
||||
asin = sys.argv[1]
|
||||
auto_confirm = '--yes' in sys.argv or '-y' in sys.argv
|
||||
output_base = Path(f'downloads/{asin}')
|
||||
output_base.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Load credentials
|
||||
headers_file = Path('headers.json')
|
||||
if not headers_file.exists():
|
||||
print("[✗] headers.json not found!")
|
||||
sys.exit(1)
|
||||
|
||||
with open(headers_file) as f:
|
||||
headers_data = json.load(f)
|
||||
|
||||
cookies = headers_data.get('cookies', '')
|
||||
adp_token = headers_data['headers'].get('x-adp-session-token') if 'headers' in headers_data else None
|
||||
|
||||
# Initialize downloader (single session for entire book)
|
||||
print(f"\n{'='*80}")
|
||||
print(f"DOWNLOADING COMPLETE BOOK: {asin}")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
downloader = KindleDownloader(cookies, adp_token)
|
||||
|
||||
# Get book metadata
|
||||
print("[*] Getting book metadata...")
|
||||
metadata = downloader.start_reading(asin)
|
||||
|
||||
title = metadata.get('deliveredAsin', asin)
|
||||
revision = metadata.get('contentVersion', '')
|
||||
start_pos = metadata.get('srl', 0)
|
||||
|
||||
print(f"[*] Title: {title}")
|
||||
print(f"[*] Revision: {revision}")
|
||||
print(f"[*] Default start position (srl): {start_pos}")
|
||||
print(f"[*] Downloading from position 0 to include front matter (TOC, cover, etc)")
|
||||
|
||||
# Save karamelToken for image decryption
|
||||
if 'karamelToken' in metadata:
|
||||
karamel_token = {
|
||||
'token': metadata['karamelToken']['token'],
|
||||
'expiresAt': metadata['karamelToken']['expiresAt']
|
||||
}
|
||||
token_file = output_base / 'karamel_token.json'
|
||||
with open(token_file, 'w') as f:
|
||||
json.dump(karamel_token, f, indent=2)
|
||||
print(f"[✓] Saved karamelToken to {token_file}")
|
||||
|
||||
# Download from position 0 to get the complete book including front matter
|
||||
print(f"\n[*] Batch 0: position 0...")
|
||||
first_tar = downloader.render_pages(asin, revision, start_position=0, num_pages=5)
|
||||
first_files = downloader.extract_tar(first_tar, output_base / 'batch_0')
|
||||
|
||||
# Get position range from batch 0
|
||||
page_data_file = list((output_base / 'batch_0').glob('page_data_*.json'))[0]
|
||||
with open(page_data_file) as f:
|
||||
first_pages = json.load(f)
|
||||
|
||||
batch_0_start = first_pages[0]['startPositionId']
|
||||
batch_0_end = first_pages[-1]['endPositionId']
|
||||
print(f"[✓] Batch 0: {batch_0_start} to {batch_0_end} ({len(first_files)} files)")
|
||||
|
||||
# Load TOC to estimate book length
|
||||
toc_file = output_base / 'batch_0' / 'toc.json'
|
||||
with open(toc_file) as f:
|
||||
toc = json.load(f)
|
||||
|
||||
last_toc_pos = max(entry['tocPositionId'] for entry in toc)
|
||||
print(f"[*] Book ends around position {last_toc_pos}")
|
||||
|
||||
# Estimate number of batches
|
||||
positions_per_batch = batch_0_end - batch_0_start
|
||||
estimated_batches = int((last_toc_pos - start_pos) / positions_per_batch) + 1
|
||||
|
||||
print(f"[*] Estimated {estimated_batches} batches needed (~{positions_per_batch} positions per 5 pages)")
|
||||
print(f"\n[!] WARNING: This will download the entire book!")
|
||||
print(f"[!] Estimated total: {estimated_batches * 5} pages")
|
||||
|
||||
if not auto_confirm:
|
||||
response = input(f"\nContinue? [y/N]: ")
|
||||
if response.lower() != 'y':
|
||||
print("[*] Aborted")
|
||||
sys.exit(0)
|
||||
else:
|
||||
print("[*] Auto-confirmed with --yes flag")
|
||||
|
||||
# Download remaining batches starting from where batch_0 ended
|
||||
current_pos = batch_0_end + 1
|
||||
batch_num = 1
|
||||
|
||||
print(f"\n[*] Downloading remaining batches...")
|
||||
|
||||
while current_pos < last_toc_pos:
|
||||
try:
|
||||
print(f"\n[*] Batch {batch_num}: position {current_pos}...")
|
||||
tar_data = downloader.render_pages(asin, revision, start_position=current_pos, num_pages=5)
|
||||
files = downloader.extract_tar(tar_data, output_base / f'batch_{batch_num}')
|
||||
|
||||
# Get end position from this batch
|
||||
page_file = list((output_base / f'batch_{batch_num}').glob('page_data_*.json'))[0]
|
||||
with open(page_file) as f:
|
||||
pages = json.load(f)
|
||||
|
||||
if pages:
|
||||
batch_end = pages[-1]['endPositionId']
|
||||
print(f"[✓] Batch {batch_num}: {pages[0]['startPositionId']} to {batch_end}")
|
||||
current_pos = batch_end + 1
|
||||
else:
|
||||
print(f"[!] Batch {batch_num}: No pages returned, stopping")
|
||||
break
|
||||
|
||||
batch_num += 1
|
||||
|
||||
except Exception as e:
|
||||
print(f"[✗] Error downloading batch {batch_num}: {e}")
|
||||
break
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"[✓] DOWNLOAD COMPLETE")
|
||||
print(f"[✓] Downloaded {batch_num} batches")
|
||||
print(f"[✓] Saved to: {output_base}/")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
# Save download metadata
|
||||
download_info = {
|
||||
'asin': asin,
|
||||
'revision': revision,
|
||||
'start_position': start_pos,
|
||||
'total_batches': batch_num,
|
||||
'pages_per_batch': 5,
|
||||
'estimated_positions': f'{start_pos} to {current_pos}'
|
||||
}
|
||||
|
||||
with open(output_base / 'download_info.json', 'w') as f:
|
||||
json.dump(download_info, f, indent=2)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
282
downloader.py
Normal file
282
downloader.py
Normal file
|
|
@ -0,0 +1,282 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Kindle Book Downloader
|
||||
Downloads raw page data from Kindle Cloud Reader (Stage 1)
|
||||
|
||||
Usage:
|
||||
python3 downloader.py <ASIN> [--pages N] [--output DIR]
|
||||
|
||||
Example:
|
||||
python3 downloader.py B0FLBTR2FS --pages 10 --output downloads/
|
||||
"""
|
||||
import requests
|
||||
import json
|
||||
import tarfile
|
||||
import io
|
||||
import sys
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
class KindleDownloader:
|
||||
"""Downloads raw encrypted book data from Kindle Cloud Reader"""
|
||||
|
||||
def __init__(self, cookies_string, adp_session_token=None):
|
||||
"""
|
||||
Initialize with authentication credentials
|
||||
|
||||
Args:
|
||||
cookies_string: Cookie string from browser
|
||||
adp_session_token: x-adp-session-token header value
|
||||
"""
|
||||
self.session = requests.Session()
|
||||
|
||||
# Parse cookies
|
||||
for cookie in cookies_string.split('; '):
|
||||
if '=' in cookie:
|
||||
name, value = cookie.split('=', 1)
|
||||
self.session.cookies.set(name, value, domain='.amazon.com')
|
||||
|
||||
self.adp_session_token = adp_session_token
|
||||
self.rendering_token = None
|
||||
self.token_expires = None
|
||||
|
||||
def start_reading(self, asin):
|
||||
"""
|
||||
Initialize reading session and get rendering token
|
||||
|
||||
Args:
|
||||
asin: Book ASIN
|
||||
|
||||
Returns:
|
||||
dict: Book metadata including token, revision, srl
|
||||
"""
|
||||
url = 'https://read.amazon.com/service/mobile/reader/startReading'
|
||||
params = {
|
||||
'asin': asin,
|
||||
'clientVersion': '20000100'
|
||||
}
|
||||
|
||||
headers = {}
|
||||
if self.adp_session_token:
|
||||
headers['x-adp-session-token'] = self.adp_session_token
|
||||
|
||||
print(f"[*] Requesting reading session for {asin}...")
|
||||
response = self.session.get(url, params=params, headers=headers)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
|
||||
# Store token
|
||||
if 'karamelToken' in data:
|
||||
self.rendering_token = data['karamelToken']['token']
|
||||
self.token_expires = data['karamelToken']['expiresAt']
|
||||
print(f"[✓] Got rendering token (expires: {self.token_expires})")
|
||||
|
||||
return data
|
||||
|
||||
def render_pages(self, asin, revision, start_position=0, num_pages=2):
|
||||
"""
|
||||
Download raw page data from Kindle renderer
|
||||
|
||||
Args:
|
||||
asin: Book ASIN
|
||||
revision: Content revision ID
|
||||
start_position: Starting position ID
|
||||
num_pages: Number of pages to fetch
|
||||
|
||||
Returns:
|
||||
bytes: Raw TAR archive containing page data
|
||||
"""
|
||||
url = 'https://read.amazon.com/renderer/render'
|
||||
|
||||
params = {
|
||||
'version': '3.0',
|
||||
'asin': asin,
|
||||
'contentType': 'FullBook',
|
||||
'revision': revision,
|
||||
'fontFamily': 'Bookerly',
|
||||
'fontSize': '8.91',
|
||||
'lineHeight': '1.4',
|
||||
'dpi': '160',
|
||||
'height': '1600',
|
||||
'width': '1000',
|
||||
'marginBottom': '0',
|
||||
'marginLeft': '9',
|
||||
'marginRight': '9',
|
||||
'marginTop': '0',
|
||||
'maxNumberColumns': '1',
|
||||
'theme': 'dark',
|
||||
'locationMap': 'false',
|
||||
'packageType': 'TAR',
|
||||
'encryptionVersion': 'NONE',
|
||||
'numPage': str(num_pages),
|
||||
'skipPageCount': '0',
|
||||
'startingPosition': str(start_position),
|
||||
'bundleImages': 'false'
|
||||
}
|
||||
|
||||
headers = {
|
||||
'x-amz-rendering-token': self.rendering_token
|
||||
}
|
||||
|
||||
print(f"[*] Downloading {num_pages} pages from position {start_position}...")
|
||||
response = self.session.get(url, params=params, headers=headers)
|
||||
|
||||
if response.status_code != 200:
|
||||
print(f"[✗] Error {response.status_code}: {response.text[:200]}")
|
||||
|
||||
response.raise_for_status()
|
||||
|
||||
return response.content
|
||||
|
||||
def extract_tar(self, tar_bytes, output_dir):
|
||||
"""
|
||||
Extract TAR archive to directory
|
||||
|
||||
Args:
|
||||
tar_bytes: Raw TAR data
|
||||
output_dir: Directory to extract to
|
||||
|
||||
Returns:
|
||||
list: Names of extracted files
|
||||
"""
|
||||
output_path = Path(output_dir)
|
||||
output_path.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
extracted_files = []
|
||||
|
||||
with tarfile.open(fileobj=io.BytesIO(tar_bytes)) as tar:
|
||||
for member in tar.getmembers():
|
||||
if member.isfile():
|
||||
content = tar.extractfile(member).read()
|
||||
file_path = output_path / member.name
|
||||
# Create parent directories if they don't exist
|
||||
file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
file_path.write_bytes(content)
|
||||
extracted_files.append(member.name)
|
||||
|
||||
return extracted_files
|
||||
|
||||
def download(self, asin, num_pages=2, output_dir=None):
|
||||
"""
|
||||
Download book pages and save raw data
|
||||
|
||||
Args:
|
||||
asin: Book ASIN
|
||||
num_pages: Number of pages to download
|
||||
output_dir: Output directory (default: downloads/<asin>/)
|
||||
|
||||
Returns:
|
||||
dict: Download metadata
|
||||
"""
|
||||
print(f"\n{'='*80}")
|
||||
print(f"KINDLE DOWNLOADER")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
# Get metadata and token
|
||||
metadata = self.start_reading(asin)
|
||||
|
||||
title = metadata.get('deliveredAsin', asin)
|
||||
revision = metadata.get('contentVersion', '')
|
||||
srl = metadata.get('srl', 0)
|
||||
|
||||
print(f"[*] ASIN: {title}")
|
||||
print(f"[*] Revision: {revision}")
|
||||
print(f"[*] SRL (start position): {srl}")
|
||||
|
||||
# Download pages
|
||||
tar_data = self.render_pages(asin, revision, start_position=srl, num_pages=num_pages)
|
||||
print(f"[✓] Downloaded {len(tar_data)} bytes")
|
||||
|
||||
# Extract to directory
|
||||
if output_dir is None:
|
||||
output_dir = f"downloads/{asin}"
|
||||
|
||||
print(f"[*] Extracting to {output_dir}/...")
|
||||
extracted_files = self.extract_tar(tar_data, output_dir)
|
||||
print(f"[✓] Extracted {len(extracted_files)} files:")
|
||||
for filename in extracted_files:
|
||||
print(f" - {filename}")
|
||||
|
||||
# Save metadata
|
||||
metadata_file = Path(output_dir) / 'download_metadata.json'
|
||||
download_info = {
|
||||
'asin': asin,
|
||||
'revision': revision,
|
||||
'srl': srl,
|
||||
'start_position': srl,
|
||||
'num_pages': num_pages,
|
||||
'extracted_files': extracted_files
|
||||
}
|
||||
metadata_file.write_text(json.dumps(download_info, indent=2))
|
||||
print(f"[✓] Saved metadata to {metadata_file}")
|
||||
|
||||
print(f"\n{'='*80}")
|
||||
print(f"[✓] DOWNLOAD COMPLETE")
|
||||
print(f"[✓] Data saved to: {output_dir}/")
|
||||
print(f"{'='*80}\n")
|
||||
|
||||
return download_info
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Download raw page data from Kindle Cloud Reader',
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
Examples:
|
||||
python3 downloader.py B0FLBTR2FS
|
||||
python3 downloader.py B0FLBTR2FS --pages 10
|
||||
python3 downloader.py B0FLBTR2FS --output my_books/
|
||||
"""
|
||||
)
|
||||
parser.add_argument('asin', help='Book ASIN to download')
|
||||
parser.add_argument('--pages', type=int, default=2, help='Number of pages to download (default: 2)')
|
||||
parser.add_argument('--output', help='Output directory (default: downloads/<asin>/)')
|
||||
parser.add_argument('--start-position', type=int, help='Override start position (default: use SRL from metadata)')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Load credentials from headers.json
|
||||
headers_file = Path('headers.json')
|
||||
if not headers_file.exists():
|
||||
print("[✗] headers.json not found!")
|
||||
print("\nCreate headers.json with:")
|
||||
print(' {')
|
||||
print(' "headers": {"x-adp-session-token": "..."},')
|
||||
print(' "cookies": "session-id=...; ..."')
|
||||
print(' }')
|
||||
sys.exit(1)
|
||||
|
||||
with open(headers_file) as f:
|
||||
headers_data = json.load(f)
|
||||
|
||||
cookies = headers_data.get('cookies', '')
|
||||
if not cookies:
|
||||
print("[✗] No cookies found in headers.json!")
|
||||
sys.exit(1)
|
||||
|
||||
adp_token = None
|
||||
if 'headers' in headers_data:
|
||||
adp_token = headers_data['headers'].get('x-adp-session-token')
|
||||
|
||||
# Download
|
||||
downloader = KindleDownloader(cookies, adp_token)
|
||||
|
||||
# Override start position if specified
|
||||
if args.start_position is not None:
|
||||
metadata = downloader.start_reading(args.asin)
|
||||
revision = metadata.get('contentVersion', '')
|
||||
|
||||
# Download from custom position
|
||||
tar_data = downloader.render_pages(args.asin, revision, start_position=args.start_position, num_pages=args.pages)
|
||||
|
||||
# Extract
|
||||
output_dir = args.output or f"downloads/{args.asin}"
|
||||
print(f"[*] Extracting to {output_dir}/...")
|
||||
extracted_files = downloader.extract_tar(tar_data, output_dir)
|
||||
print(f"[✓] Extracted {len(extracted_files)} files")
|
||||
else:
|
||||
downloader.download(args.asin, num_pages=args.pages, output_dir=args.output)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
BIN
fonts/Bookerly Bold Italic.ttf
Normal file
BIN
fonts/Bookerly Bold Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Bold.ttf
Normal file
BIN
fonts/Bookerly Bold.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Display Bold Italic.ttf
Normal file
BIN
fonts/Bookerly Display Bold Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Display Bold.ttf
Normal file
BIN
fonts/Bookerly Display Bold.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Display Italic.ttf
Normal file
BIN
fonts/Bookerly Display Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Display.ttf
Normal file
BIN
fonts/Bookerly Display.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Italic.ttf
Normal file
BIN
fonts/Bookerly Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly LCD Italic.ttf
Normal file
BIN
fonts/Bookerly LCD Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly LCD Light Italic.ttf
Normal file
BIN
fonts/Bookerly LCD Light Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Light Italic.ttf
Normal file
BIN
fonts/Bookerly Light Italic.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly Light.ttf
Normal file
BIN
fonts/Bookerly Light.ttf
Normal file
Binary file not shown.
BIN
fonts/Bookerly.ttf
Normal file
BIN
fonts/Bookerly.ttf
Normal file
Binary file not shown.
Loading…
Add table
Add a link
Reference in a new issue