Deduplicate PID/unpad helpers and generalize the Topaz path fix
Move the copies of unpad, crc32, checksumPid and pidFromSerial that were scattered across the plugin into utilities.py, and add a shared safe_join that generalizes the Topaz extraction path-traversal fix. - unpad: adobekey, ineptepub and ineptpdf import the shared helper. The four bare-script key tools keep their local copies since they run without a package context. - checksumPid is type-preserving (bytes for kgenpids/kindlepid, str for mobidedrm) so every call site keeps its exact behavior. - kgenpids falls back to an absolute import because it is imported top-level by the worker modules. - topazextract.extractFiles uses safe_join; genbook is unchanged as it only handles already-sanitized names. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -47,7 +47,7 @@ from base64 import b64decode
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, unpad
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
|
||||
@@ -75,11 +75,6 @@ if iswindows:
|
||||
except ImportError:
|
||||
from Crypto.Cipher import AES
|
||||
|
||||
def unpad(data, padding=16):
|
||||
pad_len = data[-1]
|
||||
|
||||
return data[:-pad_len]
|
||||
|
||||
DEVICE_KEY_PATH = r'Software\Adobe\Adept\Device'
|
||||
PRIVATE_LICENCE_KEY_PATH = r'Software\Adobe\Adept\Activation'
|
||||
|
||||
|
||||
@@ -62,14 +62,9 @@ except ImportError:
|
||||
from Crypto.PublicKey import RSA
|
||||
|
||||
|
||||
def unpad(data, padding=16):
|
||||
pad_len = data[-1]
|
||||
|
||||
return data[:-pad_len]
|
||||
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, unpad
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
|
||||
|
||||
@@ -84,14 +84,9 @@ except ImportError:
|
||||
from Crypto.PublicKey import RSA
|
||||
|
||||
|
||||
def unpad(data, padding=16):
|
||||
pad_len = data[-1]
|
||||
|
||||
return data[:-pad_len]
|
||||
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, unpad
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
iswindows = sys.platform.startswith('win')
|
||||
|
||||
+10
-36
@@ -15,12 +15,20 @@ __version__ = '3.0'
|
||||
|
||||
|
||||
import os, csv
|
||||
import binascii
|
||||
import zlib
|
||||
import re
|
||||
from struct import pack, unpack, unpack_from
|
||||
import traceback
|
||||
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
try:
|
||||
from .utilities import checksumPid, pidFromSerial
|
||||
except ImportError:
|
||||
# kgenpids is also imported as a top-level module (``import kgenpids`` from
|
||||
# topazextract / k4mobidedrm), so fall back to an absolute import when there
|
||||
# is no package context.
|
||||
from utilities import checksumPid, pidFromSerial
|
||||
|
||||
class DrmException(Exception):
|
||||
pass
|
||||
|
||||
@@ -141,40 +149,6 @@ def generateDevicePID(table,dsn,nbRoll):
|
||||
pidAscii += bytes(bytearray([charMap4[index]]))
|
||||
return pidAscii
|
||||
|
||||
def crc32(s):
|
||||
return (~binascii.crc32(s,-1))&0xFFFFFFFF
|
||||
|
||||
# convert from 8 digit PID to 10 digit PID with checksum
|
||||
def checksumPid(s):
|
||||
global charMap4
|
||||
crc = crc32(s)
|
||||
crc = crc ^ (crc >> 16)
|
||||
res = s
|
||||
l = len(charMap4)
|
||||
for i in (0,1):
|
||||
b = crc & 0xff
|
||||
pos = (b // l) ^ (b % l)
|
||||
res += bytes(bytearray([charMap4[pos%l]]))
|
||||
crc >>= 8
|
||||
return res
|
||||
|
||||
|
||||
# old kindle serial number to fixed pid
|
||||
def pidFromSerial(s, l):
|
||||
global charMap4
|
||||
crc = crc32(s)
|
||||
arr1 = [0]*l
|
||||
for i in range(len(s)):
|
||||
arr1[i%l] ^= s[i]
|
||||
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
|
||||
for i in range(l):
|
||||
arr1[i] ^= crc_bytes[i&3]
|
||||
pid = b""
|
||||
for i in range(l):
|
||||
b = arr1[i] & 0xff
|
||||
pid += bytes(bytearray([charMap4[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
|
||||
return pid
|
||||
|
||||
|
||||
# Parse the EXTH header records and use the Kindle serial number to calculate the book pid.
|
||||
def getKindlePids(rec209, token, serialnum):
|
||||
|
||||
@@ -14,49 +14,12 @@
|
||||
|
||||
|
||||
import sys
|
||||
import binascii
|
||||
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, checksumPid, pidFromSerial
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
|
||||
|
||||
def crc32(s):
|
||||
return (~binascii.crc32(s,-1))&0xFFFFFFFF
|
||||
|
||||
def checksumPid(s):
|
||||
crc = crc32(s)
|
||||
crc = crc ^ (crc >> 16)
|
||||
res = s
|
||||
l = len(letters)
|
||||
for i in (0,1):
|
||||
b = crc & 0xff
|
||||
pos = (b // l) ^ (b % l)
|
||||
res += bytes(bytearray([letters[pos%l]]))
|
||||
crc >>= 8
|
||||
|
||||
return res
|
||||
|
||||
def pidFromSerial(s, l):
|
||||
crc = crc32(s)
|
||||
|
||||
arr1 = [0]*l
|
||||
for i in range(len(s)):
|
||||
arr1[i%l] ^= s[i]
|
||||
|
||||
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
|
||||
for i in range(l):
|
||||
arr1[i] ^= crc_bytes[i&3]
|
||||
|
||||
pid = b""
|
||||
for i in range(l):
|
||||
b = arr1[i] & 0xff
|
||||
pid+=bytes(bytearray([letters[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
|
||||
|
||||
return pid
|
||||
|
||||
def cli_main():
|
||||
print("Mobipocket PID calculator for Amazon Kindle. Copyright © 2007, 2009 Igor Skochinsky")
|
||||
argv=unicode_argv("kindlepid.py")
|
||||
|
||||
@@ -78,14 +78,13 @@ __version__ = "1.1"
|
||||
import sys
|
||||
import os
|
||||
import struct
|
||||
import binascii
|
||||
|
||||
|
||||
#@@CALIBRE_COMPAT_CODE@@
|
||||
|
||||
|
||||
from .alfcrypto import Pukall_Cipher
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, checksumPid
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
|
||||
@@ -105,27 +104,6 @@ def PC1(key, src, decryption=True):
|
||||
except:
|
||||
raise
|
||||
|
||||
letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
|
||||
|
||||
def crc32(s):
|
||||
return (~binascii.crc32(s,-1))&0xFFFFFFFF
|
||||
|
||||
def checksumPid(s):
|
||||
|
||||
s = s.encode()
|
||||
|
||||
|
||||
crc = crc32(s)
|
||||
crc = crc ^ (crc >> 16)
|
||||
res = s
|
||||
l = len(letters)
|
||||
for i in (0,1):
|
||||
b = crc & 0xff
|
||||
pos = (b // l) ^ (b % l)
|
||||
res += bytes(bytearray([letters[pos%l]]))
|
||||
crc >>= 8
|
||||
return res.decode()
|
||||
|
||||
# expects bytearray
|
||||
def getSizeOfTrailingDataEntries(ptr, size, flags):
|
||||
def getSizeOfTrailingDataEntry(ptr, size):
|
||||
|
||||
@@ -24,7 +24,7 @@ from struct import pack
|
||||
from struct import unpack
|
||||
|
||||
from .alfcrypto import Topaz_Cipher
|
||||
from .utilities import SafeUnbuffered
|
||||
from .utilities import SafeUnbuffered, safe_join
|
||||
|
||||
from .argv_utils import unicode_argv
|
||||
|
||||
@@ -358,15 +358,9 @@ class TopazBook:
|
||||
if name == b'glyphs':
|
||||
destdir = os.path.join(outdir,"glyphs")
|
||||
# The record name is read verbatim from the (untrusted) book
|
||||
# file, so strip any directory components to stop a crafted
|
||||
# name (e.g. "../../foo") from writing outside of destdir.
|
||||
fname = os.path.basename(fname)
|
||||
outputFile = os.path.join(destdir,fname)
|
||||
# Defense in depth: never let the resolved path escape destdir.
|
||||
real_out = os.path.abspath(outputFile)
|
||||
real_dir = os.path.abspath(destdir)
|
||||
if not (real_out == real_dir or real_out.startswith(real_dir + os.sep)):
|
||||
raise DrmException("Invalid book record name: {0}".format(repr(name)))
|
||||
# file; safe_join strips any directory components and refuses
|
||||
# a name that would escape destdir (e.g. "../../foo").
|
||||
outputFile = safe_join(destdir, fname)
|
||||
print(".", end=' ')
|
||||
record = self.getBookPayloadRecord(name,index)
|
||||
if isinstance(record, str):
|
||||
|
||||
@@ -5,6 +5,10 @@
|
||||
|
||||
__license__ = 'GPL v3'
|
||||
|
||||
import binascii
|
||||
import os
|
||||
|
||||
|
||||
def uStrCmp (s1, s2, caseless=False):
|
||||
import unicodedata as ud
|
||||
str1 = s1 if isinstance(s1, str) else str(s1)
|
||||
@@ -39,4 +43,81 @@ class SafeUnbuffered:
|
||||
raise
|
||||
def __getattr__(self, attr):
|
||||
return getattr(self.stream, attr)
|
||||
|
||||
|
||||
|
||||
def unpad(data, padding=16):
|
||||
"""Strip PKCS#7-style padding by trusting the final byte as the pad length.
|
||||
|
||||
This matches the historical inline implementations across the plugin: it
|
||||
does not validate that every pad byte is equal, it simply removes the
|
||||
number of trailing bytes given by the last byte.
|
||||
"""
|
||||
pad_len = data[-1]
|
||||
|
||||
return data[:-pad_len]
|
||||
|
||||
|
||||
# Alphabet used to encode Kindle/Mobipocket PID checksum characters.
|
||||
PID_ALPHABET = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
|
||||
|
||||
|
||||
def crc32(s):
|
||||
return (~binascii.crc32(s, -1)) & 0xFFFFFFFF
|
||||
|
||||
|
||||
def checksumPid(s):
|
||||
"""Convert an 8-digit PID into a 10-digit PID with a 2-character checksum.
|
||||
|
||||
The result is the same type (``str`` or ``bytes``) as the argument, which
|
||||
preserves the historical contracts of every call site: kgenpids and
|
||||
kindlepid pass and expect ``bytes`` while mobidedrm passes and expects
|
||||
``str``.
|
||||
"""
|
||||
want_str = isinstance(s, str)
|
||||
if want_str:
|
||||
s = s.encode()
|
||||
crc = crc32(s)
|
||||
crc = crc ^ (crc >> 16)
|
||||
res = s
|
||||
length = len(PID_ALPHABET)
|
||||
for _ in (0, 1):
|
||||
b = crc & 0xff
|
||||
pos = (b // length) ^ (b % length)
|
||||
res += bytes(bytearray([PID_ALPHABET[pos % length]]))
|
||||
crc >>= 8
|
||||
return res.decode() if want_str else res
|
||||
|
||||
|
||||
def pidFromSerial(s, l):
|
||||
"""Convert an (old Kindle) serial number into a fixed-length PID.
|
||||
|
||||
Takes and returns ``bytes``.
|
||||
"""
|
||||
crc = crc32(s)
|
||||
arr1 = [0] * l
|
||||
for i in range(len(s)):
|
||||
arr1[i % l] ^= s[i]
|
||||
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
|
||||
for i in range(l):
|
||||
arr1[i] ^= crc_bytes[i & 3]
|
||||
pid = b""
|
||||
for i in range(l):
|
||||
b = arr1[i] & 0xff
|
||||
pid += bytes(bytearray([PID_ALPHABET[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
|
||||
return pid
|
||||
|
||||
|
||||
def safe_join(directory, filename):
|
||||
"""Join *filename* onto *directory*, guaranteeing the result stays inside it.
|
||||
|
||||
Any directory components in *filename* are stripped (defeating ``../``
|
||||
traversal) and the resolved path is verified to be contained within
|
||||
*directory*. Raises ``ValueError`` for a name that would still escape.
|
||||
"""
|
||||
filename = os.path.basename(filename)
|
||||
target = os.path.join(directory, filename)
|
||||
real_target = os.path.abspath(target)
|
||||
real_dir = os.path.abspath(directory)
|
||||
if not (real_target == real_dir or real_target.startswith(real_dir + os.sep)):
|
||||
raise ValueError("Unsafe path: {0!r} escapes {1!r}".format(filename, directory))
|
||||
return target
|
||||
|
||||
Reference in New Issue
Block a user