Deduplicate PID/unpad helpers and generalize the Topaz path fix

Move the copies of unpad, crc32, checksumPid and pidFromSerial that were
scattered across the plugin into utilities.py, and add a shared safe_join
that generalizes the Topaz extraction path-traversal fix.

- unpad: adobekey, ineptepub and ineptpdf import the shared helper. The
  four bare-script key tools keep their local copies since they run
  without a package context.
- checksumPid is type-preserving (bytes for kgenpids/kindlepid, str for
  mobidedrm) so every call site keeps its exact behavior.
- kgenpids falls back to an absolute import because it is imported
  top-level by the worker modules.
- topazextract.extractFiles uses safe_join; genbook is unchanged as it
  only handles already-sanitized names.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-24 09:17:48 -05:00
co-authored by Claude Opus 4.8
parent 1ca134f5c3
commit bb96fe90ac
8 changed files with 101 additions and 126 deletions
+1 -6
View File
@@ -47,7 +47,7 @@ from base64 import b64decode
#@@CALIBRE_COMPAT_CODE@@
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, unpad
from .argv_utils import unicode_argv
@@ -75,11 +75,6 @@ if iswindows:
except ImportError:
from Crypto.Cipher import AES
def unpad(data, padding=16):
pad_len = data[-1]
return data[:-pad_len]
DEVICE_KEY_PATH = r'Software\Adobe\Adept\Device'
PRIVATE_LICENCE_KEY_PATH = r'Software\Adobe\Adept\Activation'
+1 -6
View File
@@ -62,14 +62,9 @@ except ImportError:
from Crypto.PublicKey import RSA
def unpad(data, padding=16):
pad_len = data[-1]
return data[:-pad_len]
#@@CALIBRE_COMPAT_CODE@@
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, unpad
from .argv_utils import unicode_argv
+1 -6
View File
@@ -84,14 +84,9 @@ except ImportError:
from Crypto.PublicKey import RSA
def unpad(data, padding=16):
pad_len = data[-1]
return data[:-pad_len]
#@@CALIBRE_COMPAT_CODE@@
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, unpad
from .argv_utils import unicode_argv
iswindows = sys.platform.startswith('win')
+10 -36
View File
@@ -15,12 +15,20 @@ __version__ = '3.0'
import os, csv
import binascii
import zlib
import re
from struct import pack, unpack, unpack_from
import traceback
#@@CALIBRE_COMPAT_CODE@@
try:
from .utilities import checksumPid, pidFromSerial
except ImportError:
# kgenpids is also imported as a top-level module (``import kgenpids`` from
# topazextract / k4mobidedrm), so fall back to an absolute import when there
# is no package context.
from utilities import checksumPid, pidFromSerial
class DrmException(Exception):
pass
@@ -141,40 +149,6 @@ def generateDevicePID(table,dsn,nbRoll):
pidAscii += bytes(bytearray([charMap4[index]]))
return pidAscii
def crc32(s):
return (~binascii.crc32(s,-1))&0xFFFFFFFF
# convert from 8 digit PID to 10 digit PID with checksum
def checksumPid(s):
global charMap4
crc = crc32(s)
crc = crc ^ (crc >> 16)
res = s
l = len(charMap4)
for i in (0,1):
b = crc & 0xff
pos = (b // l) ^ (b % l)
res += bytes(bytearray([charMap4[pos%l]]))
crc >>= 8
return res
# old kindle serial number to fixed pid
def pidFromSerial(s, l):
global charMap4
crc = crc32(s)
arr1 = [0]*l
for i in range(len(s)):
arr1[i%l] ^= s[i]
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
for i in range(l):
arr1[i] ^= crc_bytes[i&3]
pid = b""
for i in range(l):
b = arr1[i] & 0xff
pid += bytes(bytearray([charMap4[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
return pid
# Parse the EXTH header records and use the Kindle serial number to calculate the book pid.
def getKindlePids(rec209, token, serialnum):
+1 -38
View File
@@ -14,49 +14,12 @@
import sys
import binascii
#@@CALIBRE_COMPAT_CODE@@
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, checksumPid, pidFromSerial
from .argv_utils import unicode_argv
letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
def crc32(s):
return (~binascii.crc32(s,-1))&0xFFFFFFFF
def checksumPid(s):
crc = crc32(s)
crc = crc ^ (crc >> 16)
res = s
l = len(letters)
for i in (0,1):
b = crc & 0xff
pos = (b // l) ^ (b % l)
res += bytes(bytearray([letters[pos%l]]))
crc >>= 8
return res
def pidFromSerial(s, l):
crc = crc32(s)
arr1 = [0]*l
for i in range(len(s)):
arr1[i%l] ^= s[i]
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
for i in range(l):
arr1[i] ^= crc_bytes[i&3]
pid = b""
for i in range(l):
b = arr1[i] & 0xff
pid+=bytes(bytearray([letters[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
return pid
def cli_main():
print("Mobipocket PID calculator for Amazon Kindle. Copyright © 2007, 2009 Igor Skochinsky")
argv=unicode_argv("kindlepid.py")
+1 -23
View File
@@ -78,14 +78,13 @@ __version__ = "1.1"
import sys
import os
import struct
import binascii
#@@CALIBRE_COMPAT_CODE@@
from .alfcrypto import Pukall_Cipher
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, checksumPid
from .argv_utils import unicode_argv
@@ -105,27 +104,6 @@ def PC1(key, src, decryption=True):
except:
raise
letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
def crc32(s):
return (~binascii.crc32(s,-1))&0xFFFFFFFF
def checksumPid(s):
s = s.encode()
crc = crc32(s)
crc = crc ^ (crc >> 16)
res = s
l = len(letters)
for i in (0,1):
b = crc & 0xff
pos = (b // l) ^ (b % l)
res += bytes(bytearray([letters[pos%l]]))
crc >>= 8
return res.decode()
# expects bytearray
def getSizeOfTrailingDataEntries(ptr, size, flags):
def getSizeOfTrailingDataEntry(ptr, size):
+4 -10
View File
@@ -24,7 +24,7 @@ from struct import pack
from struct import unpack
from .alfcrypto import Topaz_Cipher
from .utilities import SafeUnbuffered
from .utilities import SafeUnbuffered, safe_join
from .argv_utils import unicode_argv
@@ -358,15 +358,9 @@ class TopazBook:
if name == b'glyphs':
destdir = os.path.join(outdir,"glyphs")
# The record name is read verbatim from the (untrusted) book
# file, so strip any directory components to stop a crafted
# name (e.g. "../../foo") from writing outside of destdir.
fname = os.path.basename(fname)
outputFile = os.path.join(destdir,fname)
# Defense in depth: never let the resolved path escape destdir.
real_out = os.path.abspath(outputFile)
real_dir = os.path.abspath(destdir)
if not (real_out == real_dir or real_out.startswith(real_dir + os.sep)):
raise DrmException("Invalid book record name: {0}".format(repr(name)))
# file; safe_join strips any directory components and refuses
# a name that would escape destdir (e.g. "../../foo").
outputFile = safe_join(destdir, fname)
print(".", end=' ')
record = self.getBookPayloadRecord(name,index)
if isinstance(record, str):
+82 -1
View File
@@ -5,6 +5,10 @@
__license__ = 'GPL v3'
import binascii
import os
def uStrCmp (s1, s2, caseless=False):
import unicodedata as ud
str1 = s1 if isinstance(s1, str) else str(s1)
@@ -39,4 +43,81 @@ class SafeUnbuffered:
raise
def __getattr__(self, attr):
return getattr(self.stream, attr)
def unpad(data, padding=16):
"""Strip PKCS#7-style padding by trusting the final byte as the pad length.
This matches the historical inline implementations across the plugin: it
does not validate that every pad byte is equal, it simply removes the
number of trailing bytes given by the last byte.
"""
pad_len = data[-1]
return data[:-pad_len]
# Alphabet used to encode Kindle/Mobipocket PID checksum characters.
PID_ALPHABET = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789'
def crc32(s):
return (~binascii.crc32(s, -1)) & 0xFFFFFFFF
def checksumPid(s):
"""Convert an 8-digit PID into a 10-digit PID with a 2-character checksum.
The result is the same type (``str`` or ``bytes``) as the argument, which
preserves the historical contracts of every call site: kgenpids and
kindlepid pass and expect ``bytes`` while mobidedrm passes and expects
``str``.
"""
want_str = isinstance(s, str)
if want_str:
s = s.encode()
crc = crc32(s)
crc = crc ^ (crc >> 16)
res = s
length = len(PID_ALPHABET)
for _ in (0, 1):
b = crc & 0xff
pos = (b // length) ^ (b % length)
res += bytes(bytearray([PID_ALPHABET[pos % length]]))
crc >>= 8
return res.decode() if want_str else res
def pidFromSerial(s, l):
"""Convert an (old Kindle) serial number into a fixed-length PID.
Takes and returns ``bytes``.
"""
crc = crc32(s)
arr1 = [0] * l
for i in range(len(s)):
arr1[i % l] ^= s[i]
crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff]
for i in range(l):
arr1[i] ^= crc_bytes[i & 3]
pid = b""
for i in range(l):
b = arr1[i] & 0xff
pid += bytes(bytearray([PID_ALPHABET[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]]))
return pid
def safe_join(directory, filename):
"""Join *filename* onto *directory*, guaranteeing the result stays inside it.
Any directory components in *filename* are stripped (defeating ``../``
traversal) and the resolved path is verified to be contained within
*directory*. Raises ``ValueError`` for a name that would still escape.
"""
filename = os.path.basename(filename)
target = os.path.join(directory, filename)
real_target = os.path.abspath(target)
real_dir = os.path.abspath(directory)
if not (real_target == real_dir or real_target.startswith(real_dir + os.sep)):
raise ValueError("Unsafe path: {0!r} escapes {1!r}".format(filename, directory))
return target