From bb96fe90ac1df5bf58f82bf1aa5b382aa787ce0a Mon Sep 17 00:00:00 2001 From: Jason Ross Date: Wed, 24 Jun 2026 09:17:48 -0500 Subject: [PATCH] Deduplicate PID/unpad helpers and generalize the Topaz path fix Move the copies of unpad, crc32, checksumPid and pidFromSerial that were scattered across the plugin into utilities.py, and add a shared safe_join that generalizes the Topaz extraction path-traversal fix. - unpad: adobekey, ineptepub and ineptpdf import the shared helper. The four bare-script key tools keep their local copies since they run without a package context. - checksumPid is type-preserving (bytes for kgenpids/kindlepid, str for mobidedrm) so every call site keeps its exact behavior. - kgenpids falls back to an absolute import because it is imported top-level by the worker modules. - topazextract.extractFiles uses safe_join; genbook is unchanged as it only handles already-sanitized names. Co-Authored-By: Claude Opus 4.8 --- DeDRM_plugin/adobekey.py | 7 +-- DeDRM_plugin/ineptepub.py | 7 +-- DeDRM_plugin/ineptpdf.py | 7 +-- DeDRM_plugin/kgenpids.py | 46 +++++--------------- DeDRM_plugin/kindlepid.py | 39 +---------------- DeDRM_plugin/mobidedrm.py | 24 +---------- DeDRM_plugin/topazextract.py | 14 ++---- DeDRM_plugin/utilities.py | 83 +++++++++++++++++++++++++++++++++++- 8 files changed, 101 insertions(+), 126 deletions(-) diff --git a/DeDRM_plugin/adobekey.py b/DeDRM_plugin/adobekey.py index b212db9..7465d37 100644 --- a/DeDRM_plugin/adobekey.py +++ b/DeDRM_plugin/adobekey.py @@ -47,7 +47,7 @@ from base64 import b64decode #@@CALIBRE_COMPAT_CODE@@ -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, unpad from .argv_utils import unicode_argv @@ -75,11 +75,6 @@ if iswindows: except ImportError: from Crypto.Cipher import AES - def unpad(data, padding=16): - pad_len = data[-1] - - return data[:-pad_len] - DEVICE_KEY_PATH = r'Software\Adobe\Adept\Device' PRIVATE_LICENCE_KEY_PATH = r'Software\Adobe\Adept\Activation' diff --git a/DeDRM_plugin/ineptepub.py b/DeDRM_plugin/ineptepub.py index d9ffac2..8186ada 100644 --- a/DeDRM_plugin/ineptepub.py +++ b/DeDRM_plugin/ineptepub.py @@ -62,14 +62,9 @@ except ImportError: from Crypto.PublicKey import RSA -def unpad(data, padding=16): - pad_len = data[-1] - - return data[:-pad_len] - #@@CALIBRE_COMPAT_CODE@@ -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, unpad from .argv_utils import unicode_argv diff --git a/DeDRM_plugin/ineptpdf.py b/DeDRM_plugin/ineptpdf.py index 7680b8c..fb8938d 100755 --- a/DeDRM_plugin/ineptpdf.py +++ b/DeDRM_plugin/ineptpdf.py @@ -84,14 +84,9 @@ except ImportError: from Crypto.PublicKey import RSA -def unpad(data, padding=16): - pad_len = data[-1] - - return data[:-pad_len] - #@@CALIBRE_COMPAT_CODE@@ -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, unpad from .argv_utils import unicode_argv iswindows = sys.platform.startswith('win') diff --git a/DeDRM_plugin/kgenpids.py b/DeDRM_plugin/kgenpids.py index b133484..90d1d13 100644 --- a/DeDRM_plugin/kgenpids.py +++ b/DeDRM_plugin/kgenpids.py @@ -15,12 +15,20 @@ __version__ = '3.0' import os, csv -import binascii -import zlib import re from struct import pack, unpack, unpack_from import traceback +#@@CALIBRE_COMPAT_CODE@@ + +try: + from .utilities import checksumPid, pidFromSerial +except ImportError: + # kgenpids is also imported as a top-level module (``import kgenpids`` from + # topazextract / k4mobidedrm), so fall back to an absolute import when there + # is no package context. + from utilities import checksumPid, pidFromSerial + class DrmException(Exception): pass @@ -141,40 +149,6 @@ def generateDevicePID(table,dsn,nbRoll): pidAscii += bytes(bytearray([charMap4[index]])) return pidAscii -def crc32(s): - return (~binascii.crc32(s,-1))&0xFFFFFFFF - -# convert from 8 digit PID to 10 digit PID with checksum -def checksumPid(s): - global charMap4 - crc = crc32(s) - crc = crc ^ (crc >> 16) - res = s - l = len(charMap4) - for i in (0,1): - b = crc & 0xff - pos = (b // l) ^ (b % l) - res += bytes(bytearray([charMap4[pos%l]])) - crc >>= 8 - return res - - -# old kindle serial number to fixed pid -def pidFromSerial(s, l): - global charMap4 - crc = crc32(s) - arr1 = [0]*l - for i in range(len(s)): - arr1[i%l] ^= s[i] - crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff] - for i in range(l): - arr1[i] ^= crc_bytes[i&3] - pid = b"" - for i in range(l): - b = arr1[i] & 0xff - pid += bytes(bytearray([charMap4[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]])) - return pid - # Parse the EXTH header records and use the Kindle serial number to calculate the book pid. def getKindlePids(rec209, token, serialnum): diff --git a/DeDRM_plugin/kindlepid.py b/DeDRM_plugin/kindlepid.py index 3700fe9..f68b77e 100644 --- a/DeDRM_plugin/kindlepid.py +++ b/DeDRM_plugin/kindlepid.py @@ -14,49 +14,12 @@ import sys -import binascii #@@CALIBRE_COMPAT_CODE@@ -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, checksumPid, pidFromSerial from .argv_utils import unicode_argv -letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789' - -def crc32(s): - return (~binascii.crc32(s,-1))&0xFFFFFFFF - -def checksumPid(s): - crc = crc32(s) - crc = crc ^ (crc >> 16) - res = s - l = len(letters) - for i in (0,1): - b = crc & 0xff - pos = (b // l) ^ (b % l) - res += bytes(bytearray([letters[pos%l]])) - crc >>= 8 - - return res - -def pidFromSerial(s, l): - crc = crc32(s) - - arr1 = [0]*l - for i in range(len(s)): - arr1[i%l] ^= s[i] - - crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff] - for i in range(l): - arr1[i] ^= crc_bytes[i&3] - - pid = b"" - for i in range(l): - b = arr1[i] & 0xff - pid+=bytes(bytearray([letters[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]])) - - return pid - def cli_main(): print("Mobipocket PID calculator for Amazon Kindle. Copyright © 2007, 2009 Igor Skochinsky") argv=unicode_argv("kindlepid.py") diff --git a/DeDRM_plugin/mobidedrm.py b/DeDRM_plugin/mobidedrm.py index 96c20ba..f89ba0f 100755 --- a/DeDRM_plugin/mobidedrm.py +++ b/DeDRM_plugin/mobidedrm.py @@ -78,14 +78,13 @@ __version__ = "1.1" import sys import os import struct -import binascii #@@CALIBRE_COMPAT_CODE@@ from .alfcrypto import Pukall_Cipher -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, checksumPid from .argv_utils import unicode_argv @@ -105,27 +104,6 @@ def PC1(key, src, decryption=True): except: raise -letters = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789' - -def crc32(s): - return (~binascii.crc32(s,-1))&0xFFFFFFFF - -def checksumPid(s): - - s = s.encode() - - - crc = crc32(s) - crc = crc ^ (crc >> 16) - res = s - l = len(letters) - for i in (0,1): - b = crc & 0xff - pos = (b // l) ^ (b % l) - res += bytes(bytearray([letters[pos%l]])) - crc >>= 8 - return res.decode() - # expects bytearray def getSizeOfTrailingDataEntries(ptr, size, flags): def getSizeOfTrailingDataEntry(ptr, size): diff --git a/DeDRM_plugin/topazextract.py b/DeDRM_plugin/topazextract.py index 59b892d..4118743 100644 --- a/DeDRM_plugin/topazextract.py +++ b/DeDRM_plugin/topazextract.py @@ -24,7 +24,7 @@ from struct import pack from struct import unpack from .alfcrypto import Topaz_Cipher -from .utilities import SafeUnbuffered +from .utilities import SafeUnbuffered, safe_join from .argv_utils import unicode_argv @@ -358,15 +358,9 @@ class TopazBook: if name == b'glyphs': destdir = os.path.join(outdir,"glyphs") # The record name is read verbatim from the (untrusted) book - # file, so strip any directory components to stop a crafted - # name (e.g. "../../foo") from writing outside of destdir. - fname = os.path.basename(fname) - outputFile = os.path.join(destdir,fname) - # Defense in depth: never let the resolved path escape destdir. - real_out = os.path.abspath(outputFile) - real_dir = os.path.abspath(destdir) - if not (real_out == real_dir or real_out.startswith(real_dir + os.sep)): - raise DrmException("Invalid book record name: {0}".format(repr(name))) + # file; safe_join strips any directory components and refuses + # a name that would escape destdir (e.g. "../../foo"). + outputFile = safe_join(destdir, fname) print(".", end=' ') record = self.getBookPayloadRecord(name,index) if isinstance(record, str): diff --git a/DeDRM_plugin/utilities.py b/DeDRM_plugin/utilities.py index 9cdf3c4..f89942e 100644 --- a/DeDRM_plugin/utilities.py +++ b/DeDRM_plugin/utilities.py @@ -5,6 +5,10 @@ __license__ = 'GPL v3' +import binascii +import os + + def uStrCmp (s1, s2, caseless=False): import unicodedata as ud str1 = s1 if isinstance(s1, str) else str(s1) @@ -39,4 +43,81 @@ class SafeUnbuffered: raise def __getattr__(self, attr): return getattr(self.stream, attr) - \ No newline at end of file + + +def unpad(data, padding=16): + """Strip PKCS#7-style padding by trusting the final byte as the pad length. + + This matches the historical inline implementations across the plugin: it + does not validate that every pad byte is equal, it simply removes the + number of trailing bytes given by the last byte. + """ + pad_len = data[-1] + + return data[:-pad_len] + + +# Alphabet used to encode Kindle/Mobipocket PID checksum characters. +PID_ALPHABET = b'ABCDEFGHIJKLMNPQRSTUVWXYZ123456789' + + +def crc32(s): + return (~binascii.crc32(s, -1)) & 0xFFFFFFFF + + +def checksumPid(s): + """Convert an 8-digit PID into a 10-digit PID with a 2-character checksum. + + The result is the same type (``str`` or ``bytes``) as the argument, which + preserves the historical contracts of every call site: kgenpids and + kindlepid pass and expect ``bytes`` while mobidedrm passes and expects + ``str``. + """ + want_str = isinstance(s, str) + if want_str: + s = s.encode() + crc = crc32(s) + crc = crc ^ (crc >> 16) + res = s + length = len(PID_ALPHABET) + for _ in (0, 1): + b = crc & 0xff + pos = (b // length) ^ (b % length) + res += bytes(bytearray([PID_ALPHABET[pos % length]])) + crc >>= 8 + return res.decode() if want_str else res + + +def pidFromSerial(s, l): + """Convert an (old Kindle) serial number into a fixed-length PID. + + Takes and returns ``bytes``. + """ + crc = crc32(s) + arr1 = [0] * l + for i in range(len(s)): + arr1[i % l] ^= s[i] + crc_bytes = [crc >> 24 & 0xff, crc >> 16 & 0xff, crc >> 8 & 0xff, crc & 0xff] + for i in range(l): + arr1[i] ^= crc_bytes[i & 3] + pid = b"" + for i in range(l): + b = arr1[i] & 0xff + pid += bytes(bytearray([PID_ALPHABET[(b >> 7) + ((b >> 5 & 3) ^ (b & 0x1f))]])) + return pid + + +def safe_join(directory, filename): + """Join *filename* onto *directory*, guaranteeing the result stays inside it. + + Any directory components in *filename* are stripped (defeating ``../`` + traversal) and the resolved path is verified to be contained within + *directory*. Raises ``ValueError`` for a name that would still escape. + """ + filename = os.path.basename(filename) + target = os.path.join(directory, filename) + real_target = os.path.abspath(target) + real_dir = os.path.abspath(directory) + if not (real_target == real_dir or real_target.startswith(real_dir + os.sep)): + raise ValueError("Unsafe path: {0!r} escapes {1!r}".format(filename, directory)) + return target