Add best-effort PII redaction pass (internal/scrub) (#8)

Introduce package internal/scrub, a schema-agnostic, regex/heuristic
based redaction pass to run over raw bug-report payloads before storage
(#9). It masks (never deletes) matches with bracketed placeholders so
payload structure is preserved for triage.

Categories:
- Emails: robust address regex; ignores @handles and "meet @ 3pm".
- Auth tokens/secrets: Authorization/Proxy-Authorization header values,
  standalone Bearer tokens, eyJ-anchored JWTs, well-known provider key
  formats (GitHub, GitLab, Slack, Stripe, OpenAI, Google, AWS), and
  values under secret-named keys (password, api_key, token, ...).
- IP addresses: octet-validated IPv4 and comprehensive IPv6 (full,
  compressed, loopback, IPv4-mapped), ordered for correct extraction.
- Names: deliberately weak, key-directed heuristic (name/user/...),
  \b-anchored to avoid filename/hostname collisions. Documented in code
  as best-effort and NOT to be relied upon.

API: Scrub([]byte) []byte, ScrubString(string) string, plus composable
per-category RedactEmails/RedactTokens/RedactIPs/RedactNames and exported
Placeholder* constants. Non-mutating and idempotent. Build-tag-free so it
compiles for host and the Wasm target.

Tests cover each category with positive and over-redaction-guard cases
(89 passing checks); go test ./... is green.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-02 13:46:45 -05:00
co-authored by Claude Opus 4.8
parent 8e1dc66c54
commit 6508679d86
2 changed files with 592 additions and 0 deletions
+256
View File
@@ -0,0 +1,256 @@
// Package scrub performs a BEST-EFFORT PII redaction pass over raw bug-report
// payloads before they are persisted.
//
// # Best-effort, not a guarantee
//
// This package makes a good-faith attempt to remove obviously sensitive data
// (email addresses, auth tokens/secrets, IP addresses, and — very weakly —
// personal names). It is a defence-in-depth layer, NOT a guarantee: it WILL
// miss things and it MAY over-redact non-sensitive data that merely resembles a
// sensitive pattern. Callers and operators MUST NOT treat a scrubbed payload as
// certified free of PII. The authoritative statement of this limitation lives in
// the privacy documentation (see issue #12); keep this comment aligned with it.
//
// # Why regex/heuristic based
//
// The bug-report payload schema is not finalised (see issue #7), so redaction
// operates on raw text/bytes using regular expressions and lightweight
// heuristics. This keeps it schema-agnostic: it works the same whether the
// payload is JSON, form-encoded, a log excerpt, or free text, and it can be
// dropped in front of storage (see issue #9) without a schema dependency.
//
// # Design
//
// Redaction is expressed as an ordered list of (regexp, replacement) rules
// grouped into categories. Each rule MASKS rather than deletes — matched spans
// are replaced with a bracketed placeholder such as "[REDACTED_EMAIL]" — so the
// surrounding structure of the payload is preserved for triage. The categories
// are exposed individually (RedactEmails, RedactTokens, RedactIPs, RedactNames)
// so a caller can compose only the passes it wants; ScrubString / Scrub run all
// of them in an order chosen to avoid rules clobbering each other's output.
//
// The pass is idempotent: running it twice yields the same result, because the
// placeholders it emits are constructed so that no rule matches them.
//
// This package carries no build constraints, so it compiles and is unit-tested
// with the standard Go toolchain on the host and is reused verbatim by the
// Cloudflare Worker Wasm build.
package scrub
import "regexp"
// Placeholder tokens substituted in place of redacted spans. They are exported
// so that downstream code (e.g. the storage layer in #9) and tests can refer to
// them without hard-coding string literals. Each is deliberately bracketed and
// upper-snake-case so that no redaction rule re-matches it, keeping the pass
// idempotent.
const (
PlaceholderEmail = "[REDACTED_EMAIL]"
PlaceholderToken = "[REDACTED_TOKEN]"
PlaceholderAuth = "[REDACTED_AUTH]"
PlaceholderIP = "[REDACTED_IP]"
PlaceholderName = "[REDACTED_NAME]"
)
// redactor is a single compiled redaction rule. replacement may reference
// capture groups from re using the ${n} syntax (see regexp.Regexp.ReplaceAllString).
type redactor struct {
re *regexp.Regexp
replacement string
}
// apply runs a sequence of redactors over s in order.
func apply(rules []redactor, s string) string {
for _, r := range rules {
s = r.re.ReplaceAllString(s, r.replacement)
}
return s
}
// Scrub returns a best-effort PII-redacted copy of payload. The input is never
// mutated; a fresh slice is returned (a nil/empty input is returned unchanged).
//
// This is the primary entry point for the storage path (#9): call it on the raw
// payload bytes immediately before persistence.
func Scrub(payload []byte) []byte {
if len(payload) == 0 {
return payload
}
return []byte(ScrubString(string(payload)))
}
// ScrubString is the string-typed equivalent of Scrub. It applies every
// redaction category in a fixed order:
//
// 1. tokens/secrets — so an Authorization header or "password": "…" value is
// masked as a whole before narrower rules (email, IP) can nibble at it;
// 2. emails;
// 3. IP addresses;
// 4. names — last, and deliberately skipping values that are already a
// placeholder, so it never relabels e.g. an email it cannot see past a key.
func ScrubString(s string) string {
s = RedactTokens(s)
s = RedactEmails(s)
s = RedactIPs(s)
s = RedactNames(s)
return s
}
// RedactEmails masks email addresses. Exposed for composition.
func RedactEmails(s string) string { return apply(emailRedactors, s) }
// RedactTokens masks auth tokens and secrets: Authorization/auth headers, bare
// bearer tokens, JWTs, well-known API-key/secret formats, and values assigned to
// obviously secret-named keys (password, api_key, token, …). Exposed for composition.
func RedactTokens(s string) string { return apply(tokenRedactors, s) }
// RedactIPs masks IPv4 and IPv6 addresses. Exposed for composition.
func RedactIPs(s string) string { return apply(ipRedactors, s) }
// RedactNames applies the BEST-EFFORT, deliberately weak name heuristic. See the
// nameRedactors documentation for its (significant) limitations. Exposed for composition.
func RedactNames(s string) string { return apply(nameRedactors, s) }
// ---------------------------------------------------------------------------
// Email
// ---------------------------------------------------------------------------
var emailRedactors = []redactor{
// Pragmatic RFC-5321-ish address: a local part, "@", a dotted domain, and a
// 2+ letter TLD. This intentionally does NOT match bare "@handle" mentions
// (no local part) or "meet @ 3pm" (space around @), guarding against
// over-redaction of social handles and prose.
{
re: regexp.MustCompile(`[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}`),
replacement: PlaceholderEmail,
},
}
// ---------------------------------------------------------------------------
// Tokens & secrets
// ---------------------------------------------------------------------------
var tokenRedactors = []redactor{
// Authorization / Proxy-Authorization / X-...-Authorization header values,
// in both header form ("Authorization: Bearer <t>") and JSON form
// ("\"authorization\": \"Bearer <t>\""). The whole credential (including any
// scheme word) is replaced; the key and separator are preserved via ${1}.
{
re: regexp.MustCompile(`(?i)((?:proxy-)?authorization\s*"?\s*[:=]\s*"?)(?:(?:bearer|basic|digest|token|negotiate)\s+)?[A-Za-z0-9._~+/=\-]+`),
replacement: "${1}" + PlaceholderAuth,
},
// Standalone "Bearer <token>" not attached to an Authorization key. Requires
// a fairly long credential so the English word "bearer" in prose (e.g.
// "bearer of bad news") is not redacted.
{
re: regexp.MustCompile(`(?i)\bbearer\s+[A-Za-z0-9._~+/=\-]{16,}`),
replacement: PlaceholderToken,
},
// JWTs, anchored on the "eyJ" header prefix (base64url of `{"`). Anchoring
// keeps false positives near-zero: dotted identifiers such as Java package
// names or "www.example.com" never start with eyJ and never carry base64url
// payload/signature segments. Non-standard JWTs that do NOT start with eyJ
// are only caught if they happen to hit one of the rules below.
{
re: regexp.MustCompile(`eyJ[A-Za-z0-9_\-]{5,}\.[A-Za-z0-9_\-]{5,}\.[A-Za-z0-9_\-]{5,}`),
replacement: PlaceholderToken,
},
// Well-known provider API-key / secret formats. These are case-sensitive by
// design (their prefixes are fixed-case) and the random tails are restricted
// to base62 where possible so kebab-case identifiers like the CSS class
// "sk-loading-spinner" are not mistaken for an OpenAI "sk-" key.
{re: regexp.MustCompile(`\bgh[pousr]_[A-Za-z0-9]{20,}`), replacement: PlaceholderToken}, // GitHub tokens
{re: regexp.MustCompile(`\bgithub_pat_[A-Za-z0-9_]{20,}`), replacement: PlaceholderToken}, // GitHub fine-grained PAT
{re: regexp.MustCompile(`\bglpat-[A-Za-z0-9_\-]{16,}`), replacement: PlaceholderToken}, // GitLab PAT
{re: regexp.MustCompile(`\bxox[baprs]-[A-Za-z0-9-]{10,}`), replacement: PlaceholderToken}, // Slack tokens
{re: regexp.MustCompile(`\b(?:sk|pk|rk)_(?:live|test)_[A-Za-z0-9]{10,}`), replacement: PlaceholderToken}, // Stripe keys
{re: regexp.MustCompile(`\bsk-(?:proj-)?[A-Za-z0-9]{20,}`), replacement: PlaceholderToken}, // OpenAI keys
{re: regexp.MustCompile(`\bAIza[0-9A-Za-z_\-]{35}`), replacement: PlaceholderToken}, // Google API key
{re: regexp.MustCompile(`\b(?:AKIA|ASIA|AGPA|AIDA|AROA|AIPA|ANPA|ANVA|APKA)[0-9A-Z]{16}`), replacement: PlaceholderToken}, // AWS access key IDs
// Values assigned to obviously-secret keys, in JSON ("token": "…") or
// flag/env form (api_key=…). Key + separator (+ optional opening quote) are
// preserved via ${1}; the value up to the next quote/space/comma/brace is
// masked. This is the catch-all for opaque secrets that have no recognisable
// standalone shape.
{
re: regexp.MustCompile(`(?i)\b((?:passwords?|passwd|pwd|secret[_-]?key|client[_-]?secret|api[_-]?keys?|apikeys?|access[_-]?tokens?|refresh[_-]?tokens?|auth[_-]?tokens?|private[_-]?keys?|secrets?|tokens?)"?\s*[:=]\s*"?)[^"\s,}]+`),
replacement: "${1}" + PlaceholderToken,
},
}
// ---------------------------------------------------------------------------
// IP addresses
// ---------------------------------------------------------------------------
// IPv6 is matched before IPv4 so that an IPv4-mapped IPv6 address is masked as a
// single unit rather than leaving an "::ffff:[REDACTED_IP]" fragment.
var ipRedactors = []redactor{
{re: regexp.MustCompile(ipv6Pattern), replacement: PlaceholderIP},
// IPv4 with per-octet 0-255 validation, so "999.1.1.1" and 3-part semantic
// versions ("v1.2.3") are not matched. A genuine 4-part version like
// "1.2.3.4" is indistinguishable from an IP by regex and IS masked; this
// ambiguity is documented as an accepted best-effort limitation.
{re: regexp.MustCompile(`\b(?:(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])\.){3}(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])\b`), replacement: PlaceholderIP},
}
// ipv6Pattern is a comprehensive IPv6 matcher (RE2-safe: no back-references or
// look-around). It covers full, compressed ("::"), loopback ("::1") and
// IPv4-mapped forms. It deliberately requires either 8 groups or a "::", so
// single-colon sequences such as clock times ("12:34:56") and MAC addresses are
// not matched.
//
// Ordering matters: Go's regexp is leftmost-FIRST (Perl semantics), and this
// pattern is used unanchored for extraction, so the alternatives are ordered
// from most-consuming to least. The IPv4-embedding and multi-trailing-group
// forms come before the bare "…::" form; otherwise an address like
// "fe80::1ff:fe23:4567:890a" would match only its "fe80::" prefix.
const ipv6Pattern = `(?:[0-9A-Fa-f]{1,4}:){7}[0-9A-Fa-f]{1,4}` + // 1:2:3:4:5:6:7:8
`|(?:[0-9A-Fa-f]{1,4}:){1,4}:(?:(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])\.){3}(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])` + // …::IPv4
`|::(?:[Ff]{4}(?::0{1,4})?:)?(?:(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])\.){3}(?:25[0-5]|2[0-4][0-9]|1[0-9][0-9]|[1-9]?[0-9])` + // ::ffff:IPv4
`|(?:[0-9A-Fa-f]{1,4}:){1,2}(?::[0-9A-Fa-f]{1,4}){1,5}` +
`|(?:[0-9A-Fa-f]{1,4}:){1,3}(?::[0-9A-Fa-f]{1,4}){1,4}` +
`|(?:[0-9A-Fa-f]{1,4}:){1,4}(?::[0-9A-Fa-f]{1,4}){1,3}` +
`|(?:[0-9A-Fa-f]{1,4}:){1,5}(?::[0-9A-Fa-f]{1,4}){1,2}` +
`|(?:[0-9A-Fa-f]{1,4}:){1,6}:[0-9A-Fa-f]{1,4}` +
`|[0-9A-Fa-f]{1,4}:(?::[0-9A-Fa-f]{1,4}){1,6}` +
`|(?:[0-9A-Fa-f]{1,4}:){1,7}:` + // 1:2:3:4:5:6:7::
`|:(?:(?::[0-9A-Fa-f]{1,4}){1,7}|:)` // ::8 / ::
// ---------------------------------------------------------------------------
// Names (BEST-EFFORT — deliberately weak)
// ---------------------------------------------------------------------------
// nameRedactors implement a deliberately LIMITED, key-directed name heuristic.
//
// Reliable name detection is an unsolved problem, so this makes no attempt at
// it. It only masks the value that immediately follows an obvious name-ish key
// (name, full_name, first_name, last_name, display_name, username, user, …). It
// is anchored with \b so it does NOT fire on suffix collisions such as
// "filename:" or "hostname:".
//
// KNOWN, ACCEPTED LIMITATIONS (capture these in the privacy doc, #12):
// - It MISSES every human name that appears in free-form prose, in an
// unrecognised key, or in a nested structure it cannot parse.
// - It OVER-redacts non-personal values that happen to sit under a name-like
// key, e.g. `name: my-service` in a Kubernetes manifest becomes
// `name: [REDACTED_NAME]`. This is an accepted trade-off, not a bug.
//
// Do NOT rely on this pass to remove names.
var nameRedactors = []redactor{
// Quoted value: "name": "John Doe". The value's first character must not be
// '[', so an existing placeholder (e.g. "[REDACTED_EMAIL]" produced by an
// earlier pass) is left intact rather than relabelled.
{
re: regexp.MustCompile(`(?i)\b((?:first[_ ]?name|last[_ ]?name|full[_ ]?name|display[_ ]?name|user[_ ]?name|nick[_ ]?name|sur[_ ]?name|username|name|user)"?\s*[:=]\s*")[^"\n\[][^"\n]{0,119}"`),
replacement: "${1}" + PlaceholderName + `"`,
},
// Unquoted value: name: John Doe / user=jsmith. Consumes up to the next
// comma, brace or line break. The first character excludes quotes, brackets,
// braces and whitespace so quoted values (handled above), placeholders and
// nested objects/arrays are skipped.
{
re: regexp.MustCompile(`(?i)\b((?:first[_ ]?name|last[_ ]?name|full[_ ]?name|display[_ ]?name|user[_ ]?name|nick[_ ]?name|sur[_ ]?name|username|name|user)\s*[:=]\s*)[^"\s\[',}{\r\n][^,\r\n}]*`),
replacement: "${1}" + PlaceholderName,
},
}
+336
View File
@@ -0,0 +1,336 @@
package scrub
import (
"bytes"
"strings"
"testing"
)
// redactedOK asserts that got differs from in, contains the expected
// placeholder, and no longer contains any of the sensitive substrings.
func redactedOK(t *testing.T, got, in, placeholder string, secrets ...string) {
t.Helper()
if got == in {
t.Errorf("expected redaction but output was unchanged: %q", in)
}
if !strings.Contains(got, placeholder) {
t.Errorf("output %q is missing placeholder %q", got, placeholder)
}
for _, s := range secrets {
if s != "" && strings.Contains(got, s) {
t.Errorf("output %q still leaks sensitive substring %q", got, s)
}
}
}
// unchanged asserts that a supposedly-safe input is passed through verbatim
// (guards against over-redaction).
func unchanged(t *testing.T, got, in string) {
t.Helper()
if got != in {
t.Errorf("expected no change (over-redaction guard):\n in = %q\n got = %q", in, got)
}
}
// ---------------------------------------------------------------------------
// Email
// ---------------------------------------------------------------------------
func TestRedactEmails(t *testing.T) {
redact := []struct {
name, in, secret string
}{
{"simple", `contact me at alice@example.com please`, "alice@example.com"},
{"plus tag and subdomain", `from john.doe+tag@mail.sub.example.co.uk`, "john.doe+tag@mail.sub.example.co.uk"},
{"uppercase", `ALICE@EXAMPLE.COM`, "ALICE@EXAMPLE.COM"},
{"digits and dashes", `user-123.name@my-host.io`, "user-123.name@my-host.io"},
{"inside json", `{"reporter":"bob@example.org"}`, "bob@example.org"},
{"inside angle brackets", `Bob <bob@example.org>`, "bob@example.org"},
}
for _, tc := range redact {
t.Run("redact/"+tc.name, func(t *testing.T) {
redactedOK(t, RedactEmails(tc.in), tc.in, PlaceholderEmail, tc.secret)
})
}
keep := []struct {
name, in string
}{
{"social handle", `follow @acmecorp for updates`},
{"cc mention", `ping @channel now`},
{"meet at time", `let's meet @ 3pm tomorrow`},
{"java annotation", `@Override public void run()`},
{"local only no tld", `login as user@localhost works`},
{"at with no domain", `rate is 5@each item`},
}
for _, tc := range keep {
t.Run("keep/"+tc.name, func(t *testing.T) {
unchanged(t, RedactEmails(tc.in), tc.in)
})
}
}
// ---------------------------------------------------------------------------
// Tokens & secrets
// ---------------------------------------------------------------------------
func TestRedactTokens(t *testing.T) {
jwt := `eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w5N_XgL0n3I9PlFUP0THsR8U`
// gitlabPAT is assembled at runtime rather than written as a literal: the
// GitLab token pattern has no checksum, so secret scanners flag any
// contiguous "glpat-<20 chars>" string on sight — even obvious test data.
// Concatenation keeps the literal out of the source blob while still
// producing a value that exercises the redaction regex.
gitlabPAT := "glpat-" + strings.Repeat("A", 20)
redact := []struct {
name, in, placeholder, secret string
}{
// Authorization headers (header + JSON forms, several schemes).
{"auth header bearer", `Authorization: Bearer ` + jwt, PlaceholderAuth, jwt},
{"auth header basic", `Authorization: Basic dXNlcjpwYXNzd29yZA==`, PlaceholderAuth, "dXNlcjpwYXNzd29yZA=="},
{"auth header json", `"authorization": "Bearer abc.def.ghijklmnop"`, PlaceholderAuth, "abc.def.ghijklmnop"},
{"proxy auth header", `Proxy-Authorization: Bearer sometokenvalue1234`, PlaceholderAuth, "sometokenvalue1234"},
// Standalone bearer.
{"bearer standalone", `sent header Bearer 0123456789abcdefghij done`, PlaceholderToken, "0123456789abcdefghij"},
// JWT anchored on eyJ.
{"jwt bare", `id_token=` + jwt, PlaceholderToken, jwt},
// Well-known provider key formats.
{"github token", `token ghp_0123456789abcdefghijklmnopqrstuvwxyz`, PlaceholderToken, "ghp_0123456789abcdefghijklmnopqrstuvwxyz"},
{"github pat", `github_pat_11ABCDEFG0aBcDeFgHiJkL_mNoPqRsTuVwXyZ012345`, PlaceholderToken, "github_pat_11ABCDEFG0aBcDeFgHiJkL_mNoPqRsTuVwXyZ012345"},
{"gitlab pat", gitlabPAT, PlaceholderToken, gitlabPAT},
{"slack token", `xoxb-1234567890-abcdefghijkl`, PlaceholderToken, "xoxb-1234567890-abcdefghijkl"},
{"stripe secret key", `sk_live_0123456789abcdefABCDEF`, PlaceholderToken, "sk_live_0123456789abcdefABCDEF"},
{"openai key", `sk-abcdefghijklmnopqrstuvwxyz0123`, PlaceholderToken, "sk-abcdefghijklmnopqrstuvwxyz0123"},
{"google api key", `AIzaSyA0123456789abcdefghijklmnopqrstuv`, PlaceholderToken, "AIzaSyA0123456789abcdefghijklmnopqrstuv"},
{"aws access key id", `AKIAIOSFODNN7EXAMPLE`, PlaceholderToken, "AKIAIOSFODNN7EXAMPLE"},
// Secret-named key assignments (JSON / env / header forms).
{"password json", `{"password": "hunter2!"}`, PlaceholderToken, "hunter2!"},
{"api_key env", `api_key=super-secret-value-42`, PlaceholderToken, "super-secret-value-42"},
{"access_token colon", `access_token: aQ9xZ_opaquevalue`, PlaceholderToken, "aQ9xZ_opaquevalue"},
{"client secret", `client_secret = 0f1e2d3c4b5a`, PlaceholderToken, "0f1e2d3c4b5a"},
{"x-auth-token header", `X-Auth-Token: s3cr3tvalue123`, PlaceholderToken, "s3cr3tvalue123"},
{"private key kv", `"private_key":"MIIBVerySecretMaterial"`, PlaceholderToken, "MIIBVerySecretMaterial"},
}
for _, tc := range redact {
t.Run("redact/"+tc.name, func(t *testing.T) {
redactedOK(t, RedactTokens(tc.in), tc.in, tc.placeholder, tc.secret)
})
}
keep := []struct {
name, in string
}{
{"bearer in prose", `he was the bearer of bad news`},
{"css sk prefix", `class="sk-loading-spinner-container-large"`},
{"short aws-like", `AKIASHORT is not a key`},
{"jwt-like domain", `visit www.example.com and analytics.google.com`},
{"java package", `at com.example.myapp.service.Handler.run(Handler.java:42)`},
{"secret word in prose", `there is nothing secret here at all`},
{"semver token-ish", `upgraded to version 1.2.3 today`},
}
for _, tc := range keep {
t.Run("keep/"+tc.name, func(t *testing.T) {
unchanged(t, RedactTokens(tc.in), tc.in)
})
}
}
// ---------------------------------------------------------------------------
// IP addresses
// ---------------------------------------------------------------------------
func TestRedactIPs(t *testing.T) {
redact := []struct {
name, in, secret string
}{
// IPv4
{"ipv4 private", `client 192.168.0.42 connected`, "192.168.0.42"},
{"ipv4 public", `resolved to 8.8.8.8`, "8.8.8.8"},
{"ipv4 broadcast bound", `mask 255.255.255.255 here`, "255.255.255.255"},
{"ipv4 zero", `bound 0.0.0.0:8080`, "0.0.0.0"},
{"ipv4 in json", `{"ip":"10.0.0.1"}`, "10.0.0.1"},
// IPv6
{"ipv6 full", `addr 2001:0db8:85a3:0000:0000:8a2e:0370:7334 up`, "2001:0db8:85a3:0000:0000:8a2e:0370:7334"},
{"ipv6 compressed", `peer 2001:db8::8a2e:370:7334 seen`, "2001:db8::8a2e:370:7334"},
{"ipv6 loopback", `bind ::1 only`, "::1"},
{"ipv6 link local", `via fe80::1ff:fe23:4567:890a here`, "fe80::1ff:fe23:4567:890a"},
{"ipv6 trailing colons", `route 1:2:3:4:5:6:7:: set`, "1:2:3:4:5:6:7::"},
{"ipv6 mapped ipv4", `mapped ::ffff:192.168.1.1 shown`, "::ffff:192.168.1.1"},
}
for _, tc := range redact {
t.Run("redact/"+tc.name, func(t *testing.T) {
redactedOK(t, RedactIPs(tc.in), tc.in, PlaceholderIP, tc.secret)
})
}
keep := []struct {
name, in string
}{
{"clock time", `event at 12:34:56 today`},
{"clock time ms", `stamp 12:00:00 exactly`},
{"mac address", `nic 00:1A:2B:3C:4D:5E present`},
{"semver three part", `running v1.2.3 build`},
{"octet over 255", `not an ip 999.1.1.1 here`},
{"octet 256", `bad 256.100.100.100 value`},
{"key value colon", `config foo:bar baz:qux`},
}
for _, tc := range keep {
t.Run("keep/"+tc.name, func(t *testing.T) {
unchanged(t, RedactIPs(tc.in), tc.in)
})
}
}
// ---------------------------------------------------------------------------
// Names (best-effort)
// ---------------------------------------------------------------------------
func TestRedactNames(t *testing.T) {
redact := []struct {
name, in, secret string
}{
{"json name", `{"name": "John Doe"}`, "John Doe"},
{"unquoted name", `name: Jane Roe`, "Jane Roe"},
{"username", `username: jsmith`, "jsmith"},
{"user", `user=administrator`, "administrator"},
{"first name camel", `"firstName": "Grace"`, "Grace"},
{"last name snake", `last_name: Hopper`, "Hopper"},
{"full name", `"full_name":"Ada Lovelace"`, "Ada Lovelace"},
{"display name", `display_name: coolcat99`, "coolcat99"},
{"nickname", `nickname: Ace`, "Ace"},
}
for _, tc := range redact {
t.Run("redact/"+tc.name, func(t *testing.T) {
redactedOK(t, RedactNames(tc.in), tc.in, PlaceholderName, tc.secret)
})
}
keep := []struct {
name, in string
}{
// Suffix collisions must NOT trigger the bare "name" key.
{"filename", `filename: report.pdf`},
{"hostname", `hostname: web01.internal`},
{"codename", `codename: falcon`},
{"pathname", `pathname: /var/log/app`},
{"user-agent header", `user-agent: Mozilla/5.0`},
{"user_id key", `user_id: 12345`},
// Best-effort limitation: names in free-form prose are NOT detected.
{"name in prose", `my name is Robert and I like tea`},
}
for _, tc := range keep {
t.Run("keep/"+tc.name, func(t *testing.T) {
unchanged(t, RedactNames(tc.in), tc.in)
})
}
}
// ---------------------------------------------------------------------------
// Structure preservation (exact output)
// ---------------------------------------------------------------------------
// TestExactOutput locks the exact, structure-preserving replacements for a
// representative case per category, proving redaction masks rather than deletes.
func TestExactOutput(t *testing.T) {
cases := []struct {
name, in, want string
}{
{"email", `email=alice@example.com;`, `email=` + PlaceholderEmail + `;`},
{"auth header", `Authorization: Bearer abcdef.ghijk.lmnop`, `Authorization: ` + PlaceholderAuth},
{"password json", `{"password": "hunter2"}`, `{"password": "` + PlaceholderToken + `"}`},
{"ipv4", `[10.1.2.3]`, `[` + PlaceholderIP + `]`},
{"ipv6 loopback", `(::1)`, `(` + PlaceholderIP + `)`},
{"name json", `{"name": "John Doe"}`, `{"name": "` + PlaceholderName + `"}`},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
if got := ScrubString(tc.in); got != tc.want {
t.Errorf("ScrubString(%q)\n got = %q\n want = %q", tc.in, got, tc.want)
}
})
}
}
// ---------------------------------------------------------------------------
// Integration, API surface, idempotency, immutability
// ---------------------------------------------------------------------------
const sampleReport = `Bug report:
User contact: alice@example.com, backup bob@work.co.uk
Session: Authorization: Bearer eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.c2lnbmF0dXJlX3ZhbHVl
Config: {"api_key": "sk_live_abcd1234efgh5678ijkl", "name": "Carol Smith"}
Env: AWS_KEY=AKIAIOSFODNN7EXAMPLE password=p@ssw0rd!
Network: connected from 203.0.113.7 via gateway 2001:db8::1
Stack: at com.example.app.Main.run(Main.java:99)`
func TestScrubStringIntegration(t *testing.T) {
got := ScrubString(sampleReport)
mustBeGone := []string{
"alice@example.com", "bob@work.co.uk",
"eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.c2lnbmF0dXJlX3ZhbHVl",
"sk_live_abcd1234efgh5678ijkl", "Carol Smith",
"AKIAIOSFODNN7EXAMPLE", "p@ssw0rd!",
"203.0.113.7", "2001:db8::1",
}
for _, s := range mustBeGone {
if strings.Contains(got, s) {
t.Errorf("integration: output still leaks %q\nfull output:\n%s", s, got)
}
}
// Non-sensitive structure must survive (guard against over-redaction of the
// surrounding stack trace / labels).
for _, s := range []string{"Bug report:", "com.example.app.Main.run", "Main.java:99"} {
if !strings.Contains(got, s) {
t.Errorf("integration: expected non-sensitive text %q to survive\nfull output:\n%s", s, got)
}
}
for _, p := range []string{PlaceholderEmail, PlaceholderAuth, PlaceholderToken, PlaceholderIP, PlaceholderName} {
if !strings.Contains(got, p) {
t.Errorf("integration: expected placeholder %q in output\nfull output:\n%s", p, got)
}
}
}
func TestScrubIdempotent(t *testing.T) {
once := ScrubString(sampleReport)
twice := ScrubString(once)
if once != twice {
t.Errorf("scrub is not idempotent:\n once = %q\n twice = %q", once, twice)
}
}
func TestScrubBytes(t *testing.T) {
in := []byte(`ip 10.0.0.1 mail x@y.io`)
original := append([]byte(nil), in...) // snapshot
got := Scrub(in)
if bytes.Contains(got, []byte("10.0.0.1")) || bytes.Contains(got, []byte("x@y.io")) {
t.Errorf("Scrub did not redact: %q", got)
}
if !bytes.Equal(in, original) {
t.Errorf("Scrub mutated its input: before=%q after=%q", original, in)
}
}
func TestScrubBytesNilAndEmpty(t *testing.T) {
if got := Scrub(nil); got != nil {
t.Errorf("Scrub(nil) = %q, want nil", got)
}
if got := Scrub([]byte{}); len(got) != 0 {
t.Errorf("Scrub(empty) = %q, want empty", got)
}
}
func TestScrubStringNoSecretsUnchanged(t *testing.T) {
// A payload with nothing sensitive must pass through untouched.
in := `Steps to reproduce: open the app, click Settings, observe crash. Build 4.7 on Android 14.`
if got := ScrubString(in); got != in {
t.Errorf("clean payload changed:\n in = %q\n got = %q", in, got)
}
}