feat(sanitize): in-place, length-preserving log de-identification
Adds internal/sanitize: rewrites the customer-identifying spans that internal/privacy detects (domain/FQDN/e-mail/AD/public-IP/timezone) with same-length neutral fillers, in place, without changing the file format. - Fillers keep byte length: "sigma.sbrf.ru" -> "xxxxx.xxxx.xx", IP -> "00.000.000.00", "Europe/Moscow" -> "Etc/Universal" (same-length valid neutral IANA zone), offset "180" -> "000". Timestamps are not recomputed. - Lossless recursive archive walk (tar/.sds/gz/tgz/zip): entry names, modes, and all embedded timestamps preserved; untouched zip entries copied raw; member payload length unchanged so tar headers stay byte-identical; only the .gz/.zip compression layer is rebuilt. 0 redactions -> byte-identical output. - privacy.FindSpans is the one matcher shared by detection and redaction; fillers are recognised by isRedactionFiller so a re-scan / second pass is a no-op. New privacy FPs fixed along the way: syslog selectors (local7.info), "MEVersion" firmware quads, *.conf_bak vendor templates, bundled viewer domains. - Binary members (FRU.bin, localtime, redis-dump.rdb, SOL captures) and unreadable nested archives are reported in Result.SkippedBinary, never edited. - Surfaces: POST /api/sanitize (+ GET /api/sanitize/download), the "Обезличить и скачать копию" button in the Customer-data panel, and logpile -sanitize <file> (restores mtime/atime). Verified: re-parsing a sanitized Dell TSR / xFusion / Inspur onekeylog / H3C .sds yields the identical hardware inventory; re-scan is clean. ADL-067, bible-local/docs/log-sanitization.md. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
e74e01ad05
commit
a63bb17438
@@ -0,0 +1,223 @@
|
||||
package sanitize
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"strings"
|
||||
|
||||
"git.mchus.pro/mchus/logpile/internal/privacy"
|
||||
)
|
||||
|
||||
// Span categories that get rewritten. fru_location is intentionally excluded
|
||||
// (often a serial / manufacturing code, low value, and may live in a binary
|
||||
// FRU area); it is only reported by the scan, never edited.
|
||||
var redactableCat = map[string]bool{
|
||||
"domain": true, "mgmt_subdomain": true, "resolv": true, "ad_ldap": true,
|
||||
"email": true, "nsupdate": true, "collector": true, "dhcp": true,
|
||||
"cert": true, "public_ip": true, "timezone": true,
|
||||
}
|
||||
|
||||
type change struct {
|
||||
path string
|
||||
category string
|
||||
before string
|
||||
after string
|
||||
}
|
||||
|
||||
// redactText rewrites every redactable span in content, keeping each
|
||||
// replacement byte-for-byte the same length as the original so the total
|
||||
// content length never changes. Line terminators are preserved exactly.
|
||||
func redactText(content []byte, certFile bool) ([]byte, []change) {
|
||||
var out bytes.Buffer
|
||||
out.Grow(len(content))
|
||||
var changes []change
|
||||
|
||||
for _, s := range splitKeepEOL(content) {
|
||||
spans := redactableSpans(s.line, certFile)
|
||||
if len(spans) == 0 {
|
||||
out.WriteString(s.line)
|
||||
out.Write(s.eol)
|
||||
continue
|
||||
}
|
||||
prev := 0
|
||||
for _, sp := range spans {
|
||||
out.WriteString(s.line[prev:sp.Start])
|
||||
orig := s.line[sp.Start:sp.End]
|
||||
repl := fillerFor(sp.Category, orig)
|
||||
if len(repl) != len(orig) {
|
||||
repl = xFill(orig)
|
||||
}
|
||||
out.WriteString(repl)
|
||||
changes = append(changes, change{category: sp.Category, before: orig, after: repl})
|
||||
prev = sp.End
|
||||
}
|
||||
out.WriteString(s.line[prev:])
|
||||
out.Write(s.eol)
|
||||
}
|
||||
return out.Bytes(), changes
|
||||
}
|
||||
|
||||
// redactableSpans returns the redactable spans of one line, sorted by start and
|
||||
// with overlaps merged (so a nested domain inside an e-mail is redacted once).
|
||||
func redactableSpans(line string, certFile bool) []privacy.Span {
|
||||
raw := privacy.FindSpans(line, certFile)
|
||||
kept := raw[:0]
|
||||
for _, sp := range raw {
|
||||
if redactableCat[sp.Category] {
|
||||
kept = append(kept, sp)
|
||||
}
|
||||
}
|
||||
if len(kept) < 2 {
|
||||
return kept
|
||||
}
|
||||
sortSpans(kept)
|
||||
merged := kept[:1]
|
||||
for _, sp := range kept[1:] {
|
||||
last := &merged[len(merged)-1]
|
||||
if sp.Start <= last.End {
|
||||
if sp.End > last.End {
|
||||
last.End = sp.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
merged = append(merged, sp)
|
||||
}
|
||||
return merged
|
||||
}
|
||||
|
||||
func sortSpans(s []privacy.Span) {
|
||||
for i := 1; i < len(s); i++ {
|
||||
for j := i; j > 0 && s[j-1].Start > s[j].Start; j-- {
|
||||
s[j-1], s[j] = s[j], s[j-1]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// fillerFor returns a same-length neutral replacement for one matched token.
|
||||
func fillerFor(category, orig string) string {
|
||||
switch category {
|
||||
case "public_ip":
|
||||
return digitZero(orig) // 93.184.216.34 -> 00.000.000.00
|
||||
case "timezone":
|
||||
return tzFiller(orig)
|
||||
default:
|
||||
return xFill(orig) // sigma.sbrf.ru -> xxxxx.xxxx.xx
|
||||
}
|
||||
}
|
||||
|
||||
// xFill replaces every ASCII letter/digit with 'x', keeping punctuation
|
||||
// (dots, hyphens, '@', ':', '*', '_') in place.
|
||||
func xFill(s string) string {
|
||||
b := []byte(s)
|
||||
for i, c := range b {
|
||||
if (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') || (c >= '0' && c <= '9') {
|
||||
b[i] = 'x'
|
||||
}
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// digitZero replaces every digit with '0', keeping dots.
|
||||
func digitZero(s string) string {
|
||||
b := []byte(s)
|
||||
for i, c := range b {
|
||||
if c >= '0' && c <= '9' {
|
||||
b[i] = '0'
|
||||
}
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// tzFiller neutralises a timezone value: a numeric UTC offset becomes zeros, a
|
||||
// Region/City name becomes a same-length valid IANA zone, an abbreviation
|
||||
// becomes "UTC" (len 3) or an x-fill.
|
||||
func tzFiller(orig string) string {
|
||||
if isOffset(orig) {
|
||||
return digitZero(orig) // 180 -> 000, -300 -> -000
|
||||
}
|
||||
if strings.Contains(orig, "/") {
|
||||
if z, ok := neutralZoneByLen[len(orig)]; ok {
|
||||
return z
|
||||
}
|
||||
return keepSlashXFill(orig)
|
||||
}
|
||||
if len(orig) == 3 {
|
||||
return "UTC"
|
||||
}
|
||||
return xFill(orig)
|
||||
}
|
||||
|
||||
func isOffset(s string) bool {
|
||||
if s == "" {
|
||||
return false
|
||||
}
|
||||
for i, c := range s {
|
||||
if c >= '0' && c <= '9' {
|
||||
continue
|
||||
}
|
||||
if (c == '+' || c == '-') && i == 0 {
|
||||
continue
|
||||
}
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func keepSlashXFill(s string) string {
|
||||
b := []byte(xFill(s))
|
||||
for i, c := range []byte(s) {
|
||||
if c == '/' {
|
||||
b[i] = '/'
|
||||
}
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// neutralZoneByLen maps a timezone-name length to a same-length, valid IANA
|
||||
// zone that carries no regional information (UTC/Etc/* where the length allows,
|
||||
// otherwise a fixed far-away decoy). Every value here is on the privacy
|
||||
// allowlist so a re-scan of the sanitized file stays clean.
|
||||
var neutralZoneByLen = map[int]string{
|
||||
3: "UTC",
|
||||
4: "Zulu",
|
||||
7: "Etc/UTC",
|
||||
8: "Etc/GMT0",
|
||||
9: "Universal",
|
||||
10: "US/Pacific",
|
||||
11: "Brazil/East",
|
||||
12: "Canada/Yukon",
|
||||
13: "Etc/Universal",
|
||||
14: "Pacific/Tarawa",
|
||||
15: "Atlantic/Azores",
|
||||
16: "Antarctica/Troll",
|
||||
17: "Antarctica/Vostok",
|
||||
18: "Antarctica/McMurdo",
|
||||
19: "Pacific/Guadalcanal",
|
||||
20: "Pacific/Bougainville",
|
||||
}
|
||||
|
||||
type eolSeg struct {
|
||||
line string
|
||||
eol []byte
|
||||
}
|
||||
|
||||
// splitKeepEOL splits content into lines while preserving each original line
|
||||
// terminator ("\n", "\r\n", or none for a final unterminated line).
|
||||
func splitKeepEOL(b []byte) []eolSeg {
|
||||
var segs []eolSeg
|
||||
i := 0
|
||||
for i < len(b) {
|
||||
j := bytes.IndexByte(b[i:], '\n')
|
||||
if j < 0 {
|
||||
segs = append(segs, eolSeg{line: string(b[i:])})
|
||||
return segs
|
||||
}
|
||||
nl := i + j
|
||||
lineEnd := nl
|
||||
if lineEnd > i && b[lineEnd-1] == '\r' {
|
||||
lineEnd--
|
||||
}
|
||||
segs = append(segs, eolSeg{line: string(b[i:lineEnd]), eol: append([]byte(nil), b[lineEnd:nl+1]...)})
|
||||
i = nl + 1
|
||||
}
|
||||
return segs
|
||||
}
|
||||
Reference in New Issue
Block a user