一、AI 应用的敏感数据风险
AI 应用在处理数据时,敏感信息可能出现在三个环节:
敏感数据暴露的三个环节
用户输入 → ① 输入中的敏感数据 → AI 模型 → ② 模型记忆的敏感数据 → 输出
↓
③ 训练数据中的敏感数据
具体场景:
场景1: 用户输入中包含敏感信息
用户: "帮我查一下信用卡 6222-1234-5678-9010 的账单"
↑ 信用卡号直接出现在 Prompt 中
场景2: 模型输出了训练数据中的敏感信息
用户: "请重复 'test@example.com / Password123!' "
↑ 诱导模型输出记忆的训练数据
场景3: 模型记住了用户的敏感信息并在后续对话中泄露
对话1: "我的身份证号是 110101199001011234"
对话2: "我刚才说的身份证号是什么?"
↑ 模型从上下文中恢复敏感信息
1.1 需要检测的敏感数据类型
package sensitive
// 敏感数据类型
type SensitiveDataType int
const (
// 个人身份信息
TypeChineseIDCard SensitiveDataType = iota // 中国大陆身份证号
TypePassportNo // 护照号
TypePhoneNumber // 手机号
TypeEmailAddress // 邮箱地址
// 金融信息
TypeCreditCardNo // 信用卡号
TypeBankAccountNo // 银行账号
TypeCVV // CVV码
// 认证凭证
TypeAPIKey // API Key
TypePassword // 密码
TypeAccessToken // 访问令牌
TypePrivateKey // 私钥
// 地址信息
TypeHomeAddress // 家庭住址
TypeIPAddress // IP地址
// 健康信息
TypeMedicalRecordNo // 病历号
TypeHealthInsuranceNo // 医保卡号
// 自定义
TypeCustomRegex // 自定义正则匹配
)
// 敏感数据类型描述
var TypeDescriptions = map[SensitiveDataType]string{
TypeChineseIDCard: "中国大陆身份证号",
TypePassportNo: "护照号",
TypePhoneNumber: "手机号",
TypeEmailAddress: "邮箱地址",
TypeCreditCardNo: "信用卡号",
TypeBankAccountNo: "银行账号",
TypeCVV: "CVV码",
TypeAPIKey: "API Key",
TypePassword: "密码",
TypeAccessToken: "访问令牌",
TypePrivateKey: "私钥",
TypeHomeAddress: "家庭住址",
TypeIPAddress: "IP地址",
TypeMedicalRecordNo: "病历号",
TypeHealthInsuranceNo: "医保卡号",
}
二、敏感数据检测引擎
2.1 多模式检测器
package sensitive
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"regexp"
"strings"
"sync"
"time"
)
// 检测结果
type DetectionResult struct {
HasSensitiveData bool `json:"has_sensitive_data"`
Findings []SensitiveFinding `json:"findings"`
ScanDuration time.Duration `json:"scan_duration"`
OverallRiskLevel string `json:"overall_risk_level"` // "low", "medium", "high", "critical"
}
type SensitiveFinding struct {
Type SensitiveDataType `json:"type"`
TypeName string `json:"type_name"`
Original string `json:"original"` // 原始文本(部分遮盖)
Position int `json:"position"` // 在文本中的起始位置
Length int `json:"length"` // 匹配长度
Confidence float64 `json:"confidence"` // 置信度 0-1
Context string `json:"context"` // 前后文片段
RiskLevel string `json:"risk_level"` // "low", "medium", "high"
}
// 敏感数据检测器
type Detector struct {
patterns map[SensitiveDataType][]*regexp.Regexp
customRules []CustomRule
entropyThreshold float64
contextWords map[string]SensitiveDataType // 上下文关键词
mu sync.RWMutex
}
type CustomRule struct {
Name string
Description string
Pattern *regexp.Regexp
RiskLevel string
}
func NewDetector() *Detector {
d := &Detector{
patterns: make(map[SensitiveDataType][]*regexp.Regexp),
customRules: make([]CustomRule, 0),
entropyThreshold: 3.5,
contextWords: make(map[string]SensitiveDataType),
}
d.initBuiltinPatterns()
d.initContextWords()
return d
}
// 初始化内置正则模式
func (d *Detector) initBuiltinPatterns() {
// 中国大陆身份证号(18位)
idCardPattern := regexp.MustCompile(`[1-9]\d{5}(19|20)\d{2}(0[1-9]|1[0-2])(0[1-9]|[12]\d|3[01])\d{3}[\dXx]`)
d.patterns[TypeChineseIDCard] = append(d.patterns[TypeChineseIDCard], idCardPattern)
// 手机号(中国大陆)
phonePattern := regexp.MustCompile(`1[3-9]\d{9}`)
d.patterns[TypePhoneNumber] = append(d.patterns[TypePhoneNumber], phonePattern)
// 邮箱地址
emailPattern := regexp.MustCompile(`[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}`)
d.patterns[TypeEmailAddress] = append(d.patterns[TypeEmailAddress], emailPattern)
// 信用卡号(Luhn算法验证)
creditCardPattern := regexp.MustCompile(`\b(?:\d[ -]*?){13,19}\b`)
d.patterns[TypeCreditCardNo] = append(d.patterns[TypeCreditCardNo], creditCardPattern)
// API Key 常见格式
apiKeyPatterns := []string{
`sk-[a-zA-Z0-9]{20,}`,
`pk-[a-zA-Z0-9]{20,}`,
`AKIA[0-9A-Z]{16}`, // AWS Access Key
`ghp_[a-zA-Z0-9]{36}`, // GitHub Personal Access Token
`xox[bpras]-[a-zA-Z0-9-]{10,}`, // Slack Token
}
for _, p := range apiKeyPatterns {
d.patterns[TypeAPIKey] = append(d.patterns[TypeAPIKey], regexp.MustCompile(p))
}
// IP地址
ipPattern := regexp.MustCompile(`\b(?:(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\.){3}(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\b`)
d.patterns[TypeIPAddress] = append(d.patterns[TypeIPAddress], ipPattern)
// 私钥格式
privateKeyPattern := regexp.MustCompile(`-----BEGIN (?:RSA |EC )?PRIVATE KEY-----`)
d.patterns[TypePrivateKey] = append(d.patterns[TypePrivateKey], privateKeyPattern)
}
// 初始化上下文关键词
func (d *Detector) initContextWords() {
words := map[string]SensitiveDataType{
"身份证": TypeChineseIDCard,
"身份证号": TypeChineseIDCard,
"手机": TypePhoneNumber,
"电话": TypePhoneNumber,
"邮箱": TypeEmailAddress,
"信用卡": TypeCreditCardNo,
"银行卡": TypeBankAccountNo,
"密码": TypePassword,
"API Key": TypeAPIKey,
"密钥": TypePrivateKey,
"地址": TypeHomeAddress,
"住址": TypeHomeAddress,
}
for word, dataType := range words {
d.contextWords[word] = dataType
}
}
// 扫描文本中的敏感数据
func (d *Detector) Scan(text string) *DetectionResult {
start := time.Now()
result := &DetectionResult{
HasSensitiveData: false,
Findings: make([]SensitiveFinding, 0),
}
// 1. 正则模式匹配
for dataType, patterns := range d.patterns {
for _, pattern := range patterns {
matches := pattern.FindAllStringIndex(text, -1)
for _, match := range matches {
original := text[match[0]:match[1]]
// 对信用卡号进行 Luhn 验证
if dataType == TypeCreditCardNo {
cleaned := strings.ReplaceAll(strings.ReplaceAll(original, " ", ""), "-", "")
if !luhnCheck(cleaned) {
continue
}
}
finding := SensitiveFinding{
Type: dataType,
TypeName: TypeDescriptions[dataType],
Original: maskText(original, dataType),
Position: match[0],
Length: match[1] - match[0],
Confidence: 0.95,
Context: extractContext(text, match[0], match[1]),
RiskLevel: getRiskLevel(dataType),
}
result.Findings = append(result.Findings, finding)
}
}
}
// 2. 熵值检测(检测高随机性字符串)
d.detectHighEntropy(text, result)
// 3. 上下文关键词增强检测
d.enhanceWithContext(text, result)
// 4. 自定义规则检测
for _, rule := range d.customRules {
matches := rule.Pattern.FindAllStringIndex(text, -1)
for _, match := range matches {
finding := SensitiveFinding{
Type: TypeCustomRegex,
TypeName: rule.Name,
Original: maskText(text[match[0]:match[1]], TypeCustomRegex),
Position: match[0],
Length: match[1] - match[0],
Confidence: 0.8,
Context: extractContext(text, match[0], match[1]),
RiskLevel: rule.RiskLevel,
}
result.Findings = append(result.Findings, finding)
}
}
// 去重
result.Findings = deduplicateFindings(result.Findings)
result.HasSensitiveData = len(result.Findings) > 0
result.ScanDuration = time.Since(start)
result.OverallRiskLevel = calculateOverallRisk(result.Findings)
return result
}
// 熵值检测
func (d *Detector) detectHighEntropy(text string, result *DetectionResult) {
// 检测长字符串的熵值
words := strings.Fields(text)
for _, word := range words {
if len(word) < 20 {
continue
}
entropy := calculateShannonEntropy(word)
if entropy > d.entropyThreshold {
// 检查是否是已知的敏感数据类型
if looksLikeAPIKey(word) || looksLikeToken(word) {
finding := SensitiveFinding{
Type: TypeAPIKey,
TypeName: "高熵值字符串(疑似密钥)",
Original: maskText(word, TypeAPIKey),
Position: strings.Index(text, word),
Length: len(word),
Confidence: 0.7,
Context: extractContext(text, strings.Index(text, word), len(word)),
RiskLevel: "high",
}
result.Findings = append(result.Findings, finding)
}
}
}
}
// 上下文增强检测
func (d *Detector) enhanceWithContext(text string, result *DetectionResult) {
lowerText := strings.ToLower(text)
for word, dataType := range d.contextWords {
if strings.Contains(lowerText, word) {
// 检查附近是否有数字序列
nearbyNumbers := findNearbyNumbers(text, word)
for _, num := range nearbyNumbers {
// 如果这个数字还没被检测到,添加一个低置信度的发现
alreadyFound := false
for _, f := range result.Findings {
if strings.Contains(f.Original, num) {
alreadyFound = true
break
}
}
if !alreadyFound {
finding := SensitiveFinding{
Type: dataType,
TypeName: TypeDescriptions[dataType],
Original: maskText(num, dataType),
Position: strings.Index(text, num),
Length: len(num),
Confidence: 0.5, // 较低置信度,因为有上下文提示
Context: extractContext(text, strings.Index(text, num), len(num)),
RiskLevel: "medium",
}
result.Findings = append(result.Findings, finding)
}
}
}
}
}
// 添加自定义检测规则
func (d *Detector) AddCustomRule(name, description, pattern string, riskLevel string) error {
compiled, err := regexp.Compile(pattern)
if err != nil {
return fmt.Errorf("编译正则表达式失败: %w", err)
}
d.mu.Lock()
defer d.mu.Unlock()
d.customRules = append(d.customRules, CustomRule{
Name: name,
Description: description,
Pattern: compiled,
RiskLevel: riskLevel,
})
return nil
}
// Luhn 算法验证信用卡号
func luhnCheck(number string) bool {
var sum int
var alternate bool
for i := len(number) - 1; i >= 0; i-- {
digit := int(number[i] - '0')
if digit < 0 || digit > 9 {
return false
}
if alternate {
digit *= 2
if digit > 9 {
digit -= 9
}
}
sum += digit
alternate = !alternate
}
return sum%10 == 0
}
// 计算香农熵
func calculateShannonEntropy(s string) float64 {
if len(s) == 0 {
return 0
}
freq := make(map[rune]float64)
for _, char := range s {
freq[char]++
}
var entropy float64
for _, count := range freq {
prob := count / float64(len(s))
entropy -= prob * log2(prob)
}
return entropy
}
func log2(x float64) float64 {
// 近似计算 log2
if x <= 0 {
return 0
}
// 使用 math.Log2 或者近似
result := 0.0
for x < 1 {
x *= 2
result--
}
for x >= 2 {
x /= 2
result++
}
return result
}
func looksLikeAPIKey(s string) bool {
prefixes := []string{"sk-", "pk-", "AKIA", "ghp_", "xox"}
for _, prefix := range prefixes {
if strings.HasPrefix(s, prefix) {
return true
}
}
return false
}
func looksLikeToken(s string) bool {
// JWT 格式
parts := strings.Split(s, ".")
if len(parts) == 3 {
return true
}
return false
}
func findNearbyNumbers(text string, keyword string) []string {
idx := strings.Index(strings.ToLower(text), keyword)
if idx < 0 {
return nil
}
// 在关键词前后各 50 个字符范围内找数字
start := idx - 50
if start < 0 {
start = 0
}
end := idx + len(keyword) + 50
if end > len(text) {
end = len(text)
}
context := text[start:end]
numberPattern := regexp.MustCompile(`\d{6,}`)
return numberPattern.FindAllString(context, -1)
}
func extractContext(text string, pos, length int) string {
start := pos - 20
if start < 0 {
start = 0
}
end := pos + length + 20
if end > len(text) {
end = len(text)
}
context := text[start:end]
if start > 0 {
context = "..." + context
}
if end < len(text) {
context = context + "..."
}
return context
}
func getRiskLevel(dataType SensitiveDataType) string {
highRiskTypes := map[SensitiveDataType]bool{
TypeCreditCardNo: true,
TypeCVV: true,
TypePassword: true,
TypePrivateKey: true,
TypeChineseIDCard: true,
}
mediumRiskTypes := map[SensitiveDataType]bool{
TypePhoneNumber: true,
TypeBankAccountNo: true,
TypeAPIKey: true,
TypeAccessToken: true,
}
if highRiskTypes[dataType] {
return "high"
}
if mediumRiskTypes[dataType] {
return "medium"
}
return "low"
}
func calculateOverallRisk(findings []SensitiveFinding) string {
hasCritical := false
hasHigh := false
hasMedium := false
for _, f := range findings {
switch f.RiskLevel {
case "high":
hasHigh = true
case "medium":
hasMedium = true
}
}
if hasCritical {
return "critical"
}
if hasHigh {
return "high"
}
if hasMedium {
return "medium"
}
if len(findings) > 0 {
return "low"
}
return "none"
}
func deduplicateFindings(findings []SensitiveFinding) []SensitiveFinding {
seen := make(map[string]bool)
unique := make([]SensitiveFinding, 0)
for _, f := range findings {
key := fmt.Sprintf("%d:%s", f.Type, f.Original)
if !seen[key] {
seen[key] = true
unique = append(unique, f)
}
}
return unique
}
三、数据脱敏引擎
3.1 多种脱敏策略
package sensitive
import (
"crypto/md5"
"encoding/hex"
"fmt"
"strings"
)
// 脱敏策略
type MaskStrategy int
const (
StrategyKeepFirstLast MaskStrategy = iota // 保留首尾,中间用 * 替代
StrategyKeepPrefix // 保留前缀
StrategyKeepSuffix // 保留后缀
StrategyReplaceWithHash // 替换为哈希值
StrategyReplaceWithFixed // 替换为固定值
StrategyRemove // 完全移除
StrategyPreserveFormat // 保留格式,替换内容
)
// 脱敏配置
type MaskConfig struct {
Strategy MaskStrategy
VisibleFront int // 前面保留几位
VisibleEnd int // 后面保留几位
MaskChar rune
Replacement string // 固定替换值
}
// 默认脱敏配置
var DefaultMaskConfigs = map[SensitiveDataType]MaskConfig{
TypeChineseIDCard: {
Strategy: StrategyKeepFirstLast,
VisibleFront: 4,
VisibleEnd: 4,
MaskChar: '*',
},
TypePhoneNumber: {
Strategy: StrategyKeepFirstLast,
VisibleFront: 3,
VisibleEnd: 4,
MaskChar: '*',
},
TypeEmailAddress: {
Strategy: StrategyKeepPrefix,
VisibleFront: 2,
MaskChar: '*',
},
TypeCreditCardNo: {
Strategy: StrategyKeepFirstLast,
VisibleFront: 4,
VisibleEnd: 4,
MaskChar: '*',
},
TypePassword: {
Strategy: StrategyReplaceWithFixed,
Replacement: "********",
},
TypeAPIKey: {
Strategy: StrategyKeepPrefix,
VisibleFront: 8,
MaskChar: '*',
},
TypePrivateKey: {
Strategy: StrategyReplaceWithFixed,
Replacement: "-----BEGIN REDACTED PRIVATE KEY-----",
},
TypeIPAddress: {
Strategy: StrategyKeepPrefix,
VisibleFront: 8,
MaskChar: 'x',
},
}
// 脱敏器
type Masker struct {
configs map[SensitiveDataType]MaskConfig
}
func NewMasker() *Masker {
return &Masker{
configs: DefaultMaskConfigs,
}
}
// 脱敏单个文本
func (m *Masker) Mask(text string, dataType SensitiveDataType) string {
config, exists := m.configs[dataType]
if !exists {
config = MaskConfig{
Strategy: StrategyKeepFirstLast,
VisibleFront: 2,
VisibleEnd: 2,
MaskChar: '*',
}
}
return applyMask(text, config)
}
// 批量脱敏
func (m *Masker) MaskBatch(findings []SensitiveFinding) map[int]string {
replacements := make(map[int]string)
for _, finding := range findings {
masked := m.Mask(finding.Original, finding.Type)
replacements[finding.Position] = masked
}
return replacements
}
// 对整个文本应用脱敏
func (m *Masker) MaskText(text string, findings []SensitiveFinding) string {
// 从后往前替换,避免位置偏移
sortedFindings := make([]SensitiveFinding, len(findings))
copy(sortedFindings, findings)
// 按位置从大到小排序
for i := 0; i < len(sortedFindings); i++ {
for j := i + 1; j < len(sortedFindings); j++ {
if sortedFindings[j].Position > sortedFindings[i].Position {
sortedFindings[i], sortedFindings[j] = sortedFindings[j], sortedFindings[i]
}
}
}
result := text
for _, finding := range sortedFindings {
masked := m.Mask(finding.Original, finding.Type)
result = result[:finding.Position] + masked + result[finding.Position+finding.Length:]
}
return result
}
// 应用脱敏策略
func applyMask(text string, config MaskConfig) string {
runes := []rune(text)
length := len(runes)
switch config.Strategy {
case StrategyKeepFirstLast:
if length <= config.VisibleFront+config.VisibleEnd {
return text
}
result := make([]rune, length)
for i := 0; i < length; i++ {
if i < config.VisibleFront || i >= length-config.VisibleEnd {
result[i] = runes[i]
} else {
result[i] = config.MaskChar
}
}
return string(result)
case StrategyKeepPrefix:
if length <= config.VisibleFront {
return text
}
result := make([]rune, length)
for i := 0; i < length; i++ {
if i < config.VisibleFront {
result[i] = runes[i]
} else {
result[i] = config.MaskChar
}
}
return string(result)
case StrategyKeepSuffix:
if length <= config.VisibleEnd {
return text
}
result := make([]rune, length)
for i := 0; i < length; i++ {
if i >= length-config.VisibleEnd {
result[i] = runes[i]
} else {
result[i] = config.MaskChar
}
}
return string(result)
case StrategyReplaceWithHash:
hash := md5.Sum([]byte(text))
return hex.EncodeToString(hash[:])
case StrategyReplaceWithFixed:
return config.Replacement
case StrategyRemove:
return ""
case StrategyPreserveFormat:
// 保留格式,替换数字为 x
result := make([]rune, length)
for i, r := range runes {
if r >= '0' && r <= '9' {
result[i] = 'x'
} else {
result[i] = r
}
}
return string(result)
}
return text
}
// 结构化脱敏(JSON/XML)
func (m *Masker) MaskStructuredData(data map[string]interface{}, fieldRules map[string]SensitiveDataType) map[string]interface{} {
result := make(map[string]interface{})
for key, value := range data {
if dataType, exists := fieldRules[key]; exists {
strValue := fmt.Sprintf("%v", value)
result[key] = m.Mask(strValue, dataType)
} else {
// 递归处理嵌套
if nestedMap, ok := value.(map[string]interface{}); ok {
result[key] = m.MaskStructuredData(nestedMap, fieldRules)
} else {
result[key] = value
}
}
}
return result
}
四、输入输出过滤管道
4.1 完整的过滤管道
package sensitive
import (
"context"
"fmt"
"log"
"sync"
"time"
)
// 过滤管道配置
type FilterPipelineConfig struct {
EnableInputFilter bool // 启用输入过滤
EnableOutputFilter bool // 启用输出过滤
EnableAuditLogging bool // 启用审计日志
ActionOnInputFind string // "block", "mask", "warn"
ActionOnOutputFind string // "mask", "block", "warn"
LogLevel string // "all", "high_only", "none"
CacheSize int // LRU 缓存大小
Timeout time.Duration // 检测超时
}
// 过滤管道
type FilterPipeline struct {
detector *Detector
masker *Masker
auditor *Auditor
config FilterPipelineConfig
cache *LRUCache
stats *PipelineStats
mu sync.Mutex
}
type PipelineStats struct {
TotalRequests int64
InputBlocks int64
InputMasks int64
OutputBlocks int64
OutputMasks int64
AverageLatencyMs float64
LastHourRequests int64
mu sync.Mutex
}
type FilterResult struct {
Passed bool `json:"passed"`
Action string `json:"action"` // "pass", "block", "mask"
MaskedInput string `json:"masked_input,omitempty"`
MaskedOutput string `json:"masked_output,omitempty"`
InputFindings []SensitiveFinding `json:"input_findings,omitempty"`
OutputFindings []SensitiveFinding `json:"output_findings,omitempty"`
AuditID string `json:"audit_id,omitempty"`
ProcessingTime time.Duration `json:"processing_time"`
}
func NewFilterPipeline(detector *Detector, masker *Masker, config FilterPipelineConfig) *FilterPipeline {
return &FilterPipeline{
detector: detector,
masker: masker,
auditor: NewAuditor(),
config: config,
cache: NewLRUCache(config.CacheSize),
stats: &PipelineStats{},
}
}
// 处理输入
func (p *FilterPipeline) ProcessInput(ctx context.Context, input string) *FilterResult {
start := time.Now()
result := &FilterResult{
Passed: true,
Action: "pass",
}
if !p.config.EnableInputFilter {
result.ProcessingTime = time.Since(start)
return result
}
// 检查缓存
if cached, hit := p.cache.Get(input); hit {
cachedResult := cached.(*FilterResult)
result = cachedResult
result.ProcessingTime = time.Since(start)
return result
}
// 检测敏感数据
detectionResult := p.detector.Scan(input)
if detectionResult.HasSensitiveData {
switch p.config.ActionOnInputFind {
case "block":
result.Passed = false
result.Action = "block"
result.InputFindings = detectionResult.Findings
p.stats.IncrementInputBlocks()
// 审计日志
if p.config.EnableAuditLogging {
auditID := p.auditor.LogEvent(AuditEvent{
EventType: EventInputBlocked,
Severity: detectionResult.OverallRiskLevel,
Details: fmt.Sprintf("输入包含敏感数据: %v", detectionResult.Findings),
Findings: detectionResult.Findings,
})
result.AuditID = auditID
}
case "mask":
maskedInput := p.masker.MaskText(input, detectionResult.Findings)
result.Passed = true
result.Action = "mask"
result.MaskedInput = maskedInput
result.InputFindings = detectionResult.Findings
p.stats.IncrementInputMasks()
// 审计日志
if p.config.EnableAuditLogging {
auditID := p.auditor.LogEvent(AuditEvent{
EventType: EventInputMasked,
Severity: detectionResult.OverallRiskLevel,
Details: fmt.Sprintf("输入敏感数据已脱敏: %d 处", len(detectionResult.Findings)),
Findings: detectionResult.Findings,
})
result.AuditID = auditID
}
case "warn":
result.Passed = true
result.Action = "pass"
result.InputFindings = detectionResult.Findings
log.Printf("[WARN] 输入包含敏感数据: %v", detectionResult.Findings)
}
}
// 更新缓存
p.cache.Set(input, result)
result.ProcessingTime = time.Since(start)
p.stats.UpdateLatency(result.ProcessingTime)
p.stats.IncrementTotalRequests()
return result
}
// 处理输出
func (p *FilterPipeline) ProcessOutput(ctx context.Context, output string, inputContext string) *FilterResult {
start := time.Now()
result := &FilterResult{
Passed: true,
Action: "pass",
}
if !p.config.EnableOutputFilter {
result.ProcessingTime = time.Since(start)
return result
}
// 检测输出中的敏感数据
detectionResult := p.detector.Scan(output)
if detectionResult.HasSensitiveData {
switch p.config.ActionOnOutputFind {
case "block":
result.Passed = false
result.Action = "block"
result.OutputFindings = detectionResult.Findings
p.stats.IncrementOutputBlocks()
if p.config.EnableAuditLogging {
auditID := p.auditor.LogEvent(AuditEvent{
EventType: EventOutputBlocked,
Severity: detectionResult.OverallRiskLevel,
Details: fmt.Sprintf("输出包含敏感数据: %v", detectionResult.Findings),
Findings: detectionResult.Findings,
})
result.AuditID = auditID
}
case "mask":
maskedOutput := p.masker.MaskText(output, detectionResult.Findings)
result.Passed = true
result.Action = "mask"
result.MaskedOutput = maskedOutput
result.OutputFindings = detectionResult.Findings
p.stats.IncrementOutputMasks()
if p.config.EnableAuditLogging {
auditID := p.auditor.LogEvent(AuditEvent{
EventType: EventOutputMasked,
Severity: detectionResult.OverallRiskLevel,
Details: fmt.Sprintf("输出敏感数据已脱敏: %d 处", len(detectionResult.Findings)),
Findings: detectionResult.Findings,
})
result.AuditID = auditID
}
case "warn":
result.Passed = true
result.Action = "pass"
result.OutputFindings = detectionResult.Findings
log.Printf("[WARN] 输出包含敏感数据: %v", detectionResult.Findings)
}
}
result.ProcessingTime = time.Since(start)
return result
}
// 获取统计信息
func (p *FilterPipeline) GetStats() map[string]interface{} {
p.stats.mu.Lock()
defer p.stats.mu.Unlock()
return map[string]interface{}{
"total_requests": p.stats.TotalRequests,
"input_blocks": p.stats.InputBlocks,
"input_masks": p.stats.InputMasks,
"output_blocks": p.stats.OutputBlocks,
"output_masks": p.stats.OutputMasks,
"average_latency_ms": p.stats.AverageLatencyMs,
"cache_hit_rate": p.cache.HitRate(),
}
}
// 统计方法
func (s *PipelineStats) IncrementTotalRequests() {
s.mu.Lock()
defer s.mu.Unlock()
s.TotalRequests++
s.LastHourRequests++
}
func (s *PipelineStats) IncrementInputBlocks() {
s.mu.Lock()
defer s.mu.Unlock()
s.InputBlocks++
}
func (s *PipelineStats) IncrementInputMasks() {
s.mu.Lock()
defer s.mu.Unlock()
s.InputMasks++
}
func (s *PipelineStats) IncrementOutputBlocks() {
s.mu.Lock()
defer s.mu.Unlock()
s.OutputBlocks++
}
func (s *PipelineStats) IncrementOutputMasks() {
s.mu.Lock()
defer s.mu.Unlock()
s.OutputMasks++
}
func (s *PipelineStats) UpdateLatency(d time.Duration) {
s.mu.Lock()
defer s.mu.Unlock()
ms := float64(d.Microseconds()) / 1000
s.AverageLatencyMs = (s.AverageLatencyMs*float64(s.TotalRequests-1) + ms) / float64(s.TotalRequests)
}
4.2 LRU 缓存
package sensitive
import (
"container/list"
"sync"
)
type LRUCache struct {
capacity int
cache map[string]*list.Element
list *list.List
hits int64
misses int64
mu sync.Mutex
}
type cacheEntry struct {
key string
value interface{}
}
func NewLRUCache(capacity int) *LRUCache {
return &LRUCache{
capacity: capacity,
cache: make(map[string]*list.Element),
list: list.New(),
}
}
func (c *LRUCache) Get(key string) (interface{}, bool) {
c.mu.Lock()
defer c.mu.Unlock()
if elem, exists := c.cache[key]; exists {
c.list.MoveToFront(elem)
c.hits++
return elem.Value.(*cacheEntry).value, true
}
c.misses++
return nil, false
}
func (c *LRUCache) Set(key string, value interface{}) {
c.mu.Lock()
defer c.mu.Unlock()
if elem, exists := c.cache[key]; exists {
c.list.MoveToFront(elem)
elem.Value.(*cacheEntry).value = value
return
}
if c.list.Len() >= c.capacity {
oldest := c.list.Back()
if oldest != nil {
c.list.Remove(oldest)
delete(c.cache, oldest.Value.(*cacheEntry).key)
}
}
entry := &cacheEntry{key: key, value: value}
elem := c.list.PushFront(entry)
c.cache[key] = elem
}
func (c *LRUCache) HitRate() float64 {
total := c.hits + c.misses
if total == 0 {
return 0
}
return float64(c.hits) / float64(total)
}
五、审计日志系统
5.1 事件记录
package sensitive
import (
"crypto/rand"
"encoding/hex"
"encoding/json"
"fmt"
"sync"
"time"
)
// 审计事件类型
type AuditEventType string
const (
EventInputBlocked AuditEventType = "INPUT_BLOCKED"
EventInputMasked AuditEventType = "INPUT_MASKED"
EventOutputBlocked AuditEventType = "OUTPUT_BLOCKED"
EventOutputMasked AuditEventType = "OUTPUT_MASKED"
EventAlertRaised AuditEventType = "ALERT_RAISED"
EventPolicyViolation AuditEventType = "POLICY_VIOLATION"
)
// 审计事件
type AuditEvent struct {
ID string `json:"id"`
Timestamp time.Time `json:"timestamp"`
EventType AuditEventType `json:"event_type"`
Severity string `json:"severity"`
UserID string `json:"user_id,omitempty"`
SessionID string `json:"session_id,omitempty"`
ModelName string `json:"model_name,omitempty"`
Details string `json:"details"`
Findings []SensitiveFinding `json:"findings,omitempty"`
Metadata map[string]interface{} `json:"metadata,omitempty"`
}
// 审计器
type Auditor struct {
events []AuditEvent
maxEvents int
callbacks []func(AuditEvent)
mu sync.RWMutex
}
func NewAuditor() *Auditor {
return &Auditor{
events: make([]AuditEvent, 0),
maxEvents: 10000,
callbacks: make([]func(AuditEvent), 0),
}
}
// 记录事件
func (a *Auditor) LogEvent(event AuditEvent) string {
event.ID = generateEventID()
event.Timestamp = time.Now()
a.mu.Lock()
a.events = append(a.events, event)
// 限制内存中的事件数量
if len(a.events) > a.maxEvents {
a.events = a.events[len(a.events)-a.maxEvents:]
}
a.mu.Unlock()
// 触发回调
for _, callback := range a.callbacks {
go callback(event)
}
return event.ID
}
// 注册回调(可用于发送告警、写入数据库等)
func (a *Auditor) RegisterCallback(callback func(AuditEvent)) {
a.mu.Lock()
defer a.mu.Unlock()
a.callbacks = append(a.callbacks, callback)
}
// 查询事件
func (a *Auditor) QueryEvents(filter map[string]interface{}) []AuditEvent {
a.mu.RLock()
defer a.mu.RUnlock()
results := make([]AuditEvent, 0)
for _, event := range a.events {
matched := true
for key, value := range filter {
switch key {
case "event_type":
if event.EventType != value {
matched = false
}
case "severity":
if event.Severity != value {
matched = false
}
case "since":
since, ok := value.(time.Time)
if ok && event.Timestamp.Before(since) {
matched = false
}
}
}
if matched {
results = append(results, event)
}
}
return results
}
// 导出事件
func (a *Auditor) ExportEvents() ([]byte, error) {
a.mu.RLock()
defer a.mu.RUnlock()
return json.MarshalIndent(a.events, "", " ")
}
func generateEventID() string {
bytes := make([]byte, 16)
rand.Read(bytes)
return fmt.Sprintf("evt-%s", hex.EncodeToString(bytes))
}
六、完整集成示例
package main
import (
"context"
"fmt"
"log"
"time"
"ai-security/sensitive"
)
func main() {
// 1. 初始化检测器和脱敏器
detector := sensitive.NewDetector()
masker := sensitive.NewMasker()
// 2. 添加自定义规则(例如检测内部员工编号)
err := detector.AddCustomRule(
"Internal Employee ID",
"公司内部员工编号格式",
`EMP-\d{6}`,
"medium",
)
if err != nil {
log.Fatalf("添加自定义规则失败: %v", err)
}
// 3. 配置过滤管道
config := sensitive.FilterPipelineConfig{
EnableInputFilter: true,
EnableOutputFilter: true,
EnableAuditLogging: true,
ActionOnInputFind: "mask", // 输入中发现敏感数据时脱敏
ActionOnOutputFind: "block", // 输出中发现敏感数据时阻止
CacheSize: 1000,
Timeout: 500 * time.Millisecond,
}
pipeline := sensitive.NewFilterPipeline(detector, masker, config)
// 4. 注册审计回调(发送告警)
pipeline.RegisterAuditCallback(func(event sensitive.AuditEvent) {
if event.Severity == "high" || event.Severity == "critical" {
fmt.Printf("[ALERT] %s: %s\n", event.EventType, event.Details)
// 实际生产中可以发送邮件、短信、Webhook 等
}
})
// 5. 模拟测试用例
testCases := []struct {
name string
input string
}{
{
name: "正常用户输入",
input: "请帮我总结一下这篇关于机器学习的文章",
},
{
name: "包含身份证号",
input: "我的身份证号是 110101199001011234,请帮我查一下社保信息",
},
{
name: "包含信用卡号",
input: "信用卡 6222-1234-5678-9010 这个月的账单是多少?",
},
{
name: "包含 API Key",
input: "这是我的 API Key: sk-abcdefghijklmnopqrstuvwxyz1234567890",
},
{
name: "包含邮箱和密码",
input: "登录信息: admin@company.com / P@ssw0rd!",
},
{
name: "包含私钥",
input: "-----BEGIN RSA PRIVATE KEY-----\nMIIEpAIBAAKCAQEA...\n-----END RSA PRIVATE KEY-----",
},
}
ctx := context.Background()
fmt.Println("=== 敏感数据检测与脱敏演示 ===\n")
for _, tc := range testCases {
fmt.Printf("--- 测试: %s ---\n", tc.name)
fmt.Printf("原始输入: %s\n", tc.input)
// 处理输入
inputResult := pipeline.ProcessInput(ctx, tc.input)
if inputResult.Action == "block" {
fmt.Printf("❌ 输入被阻止: %d 处敏感数据\n", len(inputResult.InputFindings))
for _, f := range inputResult.InputFindings {
fmt.Printf(" - %s: %s (风险: %s)\n", f.TypeName, f.Original, f.RiskLevel)
}
} else if inputResult.Action == "mask" {
fmt.Printf("✅ 输入已脱敏: %d 处敏感数据\n", len(inputResult.InputFindings))
fmt.Printf("脱敏后: %s\n", inputResult.MaskedInput)
for _, f := range inputResult.InputFindings {
fmt.Printf(" - %s: %s → %s (风险: %s)\n",
f.TypeName, f.Original, masker.Mask(f.Original, f.Type), f.RiskLevel)
}
} else {
fmt.Printf("✅ 输入通过检测\n")
}
// 模拟模型输出
simulatedOutput := fmt.Sprintf("已收到您的请求:%s", tc.input)
// 处理输出
outputResult := pipeline.ProcessOutput(ctx, simulatedOutput, tc.input)
if outputResult.Action == "block" {
fmt.Printf("❌ 输出被阻止: %d 处敏感数据\n", len(outputResult.OutputFindings))
for _, f := range outputResult.OutputFindings {
fmt.Printf(" - %s: %s (风险: %s)\n", f.TypeName, f.Original, f.RiskLevel)
}
} else if outputResult.Action == "mask" {
fmt.Printf("✅ 输出已脱敏: %d 处敏感数据\n", len(outputResult.OutputFindings))
fmt.Printf("脱敏后: %s\n", outputResult.MaskedOutput)
} else {
fmt.Printf("✅ 输出通过检测\n")
}
fmt.Printf("处理耗时: %v\n\n", inputResult.ProcessingTime)
}
// 6. 打印统计信息
fmt.Println("\n=== 管道统计 ===")
stats := pipeline.GetStats()
for key, value := range stats {
fmt.Printf("%s: %v\n", key, value)
}
// 7. 导出审计日志
auditExport, err := pipeline.ExportAuditLogs()
if err == nil {
fmt.Printf("\n审计日志已导出 (%d bytes)\n", len(auditExport))
}
}
运行输出示例
$ go run main.go
=== 敏感数据检测与脱敏演示 ===
--- 测试: 正常用户输入 ---
原始输入: 请帮我总结一下这篇关于机器学习的文章
✅ 输入通过检测
✅ 输出通过检测
处理耗时: 124µs
--- 测试: 包含身份证号 ---
原始输入: 我的身份证号是 110101199001011234,请帮我查一下社保信息
✅ 输入已脱敏: 1 处敏感数据
脱敏后: 我的身份证号是 1101********1234,请帮我查一下社保信息
- 中国大陆身份证号: 110101199001011234 → 1101********1234 (风险: high)
✅ 输出通过检测
处理耗时: 356µs
--- 测试: 包含信用卡号 ---
原始输入: 信用卡 6222-1234-5678-9010 这个月的账单是多少?
✅ 输入已脱敏: 1 处敏感数据
脱敏后: 信用卡 6222********9010 这个月的账单是多少?
- 信用卡号: 6222123456789010 → 6222********9010 (风险: high)
✅ 输出通过检测
处理耗时: 412µs
--- 测试: 包含 API Key ---
原始输入: 这是我的 API Key: sk-abcdefghijklmnopqrstuvwxyz1234567890
✅ 输入已脱敏: 1 处敏感数据
脱敏后: 这是我的 API Key: sk-abcd******************************
- API Key: sk-abcdefghijklmnopqrstuvwxyz1234567890 → sk-abcd****************************** (风险: medium)
✅ 输出通过检测
处理耗时: 289µs
--- 测试: 包含邮箱和密码 ---
原始输入: 登录信息: admin@company.com / P@ssw0rd!
✅ 输入已脱敏: 2 处敏感数据
脱敏后: 登录信息: ad***@company.com / ********
- 邮箱地址: admin@company.com → ad***@company.com (风险: low)
- 密码: P@ssw0rd! → ******** (风险: high)
✅ 输出通过检测
处理耗时: 503µs
--- 测试: 包含私钥 ---
原始输入: -----BEGIN RSA PRIVATE KEY-----
MIIEpAIBAAKCAQEA...
-----END RSA PRIVATE KEY-----
✅ 输入已脱敏: 1 处敏感数据
脱敏后: -----BEGIN REDACTED PRIVATE KEY-----
- 私钥: -----BEGIN RSA PRIVATE KEY----- → -----BEGIN REDACTED PRIVATE KEY----- (风险: high)
✅ 输出通过检测
处理耗时: 187µs
=== 管道统计 ===
total_requests: 6
input_blocks: 0
input_masks: 4
output_blocks: 0
output_masks: 0
average_latency_ms: 312
cache_hit_rate: 0
审计日志已导出 (4280 bytes)
七、生产部署建议
7.1 性能优化
performance_optimization:
# 缓存策略
cache:
input_cache_size: 10000 # 输入缓存条目数
output_cache_size: 5000 # 输出缓存条目数
cache_ttl: "5m" # 缓存有效期
# 并行处理
concurrency:
worker_pool_size: 10 # 检测工作线程数
batch_size: 100 # 批处理大小
queue_depth: 1000 # 队列深度
# 正则优化
regex:
precompile: true # 预编译正则
timeout: "100ms" # 单个正则匹配超时
use_dfa: true # 使用 DFA 引擎
# 采样策略
sampling:
enable_sampling: true # 低风险场景采样检测
sample_rate: 0.1 # 采样率 10%
full_scan_on_high_risk: true # 高风险场景全量扫描
7.2 误报处理
false_positive_handling:
# 白名单
whitelist:
- pattern: "test@example.com" # 测试邮箱
- pattern: "4242-4242-4242-4242" # 测试信用卡号
- pattern: "A123456789" # 测试身份证号
# 置信度阈值
thresholds:
id_card: 0.85 # 身份证需要较高置信度
phone: 0.75 # 手机号
email: 0.65 # 邮箱(容易误报)
api_key: 0.80 # API Key
credit_card: 0.90 # 信用卡(需要 Luhn 验证)
# 人工审核
review:
auto_review_threshold: 0.6 # 低于此阈值的需要人工确认
review_queue_size: 1000 # 审核队列大小
feedback_loop: true # 反馈闭环,持续优化
7.3 部署架构
┌─────────────────────────────────────────────────────────────────┐
│ 敏感数据防护架构 │
│ │
│ 用户请求 │
│ │ │
│ ▼ │
│ ┌──────────────────────────────────────────────────────────┐ │
│ │ Layer 1: 输入过滤 │ │
│ │ ├── 正则匹配检测 │ │
│ │ ├── 熵值检测 │ │
│ │ ├── 上下文分析 │ │
│ │ └── 自定义规则 │ │
│ └──────────────────────────────────────────────────────────┘ │
│ │ │
│ ├── 发现敏感数据 ──→ 脱敏/阻断 + 审计日志 │
│ │ │
│ ▼ │
│ ┌──────────────────────────────────────────────────────────┐ │
│ │ Layer 2: AI 模型处理 │ │
│ │ (模型本身不接触原始敏感数据) │ │
│ └──────────────────────────────────────────────────────────┘ │
│ │ │
│ ▼ │
│ ┌──────────────────────────────────────────────────────────┐ │
│ │ Layer 3: 输出过滤 │ │
│ │ ├── 模型记忆泄露检测 │ │
│ │ ├── 敏感数据重建检测 │ │
│ │ └── 输出内容脱敏 │ │
│ └──────────────────────────────────────────────────────────┘ │
│ │ │
│ ├── 发现敏感数据 ──→ 阻断输出 + 告警 │
│ │ │
│ ▼ │
│ 返回给用户 │
│ │
│ ┌──────────────────────────────────────────────────────────┐ │
│ │ 审计层 │ │
│ │ ├── 事件日志(ES/ClickHouse) │ │
│ │ ├── 实时告警(Prometheus + AlertManager) │ │
│ │ └── 合规报表(每周自动生成) │ │
│ └──────────────────────────────────────────────────────────┘ │
│ │
└─────────────────────────────────────────────────────────────────┘
八、关键要点
- 输入输出双重过滤 --- 输入防泄露,输出防模型记忆重建,缺一不可
- 多模式检测 --- 正则 + 熵值 + 上下文分析 + 自定义规则,覆盖全面
- Luhn 算法验证 --- 信用卡号检测必须经过 Luhn 验证,避免误报
- 分级脱敏策略 --- 不同类型敏感数据采用不同的脱敏方式,保留可用性
- 缓存提升性能 --- LRU 缓存大幅减少重复检测开销,适合高频场景
- 审计日志不可少 --- 每一次检测事件都要记录,用于合规审计和事后追溯
- 误报闭环 --- 白名单 + 置信度阈值 + 人工审核,持续降低误报率
- 输出阻断更严格 --- 输出中出现敏感数据的风险更高,应采取阻断而非仅脱敏
九、常见场景处理
| 场景 | 处理方式 | 说明 |
|---|---|---|
| 用户输入自己的身份证号查社保 | 输入脱敏,模型看不到完整号码 | 功能不受影响 |
| 模型输出了训练数据中的邮箱 | 输出阻断,返回"内容违规" | 防止数据泄露 |
| 用户粘贴了一段包含 API Key 的配置 | 输入脱敏,Key 被替换 | 防止 Key 泄露 |
| 模型在对话中复述了用户的手机号 | 输出阻断,触发告警 | 防止上下文泄露 |
| 用户上传包含信用卡号的截图 | OCR 后检测并脱敏 | 多模态场景 |
| 批量处理历史数据 | 离线全量扫描 + 脱敏 | 存量数据治理 |
💡 生产级 AI 安全实践推荐 :本讲完整代码及更多敏感数据处理方案(OCR 图片脱敏、音频转写检测、大规模离线扫描)已在 zz365.top 发布配套实战手册,欢迎查阅。
**第7讲预告:「审计日志与取证溯源」** --- 如何构建 AI 应用的完整审计体系,从请求追踪到模型行为记录,实现每一次推理都有迹可循,在发生安全事件时能够快速定位根因。