test(redact): corpus-level before/after coverage for value-aware gating (#96607)
This commit is contained in:
@@ -1112,3 +1112,69 @@ class TestMaskSecretControlStripping:
|
||||
def test_all_control_value_returns_empty_fallback(self):
|
||||
assert mask_secret("\n\x85\u200b") == ""
|
||||
assert mask_secret("\n\x85\u200b", empty="(not set)") == "(not set)"
|
||||
|
||||
|
||||
class TestValueAwareGatingCorpus:
|
||||
"""Issue #96607: corpus-level before/after for value-aware gating.
|
||||
|
||||
Redaction must mask a keyword-named assignment ONLY when the value has
|
||||
credential shape (vendor prefix, hex/base64/high-entropy, or a strong
|
||||
credential-specific key name). Bare technical vocabulary — ``token``,
|
||||
``key``, ``cpu`` — in ordinary technical prose/config must pass through
|
||||
byte-for-byte, on every assignment family (ENV, dotted config, JSON,
|
||||
YAML).
|
||||
"""
|
||||
|
||||
# Realistic technical prose. On pre-fix main every line was corrupted
|
||||
# to ``***`` despite containing no secret.
|
||||
TECHNICAL_CORPUS = [
|
||||
'IDENTITY_TOKEN="bailu"',
|
||||
"--override-tensor per_layer_token_embd.weight=CPU",
|
||||
"MAX_TOKENS=4096",
|
||||
"runtime.token=local",
|
||||
"The tokenizer splits on whitespace; set max_new_tokens=256.",
|
||||
"num_key_value_heads=8",
|
||||
"token: CPU",
|
||||
"llm_load_tensors: per_layer_token_embd.weight=CPU buffer",
|
||||
]
|
||||
|
||||
# Obviously-fake but shape-realistic secrets: every one of these must
|
||||
# STAY masked after the gating change (fail-closed on credential shape
|
||||
# or strong key names).
|
||||
FAKE_SECRET_CORPUS = [
|
||||
("API_KEY=sk-fakefakefakefakefake1234567890abcd", "fakefake"),
|
||||
("GITHUB_TOKEN=ghp_FAKEfakeFAKEfake1234567890fake", "FAKEfake"),
|
||||
("MY_SERVICE_TOKEN=A9f3kZq7Lm2Xw8Rt4Yv6", "A9f3kZq7"),
|
||||
("TOKEN=6f1d2a9c8b3e4f5a6d7c8b9a0e1f2d3c", "6f1d2a9c"),
|
||||
("password=hunter2", "hunter2"),
|
||||
("db_password: hunter2", "hunter2"),
|
||||
("auth_token: 9f8e7d6c5b4a39281706f5e4d3c2b1a0", "9f8e7d6c"),
|
||||
('"token": "Zx9Qw8Er7Ty6Ui5Op4As3"', "Zx9Qw8Er"),
|
||||
("SESSION_TOKEN=shrt", "shrt"),
|
||||
("client_secret=abc", "abc"),
|
||||
("spring.datasource.password=fakePass123", "fakePass123"),
|
||||
]
|
||||
|
||||
def test_technical_prose_survives_intact(self):
|
||||
for line in self.TECHNICAL_CORPUS:
|
||||
assert redact_sensitive_text(line, force=True) == line, line
|
||||
|
||||
def test_technical_corpus_as_one_block_survives_intact(self):
|
||||
# The multi-line shape a model actually reads from tool output.
|
||||
block = "\n".join(self.TECHNICAL_CORPUS)
|
||||
assert redact_sensitive_text(block, force=True) == block
|
||||
|
||||
def test_shape_realistic_fake_secrets_still_masked(self):
|
||||
for line, cleartext in self.FAKE_SECRET_CORPUS:
|
||||
result = redact_sensitive_text(line, force=True)
|
||||
assert result != line, line
|
||||
assert cleartext not in result, line
|
||||
|
||||
def test_mixed_block_masks_only_the_secret_lines(self):
|
||||
# Precondition guard: both halves must actually exercise the gate.
|
||||
secret_line = "MY_SERVICE_TOKEN=A9f3kZq7Lm2Xw8Rt4Yv6"
|
||||
prose_line = 'IDENTITY_TOKEN="bailu"'
|
||||
block = f"{prose_line}\n{secret_line}"
|
||||
result = redact_sensitive_text(block, force=True)
|
||||
assert prose_line in result
|
||||
assert "A9f3kZq7Lm2Xw8Rt4Yv6" not in result
|
||||
|
||||
Reference in New Issue
Block a user