Fix false positives in agent guardrails

Skip empty fingerprints in SystemPromptLeakGuardrail to prevent blank
values from flagging every message. Replace overly broad "sk-" pattern
in SensitiveDataGuardrail with specific LLM provider prefixes
("sk-proj-" for OpenAI, "sk-ant-" for Anthropic) to avoid false
positives on common words like "risk-based" or "task-management".

Signed-off-by: Bryan Frimin <bryan@getprobo.com>
This commit is contained in:
Bryan Frimin
2026-03-27 12:16:37 +01:00
committed by Sacha Al Himdani
parent 2f8674471b
commit 4725a1b080
5 changed files with 32 additions and 16 deletions

View File

@@ -66,4 +66,14 @@ func TestSystemPromptLeakGuardrail_Check(t *testing.T) {
require.NoError(t, err)
assert.False(t, result.Tripwire)
})
t.Run("empty fingerprints are ignored", func(t *testing.T) {
t.Parallel()
g := guardrail.NewSystemPromptLeakGuardrail([]string{"", "secret phrase", ""})
result, err := g.Check(context.Background(), assistantMessage("hello world"))
require.NoError(t, err)
assert.False(t, result.Tripwire)
})
}