Add guardrail to seahaven-alex agent #38

Merged
amoussa1229 merged 1 commit from feature/bedrock-guardrail into main 2026-06-03 19:16:07 +00:00

View file

@ -205,12 +205,65 @@ Use the knowledge base for policy, SOP, SA8000 compliance, and handbook question
Keep responses concise, professional, and actionable.`;
// ── Guardrail (audit M-19) ────────────────────────────────────────────────
// Scope: prompt-attack + content filters + masking of credential/financial
// identifiers. Names, emails, and phone numbers are deliberately NOT masked —
// returning vendor contacts and payment/WO/PO details is Alex's core job.
const guardrail = new bedrock.CfnGuardrail(this, 'Guardrail', {
name: 'seahaven-alex-guardrail',
description: 'Prompt-attack, content, and sensitive-identifier guardrail for Alex',
blockedInputMessaging:
"Sorry, I can't help with that request. If you think this was blocked in error, contact IT.",
blockedOutputsMessaging:
'Part of that response was withheld by policy. If you need this information, contact IT.',
contentPolicyConfig: {
filtersConfig: [
// outputStrength must be NONE for PROMPT_ATTACK per the Guardrails API
{ type: 'PROMPT_ATTACK', inputStrength: 'HIGH', outputStrength: 'NONE' },
{ type: 'HATE', inputStrength: 'HIGH', outputStrength: 'HIGH' },
{ type: 'INSULTS', inputStrength: 'HIGH', outputStrength: 'HIGH' },
{ type: 'SEXUAL', inputStrength: 'HIGH', outputStrength: 'HIGH' },
{ type: 'VIOLENCE', inputStrength: 'HIGH', outputStrength: 'HIGH' },
// Output at MEDIUM: Alex legitimately answers SA8000/HR questions about
// handling misconduct reports; HIGH output filtering suppresses those.
{ type: 'MISCONDUCT', inputStrength: 'HIGH', outputStrength: 'MEDIUM' },
],
},
sensitiveInformationPolicyConfig: {
piiEntitiesConfig: [
{ type: 'US_SOCIAL_SECURITY_NUMBER', action: 'ANONYMIZE' },
{ type: 'CREDIT_DEBIT_CARD_NUMBER', action: 'ANONYMIZE' },
{ type: 'US_BANK_ACCOUNT_NUMBER', action: 'ANONYMIZE' },
{ type: 'US_BANK_ROUTING_NUMBER', action: 'ANONYMIZE' },
{ type: 'PASSWORD', action: 'ANONYMIZE' },
{ type: 'AWS_ACCESS_KEY', action: 'ANONYMIZE' },
{ type: 'AWS_SECRET_KEY', action: 'ANONYMIZE' },
],
},
});
const guardrailVersion = new bedrock.CfnGuardrailVersion(this, 'GuardrailVersion', {
guardrailIdentifier: guardrail.attrGuardrailId,
description: 'Initial version — prompt-attack + content + credential/financial masking',
});
guardrailVersion.addDependency(guardrail);
agentRole.addToPolicy(new iam.PolicyStatement({
actions: ['bedrock:ApplyGuardrail'],
// Base ARN plus version-suffixed children — runtime applies the versioned guardrail
resources: [guardrail.attrGuardrailArn, `${guardrail.attrGuardrailArn}/*`],
}));
// ── CfnAgent ──────────────────────────────────────────────────────────────
this.agent = new bedrock.CfnAgent(this, 'Agent', {
agentName: 'seahaven-alex',
description: 'Alex — Sea Haven Industries internal Slack assistant',
agentResourceRoleArn: agentRole.roleArn,
foundationModel: BedrockAgentConstruct.MODEL_ID,
guardrailConfiguration: {
guardrailIdentifier: guardrail.attrGuardrailId,
guardrailVersion: guardrailVersion.attrVersion,
},
instruction,
idleSessionTtlInSeconds: 1800, // 30 min — matches the processor's session window
knowledgeBases: [
@ -357,12 +410,17 @@ Keep responses concise, professional, and actionable.`;
],
});
// Guardrail version must exist before the agent references it (review FIX)
this.agent.addDependency(guardrailVersion);
// ── Agent alias (stable ARN for invocations) ──────────────────────────────
// Creating the alias also triggers agent preparation in CloudFormation.
this.agentAlias = new bedrock.CfnAgentAlias(this, 'AgentAlias', {
agentId: this.agent.attrAgentId,
agentAliasName: 'live',
description: 'Production alias — Alex v1',
// Description bump forces an alias update so a new agent version snapshots
// the guardrail config — CfnAgentAlias does not pick up agent changes otherwise.
description: 'Production alias — Alex v2 (guardrail)',
});
}
}