Add guardrail to seahaven-alex agent
Audit finding M-19: the employee-facing assistant had no guardrail despite access to QBO, payments, WO/PO, and HR/SA8000 data. Adds prompt-attack (HIGH input), content filters, and masking of credential/financial identifiers (SSN, cards, bank numbers, keys). Names/emails/phones deliberately unmasked - vendor contact lookup is the bot's core function. MISCONDUCT output at MEDIUM so SA8000 misconduct-reporting questions are not suppressed. Cross-reviewed (1 BLOCK fixed: ApplyGuardrail now covers version- suffixed ARNs; explicit guardrail->version->agent dependencies added). Alias description bump forces a new agent version (v10) so the live alias snapshots the guardrail config.
This commit is contained in:
parent
2df2362304
commit
0de0ce8cee
1 changed files with 59 additions and 1 deletions
|
|
@ -205,12 +205,65 @@ Use the knowledge base for policy, SOP, SA8000 compliance, and handbook question
|
|||
|
||||
Keep responses concise, professional, and actionable.`;
|
||||
|
||||
// ── Guardrail (audit M-19) ────────────────────────────────────────────────
|
||||
// Scope: prompt-attack + content filters + masking of credential/financial
|
||||
// identifiers. Names, emails, and phone numbers are deliberately NOT masked —
|
||||
// returning vendor contacts and payment/WO/PO details is Alex's core job.
|
||||
const guardrail = new bedrock.CfnGuardrail(this, 'Guardrail', {
|
||||
name: 'seahaven-alex-guardrail',
|
||||
description: 'Prompt-attack, content, and sensitive-identifier guardrail for Alex',
|
||||
blockedInputMessaging:
|
||||
"Sorry, I can't help with that request. If you think this was blocked in error, contact IT.",
|
||||
blockedOutputsMessaging:
|
||||
'Part of that response was withheld by policy. If you need this information, contact IT.',
|
||||
contentPolicyConfig: {
|
||||
filtersConfig: [
|
||||
// outputStrength must be NONE for PROMPT_ATTACK per the Guardrails API
|
||||
{ type: 'PROMPT_ATTACK', inputStrength: 'HIGH', outputStrength: 'NONE' },
|
||||
{ type: 'HATE', inputStrength: 'HIGH', outputStrength: 'HIGH' },
|
||||
{ type: 'INSULTS', inputStrength: 'HIGH', outputStrength: 'HIGH' },
|
||||
{ type: 'SEXUAL', inputStrength: 'HIGH', outputStrength: 'HIGH' },
|
||||
{ type: 'VIOLENCE', inputStrength: 'HIGH', outputStrength: 'HIGH' },
|
||||
// Output at MEDIUM: Alex legitimately answers SA8000/HR questions about
|
||||
// handling misconduct reports; HIGH output filtering suppresses those.
|
||||
{ type: 'MISCONDUCT', inputStrength: 'HIGH', outputStrength: 'MEDIUM' },
|
||||
],
|
||||
},
|
||||
sensitiveInformationPolicyConfig: {
|
||||
piiEntitiesConfig: [
|
||||
{ type: 'US_SOCIAL_SECURITY_NUMBER', action: 'ANONYMIZE' },
|
||||
{ type: 'CREDIT_DEBIT_CARD_NUMBER', action: 'ANONYMIZE' },
|
||||
{ type: 'US_BANK_ACCOUNT_NUMBER', action: 'ANONYMIZE' },
|
||||
{ type: 'US_BANK_ROUTING_NUMBER', action: 'ANONYMIZE' },
|
||||
{ type: 'PASSWORD', action: 'ANONYMIZE' },
|
||||
{ type: 'AWS_ACCESS_KEY', action: 'ANONYMIZE' },
|
||||
{ type: 'AWS_SECRET_KEY', action: 'ANONYMIZE' },
|
||||
],
|
||||
},
|
||||
});
|
||||
|
||||
const guardrailVersion = new bedrock.CfnGuardrailVersion(this, 'GuardrailVersion', {
|
||||
guardrailIdentifier: guardrail.attrGuardrailId,
|
||||
description: 'Initial version — prompt-attack + content + credential/financial masking',
|
||||
});
|
||||
guardrailVersion.addDependency(guardrail);
|
||||
|
||||
agentRole.addToPolicy(new iam.PolicyStatement({
|
||||
actions: ['bedrock:ApplyGuardrail'],
|
||||
// Base ARN plus version-suffixed children — runtime applies the versioned guardrail
|
||||
resources: [guardrail.attrGuardrailArn, `${guardrail.attrGuardrailArn}/*`],
|
||||
}));
|
||||
|
||||
// ── CfnAgent ──────────────────────────────────────────────────────────────
|
||||
this.agent = new bedrock.CfnAgent(this, 'Agent', {
|
||||
agentName: 'seahaven-alex',
|
||||
description: 'Alex — Sea Haven Industries internal Slack assistant',
|
||||
agentResourceRoleArn: agentRole.roleArn,
|
||||
foundationModel: BedrockAgentConstruct.MODEL_ID,
|
||||
guardrailConfiguration: {
|
||||
guardrailIdentifier: guardrail.attrGuardrailId,
|
||||
guardrailVersion: guardrailVersion.attrVersion,
|
||||
},
|
||||
instruction,
|
||||
idleSessionTtlInSeconds: 1800, // 30 min — matches the processor's session window
|
||||
knowledgeBases: [
|
||||
|
|
@ -357,12 +410,17 @@ Keep responses concise, professional, and actionable.`;
|
|||
],
|
||||
});
|
||||
|
||||
// Guardrail version must exist before the agent references it (review FIX)
|
||||
this.agent.addDependency(guardrailVersion);
|
||||
|
||||
// ── Agent alias (stable ARN for invocations) ──────────────────────────────
|
||||
// Creating the alias also triggers agent preparation in CloudFormation.
|
||||
this.agentAlias = new bedrock.CfnAgentAlias(this, 'AgentAlias', {
|
||||
agentId: this.agent.attrAgentId,
|
||||
agentAliasName: 'live',
|
||||
description: 'Production alias — Alex v1',
|
||||
// Description bump forces an alias update so a new agent version snapshots
|
||||
// the guardrail config — CfnAgentAlias does not pick up agent changes otherwise.
|
||||
description: 'Production alias — Alex v2 (guardrail)',
|
||||
});
|
||||
}
|
||||
}
|
||||
|
|
|
|||
Reference in a new issue