mirror of
https://github.com/Sea-Haven-Industries/sh-mcp.git
synced 2026-10-01 11:43:16 +00:00
229 lines
9.4 KiB
TypeScript
229 lines
9.4 KiB
TypeScript
|
|
/**
|
||
|
|
* The single authoritative tool-execution path.
|
||
|
|
*
|
||
|
|
* Both transports — MCP (`mcp.ts`) and OpenAPI/HTTP (`http.ts`) — route every
|
||
|
|
* tool call through {@link executeTool}. Centralizing here is what makes the
|
||
|
|
* security guarantees of design.md §2.5 hold uniformly across interfaces:
|
||
|
|
*
|
||
|
|
* 1. tool lookup (unknown → `UnknownToolError` → 404)
|
||
|
|
* 2. server-side scope enforcement (`requireScope`; defense-in-depth — handlers
|
||
|
|
* also call it). Tool-hiding in the UI is NOT the boundary (design.md §2.5).
|
||
|
|
* 3. input-schema validation BEFORE the handler runs (ajv) — never pass
|
||
|
|
* unvalidated input to a handler (design.md §2.5 prompt-injection containment).
|
||
|
|
* 4. per-session cap + per-tool rate limit (design.md §7.3).
|
||
|
|
* 5. handler execution.
|
||
|
|
* 6. finance-tier redaction on egress — bank/routing/card/SSN masked before the
|
||
|
|
* value leaves the dispatcher (design.md §2.5, §7.3). Belt-and-braces over
|
||
|
|
* each finance tool's own internal redaction.
|
||
|
|
* 7. structured audit record for every finance (and future physical) call —
|
||
|
|
* args HASHED, never logged raw; no secrets (design.md §2.5, §7.3).
|
||
|
|
*
|
||
|
|
* Tool output is treated strictly as DATA, never as instructions: the dispatcher
|
||
|
|
* inspects/redacts it but never re-enters itself based on its content, so a
|
||
|
|
* prompt-injection payload in a tool response cannot trigger another tool call
|
||
|
|
* (design.md §2.5).
|
||
|
|
*/
|
||
|
|
|
||
|
|
import AjvModule from 'ajv';
|
||
|
|
import addFormatsModule from 'ajv-formats';
|
||
|
|
import type { Ajv as AjvInstance, Options as AjvOptions, ValidateFunction } from 'ajv';
|
||
|
|
|
||
|
|
// ajv / ajv-formats ship as CJS; under NodeNext ESM the callable lives on
|
||
|
|
// `.default`. Normalize so both module shapes work, then re-type the runtime
|
||
|
|
// values as the constructable class / callable plugin.
|
||
|
|
type AjvCtor = new (opts?: AjvOptions) => AjvInstance;
|
||
|
|
const Ajv = ((AjvModule as { default?: unknown }).default ?? AjvModule) as unknown as AjvCtor;
|
||
|
|
const addFormats = ((addFormatsModule as { default?: unknown }).default ?? addFormatsModule) as (
|
||
|
|
ajv: AjvInstance,
|
||
|
|
) => AjvInstance;
|
||
|
|
|
||
|
|
import { requireScope, ScopeError } from './auth.js';
|
||
|
|
import { hashArgs, type AuditLogger, type AuditRecord } from './audit.js';
|
||
|
|
import { redact } from './redact.js';
|
||
|
|
import { RateLimitError, type RateLimiter } from './rate-limit.js';
|
||
|
|
import type { ToolRegistry } from './registry.js';
|
||
|
|
import type { AuthContext, ToolDef } from './types.js';
|
||
|
|
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
// Errors (adapters translate these to status codes)
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
/** Unknown tool name → HTTP 404 / MCP "method not found"-equivalent. */
|
||
|
|
export class UnknownToolError extends Error {
|
||
|
|
readonly tool: string;
|
||
|
|
constructor(tool: string) {
|
||
|
|
super(`Unknown tool "${tool}".`);
|
||
|
|
this.name = 'UnknownToolError';
|
||
|
|
this.tool = tool;
|
||
|
|
Object.setPrototypeOf(this, new.target.prototype);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
/** Input failed JSON-Schema validation → HTTP 400. */
|
||
|
|
export class InputValidationError extends Error {
|
||
|
|
readonly tool: string;
|
||
|
|
/** Human-readable validation messages; never echoes secrets. */
|
||
|
|
readonly issues: string[];
|
||
|
|
constructor(tool: string, issues: string[]) {
|
||
|
|
super(`Input validation failed for "${tool}": ${issues.join('; ')}`);
|
||
|
|
this.name = 'InputValidationError';
|
||
|
|
this.tool = tool;
|
||
|
|
this.issues = issues;
|
||
|
|
Object.setPrototypeOf(this, new.target.prototype);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
// Dependencies injected into the dispatcher
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
export interface DispatchDeps {
|
||
|
|
/** Audit sink — finance/physical calls emit a record here. */
|
||
|
|
auditLogger: AuditLogger;
|
||
|
|
/** Per-session + per-tool limiter consulted before each handler runs. */
|
||
|
|
rateLimiter: RateLimiter;
|
||
|
|
}
|
||
|
|
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
// AJV — compiled validators cached per tool
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
// One Ajv instance for the process. Schemas are JSON Schema (draft-07 / the
|
||
|
|
// OpenAPI 3.1 subset our tools use). `strict: false` because tool authors use
|
||
|
|
// vocabulary (e.g. `description`) liberally; we only need structural validation.
|
||
|
|
//
|
||
|
|
// `allErrors: false` (the default) is deliberate and security-relevant: the
|
||
|
|
// input is UNTRUSTED, and `allErrors: true` makes ajv enumerate every schema
|
||
|
|
// violation, which an attacker can weaponize into CPU/memory exhaustion by
|
||
|
|
// sending a large/deeply-nested payload that fails many constraints at once
|
||
|
|
// (CodeQL js/resource-exhaustion). Short-circuiting on the first error caps the
|
||
|
|
// work per request; the 400 still names the first failing path, which is enough
|
||
|
|
// for a caller to fix their input.
|
||
|
|
const ajv = new Ajv({ allErrors: false, strict: false, coerceTypes: false });
|
||
|
|
addFormats(ajv);
|
||
|
|
|
||
|
|
const validatorCache = new WeakMap<object, ValidateFunction>();
|
||
|
|
|
||
|
|
function getValidator(tool: ToolDef<unknown, unknown>): ValidateFunction {
|
||
|
|
const schema = tool.inputSchema as object;
|
||
|
|
const cached = validatorCache.get(schema);
|
||
|
|
if (cached) return cached;
|
||
|
|
const validate = ajv.compile(schema);
|
||
|
|
validatorCache.set(schema, validate);
|
||
|
|
return validate;
|
||
|
|
}
|
||
|
|
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
// Finance egress redaction
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
/** Field names whose values are masked wholesale on finance egress. */
|
||
|
|
const SENSITIVE_FIELD_RE = /(account|routing|card|ssn|tax[_-]?id|iban|swift)/i;
|
||
|
|
|
||
|
|
/**
|
||
|
|
* Deep-redact a finance tool's output before it leaves the dispatcher.
|
||
|
|
*
|
||
|
|
* Two complementary passes (design.md §2.5):
|
||
|
|
* - String values run through `redact()` (pattern-based: routing/account/card/SSN).
|
||
|
|
* - Any field whose KEY looks sensitive is fully replaced with `[REDACTED]`,
|
||
|
|
* catching isolated values (e.g. a bare `bankAccountNumber: "123456789"`)
|
||
|
|
* that the inline pattern matcher would miss without keyword context.
|
||
|
|
*
|
||
|
|
* Returns a NEW structure; the handler's value is not mutated.
|
||
|
|
*/
|
||
|
|
export function redactDeep(value: unknown): unknown {
|
||
|
|
if (typeof value === 'string') return redact(value);
|
||
|
|
if (Array.isArray(value)) return value.map(redactDeep);
|
||
|
|
if (value !== null && typeof value === 'object') {
|
||
|
|
const out: Record<string, unknown> = {};
|
||
|
|
for (const [key, v] of Object.entries(value as Record<string, unknown>)) {
|
||
|
|
if (SENSITIVE_FIELD_RE.test(key) && v !== null && v !== undefined) {
|
||
|
|
// Key looks sensitive → mask the ENTIRE value wholesale, whether it is a
|
||
|
|
// scalar, an array, or a nested object. Recursing into a non-scalar here
|
||
|
|
// would lose the keyword context redact() needs, letting a bare nested
|
||
|
|
// value (e.g. { account: { number: "021000021" } }) escape unmasked.
|
||
|
|
// Over-masking on the finance tier is the correct trade: a false negative
|
||
|
|
// is a PII leak (design.md §2.5).
|
||
|
|
out[key] = '[REDACTED]';
|
||
|
|
} else {
|
||
|
|
out[key] = redactDeep(v);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return out;
|
||
|
|
}
|
||
|
|
return value;
|
||
|
|
}
|
||
|
|
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
// executeTool — the one path
|
||
|
|
// ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
/**
|
||
|
|
* Look up, authorize, validate, rate-limit, run, redact, and audit a single
|
||
|
|
* tool call. Used identically by both transports.
|
||
|
|
*
|
||
|
|
* @throws {UnknownToolError} unknown tool name (→ 404)
|
||
|
|
* @throws {ScopeError} caller lacks the required scope (→ 403)
|
||
|
|
* @throws {InputValidationError} input failed schema validation (→ 400)
|
||
|
|
* @throws {RateLimitError} limiter rejected the call (→ 429)
|
||
|
|
* @throws {Error} handler threw (→ 500; message not leaked verbatim)
|
||
|
|
*/
|
||
|
|
export async function executeTool(
|
||
|
|
registry: ToolRegistry,
|
||
|
|
ctx: AuthContext,
|
||
|
|
toolName: string,
|
||
|
|
rawInput: unknown,
|
||
|
|
deps: DispatchDeps,
|
||
|
|
): Promise<unknown> {
|
||
|
|
const tool = registry.get(toolName) as ToolDef<unknown, unknown> | undefined;
|
||
|
|
if (!tool) {
|
||
|
|
throw new UnknownToolError(toolName);
|
||
|
|
}
|
||
|
|
|
||
|
|
const audited = tool.tier === 'finance';
|
||
|
|
let decision: AuditRecord['decision'] = 'deny';
|
||
|
|
let result: AuditRecord['result'] = 'error';
|
||
|
|
|
||
|
|
try {
|
||
|
|
// 2. Server-side scope enforcement (authoritative; not UI tool-hiding).
|
||
|
|
requireScope(ctx, tool.requiredScope);
|
||
|
|
decision = 'allow';
|
||
|
|
|
||
|
|
// 3. Input-schema validation BEFORE the handler sees the input.
|
||
|
|
const validate = getValidator(tool);
|
||
|
|
if (!validate(rawInput)) {
|
||
|
|
const issues = (validate.errors ?? []).map(
|
||
|
|
(e) => `${e.instancePath || '(root)'} ${e.message ?? 'is invalid'}`,
|
||
|
|
);
|
||
|
|
throw new InputValidationError(toolName, issues.length ? issues : ['invalid input']);
|
||
|
|
}
|
||
|
|
|
||
|
|
// 4. Rate limit / session cap.
|
||
|
|
deps.rateLimiter.check(ctx.sub, toolName);
|
||
|
|
|
||
|
|
// 5. Run the handler. Its output is data only.
|
||
|
|
const output = await tool.handler(rawInput, ctx);
|
||
|
|
|
||
|
|
// 6. Finance egress redaction (belt-and-braces over internal redaction).
|
||
|
|
const safeOutput = tool.tier === 'finance' ? redactDeep(output) : output;
|
||
|
|
|
||
|
|
result = 'ok';
|
||
|
|
return safeOutput;
|
||
|
|
} finally {
|
||
|
|
// 7. Audit every finance/physical call regardless of outcome.
|
||
|
|
if (audited) {
|
||
|
|
deps.auditLogger.log({
|
||
|
|
sub: ctx.sub,
|
||
|
|
tool: toolName,
|
||
|
|
argsHash: hashArgs(rawInput),
|
||
|
|
decision,
|
||
|
|
result,
|
||
|
|
ts: new Date().toISOString(),
|
||
|
|
});
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
// Re-export the error types adapters need to translate outcomes.
|
||
|
|
export { ScopeError, RateLimitError };
|