t drift from silently bypassing compliance requirements. It also enables independent scaling: you can swap models or adjust temperature without rewriting core validation logic.
import { z } from 'zod';
const RefundRequestSchema = z.object({
orderId: z.string().uuid(),
amountCents: z.number().int().min(0),
reason: z.string().max(500),
});
type RefundRequest = z.infer<typeof RefundRequestSchema>;
export class TransactionGuard {
private readonly maxAutoRefundCents = 5000;
async validateAndRoute(request: RefundRequest): Promise<{ status: 'approved' | 'escalated' | 'rejected' }> {
const parsed = RefundRequestSchema.safeParse(request);
if (!parsed.success) {
return { status: 'rejected' };
}
const { orderId, amountCents } = parsed.data;
if (amountCents > this.maxAutoRefundCents) {
return { status: 'escalated' };
}
const hasOwnership = await this.verifyOrderOwnership(orderId);
if (!hasOwnership) {
return { status: 'rejected' };
}
return { status: 'approved' };
}
private async verifyOrderOwnership(orderId: string): Promise<boolean> {
// Deterministic database lookup
return true;
}
}
Step 2: Implement Hybrid Retrieval with Mandatory Citation Enforcement
Pure vector similarity search fails on exact identifiers, error codes, SKUs, and structured metadata. Production retrieval pipelines must combine dense embeddings with keyword-based matching (BM25 or inverted indexes) to guarantee exact-match recall. Additionally, every model response must be traceable to specific retrieved chunks to enable hallucination detection and auditability.
Architecture Rationale: Hybrid search reduces false negatives on structured queries. Citation enforcement creates a verifiable feedback loop for evaluation pipelines and enables downstream systems to validate source credibility.
import { EmbeddingModel, KeywordIndex, DocumentChunk } from '@ai-engine/core';
export class HybridRetriever {
constructor(
private readonly embeddingModel: EmbeddingModel,
private readonly keywordIndex: KeywordIndex
) {}
async search(query: string, topK: number = 5): Promise<DocumentChunk[]> {
const queryVector = await this.embeddingModel.encode(query);
const vectorResults = await this.embeddingIndex.search(queryVector, topK);
const keywordResults = await this.keywordIndex.search(query, topK);
const merged = this.rerankAndDeduplicate(vectorResults, keywordResults);
return merged.slice(0, topK);
}
private rerankAndDeduplicate(
vectorHits: DocumentChunk[],
keywordHits: DocumentChunk[]
): DocumentChunk[] {
const seen = new Set<string>();
const combined: DocumentChunk[] = [];
for (const hit of [...vectorHits, ...keywordHits]) {
if (!seen.has(hit.id)) {
seen.add(hit.id);
combined.push(hit);
}
}
return combined.sort((a, b) => (b.score ?? 0) - (a.score ?? 0));
}
}
Step 3: Deploy Versioned Evaluation Pipelines
Single-run pass/fail testing is mathematically invalid for non-deterministic systems. Evaluation must run against a versioned dataset with threshold-based metrics. Every prompt modification, model swap, or retrieval configuration change triggers a full regression run. Metrics should measure faithfulness, citation accuracy, and business-rule compliance across the entire dataset, not isolated examples.
Architecture Rationale: Threshold-based evaluation catches regression drift that single-case testing misses. Versioning ensures reproducibility and enables rollback decisions based on empirical data rather than subjective assessment.
import { EvalDataset, MetricThreshold, EvalReport } from '@ai-engine/eval';
export class EvaluationPipeline {
async run(dataset: EvalDataset, thresholds: MetricThreshold[]): Promise<EvalReport> {
const results = await Promise.all(
dataset.cases.map(async (testCase) => {
const modelOutput = await this.invokeModel(testCase.input);
return this.measure(testCase, modelOutput);
})
);
const aggregated = this.aggregateMetrics(results);
const violations = thresholds.filter(t => aggregated[t.metric] < t.minValue);
return {
passed: violations.length === 0,
metrics: aggregated,
violations,
datasetVersion: dataset.version,
};
}
private async invokeModel(input: string): Promise<string> {
// Model inference logic
return '';
}
private measure(testCase: any, output: string): Record<string, number> {
// Faithfulness, citation match, rule compliance scoring
return { faithfulness: 0.92, citationAccuracy: 0.88 };
}
private aggregateMetrics(results: Record<string, number>[]): Record<string, number> {
// Average/percentile aggregation
return { faithfulness: 0.91, citationAccuracy: 0.87 };
}
}
Step 4: Implement Deterministic Gates for AI-Generated Code
AI-assisted development accelerates code volume but does not reduce maintenance cost. The governing principle is simple: generation is cheap, ownership is expensive. Every AI-generated artifact must pass through automated static gates before entering human review. These gates enforce type safety, linting compliance, secret scanning, SAST vulnerability checks, and coverage delta validation.
Architecture Rationale: Automated gates absorb the increased review volume without degrading quality. They prevent common AI failure modesâstring-concatenated SQL, outdated API signatures, timezone mishandling, and integer truncationâfrom reaching production.
import { execSync } from 'child_process';
import { ReviewGateResult } from '@dev-tooling/gates';
export class CodeReviewGate {
async validate(changeSet: string[]): Promise<ReviewGateResult> {
const checks = [
{ name: 'type-checks', run: () => this.runCommand('tsc --noEmit') },
{ name: 'linter-clean', run: () => this.runCommand('eslint --max-warnings=0') },
{ name: 'no-vuln-patterns', run: () => this.runCommand('npm run sast') },
{ name: 'no-secrets', run: () => this.runCommand('gitleaks detect') },
{ name: 'coverage-not-reduced', run: () => this.checkCoverageDelta() },
];
const failures: string[] = [];
for (const check of checks) {
try {
await check.run();
} catch {
failures.push(check.name);
}
}
return {
ready: failures.length === 0,
blockedBy: failures,
};
}
private runCommand(cmd: string): Promise<void> {
return new Promise((resolve, reject) => {
try {
execSync(cmd, { stdio: 'inherit' });
resolve();
} catch {
reject(new Error(`Gate failed: ${cmd}`));
}
});
}
private async checkCoverageDelta(): Promise<void> {
// Compare current coverage against baseline
const delta = await this.getCoverageDiff();
if (delta < 0) throw new Error('Coverage regression detected');
}
}
Step 5: Anchor Legacy Modernization with Characterization Tests
When AI assists in refactoring or translating legacy modules, the model will inevitably drop side effects, alter error handling, or change implicit contracts. The only safe migration path is characterization testing: capturing the existing system's behavior (including quirks and edge cases) before writing modernized code. The new implementation must satisfy the characterization suite, not an idealized specification.
Architecture Rationale: Characterization tests preserve behavioral contracts during transformation. They prevent downstream system failures caused by silent logic changes and provide a deterministic safety net for incremental modernization.
Pitfall Guide
1. Prompt-Enforced Business Rules
Explanation: Embedding spending caps, permission checks, or compliance thresholds directly in system prompts. Models ignore or reinterpret instructions under distribution shift or adversarial input.
Fix: Move all invariants to deterministic code that executes post-inference. Treat prompts as data transformers, not policy engines.
2. Pure Vector Search for Exact Identifiers
Explanation: Relying exclusively on cosine similarity for retrieval. Dense embeddings struggle with exact matches on SKUs, error codes, serial numbers, and structured metadata.
Fix: Implement hybrid retrieval combining dense vectors with BM25/keyword indexing. Apply reranking to balance semantic relevance and exact-match precision.
3. Auto-Generating Tests for Auto-Generated Code
Explanation: Using the same model to write implementation and test cases. The test suite inevitably mirrors implementation bugs, encoding incorrect behavior as expected outcomes.
Fix: Generate tests only for human-authored code. When AI writes both, anchor validation to a separate human-defined specification or characterization suite.
4. Premature Multi-Agent Orchestration
Explanation: Deploying autonomous multi-agent systems for tasks that require simple classification or single-step reasoning. Each additional agent layer multiplies debugging complexity and reduces reliability.
Fix: Start with single-call, typed-output architectures. Reserve orchestration for problems requiring dynamic planning, tool chaining, or unpredictable workflow branching.
5. Skipping Versioned Evaluation Sets
Explanation: Running ad-hoc tests or relying on manual spot-checks after prompt/model changes. Non-deterministic outputs require statistical validation across consistent datasets.
Fix: Maintain versioned evaluation sets with threshold metrics. Trigger full regression runs on every configuration change. Reject deployments that violate minimum faithfulness or accuracy thresholds.
6. Bypassing Static Gates for AI Drafts
Explanation: Treating AI-generated code as "pre-validated" and routing it directly to human review without automated checks. This increases reviewer fatigue and allows subtle vulnerabilities to slip through.
Fix: Enforce identical static analysis, linting, and coverage gates for AI and human-authored code. Block PRs that fail deterministic checks before human review begins.
7. Ignoring Characterization Tests During Refactoring
Explanation: Modernizing legacy code without capturing existing behavior. AI translations silently drop side effects, alter error propagation, or change implicit contracts.
Fix: Record characterization tests before modernization. Require the new implementation to pass the exact same behavioral suite. Migrate in small, verifiable increments.
Production Bundle
Action Checklist
Decision Matrix
| Scenario | Recommended Approach | Why | Cost Impact |
|---|
| Exact-match retrieval (SKUs, error codes) | Hybrid search (dense + BM25) | Pure vectors miss structured identifiers | +15% infra, -40% support tickets |
| High-stakes transaction routing | Deterministic guardrails post-inference | Prompts cannot enforce compliance reliably | +10% dev time, -60% incident rate |
| AI-assisted legacy migration | Characterization tests + incremental refactoring | Prevents silent side-effect loss | +20% upfront testing, -50% regression bugs |
| Test generation workflow | AI tests for human code only | Avoids mirroring implementation bugs | Neutral cost, +35% defect detection |
| Multi-agent orchestration | Single-call typed output first | Reduces debugging complexity and latency | -30% infra cost, +25% reliability |
Configuration Template
# ai-pipeline.config.yaml
evaluation:
dataset_version: v2.4.1
thresholds:
faithfulness: 0.85
citation_accuracy: 0.90
business_rule_compliance: 1.0
regression_trigger: "on_prompt_or_model_change"
retrieval:
strategy: hybrid
vector_model: "text-embedding-3-large"
keyword_index: "bm25"
reranker: "cross-encoder-ms-marco"
max_chunks: 5
enforce_citation: true
guardrails:
business_rules: "deterministic_code"
idempotency: "required"
pii_filtering: "enabled"
access_control: "rbac"
code_review:
ai_draft_gates:
- type_check: true
- lint_strict: true
- sast_scan: true
- secret_detection: true
- coverage_delta: ">= 0"
human_accountability: "mandatory"
Quick Start Guide
- Initialize the evaluation dataset: Create a versioned JSON file containing 30â50 representative inputs, expected outputs, and business-rule constraints. Tag it with a semantic version.
- Deploy hybrid retrieval: Configure your vector database alongside a keyword index. Route queries through both, merge results, and enforce citation attachment on every model response.
- Add deterministic guardrails: Wrap model outputs in a validation layer that checks permissions, spending limits, and data constraints before executing state changes.
- Enable code review gates: Integrate type checking, linting, SAST, and secret scanning into your CI pipeline. Block AI-generated PRs that fail any gate before human review.
- Run the first regression: Trigger the evaluation pipeline against your dataset. Verify that faithfulness and citation accuracy meet thresholds. Commit the baseline metrics and proceed to staging.