feat(p7): FakeInference deterministic provider + golden eval registry

Task 7.2: FakeInference implements the InferencePort with keyword vocabulary
mirroring the P6 RuleEngine; deterministic (no clock/random) so AI artifacts are
replayable. Golden fixtures (AI_GOLDEN) + inference.spec.ts assert byte-identical
replay, abstention, and Scenario-7 (EVENT never exceeds DISCUSSION certainty).
Controls: abstain()/deny()/setCost().

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-07-01 16:29:51 +05:30
parent 42bb99c09f
commit e03f2f72e4
5 changed files with 252 additions and 0 deletions
+2
View File
@@ -6,8 +6,10 @@ import { FakeMdm } from './mdm';
import { FakeCrre } from './crre';
import { FakeSas } from './sas';
import { FakeCapability } from './capability';
import { FakeInference, makeFakeInference } from './inference';
export { FakeSession, FakeOpa, FakeCmp, FakeMdm, FakeCrre, FakeSas, FakeCapability };
export { FakeInference, makeFakeInference };
/** The full set of fakes, typed so tests can reach each fake's control methods. */
export interface FakePorts extends IiosPlatformPorts {
@@ -0,0 +1,152 @@
import type {
InferencePort,
InferenceRequest,
InferenceResult,
InferenceClaim,
InferenceEvidence,
InferenceModelRun,
} from '@insignia/iios-contracts';
/**
* Deterministic, golden fake of the AI inference port (P7). No network, no clock,
* no randomness — the same input always yields byte-identical output, which is what
* makes AI artifacts replayable. Keyword vocabulary intentionally mirrors the P6
* RuleEngine so AI-proposed flags line up with deterministic ones.
*
* Controls: `.abstain()` (always decline a claim), `.deny()` (simulate a provider
* error → the service must fail closed), `.setCost(n)` (force costUnits for budget
* tests).
*/
export class FakeInference implements InferencePort {
private mode: 'normal' | 'abstain' | 'deny' = 'normal';
private costOverride?: number;
abstain(): this {
this.mode = 'abstain';
return this;
}
deny(): this {
this.mode = 'deny';
return this;
}
setCost(n: number): this {
this.costOverride = n;
return this;
}
async infer(req: InferenceRequest): Promise<InferenceResult> {
if (this.mode === 'deny') throw new Error('fake inference provider error');
const text = req.text ?? '';
const evidence: InferenceEvidence[] = [
{ sourceType: 'INTERACTION', sourceId: req.interactionId ?? 'unknown', relevanceScore: 1 },
];
const modelRun = this.run(req, text);
if (this.mode === 'abstain') {
return {
artifactType: req.jobType === 'SUMMARIZE' ? 'SUMMARY' : req.jobType === 'EXTRACT' ? 'EXTRACTION' : 'CLASSIFICATION',
content: {},
confidence: 0,
abstentionReason: 'ABSTAINED',
claims: [],
evidence,
modelRun,
};
}
if (req.jobType === 'CLASSIFY') {
const flagTypes = this.flags(text);
const claims: InferenceClaim[] = flagTypes.map((flagType) => ({
claimType: 'MODERATION_FLAG',
claimJson: { flagType, explanation: `keyword rule matched for ${flagType}` },
confidence: 0.9,
}));
return {
artifactType: 'CLASSIFICATION',
content: { flags: flagTypes },
confidence: flagTypes.length > 0 ? 0.9 : 0.95,
explanation: flagTypes.length > 0 ? `flagged: ${flagTypes.join(', ')}` : 'no moderation concerns',
claims,
evidence,
modelRun,
};
}
if (req.jobType === 'SUMMARIZE') {
const first = (text.split(/[.!?]/)[0] ?? '').trim();
return {
artifactType: 'SUMMARY',
content: { text: `[ai] ${first}`, sourceInteractionId: req.interactionId ?? null },
confidence: 0.8,
explanation: 'extractive first-sentence summary (golden)',
claims: [],
evidence,
modelRun,
};
}
// EXTRACT
const event = this.extractEvent(text);
if (!event) {
return {
artifactType: 'EXTRACTION',
content: { claims: [] },
confidence: 0,
abstentionReason: 'NO_CLAIM',
claims: [],
evidence,
modelRun,
};
}
return {
artifactType: 'EXTRACTION',
content: { claims: [event] },
confidence: 0.7,
explanation: 'candidate event extracted (unconfirmed)',
// Scenario 7: never CONFIRMED/OFFICIAL without a deterministic confirmation.
claims: [{ claimType: 'EVENT', claimJson: event, confidence: 0.7 }],
evidence,
modelRun,
};
}
private flags(text: string): string[] {
const t = text.toLowerCase();
const out: string[] = [];
if (t.includes('party') || /\b(9|10|11)\s*pm\b/.test(t)) out.push('AFTER_HOURS_EVENT');
if (t.includes('sale') || t.includes('promo') || t.includes('discount')) out.push('PROMOTION');
return out;
}
private extractEvent(text: string): { title: string; when: string; certainty: string } | null {
const t = text.toLowerCase();
const timeMatch = t.match(/\b(9|10|11|12|[1-8])\s*(am|pm)\b/);
if (t.includes('party') || t.includes('event') || t.includes('meeting') || timeMatch) {
const title = t.includes('party') ? 'party' : t.includes('meeting') ? 'meeting' : 'event';
return { title, when: timeMatch ? timeMatch[0] : 'unspecified', certainty: 'DISCUSSION' };
}
return null;
}
private run(req: InferenceRequest, text: string): InferenceModelRun {
const tokensIn = text.length === 0 ? 0 : text.trim().split(/\s+/).length;
const model =
req.jobType === 'CLASSIFY' ? 'golden-classifier' : req.jobType === 'SUMMARIZE' ? 'golden-summarizer' : 'golden-extractor';
return {
model,
modelVersion: '1',
promptVersion: 'p1',
tokensIn,
tokensOut: Math.max(1, Math.ceil(tokensIn / 4)),
costUnits: this.costOverride ?? Math.max(1, Math.ceil(text.length / 10)),
latencyMs: 1,
};
}
}
export function makeFakeInference(): FakeInference {
return new FakeInference();
}
@@ -0,0 +1,52 @@
import type { AiJobType } from '@insignia/iios-contracts';
/**
* The golden prompt/eval registry for the deterministic AI provider (P7). Each row
* is an input + the invariant the eval asserts. This is the offline eval set the
* atlas requires; it also pins the abstention / Scenario-7 behaviour.
*/
export interface AiGoldenCase {
name: string;
jobType: AiJobType;
text: string;
expect: {
artifactType: string;
claimTypes: string[];
abstained: boolean;
/** For EXTRACT event cases: the certainty must never exceed DISCUSSION (Scenario 7). */
maxCertainty?: string;
};
}
export const AI_GOLDEN: AiGoldenCase[] = [
{
name: 'classify after-hours party',
jobType: 'CLASSIFY',
text: 'There is a party at 9 PM tonight!',
expect: { artifactType: 'CLASSIFICATION', claimTypes: ['MODERATION_FLAG'], abstained: false },
},
{
name: 'classify safe message',
jobType: 'CLASSIFY',
text: 'Community cleanup on Saturday morning.',
expect: { artifactType: 'CLASSIFICATION', claimTypes: [], abstained: false },
},
{
name: 'summarize',
jobType: 'SUMMARIZE',
text: 'The committee met today. We discussed the budget. Next steps to follow.',
expect: { artifactType: 'SUMMARY', claimTypes: [], abstained: false },
},
{
name: 'extract candidate event stays DISCUSSION',
jobType: 'EXTRACT',
text: 'Maybe we do a party at 9 PM?',
expect: { artifactType: 'EXTRACTION', claimTypes: ['EVENT'], abstained: false, maxCertainty: 'DISCUSSION' },
},
{
name: 'extract abstains when no claim',
jobType: 'EXTRACT',
text: 'Thanks everyone, appreciate it.',
expect: { artifactType: 'EXTRACTION', claimTypes: [], abstained: true },
},
];
+1
View File
@@ -2,3 +2,4 @@ export * from './fakes';
export * from './replay';
export { portalMessageBasic } from './fixtures/portal-message-basic';
export { emailFromUnknown } from './fixtures/email-from-unknown';
export { AI_GOLDEN, type AiGoldenCase } from './fixtures/ai-golden';