Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ This package ships **two eval cohorts** plus a documented third profile that is
| --- | --- | --- | --- | --- |
| **Golden-path** | `evals/attack_discovery_agent_builder.spec.ts` | `src/fixtures.ts` — 2 marker alerts | **Weekly** (`llm_evals.yml` sets `EVAL_GREP`) | Does the default agent **route**, **call AD tools**, and **complete the workflow**? |
| **Clean profile** | `evals/clean_profile_provided_alerts.spec.ts` | `src/scenario_registry/` — 4 chains, 16 alerts + raw events | **On-demand** (full suite or `--grep "clean profile"`) | On realistic multi-stage chains, does AD produce **quality discoveries** with context gathering? |
| **Full profile** | — (not in this package) | `ad-2.0-portable-seeder.py seed --profile full` | Manual / future follow-up | With ~150+ distractor alerts, does AD find real chains **without noise false positives**? |
| **Full profile** | `full_profile_discrimination.spec.ts` | `scenario_registry/` full seed (7 chains + 150 noise alerts) | **On-demand** — `--grep "full profile"` | With noise present, does AD find real chains **without citing noise alerts**? |

### Golden-path (`fixtures.ts`)

Expand Down Expand Up @@ -40,9 +40,21 @@ Kibana-native reimplementation of portable seeder `seed --profile clean` (does *
- **Seed label:** `ad-portable-seeder-2026-07`
- One provided-alerts eval per chain; rubric/criteria are chain-specific.

### Full profile (out of scope for this package)
### Full profile (on-demand)

Includes clean profile plus cloud scenarios (AWS, Azure, macOS) and background noise (~110 unrelated alerts + a 40-alert noisy rule cluster). Use the portable seeder locally until discrimination/FPR evaluators exist.
Includes clean profile plus cloud scenarios (`aws-compromise`, `azure-oauth`, `macos-toolkit`) and noise:

- ~110 unrelated background alerts
- 40-alert Defender signature-update cluster

**CI cadence:** on-demand only (`evals/full_profile_discrimination.spec.ts`). Not wired to weekly `llm_evals.yml`.

**FPR evaluators:** `NoiseFalsePositive`, `DiscoveryCountCap`, `MinValidatedDiscovery` — insights must not cite noise alert IDs, discovery count must stay bounded, and at least one validated discovery is required.

```bash
node scripts/evals run --suite attack-discovery-agent-builder \
--grep "full profile"
```

## Natural routing (default)

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
/*
* Copyright Elasticsearch B.V. and/or licensed to Elasticsearch B.V. under one
* or more contributor license agreements. Licensed under the Elastic License
* 2.0; you may not use this file except in compliance with the Elastic License
* 2.0.
*/

import { tags } from '@kbn/evals';
import { fullProfileDiscriminationDataset } from '../src/datasets/full_profile_discrimination';
import { evaluate } from '../src/evaluate';
import { cleanupAd2ScenarioProfile, seedAd2ScenarioProfile } from '../src/scenario_registry';

evaluate.describe(
'Attack Discovery Agent Builder — full profile (on-demand)',
{ tag: tags.stateful.classic },
() => {
evaluate.beforeAll(async ({ esClient, fetch }) => {
await seedAd2ScenarioProfile(esClient, fetch, { profile: 'full' });
await fetch('/internal/elastic_assistant/update_anonymization_fields', {
method: 'POST',
headers: { 'elastic-api-version': '1' },
});
});

evaluate.afterAll(async ({ esClient }) => {
await cleanupAd2ScenarioProfile(esClient);
});

evaluate('full profile live-retrieval noise discrimination', async ({ evaluateDataset }) => {
await evaluateDataset({ dataset: fullProfileDiscriminationDataset });
});
}
);
Original file line number Diff line number Diff line change
Expand Up @@ -19,25 +19,49 @@ export interface AgentBuilderConverseResponse {
insights?: AttackDiscovery[] | null;
}

const parseInsightsFromToolResult = (
steps: AgentBuilderConverseResponse['steps'] | undefined
): AttackDiscovery[] | null => {
// Prefer insights from the run tool's attack_discoveries field — this is the
// canonical source, immune to message block ordering issues.
if (!steps) {
return null;
}
const adStep = steps.find(
(
step
): step is typeof step & { results?: Array<{ data?: { attack_discoveries?: unknown } }> } =>
step.tool_id === 'security.attack-discovery.run' && step.type === 'tool_call'
);
const discoveries = adStep?.results?.[0]?.data?.attack_discoveries;
if (Array.isArray(discoveries) && discoveries.length > 0) {
return discoveries as AttackDiscovery[];
}
return null;
};

const parseInsightsFromMessage = (message: string): AttackDiscovery[] | null => {
// The agent returns the insights JSON inside a fenced code block at the end
// of the report. Extract the last JSON block and parse it.
// Fallback: extract insights from the last ```json fenced block in the message.
// This is fragile — if the model emits a proposed ES|QL rule after the insights
// block, this grabs the wrong one. Prefer parseInsightsFromToolResult when available.
const matches = message.match(/```json\s*([\s\S]*?)\s*```/g);
if (!matches || matches.length === 0) {
return null;
}

const lastBlock = matches[matches.length - 1].replace(/```json\s*/, '').replace(/\s*```/, '');

try {
const parsed = JSON.parse(lastBlock);
if (parsed && Array.isArray(parsed.insights)) {
return parsed.insights;
// Search from the end backwards for a block containing "insights"
for (let i = matches.length - 1; i >= 0; i--) {
const block = matches[i].replace(/```json\s*/, '').replace(/\s*```/, '');
try {
const parsed = JSON.parse(block);
if (parsed && Array.isArray(parsed.insights)) {
return parsed.insights;
}
} catch {
// not valid JSON, try next block
}
return null;
} catch {
return null;
}
return null;
};

export class AttackDiscoveryAgentBuilderChatClient {
Expand Down Expand Up @@ -77,7 +101,9 @@ export class AttackDiscoveryAgentBuilderChatClient {
steps: response.steps ?? [],
errors: [],
traceId: response.trace_id,
insights: parseInsightsFromMessage(response.response.message),
insights:
parseInsightsFromToolResult(response.steps) ??
parseInsightsFromMessage(response.response.message),
};
},
{
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
/*
* Copyright Elasticsearch B.V. and/or licensed to Elasticsearch B.V. under one
* or more contributor license agreements. Licensed under the Elastic License
* 2.0; you may not use this file except in compliance with the Elastic License
* 2.0.
*/

import { AD2_SCENARIO_SEED_LABEL } from '../scenario_registry/constants';
import { buildAd2SeedPlan } from '../scenario_registry/registry';
import type { AttackDiscoveryAgentBuilderExample } from '../types';

const fullProfilePlan = buildAd2SeedPlan({
profile: 'full',
baseTime: new Date('2026-07-01T12:00:00.000Z'),
});

const signalAlertCount =
fullProfilePlan.alerts.length - (fullProfilePlan.noiseAlertIds?.length ?? 0);

export const fullProfileDiscriminationDataset = {
name: 'attack-discovery-agent-builder: full profile (noise discrimination)',
description:
'Live-retrieval eval over portable-seeder full profile: seven signal chains plus background noise and a 40-alert Defender cluster. On-demand only — not part of weekly golden-path CI.',
examples: [
{
input: {
question: `Run Attack Discovery by retrieving open alerts seeded with label ${AD2_SCENARIO_SEED_LABEL} from the last twenty-four hours. Return validated discoveries for real attack chains and avoid turning unrelated background or Defender update alerts into discoveries.`,
triageType: 'live-retrieval',
expectedSkills: ['attack-discovery-generator'],
expectedToolPath: [
'security.attack-discovery.get_default_esql_query',
'platform.core.execute_esql',
'security.attack-discovery.run',
],
},
output: {
expectedToolPath: [
'security.attack-discovery.get_default_esql_query',
'platform.core.execute_esql',
'security.attack-discovery.run',
],
expectedWorkflowStages: ['generation', 'validation'],
expectedRetrievedAlertCount: signalAlertCount,
expectedPassedAlertCount: null,
forbiddenAlertIds: [...(fullProfilePlan.noiseAlertIds ?? [])],
maxDiscoveryCount: 12,
minValidatedDiscoveryCount: 1,
criteria: [
'At least one insight references a real attack chain host (for example wks-alice-01, dev-cloudops-04, or mbp-taylor-05).',
'Insights do not treat the Defender signature-update cluster as a coordinated attack chain.',
'Insights do not cite background-only Okta, firewall, or heartbeat noise as primary attack evidence.',
],
},
metadata: {
alertCount: fullProfilePlan.alerts.length,
fixture: 'full-profile',
seedProfile: 'full',
},
},
] satisfies AttackDiscoveryAgentBuilderExample[],
};
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,11 @@ import { createAttackDiscoveryCriteriaEvaluator } from './evaluators/attack_disc
import { createAttackDiscoveryRubricEvaluator } from './evaluators/attack_discovery_rubric_evaluator';
import { createCostPerAlertEvaluator } from './evaluators/cost_per_alert_evaluator';
import { createForbiddenToolsEvaluator } from './evaluators/forbidden_tools_evaluator';
import {
createDiscoveryCountCapEvaluator,
createMinValidatedDiscoveryEvaluator,
createNoiseFalsePositiveEvaluator,
} from './evaluators/noise_fpr_evaluator';

type AdToolResult = NonNullable<AttackDiscoveryAgentBuilderTaskOutput['adToolResult']>;

Expand Down Expand Up @@ -369,6 +374,9 @@ export const createEvaluateAttackDiscoveryAgentBuilderDataset =
createWorkflowEvidenceEvaluator(),
trajectory,
createForbiddenToolsEvaluator(),
createNoiseFalsePositiveEvaluator(),
createDiscoveryCountCapEvaluator(),
createMinValidatedDiscoveryEvaluator(),
createCostPerAlertEvaluator(),
createAttackDiscoveryBasicEvaluator(),
createAttackDiscoveryCriteriaEvaluator({ evaluators }) as Evaluator<
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
/*
* Copyright Elasticsearch B.V. and/or licensed to Elasticsearch B.V. under one
* or more contributor license agreements. Licensed under the Elastic License
* 2.0; you may not use this file except in compliance with the Elastic License
* 2.0.
*/

import {
createDiscoveryCountCapEvaluator,
createMinValidatedDiscoveryEvaluator,
createNoiseFalsePositiveEvaluator,
} from './noise_fpr_evaluator';
import type { AttackDiscoveryAgentBuilderTaskOutput } from '../types';

const baseOutput = (): AttackDiscoveryAgentBuilderTaskOutput => ({
messages: [],
steps: [],
errors: [],
insights: [
{
title: 'Chain on wks-alice-01',
summaryMarkdown: 'summary',
detailsMarkdown: 'details',
alertIds: ['ad-portable-encoded-powershell-alert-1'],
},
],
workflow: {
stages: ['generation', 'validation'],
retrievedAlertCount: 178,
passedAlertCount: null,
validatedDiscoveryCount: 1,
},
adToolResult: {
status: 'completed',
discoveryCount: 1,
},
});

describe('noise FPR evaluators', () => {
it('NoiseFalsePositive fails when insights cite forbidden noise alert IDs', async () => {
const evaluator = createNoiseFalsePositiveEvaluator();
const output = baseOutput();
output.insights = [
{
title: 'Noise',
summaryMarkdown: 'summary',
detailsMarkdown: 'details',
alertIds: ['ad-portable-loud-cluster-alert-3'],
},
];

const result = await evaluator.evaluate({
input: {} as never,
output,
expected: { forbiddenAlertIds: ['ad-portable-loud-cluster-alert-3'] },
metadata: {},
});

expect(result.score).toBe(0);
});

it('DiscoveryCountCap fails when discoveries exceed the configured cap', async () => {
const evaluator = createDiscoveryCountCapEvaluator();
const result = await evaluator.evaluate({
input: {} as never,
output: { ...baseOutput(), adToolResult: { status: 'completed', discoveryCount: 15 } },
expected: { maxDiscoveryCount: 12 },
metadata: {},
});

expect(result.score).toBe(0);
});

it('MinValidatedDiscovery passes when validated discoveries meet the floor', async () => {
const evaluator = createMinValidatedDiscoveryEvaluator();
const result = await evaluator.evaluate({
input: {} as never,
output: baseOutput(),
expected: { minValidatedDiscoveryCount: 1 },
metadata: {},
});

expect(result.score).toBe(1);
});
});
Loading