Skip to content

Scanner types

AttackHypothesis combines an objective, rationale, severity, setup primitives, trigger, check, and provenance/tags. ValidationResult adds confirmation status, rates, attempts/invalid runs, recipe/run links, diagnostics, timing, and corrections. The serialization helpers preserve recipe-shaped data.

ScanCost.total_usd uses the repository's static price table. Unknown models contribute zero; this is an estimate from captured usage, not complete provider/workflow billing.

types

EffectivePromptContext dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class EffectivePromptContext:
    workflow_id: str
    provider: str
    reconstructed_prompt: str
    tool_restrictions: list[str]
    has_persist_credentials: bool
    trigger_event: str

workflow_id instance-attribute

workflow_id: str

provider instance-attribute

provider: str

reconstructed_prompt instance-attribute

reconstructed_prompt: str

tool_restrictions instance-attribute

tool_restrictions: list[str]

has_persist_credentials instance-attribute

has_persist_credentials: bool

trigger_event instance-attribute

trigger_event: str

SetupStep dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class SetupStep:
    primitive: str
    args: dict[str, Any] = field(default_factory=dict)

primitive instance-attribute

primitive: str

args class-attribute instance-attribute

args: dict[str, Any] = field(default_factory=dict)

TriggerSpec dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class TriggerSpec:
    event_type: str
    data: dict[str, Any] = field(default_factory=dict)

event_type instance-attribute

event_type: str

data class-attribute instance-attribute

data: dict[str, Any] = field(default_factory=dict)

SuccessCheck dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class SuccessCheck:
    kind: str
    args: dict[str, Any] = field(default_factory=dict)

kind instance-attribute

kind: str

args class-attribute instance-attribute

args: dict[str, Any] = field(default_factory=dict)

AttackHypothesis dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class AttackHypothesis:
    id: str
    mitre_category: str
    attack_goal: str
    rationale: str
    severity: str
    setup: list[SetupStep] = field(default_factory=list)
    trigger: TriggerSpec | None = None
    success_check: SuccessCheck | None = None
    tags: list[str] = field(default_factory=list)
    seeded_from: str | None = None

id instance-attribute

id: str

mitre_category instance-attribute

mitre_category: str

attack_goal instance-attribute

attack_goal: str

rationale instance-attribute

rationale: str

severity instance-attribute

severity: str

setup class-attribute instance-attribute

setup: list[SetupStep] = field(default_factory=list)

trigger class-attribute instance-attribute

trigger: TriggerSpec | None = None

success_check class-attribute instance-attribute

success_check: SuccessCheck | None = None

tags class-attribute instance-attribute

tags: list[str] = field(default_factory=list)

seeded_from class-attribute instance-attribute

seeded_from: str | None = None

ValidationResult dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class ValidationResult:
    hypothesis: AttackHypothesis
    status: str
    failure_reason: str | None
    success_rate: str
    iteration: int
    discard_reason: str | None
    run_ids: list[str]
    payload_used: str
    suggested_mitigation: str
    billable_minutes: float
    wall_seconds: float
    evaluator_correction: str | None = None
    recipe_path: str | None = None
    attempted_runs: int = 0
    invalid_runs: int = 0
    diagnostics: list[dict] = field(default_factory=list)

hypothesis instance-attribute

hypothesis: AttackHypothesis

status instance-attribute

status: str

failure_reason instance-attribute

failure_reason: str | None

success_rate instance-attribute

success_rate: str

iteration instance-attribute

iteration: int

discard_reason instance-attribute

discard_reason: str | None

run_ids instance-attribute

run_ids: list[str]

payload_used instance-attribute

payload_used: str

suggested_mitigation instance-attribute

suggested_mitigation: str

billable_minutes instance-attribute

billable_minutes: float

wall_seconds instance-attribute

wall_seconds: float

evaluator_correction class-attribute instance-attribute

evaluator_correction: str | None = None

recipe_path class-attribute instance-attribute

recipe_path: str | None = None

attempted_runs class-attribute instance-attribute

attempted_runs: int = 0

invalid_runs class-attribute instance-attribute

invalid_runs: int = 0

diagnostics class-attribute instance-attribute

diagnostics: list[dict] = field(default_factory=list)

MemoryEntry dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class MemoryEntry:
    provider: str
    mitre_category: str
    recipe_fingerprint: str
    recipe_template: dict
    attack_goal: str
    status: str
    failure_reason: str | None = None
    evaluator_correction: str | None = None
    tags: list[str] = field(default_factory=list)
    workflow_ids: list[str] = field(default_factory=list)
    first_seen: str = ""
    source: str | None = None

provider instance-attribute

provider: str

mitre_category instance-attribute

mitre_category: str

recipe_fingerprint instance-attribute

recipe_fingerprint: str

recipe_template instance-attribute

recipe_template: dict

attack_goal instance-attribute

attack_goal: str

status instance-attribute

status: str

failure_reason class-attribute instance-attribute

failure_reason: str | None = None

evaluator_correction class-attribute instance-attribute

evaluator_correction: str | None = None

tags class-attribute instance-attribute

tags: list[str] = field(default_factory=list)

workflow_ids class-attribute instance-attribute

workflow_ids: list[str] = field(default_factory=list)

first_seen class-attribute instance-attribute

first_seen: str = ''

source class-attribute instance-attribute

source: str | None = None

ScanCost dataclass

Source code in src/benchmark/scanner/types.py
@dataclass
class ScanCost:
    token_usage_by_model: dict[str, dict[str, int]] = field(default_factory=dict)
    total_billable_minutes: float = 0.0
    total_wall_seconds: float = 0.0

    @property
    def total_input_tokens(self) -> int:
        return sum(v.get("input", 0) for v in self.token_usage_by_model.values())

    @property
    def total_output_tokens(self) -> int:
        return sum(v.get("output", 0) for v in self.token_usage_by_model.values())

    @property
    def total_usd(self) -> float:
        return cost_for_usage(self.token_usage_by_model)

token_usage_by_model class-attribute instance-attribute

token_usage_by_model: dict[str, dict[str, int]] = field(default_factory=dict)

total_billable_minutes class-attribute instance-attribute

total_billable_minutes: float = 0.0

total_wall_seconds class-attribute instance-attribute

total_wall_seconds: float = 0.0

total_input_tokens property

total_input_tokens: int

total_output_tokens property

total_output_tokens: int

total_usd property

total_usd: float

cost_for_usage

cost_for_usage(usage_by_model: dict[str, dict[str, int]]) -> float
Source code in src/benchmark/scanner/types.py
def cost_for_usage(usage_by_model: dict[str, dict[str, int]]) -> float:
    total = 0.0
    for model, t in usage_by_model.items():
        in_price, out_price = MODEL_PRICING.get(model, (0.0, 0.0))
        total += (t.get("input", 0) * in_price + t.get("output", 0) * out_price) / 1_000_000
    return total

roll_up_usage

roll_up_usage(log) -> dict[str, dict[str, int]]

Aggregate a list[LLMResponse] into {model: {"input": int, "output": int}}.

Source code in src/benchmark/scanner/types.py
def roll_up_usage(log) -> dict[str, dict[str, int]]:
    """Aggregate a list[LLMResponse] into {model: {"input": int, "output": int}}."""
    out: dict[str, dict[str, int]] = {}
    for r in log:
        bucket = out.setdefault(r.model, {"input": 0, "output": 0})
        bucket["input"] += int(r.input_tokens)
        bucket["output"] += int(r.output_tokens)
    return out

hypothesis_to_dict

hypothesis_to_dict(h: AttackHypothesis) -> dict
Source code in src/benchmark/scanner/types.py
def hypothesis_to_dict(h: AttackHypothesis) -> dict:
    return {
        "id": h.id,
        "mitre_category": h.mitre_category,
        "attack_goal": h.attack_goal,
        "rationale": h.rationale,
        "severity": h.severity,
        "setup": [{"primitive": s.primitive, "args": s.args} for s in h.setup],
        "trigger": {"event_type": h.trigger.event_type, "data": h.trigger.data} if h.trigger else None,
        "success_check": {"kind": h.success_check.kind, "args": h.success_check.args} if h.success_check else None,
        "tags": h.tags,
        "seeded_from": h.seeded_from,
    }

hypothesis_from_dict

hypothesis_from_dict(d: dict) -> AttackHypothesis
Source code in src/benchmark/scanner/types.py
def hypothesis_from_dict(d: dict) -> AttackHypothesis:
    trigger = None
    if d.get("trigger"):
        trigger = TriggerSpec(event_type=d["trigger"].get("event_type", ""), data=d["trigger"].get("data", {}))
    success_check = None
    if d.get("success_check"):
        success_check = SuccessCheck(kind=d["success_check"].get("kind", ""), args=d["success_check"].get("args", {}))
    setup = [SetupStep(primitive=s.get("primitive", ""), args=s.get("args", {})) for s in d.get("setup", [])]
    return AttackHypothesis(
        id=d.get("id", ""),
        mitre_category=d.get("mitre_category", ""),
        attack_goal=d.get("attack_goal", ""),
        rationale=d.get("rationale", ""),
        severity=d.get("severity", "medium"),
        setup=setup,
        trigger=trigger,
        success_check=success_check,
        tags=d.get("tags", []),
        seeded_from=d.get("seeded_from"),
    )